Hi all,
I am experiencing difficulties with SIESTA LDA+U version, parallel
compilation. I downloaded this LDA+U version (not the trunck version,
not the 3.0-b version) from the SIESTA website. The compilation was
successful, and the arch.make file with which I compiled the source
codes is attached to the end of this email. The executable works fine
with non LDA+U calculations, but when I include LDA+U, the parallel
version ended abruptly with the following error:
forrtl: severe (174): SIGSEGV, segmentation fault occurred
--------------------------------------------------------------------------
mpirun has exited due to process rank 7 with PID 29305 on
node v79 exiting without calling "finalize". This may
have caused other processes in the application to be
terminated by signals sent by mpirun (as reported here).
--------------------------------------------------------------------------
When I switched to serial version, it runs smoothly with the same
input file, though slowly. It also runs smoothly if I removed the LDAU
part. I also attached the input file (please see the end of this
message). I don't know if it is a problem with I/O in parallel
version, or memory allocation. I appreciate your suggestions about
this problem.
Haibo
The arch.make for parallel compilation:
#
# This file is part of the SIESTA package.
#
# Copyright (c) Fundacion General Universidad Autonoma de Madrid:
# E.Artacho, J.Gale, A.Garcia, J.Junquera, P.Ordejon, D.Sanchez-Portal
# and J.M.Soler, 1996-2006.
#
# Use of this software constitutes agreement with the full conditions
# given in the SIESTA license, as signed by all legitimate users.
#
.SUFFIXES:
.SUFFIXES: .f .F .o .a .f90 .F90
SIESTA_ARCH=x86_64-unknown-linux-gnu--Intel
FPP=
FPP_OUTPUT=
FC=mpif90
RANLIB=ranlib
SYS=nag
SP_KIND=4
DP_KIND=8
KINDS=$(SP_KIND) $(DP_KIND)
FFLAGS=-O2 -align
FPPFLAGS= -DMPI -DFC_HAVE_FLUSH -DFC_HAVE_ABORT -DCDF
LDFLAGS=
ARFLAGS_EXTRA=
FCFLAGS_fixed_f=-fixed
FCFLAGS_free_f90=-free
FPPFLAGS_fixed_F=
FPPFLAGS_free_F90=
#MKL=/apps/intel-mkl/10.1.3.027
MKL=/apps/intel-cmkl/10.1.1.019
BLAS_LIBS=-L$(MKL)/lib/em64t -lmkl_intel_lp64 -lmkl_intel_thread
-lmkl_core -liomp5 -lguide
-lpthread
LAPACK_LIBS=-L$(MKL)/lib/em64t -lmkl_lapack
#dc_lapack.a liblapack.a
BLACS_LIBS=-L$(MKL)/lib/em64t -lmkl_blacs_openmpi_lp64 -lmpi
SCALAPACK_LIBS=-L$(MKL)/lib/em64t -lmkl_scalapack_lp64
#COMP_LIBS=dc_lapack.a liblapack.a libblas.a
NETCDF_LIBS=/apps/netcdf/3.6.3/lib/Intel/libnetcdf.a
NETCDF_INTERFACE=libnetcdf_f90.a
LIBS=$(SCALAPACK_LIBS) $(BLACS_LIBS) $(LAPACK_LIBS) $(BLAS_LIBS) $(NETCDF_LIBS)
#SIESTA needs an F90 interface to MPI
#This will give you SIESTA's own implementation
#If your compiler vendor offers an alternative, you may change
#to it here.
MPI_INTERFACE=libmpi_f90.a
MPI_INCLUDE=.
#Dependency rules are created by autoconf according to whether
#discrete preprocessing is necessary or not.
.F.o:
$(FC) -c $(FFLAGS) $(INCFLAGS) $(FPPFLAGS) $(FPPFLAGS_fixed_F) $<
.F90.o:
$(FC) -c $(FFLAGS) $(INCFLAGS) $(FPPFLAGS) $(FPPFLAGS_free_F90) $<
.f.o:
$(FC) -c $(FFLAGS) $(INCFLAGS) $(FCFLAGS_fixed_f) $<
.f90.o:
$(FC) -c $(FFLAGS) $(INCFLAGS) $(FCFLAGS_free_f90) $<
The input .fdf file:
SystemName Goethite antiferromatic
SystemLabel Goethite
%include gtht_pao # PAOs
#kgrid_cutoff 0.0 Bohr # Gamma only
%block kgrid_Monkhorst_PacK
4 0 0 0.0
0 6 0 0.0
0 0 4 0.0
%endblock kgrid_Monkhorst_Pack
# antiferromagnetic configuration
SpinPolarized .true.
%block DM.InitSpin
1 +
2 +
3 -
4 -
%endblock DM.InitSpin
LDAU.ProjectorGenerationMethod 2
LDAU.CutoffNorm 0.9
%block LDAU.proj
Fe 1
n=3 2 E 50.0 2.5
5.5 1.0
2.3 0.15
1.0
%endblock LDAU.proj
LDAU.FirstIteration false
LDAU.ThresholdTol 1.0E-2
LDAU.PopTol 1.0E-3
LDAU.PotentialShift false
# MD part: if I know force and stress, how to move atoms, chage cell ...
MD.TypeOfRun CG
MD.VariableCell true # automatic cell optimization
MD.ConstantVolume false # automatic cell optimization
MD.NumCGSteps 120
#MD.MaxCGDispl 0.2 Bohr
MD.MaxForceTol 0.005 eV/Ang
MD.MaxStressTol 0.0005 GPa
MD.TargetPressure 0.00 GPa # GO with neglible stress
%block MD.TargetStress
1.0 1.0 1.0 0.0 0.0 0.0
%endblock MD.TargetStress
WriteMullikenPop 0 # Mulliken population is too verbose
WriteXML false
WriteCoorCerius true
# SCF loop: tolerance, density mixing ...
MeshCutoff 500.0 Ry
MaxSCFIterations 300
SCFMustConverge false
DM.Tolerance 1E-5
DM.MixingWeight 0.08
DM.NumberPulay 6
#DM.Require.Energy.Convergence .true.
#DM.EnergyTolerance 1E-6 eV # manual: change in ratio, not absolute value
#Use.New.Diagk true # should be more efficient, but under develop
OccupationFunction FD
ElectronicTemperature 0.03 eV
#OccupationFunction MP
#OccupationFunctionMPOrder 1
SolutionMethod diagon
#SolutionMethod OrderN
xc.functional GGA
xc.authors PBE
#Diag.ParallelOverK true
Diag.DivideAndConquer false
LatticeConstant 1.00000000000000 Ang
%block LatticeVectors
10.0395425562631910 0.0000000000000000 0.0000000000000000
0.0000000000000000 3.0460561429862687 0.0000000000000000
0.0000000000000000 0.0000000000000000 4.6298046706082587
%endblock LatticeVectors
NumberOfSpecies 3
%block ChemicalSpeciesLabel
1 26 Fe
2 1 H
3 8 O
%endblock ChemicalSpeciesLabel
NumberOfAtoms 16
AtomicCoordinatesFormat Fractional
%block AtomicCoordinatesAndAtomicSpecies
0.1452061082163212 0.250 0.9499402824798580 1 26 Fe
0.8547938917837783 0.750 0.0500597175201207 1 26 Fe
0.3547938799834114 0.750 0.4499400559967555 1 26 Fe
0.6452061200166881 0.250 0.5500599440032445 1 26 Fe
0.9107946071799589 0.250 0.5991885319425094 2 1 H
0.5892057141153373 0.750 0.0991850462870971 2 1 H
0.0892053928201477 0.750 0.4008114680574906 2 1 H
0.4107942858846627 0.250 0.9008149537129668 2 1 H
0.8021114098355540 0.250 0.3003995148169238 3 8 O
0.9443173854072811 0.250 0.8024275017932752 3 8 O
0.6978885215116364 0.750 0.8003988284191905 3 8 O
0.5556815982319350 0.750 0.3024331766472415 3 8 O
0.1978885901644460 0.750 0.6996004851830762 3 8 O
0.0556826145927261 0.750 0.1975724982067888 3 8 O
0.3021114784883636 0.250 0.1996011715808521 3 8 O
0.4443184017680650 0.250 0.6975668233527514 3 8 O
%endblock AtomicCoordinatesAndAtomicSpecies