我正在为相当大的代码(天气预报模型)研究英特尔 Fortran 兼容性。在 Intel Fortran(并且只有Intel Fortran)上,一些字符数据似乎会被扰乱循环并指向它的字符指针。我最终得到的字符只包含 0 和 9。字符串c
只是"00:00:00:16 (whitespace padding....)"
作为它的内容。
每当我包含任何(死)OpenMP 代码并使用-fopenmp
编译器标志时,我都能重现该问题。我对这里发生的事情有点迷茫。您能在以下最小复制器中检测到任何编程错误吗?如果没有,我想我会提交一个编译器错误。
复制器
最小的.f90
module foo
implicit none
type pointer_character
character, pointer :: c
end type
type(pointer_character), allocatable :: ptr_c(:)
integer(4) :: idx_c
contains
subroutine grpbcast_set_c(c)
implicit none
character(*), intent(in), target :: c
integer(4) :: cloc
allocate(ptr_c( 256 ))
idx_c = 0
do cloc = 1, len(c)
idx_c = idx_c + 1
ptr_c( idx_c )%c => c( cloc:cloc )
end do
print *, "testprint-2", c
print *, "testprint-3.1", ptr_c( 1 )%c
print *, "testprint-3.2", ptr_c( 2 )%c
print *, "testprint-3.3", ptr_c( 3 )%c
print *, "testprint-3.4", ptr_c( 11 )%c
end subroutine
! The following subroutine is never called, but if we include it in the module foo, it will lead to the data corruption documented in
! http://stackoverflow.com/questions/42359258/intel-fortran-substring-access-with-convert-big-endian
! If the subroutine is commented it will work
!
! In other words:
! if we comment this subroutine out, the output will be
! =====================================================
! testprint-2
! 00:00:00:16
! testprint-3.10
! testprint-3.20
! testprint-3.3:
! testprint-3.46
! =====================================================
! However if we include it, the output will be
! testprint-2
! 00:00:00:16
! testprint-3.10
! testprint-3.20
! testprint-3.30
! testprint-3.40
! =====================================================
!
! tested with: ifort 17.0.1 20161005
subroutine scatter_one_record()
implicit none
integer(4) :: k
!$OMP PARALLEL DO
do k = 1, 5
end do
!$OMP END PARALLEL DO
end subroutine
end module foo
program main
use foo, only: grpbcast_set_c
implicit none
character(len=256), target:: run_period
run_period = "00:00:00:16"
call grpbcast_set_c(run_period)
end program
生成文件
.PHONY: all
all: minimal
minimal.o: minimal.f90
ifort -fopenmp -c minimal.f90 -o minimal.o
minimal: minimal.o
ifort -fopenmp -o minimal -L./ minimal.o
福特版
> ifort --version
ifort (IFORT) 17.0.1 20161005
编译运行
make
./minimal
..
..
之前的分析,这里只为了解这个问题的评论链
有问题的循环
subroutine grpbcast_set_c(c)
use nrtype, only : rp => rp
implicit none
character(*), intent(in), target :: c
integer(4) :: cloc
do cloc = 1, len(c)
idx_c = idx_c + 1
if (idx_c > max_ptr_gbcast) stop 9
ptr_c( idx_c )%c => c( cloc:cloc )
end do
end subroutine
模块中 ptr_c 的规范
type pointer_character
character, pointer :: c
end type
type(pointer_character), private, allocatable, save :: ptr_c(:)
在调试器中看到的数据 (TotalView)
编译器命令
> mpif90 -O0 -no-ipo -g -convert big_endian -fopenmp -r8 -DUSE_MPI -I/usr/apps.sp3/mpi/openmpi/1.6.5/i2013.1.046/include -I/usr/apps.sp3/mpi/openmpi/1.6.5/i2013.1.046/lib -I [NUSDAS13_PATH]/src -I [NETCDF_PATH]/include -DUSE_MPI -c mpi_comm.f90 -o mpi_comm.o
> mpif90 --version
ifort (IFORT) 14.0.2 20140120
Copyright (C) 1985-2014 Intel Corporation. All rights reserved.
更新
我刚刚尝试了最新的英特尔编译器版本:
> mpif90 --version
> ifort (IFORT) 17.0.1 20161005
结果仍然是错误的,但不同!这次我把所有的字符都归零了。
更新 2
我一直在尝试使用最小的复制器(见下文)来重现该问题,但到目前为止没有运气,即以这种方式运行时不会出现错误。但是,有一点很有趣:TotalView 显示数组的第一个地址略有不同:
我想这一定是 input 的问题c
,即使它在调试器中正确显示在grpcast_set_c
输入点。
module foo
implicit none
type pointer_character
character, pointer :: c
end type
type(pointer_character), allocatable, save :: ptr_c(:)
integer(4), save :: idx_c
contains
subroutine parmread_bcast_a( mt, aout, chktag )
implicit none
integer(4), intent(in):: mt
character(*), intent(out):: aout
character(*), intent(in):: chktag
call filetool_cardread_a( mt, aout, chktag )
write(6,'("* ",a28," = ", a)') chktag, trim(aout)
call set_ptr_c(aout)
return
end subroutine parmread_bcast_a
subroutine filetool_cardread_a( imt, aout, chktag )
implicit none
integer(4), intent(in):: imt
character(*), intent(out):: aout
character(*), intent(in), optional:: chktag
character(len=256):: tag
character(len=256):: val
character(len=256):: buff
integer(4):: idxeql
integer(4) :: is_debug = 0
cardread: do
read( imt, "(a)" ) buff
if( buff(1:1) /= "*" ) then
idxeql = index(buff, "=")
tag = adjustl(buff(:idxeql-1))
if (is_debug == 1) then
write(6,*) 'buff', trim(buff)
endif
if ( present( chktag ) ) then
if ( trim(chktag) /= trim(tag) ) then
write(6,*)"WARNING !!! the record is for ", trim(tag), &
& ", while you specified ", trim(chktag)
cycle cardread
endif
endif
val = adjustl(buff(idxeql+1:))
aout = val
return
end if
end do cardread
end subroutine filetool_cardread_a
subroutine set_ptr_c(c)
implicit none
character(*), intent(in), target :: c
integer(4) :: cloc
do cloc = 1, len(c)
idx_c = idx_c + 1
if (idx_c > 10000) stop 9
ptr_c( idx_c )%c => c( cloc:cloc )
end do
end subroutine
end module
program main
use foo, only: ptr_c, idx_c, parmread_bcast_a
implicit none
character(len=256), save:: run_period
allocate(ptr_c( 10000 ))
idx_c = 0
open( 100, file='sample.conf', form='formatted')
call parmread_bcast_a(100, run_period, "run_period")
print *, "run period", run_period
print *, ptr_c( 1 )%c
print *, ptr_c( 2 )%c
print *, ptr_c( 3 )%c
print *, ptr_c( 11 )%c
end program
示例.conf
run_period = 00:00:00:16