GPU implementation of the vector mortar exchange; the kernels treat the (variable, direction) pairs as 3*nvar independent trace lines.
| Type | Intent | Optional | Attributes | Name | ||
|---|---|---|---|---|---|---|
| class(MappedVector3D), | intent(inout) | :: | this | |||
| type(Mesh3D), | intent(inout) | :: | mesh |
subroutine MortarExchange_MappedVector3D(this,mesh)
!! GPU implementation of the vector mortar exchange; the kernels treat the
!! (variable, direction) pairs as 3*nvar independent trace lines.
implicit none
class(MappedVector3D),intent(inout) :: this
type(Mesh3D),intent(inout) :: mesh
! Local
integer :: offset
integer(c_size_t) :: buffSize
offset = mesh%decomp%offsetElem(mesh%decomp%rankId+1)
if(.not. c_associated(this%mortarBuff_gpu)) then
! The MortarFlip_3D kernel stages one face through shared memory
! (MORTAR3D_MAXNP in SELF_Mortar.cpp bounds the block size).
if(this%interp%N+1 > 16) then
print*,__FILE__,' : Error : 3D mortar kernels support N+1 <= 16.'
stop 1
endif
buffSize = int(this%interp%N+1,c_size_t)*(this%interp%N+1)*8* &
mesh%nMortars*this%nvar*3*prec
call gpuCheck(hipMalloc(this%mortarBuff_gpu,buffSize))
endif
if(mesh%decomp%mpiEnabled) then
call this%MPIMortarExchangeAsync(mesh)
endif
call MortarGather_3D_gpu(this%mortarBuff_gpu,this%boundary_gpu, &
mesh%mortarInfo_gpu,mesh%decomp%elemToRank_gpu, &
mesh%decomp%rankId,offset,this%interp%N,3*this%nvar, &
mesh%nMortars,this%nelem)
if(mesh%decomp%mpiEnabled) then
call mesh%decomp%FinalizeMPIExchangeAsync()
call MortarFlip_3D_gpu(this%mortarBuff_gpu,mesh%mortarInfo_gpu, &
mesh%decomp%elemToRank_gpu,mesh%decomp%rankId, &
this%interp%N,3*this%nvar,mesh%nMortars)
endif
call MortarScatter_3D_gpu(this%extBoundary_gpu,this%mortarBuff_gpu, &
this%interp%mortarR_gpu,this%interp%mortarP_gpu, &
mesh%mortarInfo_gpu,mesh%decomp%elemToRank_gpu, &
mesh%decomp%rankId,offset,this%interp%N,3*this%nvar, &
mesh%nMortars,this%nelem)
endsubroutine MortarExchange_MappedVector3D