10 #define MULT_GXr(u0,u1,u2,u3,u4,u5,v0,v1,v2,v3,v4,v5) (u0*v0-u1*v1 + u2*v2-u3*v3 + u4*v4-u5*v5)
11 #define MULT_GXi(u0,u1,u2,u3,u4,u5,v0,v1,v2,v3,v4,v5) (u0*v1+u1*v0 + u2*v3+u3*v2 + u4*v5+u5*v4)
12 #define MULT_GDXr(u0,u1,u2,u3,u4,u5,v0,v1,v2,v3,v4,v5) (u0*v0+u1*v1 + u2*v2+u3*v3 + u4*v4+u5*v5)
13 #define MULT_GDXi(u0,u1,u2,u3,u4,u5,v0,v1,v2,v3,v4,v5) (u0*v1-u1*v0 + u2*v3-u3*v2 + u4*v5-u5*v4)
15 #define EXT_IMG_R(v1r,v1i,v2r,v2i,w1r,w1i,w2r,w2i) (v1r*w2r - v1i*w2i - v2r*w1r + v2i*w1i)
16 #define EXT_IMG_I(v1r,v1i,v2r,v2i,w1r,w1i,w2r,w2i) (- v1r*w2i - v1i*w2r + v2r*w1i + v2i*w1r)
23 const int ieo,
const int jeo,
24 real_t kappa,
int *Nsize,
int *bc,
int iflag)
30 int Nst = Nx * Ny * Nz * Nt;
33 int nvst =
NVC *
ND * Nst_pad;
34 int ngst =
NDF * Nst_pad * 2 *
NDIM;
39 #pragma acc data present(v2[0:nvst], u_up[0:ngst], u_dn[0:ngst], \
40 v1[0:nvst], x1[0:nvst]) \
41 copyin(bc[0:4], kappa, ieo, jeo, iflag, \
42 Nx, Ny, Nz, Nt, Nst, Nst_pad)
43 #pragma acc parallel num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
46 int Nxyz = Nx * Ny * Nz;
48 #pragma acc loop gang worker vector
49 for(
int site = 0; site < Nst; ++site){
51 int iy = (site/Nx) % Ny;
52 int iz = (site/(Nxy)) % Nz;
54 int keo = (jeo +
iy +
iz +
it) % 2;
143 const int ieo,
const int jeo,
144 real_t kappa,
int *Nsize,
int *bc,
int iflag)
150 int Nst = Nx * Ny * Nz * Nt;
153 int nvst =
NVC *
ND * Nst_pad;
154 int ngst =
NDF * Nst_pad * 2 *
NDIM;
156 real_t *RESTRICT u_up = u;
157 real_t *RESTRICT u_dn = u;
159 #pragma acc data present(v2[0:nvst], u_up[0:ngst], u_dn[0:ngst], \
160 v1[0:nvst], x1[0:nvst]) \
161 copyin(bc[0:4], kappa, ieo, jeo, iflag, \
162 Nx, Ny, Nz, Nt, Nst, Nst_pad)
163 #pragma acc parallel num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
166 #pragma acc loop gang worker vector
167 for(
int site = 0; site < Nst; ++site){
169 int iy = (site/Nx) % Ny;
170 int iz = (site/(Nx*Ny)) % Nz;
171 int it = site/(Nx*Ny*Nz);
172 int keo = (jeo +
iy +
iz +
it) % 2;
175 int Nxyz = Nx * Ny * Nz;
267 const int ieo,
const int jeo,
268 int *Nsize,
int *bc,
int *do_comm,
int Nc)
274 int Nst = Nx * Ny * Nz * Nt;
277 int size =
NVC *
ND * Nst_pad;
278 int size_u =
NDF * Nst_pad * 2 *
NDIM;
284 #pragma acc data present(buf_xp[0:size_bx], buf_xm[0:size_bx], \
285 buf_yp[0:size_by], buf_ym[0:size_by], \
286 buf_zp[0:size_bz], buf_zm[0:size_bz], \
287 buf_tp[0:size_bt], buf_tm[0:size_bt], \
288 u[0:size_u], v1[0:size]) \
289 copyin(bc[0:4], do_comm[0:4], ieo, jeo, \
290 Nst, Nst_pad, Nx, Ny, Nz, Nt)
296 #pragma acc parallel async \
297 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
299 int Nyzt = Ny * Nz * Nt;
301 #pragma acc loop gang worker vector
308 int keo = (jeo +
iy +
iz +
it) % 2;
313 for(
int ic = 0; ic <
NC; ++ic){
329 #pragma acc parallel async \
330 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
332 int Nyzt = Ny * Nz * Nt;
334 #pragma acc loop gang worker vector
341 int istu = ist + Nst_pad * (1-ieo + 2*0);
342 int keo = (jeo +
iy +
iz +
it) % 2;
348 for(
int ic = 0; ic <
NC; ++ic){
359 for(
int ic = 0; ic <
NC; ++ic){
361 for(
int ic2 = 0; ic2 <
NC; ++ic2){
362 ut[2*ic2 ] = u[
IDX2_G_R(ic, ic2, istu)];
363 ut[2*ic2+1] = - u[
IDX2_G_I(ic, ic2, istu)];
366 wt1[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
368 wt1[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
370 wt2[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
372 wt2[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
375 buf_xm[
IDXBF_R(ic, 0, iyzt2)] = wt1[0];
376 buf_xm[
IDXBF_I(ic, 0, iyzt2)] = wt1[1];
377 buf_xm[
IDXBF_R(ic, 1, iyzt2)] = wt2[0];
378 buf_xm[
IDXBF_I(ic, 1, iyzt2)] = wt2[1];
392 #pragma acc parallel async \
393 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
397 #pragma acc loop gang worker vector collapse(2)
399 for(
int ix = 0;
ix < Nx; ++
ix){
401 int ist =
ix + Nx * (
iy + Ny *
izt);
402 int ixzt =
ix + Nx *
izt;
405 for(
int ic = 0; ic <
NC; ++ic){
422 #pragma acc parallel async \
423 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
427 #pragma acc loop gang worker vector collapse(2)
429 for(
int ix = 0;
ix < Nx; ++
ix){
431 int ist =
ix + Nx * (
iy + Ny*
izt);
432 int istu = ist + Nst_pad * (1-ieo + 2*1);
433 int ixzt =
ix + Nx *
izt;
436 for(
int ic = 0; ic <
NC; ++ic){
445 for(
int ic = 0; ic <
NC; ++ic){
448 for(
int ic2 = 0; ic2 <
NC; ++ic2){
449 ut[2*ic2 ] = u[
IDX2_G_R(ic, ic2, istu)];
450 ut[2*ic2+1] = - u[
IDX2_G_I(ic, ic2, istu)];
454 wt1[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
456 wt1[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
458 wt2[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
460 wt2[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
463 buf_ym[
IDXBF_R(ic, 0, ixzt)] = wt1[0];
464 buf_ym[
IDXBF_I(ic, 0, ixzt)] = wt1[1];
465 buf_ym[
IDXBF_R(ic, 1, ixzt)] = wt2[0];
466 buf_ym[
IDXBF_I(ic, 1, ixzt)] = wt2[1];
480 #pragma acc parallel async \
481 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
485 #pragma acc loop gang worker vector collapse(2)
486 for(
int it = 0;
it < Nt; ++
it){
489 int ist =
ixy + Nxy * (
iz + Nz *
it);
490 int ixyt =
ixy + Nxy *
it;
493 for(
int ic = 0; ic <
NC; ++ic){
510 #pragma acc parallel async \
511 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
515 #pragma acc loop gang worker vector collapse(2)
516 for(
int it = 0;
it < Nt; ++
it){
519 int ist =
ixy + Nxy * (
iz + Nz *
it);
520 int istu = ist + Nst_pad * (1-ieo + 2*2);
521 int ixyt =
ixy + Nxy *
it;
524 for(
int ic = 0; ic <
NC; ++ic){
533 for(
int ic = 0; ic <
NC; ++ic){
536 for(
int ic2 = 0; ic2 <
NC; ++ic2){
537 ut[2*ic2 ] = u[
IDX2_G_R(ic, ic2, istu)];
538 ut[2*ic2+1] = - u[
IDX2_G_I(ic, ic2, istu)];
542 wt1[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
544 wt1[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
546 wt2[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
548 wt2[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
550 buf_zm[
IDXBF_R(ic, 0, ixyt)] = wt1[0];
551 buf_zm[
IDXBF_I(ic, 0, ixyt)] = wt1[1];
552 buf_zm[
IDXBF_R(ic, 1, ixyt)] = wt2[0];
553 buf_zm[
IDXBF_I(ic, 1, ixyt)] = wt2[1];
567 #pragma acc parallel async \
568 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
570 int Nxyz = Nx * Ny * Nz;
572 #pragma acc loop gang worker vector
575 int ist =
ixyz + Nxyz *
it;
577 for(
int ic = 0; ic <
NC; ++ic){
587 #pragma acc parallel async \
588 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
590 int Nxyz = Nx * Ny * Nz;
592 #pragma acc loop gang worker vector
595 int ist =
ixyz + Nxyz *
it;
596 int istu = ist + Nst_pad * (1-ieo + 2*3);
599 for(
int ic = 0; ic <
NC; ++ic){
608 for(
int ic = 0; ic <
NC; ++ic){
611 for(
int ic2 = 0; ic2 <
NC; ++ic2){
612 ut[2*ic2 ] = u[
IDX2_G_R(ic, ic2, istu)];
613 ut[2*ic2+1] = - u[
IDX2_G_I(ic, ic2, istu)];
617 wt1[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
619 wt1[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
621 wt2[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
623 wt2[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
642 #pragma acc update async host (buf_xp[0:size_bx])
643 #pragma acc update async host (buf_xm[0:size_bx])
646 #pragma acc update async host (buf_yp[0:size_by])
647 #pragma acc update async host (buf_ym[0:size_by])
650 #pragma acc update async host (buf_zp[0:size_bz])
651 #pragma acc update async host (buf_zm[0:size_bz])
654 #pragma acc update async host (buf_tp[0:size_bt])
655 #pragma acc update async host (buf_tm[0:size_bt])
669 const int ieo,
const int jeo,
670 int *Nsize,
int *bc,
int *do_comm,
int Nc)
676 int Nst = Nx * Ny * Nz * Nt;
679 int size =
NVC *
ND * Nst_pad;
680 int size_u =
NDF * 2 * Nst_pad *
NDIM;
686 #pragma acc data present(buf_xp[0:size_bx], buf_xm[0:size_bx], \
687 buf_yp[0:size_by], buf_ym[0:size_by], \
688 buf_zp[0:size_bz], buf_zm[0:size_bz], \
689 buf_tp[0:size_bt], buf_tm[0:size_bt], \
690 u[0:size_u], v1[0:size]) \
691 copyin(bc[0:4], do_comm[0:4], ieo, jeo, \
692 Nx, Ny, Nz, Nt, Nst, Nst_pad)
698 #pragma acc parallel async \
699 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
701 int Nyzt = Ny * Nz * Nt;
703 #pragma acc loop gang worker vector
710 int keo = (jeo +
iy +
iz +
it) % 2;
715 for(
int ic = 0; ic <
NC; ++ic){
731 #pragma acc parallel async \
732 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
734 int Nyzt = Ny * Nz * Nt;
736 #pragma acc loop gang worker vector
743 int istu = ist + Nst_pad * (1-ieo + 2*0);
744 int keo = (jeo +
iy +
iz +
it) % 2;
750 for(
int ic = 0; ic <
NC; ++ic){
761 for(
int ic = 0; ic <
NC; ++ic){
763 for(
int ic2 = 0; ic2 <
NC; ++ic2){
764 ut[2*ic2 ] = u[
IDX2_G_R(ic, ic2, istu)];
765 ut[2*ic2+1] = - u[
IDX2_G_I(ic, ic2, istu)];
768 wt1[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
770 wt1[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
772 wt2[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
774 wt2[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
777 buf_xm[
IDXBF_R(ic, 0, iyzt2)] = wt1[0];
778 buf_xm[
IDXBF_I(ic, 0, iyzt2)] = wt1[1];
779 buf_xm[
IDXBF_R(ic, 1, iyzt2)] = wt2[0];
780 buf_xm[
IDXBF_I(ic, 1, iyzt2)] = wt2[1];
794 #pragma acc parallel async \
795 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
799 #pragma acc loop gang worker vector collapse(2)
801 for(
int ix = 0;
ix < Nx; ++
ix){
803 int ist =
ix + Nx * (
iy + Ny *
izt);
804 int ixzt =
ix + Nx *
izt;
807 for(
int ic = 0; ic <
NC; ++ic){
824 #pragma acc parallel async \
825 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
829 #pragma acc loop gang worker vector collapse(2)
831 for(
int ix = 0;
ix < Nx; ++
ix){
833 int ist =
ix + Nx * (
iy + Ny*
izt);
834 int istu = ist + Nst_pad * (1-ieo + 2*1);
835 int ixzt =
ix + Nx *
izt;
838 for(
int ic = 0; ic <
NC; ++ic){
847 for(
int ic = 0; ic <
NC; ++ic){
850 for(
int ic2 = 0; ic2 <
NC; ++ic2){
851 ut[2*ic2 ] = u[
IDX2_G_R(ic, ic2, istu)];
852 ut[2*ic2+1] = - u[
IDX2_G_I(ic, ic2, istu)];
856 wt1[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
858 wt1[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
860 wt2[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
862 wt2[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
865 buf_ym[
IDXBF_R(ic, 0, ixzt)] = wt1[0];
866 buf_ym[
IDXBF_I(ic, 0, ixzt)] = wt1[1];
867 buf_ym[
IDXBF_R(ic, 1, ixzt)] = wt2[0];
868 buf_ym[
IDXBF_I(ic, 1, ixzt)] = wt2[1];
882 #pragma acc parallel async \
883 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
887 #pragma acc loop gang worker vector collapse(2)
888 for(
int it = 0;
it < Nt; ++
it){
891 int ist =
ixy + Nxy * (
iz + Nz *
it);
892 int ixyt =
ixy + Nxy *
it;
895 for(
int ic = 0; ic <
NC; ++ic){
912 #pragma acc parallel async \
913 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
917 #pragma acc loop gang worker vector collapse(2)
918 for(
int it = 0;
it < Nt; ++
it){
921 int ist =
ixy + Nxy * (
iz + Nz *
it);
922 int istu = ist + Nst_pad * (1-ieo + 2*2);
923 int ixyt =
ixy + Nxy *
it;
926 for(
int ic = 0; ic <
NC; ++ic){
935 for(
int ic = 0; ic <
NC; ++ic){
938 for(
int ic2 = 0; ic2 <
NC; ++ic2){
939 ut[2*ic2 ] = u[
IDX2_G_R(ic, ic2, istu)];
940 ut[2*ic2+1] = - u[
IDX2_G_I(ic, ic2, istu)];
944 wt1[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
946 wt1[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
948 wt2[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
950 wt2[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
952 buf_zm[
IDXBF_R(ic, 0, ixyt)] = wt1[0];
953 buf_zm[
IDXBF_I(ic, 0, ixyt)] = wt1[1];
954 buf_zm[
IDXBF_R(ic, 1, ixyt)] = wt2[0];
955 buf_zm[
IDXBF_I(ic, 1, ixyt)] = wt2[1];
969 #pragma acc parallel async\
970 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
972 int Nxyz = Nx * Ny * Nz;
974 #pragma acc loop gang worker vector
977 int ist =
ixyz + Nxyz *
it;
979 for(
int ivc = 0; ivc <
NVC; ++ivc){
989 #pragma acc parallel async \
990 num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
992 int Nxyz = Nx * Ny * Nz;
994 #pragma acc loop gang worker vector
997 int ist =
ixyz + Nxyz *
it;
998 int istu = ist + Nst_pad * (1-ieo + 2*3);
1001 for(
int ivc = 0; ivc <
NVC; ++ivc){
1006 for(
int ic = 0; ic <
NC; ++ic){
1009 for(
int ic2 = 0; ic2 <
NC; ++ic2){
1010 ut[2*ic2 ] = u[
IDX2_G_R(ic, ic2, istu)];
1011 ut[2*ic2+1] = - u[
IDX2_G_I(ic, ic2, istu)];
1015 wt1[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
1017 wt1[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
1019 wt2[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
1021 wt2[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
1041 #pragma acc update async host (buf_xp[0:size_bx])
1042 #pragma acc update async host (buf_xm[0:size_bx])
1045 #pragma acc update async host (buf_yp[0:size_by])
1046 #pragma acc update async host (buf_ym[0:size_by])
1049 #pragma acc update async host (buf_zp[0:size_bz])
1050 #pragma acc update async host (buf_zm[0:size_bz])
1053 #pragma acc update async host (buf_tp[0:size_bt])
1054 #pragma acc update async host (buf_tm[0:size_bt])
1067 real_t kappa,
const int ieo,
const int jeo,
1069 int *Nsize,
int *bc,
int *do_comm,
int Nc)
1075 int Nst = Nx * Ny * Nz * Nt;
1078 int size =
NVC *
ND * Nst_pad;
1079 int size_u =
NDF * Nst_pad * 2 *
NDIM;
1086 #pragma acc update async device (buf_xp[0:size_bx])
1087 #pragma acc update async device (buf_xm[0:size_bx])
1090 #pragma acc update async device (buf_yp[0:size_by])
1091 #pragma acc update async device (buf_ym[0:size_by])
1094 #pragma acc update async device (buf_zp[0:size_bz])
1095 #pragma acc update async device (buf_zm[0:size_bz])
1098 #pragma acc update async device (buf_tp[0:size_bt])
1099 #pragma acc update async device (buf_tm[0:size_bt])
1105 #pragma acc data present(buf_xp[0:size_bx], buf_xm[0:size_bx], \
1106 buf_yp[0:size_by], buf_ym[0:size_by], \
1107 buf_zp[0:size_bz], buf_zm[0:size_bz], \
1108 buf_tp[0:size_bt], buf_tm[0:size_bt], \
1109 v2[0:size], u[0:size_u]) \
1110 copyin(bc[0:4], do_comm[0:4], ieo, jeo, kappa, \
1111 Nx, Ny, Nz, Nt, Nst, Nst_pad)
1114 #pragma acc parallel num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
1116 #pragma acc loop gang worker vector
1117 for(
int ist = 0; ist < Nst; ++ist){
1119 int iy = (ist/Nx) % Ny;
1120 int iz = (ist/(Nx*Ny)) % Nz;
1121 int it = ist/(Nx*Ny*Nz);
1122 int keo = (jeo +
iy +
iz +
it) % 2;
1125 for(
int id = 0;
id <
ND; ++id){
1126 for(
int ivc = 0; ivc <
NVC; ++ivc){
1127 v2L[ivc +
NVC * id] = 0.0;
1139 int istu = ist + Nst_pad * (ieo + 2*3);
1143 for(
int ivc = 0; ivc <
NVC; ++ivc){
1148 for(
int ic = 0; ic <
NC; ++ic){
1150 for(
int ic2 = 0; ic2 <
NC; ++ic2){
1151 ut[2*ic2 ] = u[
IDX2_G_R(ic2, ic, istu)];
1152 ut[2*ic2+1] = u[
IDX2_G_I(ic2, ic, istu)];
1155 wt1[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
1157 wt1[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
1159 wt2[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
1161 wt2[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
1163 v2L[ic*2 +
ID3] += wt1[0];
1164 v2L[ic*2+1 +
ID3] += wt1[1];
1165 v2L[ic*2 +
ID4] += wt2[0];
1166 v2L[ic*2+1 +
ID4] += wt2[1];
1176 for(
int ic = 0; ic <
NC; ++ic){
1181 v2L[ic*2 +
ID1] += wt1[0];
1182 v2L[ic*2+1 +
ID1] += wt1[1];
1183 v2L[ic*2 +
ID2] += wt2[0];
1184 v2L[ic*2+1 +
ID2] += wt2[1];
1194 for(
int id = 0;
id <
ND; ++id){
1195 for(
int ivc = 0; ivc <
NVC; ++ivc){
1196 v2[
IDX2_SP(ivc,
id, ist)] += -kappa * v2L[ivc +
NVC * id];
1200 for(
int id = 0;
id <
ND; ++id){
1201 for(
int ivc = 0; ivc <
NVC; ++ivc){
1202 v2[
IDX2_SP(ivc,
id, ist)] += kappa * v2L[ivc +
NVC * id];
1221 real_t kappa,
const int ieo,
const int jeo,
1223 int *Nsize,
int *bc,
int *do_comm,
int Nc)
1229 int Nst = Nx * Ny * Nz * Nt;
1232 int size =
NVC *
ND * Nst_pad;
1233 int size_u =
NDF * Nst_pad * 2 *
NDIM;
1240 #pragma acc update async device (buf_xp[0:size_bx])
1241 #pragma acc update async device (buf_xm[0:size_bx])
1244 #pragma acc update async device (buf_yp[0:size_by])
1245 #pragma acc update async device (buf_ym[0:size_by])
1248 #pragma acc update async device (buf_zp[0:size_bz])
1249 #pragma acc update async device (buf_zm[0:size_bz])
1252 #pragma acc update async device (buf_tp[0:size_bt])
1253 #pragma acc update async device (buf_tm[0:size_bt])
1258 #pragma acc data present(buf_xp[0:size_bx], buf_xm[0:size_bx], \
1259 buf_yp[0:size_by], buf_ym[0:size_by], \
1260 buf_zp[0:size_bz], buf_zm[0:size_bz], \
1261 buf_tp[0:size_bt], buf_tm[0:size_bt], \
1262 v2[0:size], u[0:size_u]) \
1263 copyin(bc[0:4], do_comm[0:4], ieo, jeo, kappa, \
1264 Nst, Nst_pad, Nx, Ny, Nz, Nt)
1267 #pragma acc parallel num_workers(NUM_WORKERS) vector_length(VECTOR_LENGTH)
1270 #pragma acc loop gang worker vector
1271 for(
int ist = 0; ist < Nst; ++ist){
1273 int iy = (ist/Nx) % Ny;
1274 int iz = (ist/(Nx*Ny)) % Nz;
1275 int it = ist/(Nx*Ny*Nz);
1276 int keo = (jeo +
iy +
iz +
it) % 2;
1279 for(
int id = 0;
id <
ND; ++id){
1280 for(
int ivc = 0; ivc <
NVC; ++ivc){
1281 v2L[ivc +
NVC * id] = 0.0;
1293 int istu = ist + Nst_pad * (ieo + 2*3);
1298 for(
int ivc = 0; ivc <
NVC; ++ivc){
1303 for(
int ic = 0; ic <
NC; ++ic){
1305 for(
int ic2 = 0; ic2 <
NC; ++ic2){
1306 ut[2*ic2 ] = u[
IDX2_G_R(ic2, ic, istu)];
1307 ut[2*ic2+1] = u[
IDX2_G_I(ic2, ic, istu)];
1310 wt1[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
1312 wt1[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
1314 wt2[0] =
MULT_UV_R(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
1316 wt2[1] =
MULT_UV_I(ut[0], ut[1], ut[2], ut[3], ut[4], ut[5],
1319 v2L[ic*2 +
ID1] += wt1[0];
1320 v2L[ic*2+1 +
ID1] += wt1[1];
1321 v2L[ic*2 +
ID2] += wt2[0];
1322 v2L[ic*2+1 +
ID2] += wt2[1];
1323 v2L[ic*2 +
ID3] += wt1[0];
1324 v2L[ic*2+1 +
ID3] += wt1[1];
1325 v2L[ic*2 +
ID4] += wt2[0];
1326 v2L[ic*2+1 +
ID4] += wt2[1];
1336 for(
int ic = 0; ic <
NC; ++ic){
1341 v2L[ic*2 +
ID1] += wt1[0];
1342 v2L[ic*2+1 +
ID1] += wt1[1];
1343 v2L[ic*2 +
ID2] += wt2[0];
1344 v2L[ic*2+1 +
ID2] += wt2[1];
1345 v2L[ic*2 +
ID3] -= wt1[0];
1346 v2L[ic*2+1 +
ID3] -= wt1[1];
1347 v2L[ic*2 +
ID4] -= wt2[0];
1348 v2L[ic*2+1 +
ID4] -= wt2[1];
1357 for(
int id = 0;
id <
ND; ++id){
1358 for(
int ivc = 0; ivc <
NVC; ++ivc){
1359 v2[
IDX2_SP(ivc,
id, ist)] += -kappa * v2L[ivc +
NVC * id];
1363 for(
int id = 0;
id <
ND; ++id){
1364 for(
int ivc = 0; ivc <
NVC; ++ivc){
1365 v2[
IDX2_SP(ivc,
id, ist)] += kappa * v2L[ivc +
NVC * id];