Bridge++  Ver.2.1.3
mult_Wilson_inline_openacc-inc.h
Go to the documentation of this file.
1 
10 #ifndef MULT_WILSON_OPENACC_INLINE_INCLUDED
11 #define MULT_WILSON_OPENACC_INLINE_INCLUDED
12 
13 #define MULT_UV_R(u0,u1,u2,u3,u4,u5,v0,v1,v2,v3,v4,v5) (u0*v0-u1*v1 + u2*v2-u3*v3 + u4*v4-u5*v5)
14 #define MULT_UV_I(u0,u1,u2,u3,u4,u5,v0,v1,v2,v3,v4,v5) (u0*v1+u1*v0 + u2*v3+u3*v2 + u4*v5+u5*v4)
15 
16 #define MULT_GXr(u0,u1,u2,u3,u4,u5,v0,v1,v2,v3,v4,v5) (u0*v0-u1*v1 + u2*v2-u3*v3 + u4*v4-u5*v5)
17 #define MULT_GXi(u0,u1,u2,u3,u4,u5,v0,v1,v2,v3,v4,v5) (u0*v1+u1*v0 + u2*v3+u3*v2 + u4*v5+u5*v4)
18 #define MULT_GDXr(u0,u1,u2,u3,u4,u5,v0,v1,v2,v3,v4,v5) (u0*v0+u1*v1 + u2*v2+u3*v3 + u4*v4+u5*v5)
19 #define MULT_GDXi(u0,u1,u2,u3,u4,u5,v0,v1,v2,v3,v4,v5) (u0*v1-u1*v0 + u2*v3-u3*v2 + u4*v5-u5*v4)
20 
21 #define EXT_IMG_R(v1r,v1i,v2r,v2i,w1r,w1i,w2r,w2i) (v1r*w2r - v1i*w2i - v2r*w1r + v2i*w1i)
22 #define EXT_IMG_I(v1r,v1i,v2r,v2i,w1r,w1i,w2r,w2i) (- v1r*w2i - v1i*w2r + v2r*w1i + v2i*w1r)
23 
24 
25 //====================================================================
26 inline void load_u(real_t* ut, real_t* up, int site)
27 {
28 #ifdef SU3_3RD_ROW_RECONST
29  for(int idf = 0; idf < 2*NVC; ++idf){
30  ut[idf] = up[IDX2(NDF, idf, site)];
31  }
32 
33  ut[12] = EXT_IMG_R(ut[2], ut[3], ut[4], ut[5], ut[ 8], ut[ 9], ut[10], ut[11]);
34  ut[13] = EXT_IMG_I(ut[2], ut[3], ut[4], ut[5], ut[ 8], ut[ 9], ut[10], ut[11]);
35  ut[14] = EXT_IMG_R(ut[4], ut[5], ut[0], ut[1], ut[10], ut[11], ut[ 6], ut[ 7]);
36  ut[15] = EXT_IMG_I(ut[4], ut[5], ut[0], ut[1], ut[10], ut[11], ut[ 6], ut[ 7]);
37  ut[16] = EXT_IMG_R(ut[0], ut[1], ut[2], ut[3], ut[ 6], ut[ 7], ut[ 8], ut[ 9]);
38  ut[17] = EXT_IMG_I(ut[0], ut[1], ut[2], ut[3], ut[ 6], ut[ 7], ut[ 8], ut[ 9]);
39 #else
40  for(int idf = 0; idf < NDF; ++idf){
41  ut[idf] = up[IDX2(NDF, idf, site)];
42  }
43 #endif
44 
45 }
46 
47 //====================================================================
48 inline void mult_wilson_xpb(real_t* RESTRICT vt, real_t* RESTRICT ut,
49  real_t* RESTRICT wt)
50 {
53 
54  vt1_0 = wt[ID1 + 0] - wt[ID4 + 1];
55  vt1_1 = wt[ID1 + 1] + wt[ID4 + 0];
56  vt1_2 = wt[ID1 + 2] - wt[ID4 + 3];
57  vt1_3 = wt[ID1 + 3] + wt[ID4 + 2];
58  vt1_4 = wt[ID1 + 4] - wt[ID4 + 5];
59  vt1_5 = wt[ID1 + 5] + wt[ID4 + 4];
60 
61  vt2_0 = wt[ID2 + 0] - wt[ID3 + 1];
62  vt2_1 = wt[ID2 + 1] + wt[ID3 + 0];
63  vt2_2 = wt[ID2 + 2] - wt[ID3 + 3];
64  vt2_3 = wt[ID2 + 3] + wt[ID3 + 2];
65  vt2_4 = wt[ID2 + 4] - wt[ID3 + 5];
66  vt2_5 = wt[ID2 + 5] + wt[ID3 + 4];
67 
68  for(int ic = 0; ic < NC; ++ic){
69 
71  real_t* up = &ut[NVC * ic];
72 
73  wt1r = MULT_GXr(up[0], up[1], up[2], up[3], up[4], up[5],
75  wt1i = MULT_GXi(up[0], up[1], up[2], up[3], up[4], up[5],
77  wt2r = MULT_GXr(up[0], up[1], up[2], up[3], up[4], up[5],
79  wt2i = MULT_GXi(up[0], up[1], up[2], up[3], up[4], up[5],
81 
82  vt[ID1 + 2*ic ] += wt1r;
83  vt[ID1 + 2*ic+1] += wt1i;
84  vt[ID2 + 2*ic ] += wt2r;
85  vt[ID2 + 2*ic+1] += wt2i;
86  vt[ID3 + 2*ic ] += wt2i;
87  vt[ID3 + 2*ic+1] += -wt2r;
88  vt[ID4 + 2*ic ] += wt1i;
89  vt[ID4 + 2*ic+1] += -wt1r;
90  }
91 
92 }
93 
94 //====================================================================
95 inline void mult_wilson_xp1(real_t* vt, real_t* wt)
96 {
97  vt[ 0] = wt[ID1 + 0] - wt[ID4 + 1];
98  vt[ 1] = wt[ID1 + 1] + wt[ID4 + 0];
99  vt[ 2] = wt[ID1 + 2] - wt[ID4 + 3];
100  vt[ 3] = wt[ID1 + 3] + wt[ID4 + 2];
101  vt[ 4] = wt[ID1 + 4] - wt[ID4 + 5];
102  vt[ 5] = wt[ID1 + 5] + wt[ID4 + 4];
103 
104  vt[ 6] = wt[ID2 + 0] - wt[ID3 + 1];
105  vt[ 7] = wt[ID2 + 1] + wt[ID3 + 0];
106  vt[ 8] = wt[ID2 + 2] - wt[ID3 + 3];
107  vt[ 9] = wt[ID2 + 3] + wt[ID3 + 2];
108  vt[10] = wt[ID2 + 4] - wt[ID3 + 5];
109  vt[11] = wt[ID2 + 5] + wt[ID3 + 4];
110 }
111 
112 //====================================================================
113 inline void mult_wilson_xp2(real_t* vt, real_t* ut, real_t* wt)
114 {
115  real_t* wt1 = &wt[0];
116  real_t* wt2 = &wt[NVC];
117 
118  for(int ic = 0; ic < NC; ++ic){
119 
120  real_t wt1r, wt1i, wt2r, wt2i;
121  real_t* up = &ut[NVC * ic];
122 
123  wt1r = MULT_GXr( up[0], up[1], up[2], up[3], up[4], up[5],
124  wt1[0], wt1[1], wt1[2], wt1[3], wt1[4], wt1[5]);
125  wt1i = MULT_GXi( up[0], up[1], up[2], up[3], up[4], up[5],
126  wt1[0], wt1[1], wt1[2], wt1[3], wt1[4], wt1[5]);
127  wt2r = MULT_GXr( up[0], up[1], up[2], up[3], up[4], up[5],
128  wt2[0], wt2[1], wt2[2], wt2[3], wt2[4], wt2[5]);
129  wt2i = MULT_GXi( up[0], up[1], up[2], up[3], up[4], up[5],
130  wt2[0], wt2[1], wt2[2], wt2[3], wt2[4], wt2[5]);
131 
132  vt[ID1 + 2*ic ] += wt1r;
133  vt[ID1 + 2*ic+1] += wt1i;
134  vt[ID2 + 2*ic ] += wt2r;
135  vt[ID2 + 2*ic+1] += wt2i;
136  vt[ID3 + 2*ic ] += wt2i;
137  vt[ID3 + 2*ic+1] += -wt2r;
138  vt[ID4 + 2*ic ] += wt1i;
139  vt[ID4 + 2*ic+1] += -wt1r;
140  }
141 
142 }
143 
144 //====================================================================
145 inline void mult_wilson_xmb(real_t* RESTRICT vt, real_t* RESTRICT ut,
146  real_t* RESTRICT wt)
147 {
150 
151  vt1_0 = wt[ID1 + 0] + wt[ID4 + 1];
152  vt1_1 = wt[ID1 + 1] - wt[ID4 + 0];
153  vt1_2 = wt[ID1 + 2] + wt[ID4 + 3];
154  vt1_3 = wt[ID1 + 3] - wt[ID4 + 2];
155  vt1_4 = wt[ID1 + 4] + wt[ID4 + 5];
156  vt1_5 = wt[ID1 + 5] - wt[ID4 + 4];
157 
158  vt2_0 = wt[ID2 + 0] + wt[ID3 + 1];
159  vt2_1 = wt[ID2 + 1] - wt[ID3 + 0];
160  vt2_2 = wt[ID2 + 2] + wt[ID3 + 3];
161  vt2_3 = wt[ID2 + 3] - wt[ID3 + 2];
162  vt2_4 = wt[ID2 + 4] + wt[ID3 + 5];
163  vt2_5 = wt[ID2 + 5] - wt[ID3 + 4];
164 
165  for(int ic = 0; ic < NC; ++ic){
166 
167  real_t wt1r, wt1i, wt2r, wt2i;
168  real_t* up = &ut[2 * ic];
169 
170  wt1r = MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
172  wt1i = MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
174  wt2r = MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
176  wt2i = MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
178 
179  vt[ID1 + 2*ic ] += wt1r;
180  vt[ID1 + 2*ic+1] += wt1i;
181  vt[ID2 + 2*ic ] += wt2r;
182  vt[ID2 + 2*ic+1] += wt2i;
183  vt[ID3 + 2*ic ] += -wt2i;
184  vt[ID3 + 2*ic+1] += wt2r;
185  vt[ID4 + 2*ic ] += -wt1i;
186  vt[ID4 + 2*ic+1] += wt1r;
187  }
188 
189 }
190 
191 //====================================================================
192 inline void mult_wilson_xm1(real_t* vt, real_t* ut, real_t* wt)
193 {
196 
197  vt1_0 = wt[ID1 + 0] + wt[ID4 + 1];
198  vt1_1 = wt[ID1 + 1] - wt[ID4 + 0];
199  vt1_2 = wt[ID1 + 2] + wt[ID4 + 3];
200  vt1_3 = wt[ID1 + 3] - wt[ID4 + 2];
201  vt1_4 = wt[ID1 + 4] + wt[ID4 + 5];
202  vt1_5 = wt[ID1 + 5] - wt[ID4 + 4];
203 
204  vt2_0 = wt[ID2 + 0] + wt[ID3 + 1];
205  vt2_1 = wt[ID2 + 1] - wt[ID3 + 0];
206  vt2_2 = wt[ID2 + 2] + wt[ID3 + 3];
207  vt2_3 = wt[ID2 + 3] - wt[ID3 + 2];
208  vt2_4 = wt[ID2 + 4] + wt[ID3 + 5];
209  vt2_5 = wt[ID2 + 5] - wt[ID3 + 4];
210 
211  for(int ic = 0; ic < NC; ++ic){
212  real_t* up = &ut[2 * ic];
213 
214  vt[2*ic] =
215  MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
217  vt[2*ic+1] =
218  MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
220  vt[2*ic + NVC] =
221  MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
223  vt[2*ic+1 + NVC] =
224  MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
226  }
227 
228 }
229 
230 //====================================================================
231 inline void mult_wilson_xm2(real_t* vt, real_t* wt)
232 {
233  real_t* wt1 = &wt[0];
234  real_t* wt2 = &wt[NVC];
235  for(int ic = 0; ic < NC; ++ic){
236  vt[ID1 + 2*ic ] += wt1[2*ic];
237  vt[ID1 + 2*ic+1] += wt1[2*ic+1];
238  vt[ID2 + 2*ic ] += wt2[2*ic];
239  vt[ID2 + 2*ic+1] += wt2[2*ic+1];
240  vt[ID3 + 2*ic ] += -wt2[2*ic+1];
241  vt[ID3 + 2*ic+1] += wt2[2*ic];
242  vt[ID4 + 2*ic ] += -wt1[2*ic+1];
243  vt[ID4 + 2*ic+1] += wt1[2*ic];
244  }
245 
246 }
247 
248 //====================================================================
249 inline void mult_wilson_ypb(real_t* vt, real_t* ut, real_t* wt)
250 {
253 
254  vt1_0 = wt[ID1 + 0] + wt[ID4 + 0];
255  vt1_1 = wt[ID1 + 1] + wt[ID4 + 1];
256  vt1_2 = wt[ID1 + 2] + wt[ID4 + 2];
257  vt1_3 = wt[ID1 + 3] + wt[ID4 + 3];
258  vt1_4 = wt[ID1 + 4] + wt[ID4 + 4];
259  vt1_5 = wt[ID1 + 5] + wt[ID4 + 5];
260 
261  vt2_0 = wt[ID2 + 0] - wt[ID3 + 0];
262  vt2_1 = wt[ID2 + 1] - wt[ID3 + 1];
263  vt2_2 = wt[ID2 + 2] - wt[ID3 + 2];
264  vt2_3 = wt[ID2 + 3] - wt[ID3 + 3];
265  vt2_4 = wt[ID2 + 4] - wt[ID3 + 4];
266  vt2_5 = wt[ID2 + 5] - wt[ID3 + 5];
267 
268  for(int ic = 0; ic < NC; ++ic){
269 
270  real_t wt1r, wt1i, wt2r, wt2i;
271  real_t* up = &ut[NVC * ic];
272 
273  wt1r = MULT_GXr(up[0], up[1], up[2], up[3], up[4], up[5],
275  wt1i = MULT_GXi(up[0], up[1], up[2], up[3], up[4], up[5],
277  wt2r = MULT_GXr(up[0], up[1], up[2], up[3], up[4], up[5],
279  wt2i = MULT_GXi(up[0], up[1], up[2], up[3], up[4], up[5],
281 
282  vt[ID1 + 2*ic ] += wt1r;
283  vt[ID1 + 2*ic+1] += wt1i;
284  vt[ID2 + 2*ic ] += wt2r;
285  vt[ID2 + 2*ic+1] += wt2i;
286  vt[ID3 + 2*ic ] += -wt2r;
287  vt[ID3 + 2*ic+1] += -wt2i;
288  vt[ID4 + 2*ic ] += wt1r;
289  vt[ID4 + 2*ic+1] += wt1i;
290  }
291 
292 }
293 
294 //====================================================================
295 inline void mult_wilson_yp1(real_t* vt, real_t* wt)
296 {
297  vt[ 0] = wt[ID1 + 0] + wt[ID4 + 0];
298  vt[ 1] = wt[ID1 + 1] + wt[ID4 + 1];
299  vt[ 2] = wt[ID1 + 2] + wt[ID4 + 2];
300  vt[ 3] = wt[ID1 + 3] + wt[ID4 + 3];
301  vt[ 4] = wt[ID1 + 4] + wt[ID4 + 4];
302  vt[ 5] = wt[ID1 + 5] + wt[ID4 + 5];
303 
304  vt[ 6] = wt[ID2 + 0] - wt[ID3 + 0];
305  vt[ 7] = wt[ID2 + 1] - wt[ID3 + 1];
306  vt[ 8] = wt[ID2 + 2] - wt[ID3 + 2];
307  vt[ 9] = wt[ID2 + 3] - wt[ID3 + 3];
308  vt[10] = wt[ID2 + 4] - wt[ID3 + 4];
309  vt[11] = wt[ID2 + 5] - wt[ID3 + 5];
310 }
311 
312 //====================================================================
313 inline void mult_wilson_yp2(real_t* vt, real_t* ut, real_t* wt)
314 {
315  real_t* wt1 = &wt[0];
316  real_t* wt2 = &wt[NVC];
317 
318  for(int ic = 0; ic < NC; ++ic){
319 
320  real_t wt1r, wt1i, wt2r, wt2i;
321  real_t* up = &ut[NVC * ic];
322 
323  wt1r = MULT_GXr( up[0], up[1], up[2], up[3], up[4], up[5],
324  wt1[0], wt1[1], wt1[2], wt1[3], wt1[4], wt1[5]);
325  wt1i = MULT_GXi( up[0], up[1], up[2], up[3], up[4], up[5],
326  wt1[0], wt1[1], wt1[2], wt1[3], wt1[4], wt1[5]);
327  wt2r = MULT_GXr( up[0], up[1], up[2], up[3], up[4], up[5],
328  wt2[0], wt2[1], wt2[2], wt2[3], wt2[4], wt2[5]);
329  wt2i = MULT_GXi( up[0], up[1], up[2], up[3], up[4], up[5],
330  wt2[0], wt2[1], wt2[2], wt2[3], wt2[4], wt2[5]);
331 
332  vt[ID1 + 2*ic ] += wt1r;
333  vt[ID1 + 2*ic+1] += wt1i;
334  vt[ID2 + 2*ic ] += wt2r;
335  vt[ID2 + 2*ic+1] += wt2i;
336  vt[ID3 + 2*ic ] += -wt2r;
337  vt[ID3 + 2*ic+1] += -wt2i;
338  vt[ID4 + 2*ic ] += wt1r;
339  vt[ID4 + 2*ic+1] += wt1i;
340  }
341 
342 }
343 
344 //====================================================================
345 inline void mult_wilson_ymb(real_t* vt, real_t* ut, real_t* wt)
346 {
349 
350  vt1_0 = wt[ID1 + 0] - wt[ID4 + 0];
351  vt1_1 = wt[ID1 + 1] - wt[ID4 + 1];
352  vt1_2 = wt[ID1 + 2] - wt[ID4 + 2];
353  vt1_3 = wt[ID1 + 3] - wt[ID4 + 3];
354  vt1_4 = wt[ID1 + 4] - wt[ID4 + 4];
355  vt1_5 = wt[ID1 + 5] - wt[ID4 + 5];
356 
357  vt2_0 = wt[ID2 + 0] + wt[ID3 + 0];
358  vt2_1 = wt[ID2 + 1] + wt[ID3 + 1];
359  vt2_2 = wt[ID2 + 2] + wt[ID3 + 2];
360  vt2_3 = wt[ID2 + 3] + wt[ID3 + 3];
361  vt2_4 = wt[ID2 + 4] + wt[ID3 + 4];
362  vt2_5 = wt[ID2 + 5] + wt[ID3 + 5];
363 
364  for(int ic = 0; ic < NC; ++ic){
365 
366  real_t wt1r, wt1i, wt2r, wt2i;
367  real_t* up = &ut[2 * ic];
368 
369  wt1r = MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
371  wt1i = MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
373  wt2r = MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
375  wt2i = MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
377 
378  vt[ID1 + 2*ic ] += wt1r;
379  vt[ID1 + 2*ic+1] += wt1i;
380  vt[ID2 + 2*ic ] += wt2r;
381  vt[ID2 + 2*ic+1] += wt2i;
382  vt[ID3 + 2*ic ] += wt2r;
383  vt[ID3 + 2*ic+1] += wt2i;
384  vt[ID4 + 2*ic ] += -wt1r;
385  vt[ID4 + 2*ic+1] += -wt1i;
386  }
387 
388 }
389 
390 //====================================================================
391 inline void mult_wilson_ym1(real_t* vt, real_t* ut, real_t* wt)
392 {
395 
396  vt1_0 = wt[ID1 + 0] - wt[ID4 + 0];
397  vt1_1 = wt[ID1 + 1] - wt[ID4 + 1];
398  vt1_2 = wt[ID1 + 2] - wt[ID4 + 2];
399  vt1_3 = wt[ID1 + 3] - wt[ID4 + 3];
400  vt1_4 = wt[ID1 + 4] - wt[ID4 + 4];
401  vt1_5 = wt[ID1 + 5] - wt[ID4 + 5];
402 
403  vt2_0 = wt[ID2 + 0] + wt[ID3 + 0];
404  vt2_1 = wt[ID2 + 1] + wt[ID3 + 1];
405  vt2_2 = wt[ID2 + 2] + wt[ID3 + 2];
406  vt2_3 = wt[ID2 + 3] + wt[ID3 + 3];
407  vt2_4 = wt[ID2 + 4] + wt[ID3 + 4];
408  vt2_5 = wt[ID2 + 5] + wt[ID3 + 5];
409 
410  for(int ic = 0; ic < NC; ++ic){
411  real_t* up = &ut[2 * ic];
412 
413  vt[2*ic] =
414  MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
416  vt[2*ic+1] =
417  MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
419  vt[2*ic + NVC] =
420  MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
422  vt[2*ic+1 + NVC] =
423  MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
425  }
426 
427 }
428 
429 //====================================================================
430 inline void mult_wilson_ym2(real_t* vt, real_t* wt)
431 {
432  real_t* wt1 = &wt[0];
433  real_t* wt2 = &wt[NVC];
434  for(int ic = 0; ic < NC; ++ic){
435  vt[ID1 + 2*ic ] += wt1[2*ic];
436  vt[ID1 + 2*ic+1] += wt1[2*ic+1];
437  vt[ID2 + 2*ic ] += wt2[2*ic];
438  vt[ID2 + 2*ic+1] += wt2[2*ic+1];
439  vt[ID3 + 2*ic ] += wt2[2*ic];
440  vt[ID3 + 2*ic+1] += wt2[2*ic+1];
441  vt[ID4 + 2*ic ] += -wt1[2*ic];
442  vt[ID4 + 2*ic+1] += -wt1[2*ic+1];
443  }
444 
445 }
446 
447 //====================================================================
448 inline void mult_wilson_zpb(real_t* vt, real_t* ut, real_t* wt)
449 {
452 
453  vt1_0 = wt[ID1 + 0] - wt[ID3 + 1];
454  vt1_1 = wt[ID1 + 1] + wt[ID3 + 0];
455  vt1_2 = wt[ID1 + 2] - wt[ID3 + 3];
456  vt1_3 = wt[ID1 + 3] + wt[ID3 + 2];
457  vt1_4 = wt[ID1 + 4] - wt[ID3 + 5];
458  vt1_5 = wt[ID1 + 5] + wt[ID3 + 4];
459 
460  vt2_0 = wt[ID2 + 0] + wt[ID4 + 1];
461  vt2_1 = wt[ID2 + 1] - wt[ID4 + 0];
462  vt2_2 = wt[ID2 + 2] + wt[ID4 + 3];
463  vt2_3 = wt[ID2 + 3] - wt[ID4 + 2];
464  vt2_4 = wt[ID2 + 4] + wt[ID4 + 5];
465  vt2_5 = wt[ID2 + 5] - wt[ID4 + 4];
466 
467  for(int ic = 0; ic < NC; ++ic){
468 
469  real_t wt1r, wt1i, wt2r, wt2i;
470  real_t* up = &ut[NVC * ic];
471 
472  wt1r = MULT_GXr(up[0], up[1], up[2], up[3], up[4], up[5],
474  wt1i = MULT_GXi(up[0], up[1], up[2], up[3], up[4], up[5],
476  wt2r = MULT_GXr(up[0], up[1], up[2], up[3], up[4], up[5],
478  wt2i = MULT_GXi(up[0], up[1], up[2], up[3], up[4], up[5],
480 
481  vt[ID1 + 2*ic ] += wt1r;
482  vt[ID1 + 2*ic+1] += wt1i;
483  vt[ID2 + 2*ic ] += wt2r;
484  vt[ID2 + 2*ic+1] += wt2i;
485  vt[ID3 + 2*ic ] += wt1i;
486  vt[ID3 + 2*ic+1] += -wt1r;
487  vt[ID4 + 2*ic ] += -wt2i;
488  vt[ID4 + 2*ic+1] += wt2r;
489  }
490 
491 }
492 
493 //====================================================================
494 inline void mult_wilson_zp1(real_t* vt, real_t* wt)
495 {
496  vt[ 0] = wt[ID1 + 0] - wt[ID3 + 1];
497  vt[ 1] = wt[ID1 + 1] + wt[ID3 + 0];
498  vt[ 2] = wt[ID1 + 2] - wt[ID3 + 3];
499  vt[ 3] = wt[ID1 + 3] + wt[ID3 + 2];
500  vt[ 4] = wt[ID1 + 4] - wt[ID3 + 5];
501  vt[ 5] = wt[ID1 + 5] + wt[ID3 + 4];
502 
503  vt[ 6] = wt[ID2 + 0] + wt[ID4 + 1];
504  vt[ 7] = wt[ID2 + 1] - wt[ID4 + 0];
505  vt[ 8] = wt[ID2 + 2] + wt[ID4 + 3];
506  vt[ 9] = wt[ID2 + 3] - wt[ID4 + 2];
507  vt[10] = wt[ID2 + 4] + wt[ID4 + 5];
508  vt[11] = wt[ID2 + 5] - wt[ID4 + 4];
509 }
510 
511 //====================================================================
512 inline void mult_wilson_zp2(real_t* vt, real_t* ut, real_t* wt)
513 {
514  real_t* wt1 = &wt[0];
515  real_t* wt2 = &wt[NVC];
516 
517  for(int ic = 0; ic < NC; ++ic){
518 
519  real_t wt1r, wt1i, wt2r, wt2i;
520  real_t* up = &ut[NVC * ic];
521 
522  wt1r = MULT_GXr( up[0], up[1], up[2], up[3], up[4], up[5],
523  wt1[0], wt1[1], wt1[2], wt1[3], wt1[4], wt1[5]);
524  wt1i = MULT_GXi( up[0], up[1], up[2], up[3], up[4], up[5],
525  wt1[0], wt1[1], wt1[2], wt1[3], wt1[4], wt1[5]);
526  wt2r = MULT_GXr( up[0], up[1], up[2], up[3], up[4], up[5],
527  wt2[0], wt2[1], wt2[2], wt2[3], wt2[4], wt2[5]);
528  wt2i = MULT_GXi( up[0], up[1], up[2], up[3], up[4], up[5],
529  wt2[0], wt2[1], wt2[2], wt2[3], wt2[4], wt2[5]);
530 
531  vt[ID1 + 2*ic ] += wt1r;
532  vt[ID1 + 2*ic+1] += wt1i;
533  vt[ID2 + 2*ic ] += wt2r;
534  vt[ID2 + 2*ic+1] += wt2i;
535  vt[ID3 + 2*ic ] += wt1i;
536  vt[ID3 + 2*ic+1] += -wt1r;
537  vt[ID4 + 2*ic ] += -wt2i;
538  vt[ID4 + 2*ic+1] += wt2r;
539  }
540 
541 }
542 
543 //====================================================================
544 inline void mult_wilson_zmb(real_t* vt, real_t* ut, real_t* wt)
545 {
548 
549  vt1_0 = wt[ID1 + 0] + wt[ID3 + 1];
550  vt1_1 = wt[ID1 + 1] - wt[ID3 + 0];
551  vt1_2 = wt[ID1 + 2] + wt[ID3 + 3];
552  vt1_3 = wt[ID1 + 3] - wt[ID3 + 2];
553  vt1_4 = wt[ID1 + 4] + wt[ID3 + 5];
554  vt1_5 = wt[ID1 + 5] - wt[ID3 + 4];
555 
556  vt2_0 = wt[ID2 + 0] - wt[ID4 + 1];
557  vt2_1 = wt[ID2 + 1] + wt[ID4 + 0];
558  vt2_2 = wt[ID2 + 2] - wt[ID4 + 3];
559  vt2_3 = wt[ID2 + 3] + wt[ID4 + 2];
560  vt2_4 = wt[ID2 + 4] - wt[ID4 + 5];
561  vt2_5 = wt[ID2 + 5] + wt[ID4 + 4];
562 
563  for(int ic = 0; ic < NC; ++ic){
564 
565  real_t wt1r, wt1i, wt2r, wt2i;
566  real_t* up = &ut[2 * ic];
567 
568  wt1r = MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
570  wt1i = MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
572  wt2r = MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
574  wt2i = MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
576 
577  vt[ID1 + 2*ic ] += wt1r;
578  vt[ID1 + 2*ic+1] += wt1i;
579  vt[ID2 + 2*ic ] += wt2r;
580  vt[ID2 + 2*ic+1] += wt2i;
581  vt[ID3 + 2*ic ] += -wt1i;
582  vt[ID3 + 2*ic+1] += wt1r;
583  vt[ID4 + 2*ic ] += wt2i;
584  vt[ID4 + 2*ic+1] += -wt2r;
585  }
586 
587 }
588 
589 //====================================================================
590 inline void mult_wilson_zm1(real_t* vt, real_t* ut, real_t* wt)
591 {
594 
595  vt1_0 = wt[ID1 + 0] + wt[ID3 + 1];
596  vt1_1 = wt[ID1 + 1] - wt[ID3 + 0];
597  vt1_2 = wt[ID1 + 2] + wt[ID3 + 3];
598  vt1_3 = wt[ID1 + 3] - wt[ID3 + 2];
599  vt1_4 = wt[ID1 + 4] + wt[ID3 + 5];
600  vt1_5 = wt[ID1 + 5] - wt[ID3 + 4];
601 
602  vt2_0 = wt[ID2 + 0] - wt[ID4 + 1];
603  vt2_1 = wt[ID2 + 1] + wt[ID4 + 0];
604  vt2_2 = wt[ID2 + 2] - wt[ID4 + 3];
605  vt2_3 = wt[ID2 + 3] + wt[ID4 + 2];
606  vt2_4 = wt[ID2 + 4] - wt[ID4 + 5];
607  vt2_5 = wt[ID2 + 5] + wt[ID4 + 4];
608 
609  for(int ic = 0; ic < NC; ++ic){
610  real_t* up = &ut[2 * ic];
611 
612  vt[2*ic] =
613  MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
615  vt[2*ic+1] =
616  MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
618  vt[2*ic + NVC] =
619  MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
621  vt[2*ic+1 + NVC] =
622  MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
624  }
625 
626 }
627 
628 //====================================================================
629 inline void mult_wilson_zm2(real_t* vt, real_t* wt)
630 {
631  real_t* wt1 = &wt[0];
632  real_t* wt2 = &wt[NVC];
633  for(int ic = 0; ic < NC; ++ic){
634  vt[ID1 + 2*ic ] += wt1[2*ic];
635  vt[ID1 + 2*ic+1] += wt1[2*ic+1];
636  vt[ID2 + 2*ic ] += wt2[2*ic];
637  vt[ID2 + 2*ic+1] += wt2[2*ic+1];
638  vt[ID3 + 2*ic ] += -wt1[2*ic+1];
639  vt[ID3 + 2*ic+1] += wt1[2*ic];
640  vt[ID4 + 2*ic ] += wt2[2*ic+1];
641  vt[ID4 + 2*ic+1] += -wt2[2*ic];
642  }
643 
644 }
645 
646 //====================================================================
647 inline void mult_wilson_tpb_dirac(real_t* vt, real_t* ut, real_t* wt)
648 {
651 
652  vt1_0 = 2.0 * wt[ID3 + 0];
653  vt1_1 = 2.0 * wt[ID3 + 1];
654  vt1_2 = 2.0 * wt[ID3 + 2];
655  vt1_3 = 2.0 * wt[ID3 + 3];
656  vt1_4 = 2.0 * wt[ID3 + 4];
657  vt1_5 = 2.0 * wt[ID3 + 5];
658 
659  vt2_0 = 2.0 * wt[ID4 + 0];
660  vt2_1 = 2.0 * wt[ID4 + 1];
661  vt2_2 = 2.0 * wt[ID4 + 2];
662  vt2_3 = 2.0 * wt[ID4 + 3];
663  vt2_4 = 2.0 * wt[ID4 + 4];
664  vt2_5 = 2.0 * wt[ID4 + 5];
665 
666  for(int ic = 0; ic < NC; ++ic){
667 
668  real_t wt1r, wt1i, wt2r, wt2i;
669  real_t* up = &ut[NVC * ic];
670 
671  wt1r = MULT_GXr(up[0], up[1], up[2], up[3], up[4], up[5],
673  wt1i = MULT_GXi(up[0], up[1], up[2], up[3], up[4], up[5],
675  wt2r = MULT_GXr(up[0], up[1], up[2], up[3], up[4], up[5],
677  wt2i = MULT_GXi(up[0], up[1], up[2], up[3], up[4], up[5],
679 
680  vt[ID3 + 2*ic ] += wt1r;
681  vt[ID3 + 2*ic+1] += wt1i;
682  vt[ID4 + 2*ic ] += wt2r;
683  vt[ID4 + 2*ic+1] += wt2i;
684  }
685 
686 }
687 
688 //====================================================================
690 {
691  vt[ 0] = 2.0 * wt[ID3 + 0];
692  vt[ 1] = 2.0 * wt[ID3 + 1];
693  vt[ 2] = 2.0 * wt[ID3 + 2];
694  vt[ 3] = 2.0 * wt[ID3 + 3];
695  vt[ 4] = 2.0 * wt[ID3 + 4];
696  vt[ 5] = 2.0 * wt[ID3 + 5];
697 
698  vt[ 6] = 2.0 * wt[ID4 + 0];
699  vt[ 7] = 2.0 * wt[ID4 + 1];
700  vt[ 8] = 2.0 * wt[ID4 + 2];
701  vt[ 9] = 2.0 * wt[ID4 + 3];
702  vt[10] = 2.0 * wt[ID4 + 4];
703  vt[11] = 2.0 * wt[ID4 + 5];
704 }
705 
706 //====================================================================
707 inline void mult_wilson_tp2_dirac(real_t* vt, real_t* ut, real_t* wt)
708 {
709  real_t* wt1 = &wt[0];
710  real_t* wt2 = &wt[NVC];
711 
712  for(int ic = 0; ic < NC; ++ic){
713 
714  real_t wt1r, wt1i, wt2r, wt2i;
715  real_t* up = &ut[NVC * ic];
716 
717  wt1r = MULT_GXr( up[0], up[1], up[2], up[3], up[4], up[5],
718  wt1[0], wt1[1], wt1[2], wt1[3], wt1[4], wt1[5]);
719  wt1i = MULT_GXi( up[0], up[1], up[2], up[3], up[4], up[5],
720  wt1[0], wt1[1], wt1[2], wt1[3], wt1[4], wt1[5]);
721  wt2r = MULT_GXr( up[0], up[1], up[2], up[3], up[4], up[5],
722  wt2[0], wt2[1], wt2[2], wt2[3], wt2[4], wt2[5]);
723  wt2i = MULT_GXi( up[0], up[1], up[2], up[3], up[4], up[5],
724  wt2[0], wt2[1], wt2[2], wt2[3], wt2[4], wt2[5]);
725 
726  vt[ID3 + 2*ic ] += wt1r;
727  vt[ID3 + 2*ic+1] += wt1i;
728  vt[ID4 + 2*ic ] += wt2r;
729  vt[ID4 + 2*ic+1] += wt2i;
730  }
731 
732 }
733 
734 //====================================================================
735 inline void mult_wilson_tmb_dirac(real_t* vt, real_t* ut, real_t* wt)
736 {
739 
740  vt1_0 = 2.0 * wt[ID1 + 0];
741  vt1_1 = 2.0 * wt[ID1 + 1];
742  vt1_2 = 2.0 * wt[ID1 + 2];
743  vt1_3 = 2.0 * wt[ID1 + 3];
744  vt1_4 = 2.0 * wt[ID1 + 4];
745  vt1_5 = 2.0 * wt[ID1 + 5];
746 
747  vt2_0 = 2.0 * wt[ID2 + 0];
748  vt2_1 = 2.0 * wt[ID2 + 1];
749  vt2_2 = 2.0 * wt[ID2 + 2];
750  vt2_3 = 2.0 * wt[ID2 + 3];
751  vt2_4 = 2.0 * wt[ID2 + 4];
752  vt2_5 = 2.0 * wt[ID2 + 5];
753 
754  for(int ic = 0; ic < NC; ++ic){
755 
756  real_t wt1r, wt1i, wt2r, wt2i;
757  real_t* up = &ut[2 * ic];
758 
759  wt1r = MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
761  wt1i = MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
763  wt2r = MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
765  wt2i = MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
767 
768  vt[ID1 + 2*ic ] += wt1r;
769  vt[ID1 + 2*ic+1] += wt1i;
770  vt[ID2 + 2*ic ] += wt2r;
771  vt[ID2 + 2*ic+1] += wt2i;
772  }
773 
774 }
775 
776 //====================================================================
777 inline void mult_wilson_tm1_dirac(real_t* vt, real_t* ut, real_t* wt)
778 {
781 
782  vt1_0 = 2.0 * wt[ID1 + 0];
783  vt1_1 = 2.0 * wt[ID1 + 1];
784  vt1_2 = 2.0 * wt[ID1 + 2];
785  vt1_3 = 2.0 * wt[ID1 + 3];
786  vt1_4 = 2.0 * wt[ID1 + 4];
787  vt1_5 = 2.0 * wt[ID1 + 5];
788 
789  vt2_0 = 2.0 * wt[ID2 + 0];
790  vt2_1 = 2.0 * wt[ID2 + 1];
791  vt2_2 = 2.0 * wt[ID2 + 2];
792  vt2_3 = 2.0 * wt[ID2 + 3];
793  vt2_4 = 2.0 * wt[ID2 + 4];
794  vt2_5 = 2.0 * wt[ID2 + 5];
795 
796  for(int ic = 0; ic < NC; ++ic){
797  real_t* up = &ut[2 * ic];
798 
799  vt[2*ic] =
800  MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
802  vt[2*ic+1] =
803  MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
805  vt[2*ic + NVC] =
806  MULT_GDXr(up[0], up[1], up[6], up[7], up[12], up[13],
808  vt[2*ic+1 + NVC] =
809  MULT_GDXi(up[0], up[1], up[6], up[7], up[12], up[13],
811  }
812 
813 }
814 
815 //====================================================================
817 {
818  real_t* wt1 = &wt[0];
819  real_t* wt2 = &wt[NVC];
820  for(int ic = 0; ic < NC; ++ic){
821  vt[ID1 + 2*ic ] += wt1[2*ic];
822  vt[ID1 + 2*ic+1] += wt1[2*ic+1];
823  vt[ID2 + 2*ic ] += wt2[2*ic];
824  vt[ID2 + 2*ic+1] += wt2[2*ic+1];
825  }
826 
827 }
828 
829 #endif
830 //============================================================END=====
mult_wilson_zm2
void mult_wilson_zm2(real_t *vt, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:629
mult_wilson_ymb
void mult_wilson_ymb(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:345
vt2_1
vt2_1
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:40
mult_wilson_xp2
void mult_wilson_xp2(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:113
ID1
#define ID1
Definition: fopr_Wilson_impl_SU2-inc.h:18
wt
real_t wt[NVCD]
Definition: mult_Clover_csw_chiral_openacc-inc.h:9
mult_wilson_xm2
void mult_wilson_xm2(real_t *vt, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:231
mult_wilson_xm1
void mult_wilson_xm1(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:192
mult_wilson_xp1
void mult_wilson_xp1(real_t *vt, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:95
mult_wilson_zpb
void mult_wilson_zpb(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:448
NDF
#define NDF
Definition: field_F_imp_SU2-inc.h:17
mult_wilson_tmb_dirac
void mult_wilson_tmb_dirac(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:735
vt2_5
vt2_5
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:44
vt1_2
vt1_2
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:34
mult_wilson_tp1_dirac
void mult_wilson_tp1_dirac(real_t *vt, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:689
wt1i
wt1i
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:55
vt2_2
vt2_2
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:41
mult_wilson_tpb_dirac
void mult_wilson_tpb_dirac(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:647
wt2r
wt2r
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:57
mult_wilson_zm1
void mult_wilson_zm1(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:590
MULT_GXr
#define MULT_GXr(u0, u1, u2, u3, u4, u5, v0, v1, v2, v3, v4, v5)
Definition: mult_Wilson_inline_openacc-inc.h:16
mult_wilson_yp2
void mult_wilson_yp2(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:313
ID2
#define ID2
Definition: fopr_Wilson_impl_SU2-inc.h:19
ID4
#define ID4
Definition: fopr_Wilson_impl_SU2-inc.h:21
vt1_0
vt1_0
Definition: mult_Wilson_eo_t_chiral_openacc-inc.h:18
MULT_GDXi
#define MULT_GDXi(u0, u1, u2, u3, u4, u5, v0, v1, v2, v3, v4, v5)
Definition: mult_Wilson_inline_openacc-inc.h:19
EXT_IMG_R
#define EXT_IMG_R(v1r, v1i, v2r, v2i, w1r, w1i, w2r, w2i)
Definition: mult_Wilson_inline_openacc-inc.h:21
mult_wilson_tp2_dirac
void mult_wilson_tp2_dirac(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:707
NC
#define NC
Definition: field_F_imp_SU2-inc.h:15
mult_wilson_yp1
void mult_wilson_yp1(real_t *vt, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:295
mult_wilson_ypb
void mult_wilson_ypb(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:249
wt1r
wt1r
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:53
wt2i
wt2i
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:59
mult_wilson_xpb
void mult_wilson_xpb(real_t *RESTRICT vt, real_t *RESTRICT ut, real_t *RESTRICT wt)
Definition: mult_Wilson_inline_openacc-inc.h:48
mult_wilson_zp1
void mult_wilson_zp1(real_t *vt, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:494
MULT_GXi
#define MULT_GXi(u0, u1, u2, u3, u4, u5, v0, v1, v2, v3, v4, v5)
Definition: mult_Wilson_inline_openacc-inc.h:17
mult_wilson_zmb
void mult_wilson_zmb(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:544
real_t
double real_t
Definition: bridgeACC_AField_double.cpp:14
ID3
#define ID3
Definition: fopr_Wilson_impl_SU2-inc.h:20
vt1_5
vt1_5
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:37
IDX2
#define IDX2(nin, in, ist)
Definition: define_index.h:28
mult_wilson_tm2_dirac
void mult_wilson_tm2_dirac(real_t *vt, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:816
vt2_3
vt2_3
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:42
NVC
#define NVC
Definition: fopr_Wilson_impl_SU2-inc.h:15
mult_wilson_ym1
void mult_wilson_ym1(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:391
load_u
void load_u(real_t *ut, real_t *up, int site)
Definition: mult_Wilson_inline_openacc-inc.h:26
vt1_3
vt1_3
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:35
mult_wilson_tm1_dirac
void mult_wilson_tm1_dirac(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:777
mult_wilson_ym2
void mult_wilson_ym2(real_t *vt, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:430
mult_wilson_zp2
void mult_wilson_zp2(real_t *vt, real_t *ut, real_t *wt)
Definition: mult_Wilson_inline_openacc-inc.h:512
mult_wilson_xmb
void mult_wilson_xmb(real_t *RESTRICT vt, real_t *RESTRICT ut, real_t *RESTRICT wt)
Definition: mult_Wilson_inline_openacc-inc.h:145
EXT_IMG_I
#define EXT_IMG_I(v1r, v1i, v2r, v2i, w1r, w1i, w2r, w2i)
Definition: mult_Wilson_inline_openacc-inc.h:22
vt2_4
vt2_4
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:43
vt2_0
vt2_0
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:39
MULT_GDXr
#define MULT_GDXr(u0, u1, u2, u3, u4, u5, v0, v1, v2, v3, v4, v5)
Definition: mult_Wilson_inline_openacc-inc.h:18
vt1_4
vt1_4
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:36
vt1_1
vt1_1
Definition: mult_Domainwall_eo_t_dirac_openacc-inc.h:33