Go to the documentation of this file.
12 template<
typename AFIELD>
16 template<
typename AFIELD>
32 vout.
general(m_vl,
"%s: construction\n", class_name.c_str());
38 if (repr !=
"Dirac") {
39 vout.
crucial(
" Error at %s: unsupported gamma-matrix type: %s\n",
40 class_name.c_str(), repr.c_str());
54 m_Ndf = 2 * m_Nc * m_Nc;
65 m_Nstv = m_Nst /
VLEN;
67 if (
VLENX * m_Nxv != m_Nx) {
68 vout.
crucial(m_vl,
"%s: Nx must be multiple of VLENX.\n",
72 if (
VLENY * m_Nyv != m_Ny) {
73 vout.
crucial(m_vl,
"%s: Ny must be multiple of VLENY.\n",
88 for (
int mu = 0; mu < m_Ndim; ++mu) {
91 do_comm_any += do_comm[mu];
92 vout.
general(
" do_comm[%d] = %d\n", mu, do_comm[mu]);
95 m_bdsize.resize(m_Ndim);
97 m_bdsize[0] = m_Nvc * Nd2 * m_Ny * m_Nz * m_Nt;
98 m_bdsize[1] = m_Nvc * Nd2 * m_Nx * m_Nz * m_Nt;
99 m_bdsize[2] = m_Nvc * Nd2 * m_Nx * m_Ny * m_Nt;
100 m_bdsize[3] = m_Nvc * Nd2 * m_Nx * m_Ny * m_Nz;
105 m_U.reset(
NDF, m_Nst, m_Ndim);
108 int NinF = 2 * m_Nc * m_Nd;
109 m_v2.reset(NinF, m_Nst, 1);
113 set_parameters(params);
123 template<
typename AFIELD>
126 chsend_up.resize(m_Ndim);
127 chrecv_up.resize(m_Ndim);
128 chsend_dn.resize(m_Ndim);
129 chrecv_dn.resize(m_Ndim);
131 for (
int mu = 0; mu < m_Ndim; ++mu) {
133 size_t Nvsize = m_bdsize[mu] *
sizeof(
real_t);
135 chsend_dn[mu].send_init(Nvsize, mu, -1);
136 chsend_up[mu].send_init(Nvsize, mu, 1);
138 chrecv_up[mu].recv_init(Nvsize, mu, 1);
139 chrecv_dn[mu].recv_init(Nvsize, mu, -1);
141 void *buf_up = (
void *)chsend_dn[mu].ptr();
142 chrecv_up[mu].recv_init(Nvsize, mu, 1, buf_up);
143 void *buf_dn = (
void *)chsend_up[mu].ptr();
144 chrecv_dn[mu].recv_init(Nvsize, mu, -1, buf_dn);
147 if (do_comm[mu] == 1) {
148 chset_send.append(chsend_up[mu]);
149 chset_send.append(chsend_dn[mu]);
150 chset_recv.append(chrecv_up[mu]);
151 chset_recv.append(chrecv_dn[mu]);
158 template<
typename AFIELD>
168 template<
typename AFIELD>
178 vout.
crucial(m_vl,
"Error at %s: input parameter not found.\n",
183 set_parameters(
real_t(kappa), bc);
188 template<
typename AFIELD>
190 const std::vector<int> bc)
192 assert(bc.size() == m_Ndim);
200 m_boundary.resize(m_Ndim);
201 for (
int mu = 0; mu < m_Ndim; ++mu) {
202 m_boundary[mu] = bc[mu];
206 vout.
general(m_vl,
"%s: set parameters\n", class_name.c_str());
207 vout.
general(m_vl,
" gamma-matrix type = %s\n", m_repr.c_str());
209 for (
int mu = 0; mu < m_Ndim; ++mu) {
210 vout.
general(m_vl,
" boundary[%d] = %2d\n", mu, m_boundary[mu]);
218 template<
typename AFIELD>
221 params.
set_double(
"hopping_parameter",
double(m_CKs));
223 params.
set_string(
"gamma_matrix_type", m_repr);
230 template<
typename AFIELD>
235 vout.
detailed(m_vl,
"%s: set_config is called: num_threads = %d\n",
236 class_name.c_str(), nth);
244 vout.
detailed(m_vl,
"%s: set_config finished\n", class_name.c_str());
249 template<
typename AFIELD>
262 template<
typename AFIELD>
269 if (ith == 0) m_conf = u;
280 template<
typename AFIELD>
291 template<
typename AFIELD>
302 template<
typename AFIELD>
310 }
else if (mu == 1) {
312 }
else if (mu == 2) {
314 }
else if (mu == 3) {
317 vout.
crucial(m_vl,
"%s: mult_up for %d direction is undefined.",
318 class_name.c_str(), mu);
325 template<
typename AFIELD>
333 }
else if (mu == 1) {
335 }
else if (mu == 2) {
337 }
else if (mu == 3) {
340 vout.
crucial(m_vl,
"%s: mult_dn for %d direction is undefined.",
341 class_name.c_str(), mu);
348 template<
typename AFIELD>
354 if (ith == 0) m_mode = mode;
361 template<
typename AFIELD>
369 template<
typename AFIELD>
374 }
else if (m_mode ==
"DdagD") {
376 }
else if (m_mode ==
"Ddag") {
378 }
else if (m_mode ==
"H") {
381 vout.
crucial(m_vl,
"%s: mode undefined.\n", class_name.c_str());
388 template<
typename AFIELD>
393 }
else if (m_mode ==
"DdagD") {
395 }
else if (m_mode ==
"Ddag") {
397 }
else if (m_mode ==
"H") {
400 vout.
crucial(m_vl,
"%s: mode undefined.\n", class_name.c_str());
407 template<
typename AFIELD>
415 template<
typename AFIELD>
426 template<
typename AFIELD>
436 template<
typename AFIELD>
453 template<
typename AFIELD>
459 int ith, nth, is, ns;
460 set_threadtask_mult(ith, nth, is, ns, m_Nstv);
464 for (
int site = is; site < ns; ++site) {
465 for (
int ic = 0; ic <
NC; ++ic) {
466 for (
int id = 0;
id <
ND2; ++id) {
467 int idx1 = 2 * (
id +
ND * ic) +
NVCD * site;
468 load_vec(
wt, &wp[
VLEN * idx1], 2);
469 save_vec(&vp[
VLEN * idx1],
wt, 2);
472 for (
int id =
ND2;
id <
ND; ++id) {
473 int idx1 = 2 * (
id +
ND * ic) +
NVCD * site;
474 load_vec(
wt, &wp[
VLEN * idx1], 2);
476 save_vec(&vp[
VLEN * idx1],
wt, 2);
486 template<
typename AFIELD>
497 if (do_comm_any > 0) {
498 if (ith == 0) chset_recv.start();
510 buf1_zp, buf1_zm, buf1_tp, buf1_tm,
511 up, v1, &m_boundary[0], m_Nsize, do_comm);
515 if (ith == 0) chset_send.start();
519 m_CKs, &m_boundary[0], m_Nsize, do_comm);
521 if (do_comm_any > 0) {
522 if (ith == 0) chset_recv.wait();
536 buf2_xp, buf2_xm, buf2_yp, buf2_ym,
537 buf2_zp, buf2_zm, buf2_tp, buf2_tm,
538 m_CKs, &m_boundary[0], m_Nsize, do_comm);
540 if (ith == 0) chset_send.wait();
548 template<
typename AFIELD>
565 aypx(-m_CKs, vp, wp);
572 template<
typename AFIELD>
581 template<
typename AFIELD>
584 int ith, nth, is, ns;
585 set_threadtask_mult(ith, nth, is, ns, m_Nstv);
589 for (
int site = is; site < ns; ++site) {
592 aypx_vec(a, vt,
wt,
NVCD);
599 template<
typename AFIELD>
602 int ith, nth, is, ns;
603 set_threadtask_mult(ith, nth, is, ns, m_Nstv);
608 for (
int site = is; site < ns; ++site) {
615 template<
typename AFIELD>
620 int ith, nth, is, ns;
621 set_threadtask_mult(ith, nth, is, ns, m_Nstv);
632 if (do_comm[0] > 0) {
633 for (
int site = is; site < ns; ++site) {
634 int ix = site % m_Nxv;
635 int iyzt = site / m_Nxv;
646 chrecv_up[0].start();
647 chsend_dn[0].start();
654 for (
int site = is; site < ns; ++site) {
655 int ix = site % m_Nxv;
656 int iyzt = site / m_Nxv;
659 clear_vec(v2v,
NVCD);
663 if ((
ix < m_Nxv - 1) || (do_comm[0] == 0)) {
664 int nei =
ix + 1 + m_Nxv *
iyzt;
665 if (
ix == m_Nxv - 1) nei = 0 + m_Nxv *
iyzt;
683 template<
typename AFIELD>
688 int ith, nth, is, ns;
689 set_threadtask_mult(ith, nth, is, ns, m_Nstv);
700 if (do_comm[0] > 0) {
701 for (
int site = is; site < ns; ++site) {
702 int ix = site % m_Nxv;
703 int iyzt = site / m_Nxv;
704 if (
ix == m_Nxv - 1) {
714 chrecv_dn[0].start();
715 chsend_up[0].start();
722 for (
int site = is; site < ns; ++site) {
723 int ix = site % m_Nxv;
724 int iyzt = site / m_Nxv;
729 clear_vec(v2v,
NVCD);
731 if ((
ix > 0) || (do_comm[0] == 0)) {
732 int nei =
ix - 1 + m_Nxv *
iyzt;
733 if (
ix == 0) nei = m_Nxv - 1 + m_Nxv *
iyzt;
740 shift_vec0_xfw(uL, &u[
VLEN *
NDF * site],
NDF);
753 template<
typename AFIELD>
757 int Nxy = m_Nxv * m_Nyv;
759 int ith, nth, is, ns;
760 set_threadtask_mult(ith, nth, is, ns, m_Nstv);
769 if (do_comm[1] > 0) {
770 for (
int site = is; site < ns; ++site) {
771 int ix = site % m_Nxv;
772 int iy = (site / m_Nxv) % m_Nyv;
773 int izt = site / Nxy;
784 chrecv_up[1].start();
785 chsend_dn[1].start();
793 for (
int site = is; site < ns; ++site) {
794 int ix = site % m_Nxv;
795 int iy = (site / m_Nxv) % m_Nyv;
796 int izt = site / Nxy;
799 clear_vec(v2v,
NVCD);
803 if ((
iy < m_Nyv - 1) || (do_comm[1] == 0)) {
804 int iy2 = (
iy + 1) % m_Nyv;
805 int nei =
ix + m_Nxv * (iy2 + m_Nyv *
izt);
823 template<
typename AFIELD>
827 int Nxy = m_Nxv * m_Nyv;
829 int ith, nth, is, ns;
830 set_threadtask_mult(ith, nth, is, ns, m_Nstv);
839 if (do_comm[1] > 0) {
840 for (
int site = is; site < ns; ++site) {
841 int ix = site % m_Nxv;
842 int iy = (site / m_Nxv) % m_Nyv;
843 int izt = site / Nxy;
844 if (
iy == m_Nyv - 1) {
855 chrecv_dn[1].start();
856 chsend_up[1].start();
864 for (
int site = is; site < ns; ++site) {
865 int ix = site % m_Nxv;
866 int iy = (site / m_Nxv) % m_Nyv;
867 int izt = site / Nxy;
870 clear_vec(v2v,
NVCD);
875 if ((
iy != 0) || (do_comm[
idir] == 0)) {
876 int iy2 = (
iy - 1 + m_Nyv) % m_Nyv;
877 int nei =
ix + m_Nxv * (iy2 + m_Nyv *
izt);
884 shift_vec0_yfw(uL, &u[
VLEN *
NDF * site],
NDF);
897 template<
typename AFIELD>
901 int Nxy = m_Nxv * m_Nyv;
903 int ith, nth, is, ns;
904 set_threadtask_mult(ith, nth, is, ns, m_Nstv);
913 if (do_comm[2] > 0) {
914 for (
int site = is; site < ns; ++site) {
915 int ixy = site % Nxy;
916 int iz = (site / Nxy) % m_Nz;
917 int it = site / (Nxy * m_Nz);
928 chrecv_up[2].start();
929 chsend_dn[2].start();
937 for (
int site = is; site < ns; ++site) {
938 int ixy = site % Nxy;
939 int iz = (site / Nxy) % m_Nz;
940 int it = site / (Nxy * m_Nz);
943 clear_vec(v2v,
NVCD);
945 if ((
iz != m_Nz - 1) || (do_comm[2] == 0)) {
946 int iz2 = (
iz + 1) % m_Nz;
947 int nei =
ixy + Nxy * (iz2 + m_Nz *
it);
962 template<
typename AFIELD>
966 int Nxy = m_Nxv * m_Nyv;
968 int ith, nth, is, ns;
969 set_threadtask_mult(ith, nth, is, ns, m_Nstv);
978 if (do_comm[2] > 0) {
979 for (
int site = is; site < ns; ++site) {
980 int ixy = site % Nxy;
981 int iz = (site / Nxy) % m_Nz;
982 int it = site / (Nxy * m_Nz);
983 if (
iz == m_Nz - 1) {
994 chrecv_dn[2].start();
995 chsend_up[2].start();
1003 for (
int site = is; site < ns; ++site) {
1004 int ixy = site % Nxy;
1005 int iz = (site / Nxy) % m_Nz;
1006 int it = site / (Nxy * m_Nz);
1009 clear_vec(v2v,
NVCD);
1011 if ((
iz > 0) || (do_comm[2] == 0)) {
1012 int iz2 = (
iz - 1 + m_Nz) % m_Nz;
1013 int nei =
ixy + Nxy * (iz2 + m_Nz *
it);
1028 template<
typename AFIELD>
1032 int Nxyz = m_Nxv * m_Nyv * m_Nz;
1034 int ith, nth, is, ns;
1035 set_threadtask_mult(ith, nth, is, ns, m_Nstv);
1044 if (do_comm[3] > 0) {
1045 for (
int site = is; site < ns; ++site) {
1046 int ixyz = site % Nxyz;
1047 int it = site / Nxyz;
1058 chrecv_up[3].start();
1059 chsend_dn[3].start();
1060 chrecv_up[3].wait();
1061 chsend_dn[3].wait();
1067 for (
int site = is; site < ns; ++site) {
1068 int ixyz = site % Nxyz;
1069 int it = site / Nxyz;
1072 clear_vec(v2v,
NVCD);
1074 if ((
it < m_Nt - 1) || (do_comm[3] == 0)) {
1075 int it2 = (
it + 1) % m_Nt;
1076 int nei =
ixyz + Nxyz * it2;
1092 template<
typename AFIELD>
1096 int Nxyz = m_Nxv * m_Nyv * m_Nz;
1098 int ith, nth, is, ns;
1099 set_threadtask_mult(ith, nth, is, ns, m_Nstv);
1108 if (do_comm[3] > 0) {
1109 for (
int site = is; site < ns; ++site) {
1110 int ixyz = site % Nxyz;
1111 int it = site / Nxyz;
1112 if (
it == m_Nt - 1) {
1122 chrecv_dn[3].start();
1123 chsend_up[3].start();
1124 chrecv_dn[3].wait();
1125 chsend_up[3].wait();
1130 for (
int site = is; site < ns; ++site) {
1131 int ixyz = site % Nxyz;
1132 int it = site / Nxyz;
1135 clear_vec(v2v,
NVCD);
1137 if ((
it > 0) || (do_comm[3] == 0)) {
1138 int it2 = (
it - 1 + m_Nt) % m_Nt;
1139 int nei =
ixyz + Nxyz * it2;
1154 template<
typename AFIELD>
1161 double flop_site, flop;
1163 if (m_repr ==
"Dirac") {
1164 flop_site =
static_cast<double>(
1165 m_Nc * m_Nd * (4 + 6 * (4 * m_Nc + 2) + 2 * (4 * m_Nc + 1)));
1166 }
else if (m_repr ==
"Chiral") {
1167 flop_site =
static_cast<double>(
1168 m_Nc * m_Nd * (4 + 8 * (4 * m_Nc + 2)));
1170 vout.
crucial(m_vl,
"%s: input repr is undefined.\n");
1174 flop = flop_site *
static_cast<double>(Lvol);
1175 if ((mode ==
"DdagD") || (mode ==
"DDdag")) flop *= 2.0;
void mult_xm(real_t *, real_t *)
void mult_wilson_gm5_dirac(double *v2, double *v1, int *Nsize)
void mult_tm(real_t *, real_t *)
void convert_gauge(INDEX &index, AFIELD &v, const Field &w)
void mult_wilson_yp2(double *RESTRICT v2, double *RESTRICT u, double *RESTRICT buf, int *Nsize, int *bc, int Nc)
void set_string(const string &key, const string &value)
void mult_xp(real_t *, real_t *)
void mult_wilson_2_dirac(double *v2, double *up, double *v1, double *buf_xp, double *buf_xm, double *buf_yp, double *buf_ym, double *buf_zp, double *buf_zm, double *buf_tp, double *buf_tm, double kappa, int *bc, int *Nsize, int *do_comm)
void mult_wilson_1_dirac(double *buf_xp, double *buf_xm, double *buf_yp, double *buf_ym, double *buf_zp, double *buf_zm, double *buf_tp, double *buf_tm, double *up, double *v1, int *bc, int *Nsize, int *do_comm)
void DdagD(AFIELD &, const AFIELD &)
static int get_num_threads()
returns available number of threads.
void mult_dag(AFIELD &, const AFIELD &)
hermitian conjugate of mult.
void mult_wilson_xpb(double *RESTRICT v2, double *RESTRICT u, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
void set_double(const string &key, const double value)
void mult_D(AFIELD &, const AFIELD &)
void set_mode(std::string mode)
setting mult mode.
void mult_D_alt(AFIELD &, const AFIELD &)
void mult_gm5(AFIELD &, const AFIELD &)
multiplies gamma_5 matrix.
virtual void mult_dn(int mu, AFIELD &, const AFIELD &)
downward nearest neighbor hopping term.
void detailed(const char *format,...)
void mult_wilson_xm2(double *RESTRICT v2, double *RESTRICT buf, int *Nsize, int *bc, int Nc)
void mult_zm(real_t *, real_t *)
void set_parameters(const Parameters ¶ms)
setting parameters by a Parameter object.
void mult_wilson_yp1(double *RESTRICT buf, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
void reverse(Field &v, const AFIELD &w)
reverse of spinor field.
void aypx(const double a, Field &y, const Field &x)
aypx(y, a, x): y := a * y + x
void mult_yp(real_t *, real_t *)
virtual void mult_up(int mu, AFIELD &, const AFIELD &)
upward nearest neighbor hopping term.
void mult_wilson_zp1(double *RESTRICT buf, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
void init(const Parameters ¶ms)
initial setup.
void get_parameters(Parameters ¶ms) const
getting parameters as a Parameters object.
void mult_wilson_tp2_dirac(double *RESTRICT v2, double *RESTRICT u, double *RESTRICT buf, int *Nsize, int *bc, int Nc)
void mult_wilson_tm2_dirac(double *RESTRICT v2, double *RESTRICT buf, int *Nsize, int *bc, int Nc)
void set_config(Field *u)
setting gauge configuration.
void H(AFIELD &, const AFIELD &)
void mult_wilson_tmb_dirac(double *RESTRICT v2, double *RESTRICT u, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
void mult_wilson_xmb(double *RESTRICT v2, double *RESTRICT u, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
void Ddag(AFIELD &, const AFIELD &)
void setup_channels()
setup channels for communication.
void mult_wilson_xm1(double *RESTRICT buf, double *RESTRICT u, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
void mult_wilson_ym2(double *RESTRICT v2, double *RESTRICT buf, int *Nsize, int *bc, int Nc)
void set_boundary(AField< REALTYPE, QXS > &ulex, const std::vector< int > &boundary)
void mult_wilson_zpb(double *RESTRICT v2, double *RESTRICT u, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
void mult_wilson_tpb_dirac(double *RESTRICT v2, double *RESTRICT u, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
int fetch_int_vector(const string &key, vector< int > &value) const
void mult_gm4(AFIELD &, const AFIELD &)
static int npe(const int dir)
logical grid extent
void convert(AFIELD &v, const Field &w)
convert of spinor field.
Wilson fermion operator for Accel branch.
void reverse_spinor(INDEX &index, Field &v, const AFIELD &w)
void aypx(real_t, real_t *, real_t *)
void mult_wilson_bulk_dirac(double *v2, double *up, double *v1, double kappa, int *bc, int *Nsize, int *do_comm)
void mult_tp(real_t *, real_t *)
void mult_ym(real_t *, real_t *)
void mult_wilson_tm1_dirac(double *RESTRICT buf, double *RESTRICT u, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
void mult_wilson_zmb(double *RESTRICT v2, double *RESTRICT u, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
void mult_wilson_tp1_dirac(double *RESTRICT buf, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
void set_int_vector(const string &key, const vector< int > &value)
void mult_wilson_xp2(double *RESTRICT v2, double *RESTRICT u, double *RESTRICT buf, int *Nsize, int *bc, int Nc)
void D(AFIELD &, const AFIELD &)
void mult_wilson_ypb(double *RESTRICT v2, double *RESTRICT u, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
void set_config_omp(Field *u)
const double * ptr(const int jin, const int site, const int jex) const
void mult_wilson_zm2(double *RESTRICT v2, double *RESTRICT buf, int *Nsize, int *bc, int Nc)
static Bridge::VerboseLevel Vlevel()
void mult_wilson_ymb(double *RESTRICT v2, double *RESTRICT u, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
static VerboseLevel set_verbose_level(const std::string &str)
void set_config_impl(Field *u)
void tidyup()
final tidy-up.
void mult_wilson_zp2(double *RESTRICT v2, double *RESTRICT u, double *RESTRICT buf, int *Nsize, int *bc, int Nc)
void mult_wilson_ym1(double *RESTRICT buf, double *RESTRICT u, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
int fetch_string(const string &key, string &value) const
int fetch_double(const string &key, double &value) const
double flop_count()
returns floating operation counts.
void crucial(const char *format,...)
void mult_wilson_zm1(double *RESTRICT buf, double *RESTRICT u, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
std::string get_mode() const
returns mult mode.
void mult_wilson_xp1(double *RESTRICT buf, double *RESTRICT v1, int *Nsize, int *bc, int Nc)
Container of Field-type object.
static int get_thread_id()
returns thread id.
void mult(AFIELD &, const AFIELD &)
multiplies fermion operator to a given field.
void convert_spinor(INDEX &index, AFIELD &v, const Field &w)
void general(const char *format,...)
static void assert_single_thread(const std::string &class_name)
assert currently running on single thread.
void mult_zp(real_t *, real_t *)
static std::string get_verbose_level(const VerboseLevel vl)