49 void *
dx,
void *
dy,
void *
dz,
62 void *
dx,
void *
dy,
void *
dz,
66 void *
w3,
int *nel,
int *lx) {
83#define CASE_1D(LX, C) \
84 opgrad_kernel_1d<real, LX, NEKO_CHUNKS(LX, C)> \
85 <<<nblcks, NEKO_CHUNKS_NTHRDS(LX, C), 0, stream>>> \
86 ((real *) ux, (real *) uy, (real *) uz, (real *) u, \
87 (real *) dx, (real *) dy, (real *) dz, \
88 (real *) drdx, (real *) dsdx, (real *) dtdx, \
89 (real *) drdy, (real *) dsdy, (real *) dtdy, \
90 (real *) drdz, (real *) dsdz, (real *) dtdz, \
92 CUDA_CHECK(cudaGetLastError());
95#define CASE_1D_SEL(LX, SEL) \
97 case 0: CASE_1D(LX, 0); break; \
98 case 1: CASE_1D(LX, 1); break; \
99 case 2: CASE_1D(LX, 2); break; \
100 default: CASE_1D(LX, 3); break; \
104#define CASE_KSTEP(LX, C) \
105 opgrad_kernel_kstep<real, LX, NEKO_EB(LX, C)> \
106 <<<NEKO_EB_NBLCKS(*nel, LX, C), NEKO_EB_NTHRDS(LX, C), 0, stream>>> \
107 ((real *) ux, (real *) uy, (real *) uz, (real *) u, \
108 (real *) dx, (real *) dy, (real *) dz, \
109 (real *) drdx, (real *) dsdx, (real *) dtdx, \
110 (real *) drdy, (real *) dsdy, (real *) dtdy, \
111 (real *) drdz, (real *) dsdz, (real *) dtdz, \
112 (real *) w3, *nel); \
113 CUDA_CHECK(cudaGetLastError());
116#define CASE_KSTEP_SEL(LX, SEL) \
118 case 0: CASE_KSTEP(LX, 0); break; \
119 case 1: CASE_KSTEP(LX, 1); break; \
120 default: CASE_KSTEP(LX, 2); break; \
123#define CASE_DMMA(LX, C) \
124 opgrad_kernel_dmma<real, LX, NEKO_DMMA_NW(C)> \
125 <<<NEKO_DMMA_NBLCKS(*nel, LX), NEKO_DMMA_NTHRDS(C), 0, stream>>> \
126 ((real *) ux, (real *) uy, (real *) uz, (real *) u, \
127 (real *) dx, (real *) dy, (real *) dz, \
128 (real *) drdx, (real *) dsdx, (real *) dtdx, \
129 (real *) drdy, (real *) dsdy, (real *) dtdy, \
130 (real *) drdz, (real *) dsdz, (real *) dtdz, \
131 (real *) w3, *nel); \
132 CUDA_CHECK(cudaGetLastError());
135#define CASE_DMMA_SEL(LX, SEL) \
137 case 0: CASE_DMMA(LX, 0); break; \
138 case 1: CASE_DMMA(LX, 1); break; \
139 default: CASE_DMMA(LX, 2); break; \
150#define CASE_DMMA_TMA(LX, C) \
151 (void) opgrad_dmma_tma_optin<real, LX, NEKO_DMMA_NW(C)>(); \
152 opgrad_kernel_dmma_tma<real, LX, NEKO_DMMA_NW(C)> \
153 <<<NEKO_DMMA_NBLCKS(*nel, LX), NEKO_DMMA_NTHRDS(C), \
154 NEKO_OPGRAD_TMA_SMEM, stream>>> \
155 ((real *) ux, (real *) uy, (real *) uz, (real *) u, \
156 (real *) dx, (real *) dy, (real *) dz, \
157 (real *) drdx, (real *) dsdx, (real *) dtdx, \
158 (real *) drdy, (real *) dsdy, (real *) dtdy, \
159 (real *) drdz, (real *) dsdz, (real *) dtdz, \
161 CUDA_CHECK(cudaGetLastError());
164#define CASE_DMMA_TMA_SEL(LX, SEL) \
166 case 0: CASE_DMMA_TMA(LX, 0); break; \
167 case 1: CASE_DMMA_TMA(LX, 1); break; \
168 default: CASE_DMMA_TMA(LX, 2); break; \
173 if(autotune[LX] == 0 ) { \
174 autotune[LX]=tune_opgrad<LX>(ux, uy, uz, u, \
179 w3, nel, lx, &autotune_eb[LX], \
180 &autotune_ch[LX], &autotune_nw[LX], \
182 } else if (autotune[LX] == 1 ) { \
183 CASE_1D_SEL(LX, autotune_ch[LX]); \
184 } else if (autotune[LX] == 2 ) { \
185 CASE_KSTEP_SEL(LX, autotune_eb[LX]); \
186 } else if (autotune[LX] == 3 ) { \
187 CASE_DMMA_SEL(LX, autotune_nw[LX]); \
188 } else if (autotune[LX] == 4 ) { \
189 CASE_DMMA_TMA_SEL(LX, autotune_tw[LX]); \
218template < const
int LX >
220 void *
dx,
void *
dy,
void *
dz,
325 "DMMA_TMA strategy not available for this config");
363 for (
int r = 0; r <
rounds; r++) {
__global__ void ale_add_kinematics_kernel(const int n, T *__restrict__ wx, T *__restrict__ wy, T *__restrict__ wz, const T *__restrict__ x_ref, const T *__restrict__ y_ref, const T *__restrict__ z_ref, const T *__restrict__ phi, const T *__restrict__ x, const T *__restrict__ y, const T *__restrict__ z, const kinematics_params_t kin_params)
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ dtdy
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ dtdx
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ dtdz
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ dz
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ dsdz
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ dx
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ u
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ dy
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ drdz
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ drdx
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ dsdx
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ dsdy
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ drdy
__global__ void const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ w3
#define NEKO_CHUNKS_CANDIDATES
#define NEKO_EB_CANDIDATES
#define NEKO_EB_SEL(LX, SEL)
#define NEKO_CHUNKS_SEL(LX, SEL)
#define NEKO_TUNE_TIME(T, LAUNCH, LX, C, ITERS)
static int neko_tune_rounds()
#define NEKO_TUNE_LOG(LX, T1, T2)
#define NEKO_TUNE_BEST(T, BEST, N)
static int neko_tune_iters()
static int neko_chunks_env()
static int neko_eb_sweep()
__global__ void T *__restrict__ uy
__global__ void T *__restrict__ T *__restrict__ uz
#define NEKO_DMMA_CANDIDATES
static bool cuda_have_dmma()
static int neko_dmma_env()
#define NEKO_DMMA_PACK(LX)
#define NEKO_TUNE_LOG_DMMA(LX, T3)
#define NEKO_TUNE_LOG_DMMA_TMA(LX, T4)
static bool cuda_have_tma_opgrad()
static int neko_dmma_tma_env()
static bool dmma_tma_opgrad_aligned(const void *u, const void *drdx, const void *dsdx, const void *dtdx, const void *drdy, const void *dsdy, const void *dtdy, const void *drdz, const void *dsdz, const void *dtdz)
void log_error(char *msg)
void log_message(char *msg)
void log_section(char *msg)
#define CASE_DMMA_SEL(LX, SEL)
#define CASE_KSTEP_SEL(LX, SEL)
#define CASE_1D_SEL(LX, SEL)
int tune_opgrad(void *ux, void *uy, void *uz, void *u, void *dx, void *dy, void *dz, void *drdx, void *dsdx, void *dtdx, void *drdy, void *dsdy, void *dtdy, void *drdz, void *dsdz, void *dtdz, void *w3, int *nel, int *lx, int *eb_sel, int *ch_sel, int *nw_sel, int *tw_sel)
void cuda_opgrad(void *ux, void *uy, void *uz, void *u, void *dx, void *dy, void *dz, void *drdx, void *dsdx, void *dtdx, void *drdy, void *dsdy, void *dtdy, void *drdz, void *dsdz, void *dtdz, void *w3, int *nel, int *lx)
#define CASE_KSTEP(LX, C)
#define CASE_DMMA_TMA_SEL(LX, SEL)
#define CASE_DMMA_TMA(LX, C)