Neko 1.99.9
A portable framework for high-order spectral element flow simulations
Loading...
Searching...
No Matches
opr_cdtp.hip
Go to the documentation of this file.
1/*
2 Copyright (c) 2021-2026, The Neko Authors
3 All rights reserved.
4
5 Redistribution and use in source and binary forms, with or without
6 modification, are permitted provided that the following conditions
7 are met:
8
9 * Redistributions of source code must retain the above copyright
10 notice, this list of conditions and the following disclaimer.
11
12 * Redistributions in binary form must reproduce the above
13 copyright notice, this list of conditions and the following
14 disclaimer in the documentation and/or other materials provided
15 with the distribution.
16
17 * Neither the name of the authors nor the names of its
18 contributors may be used to endorse or promote products derived
19 from this software without specific prior written permission.
20
21 THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
22 "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
23 LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS
24 FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE
25 COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT,
26 INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
27 BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
28 LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
29 CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
30 LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN
31 ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
32 POSSIBILITY OF SUCH DAMAGE.
33*/
34
35#include <string.h>
36#include <stdlib.h>
37#include <stdio.h>
38#include <hip/hip_runtime.h>
40#include <device/hip/check.h>
41#include "cdtp_kernel.h"
42#include "elem_block_tune.h"
43
44extern "C" {
45 #include <common/neko_log.h>
46}
47
48template < const int >
49int tune_cdtp(void *dtx, void *x,
50 void *dr, void *ds, void *dt,
51 void *dxt, void *dyt, void *dzt,
52 void *w3, int *nel, int *lx, int *eb_sel, int *ch_sel,
53 int *nwf_sel);
54
55extern "C" {
56
60 void hip_cdtp(void *dtx, void *x,
61 void *dr, void *ds, void *dt,
62 void *dxt, void *dyt, void *dzt,
63 void *w3, int *nel, int *lx) {
64
65 static int autotune[17] = { 0 };
66 /* elements per block candidate chosen by the tuner */
67 static int autotune_eb[17] = { 0 };
68 /* chunk candidate chosen for the 1d variant */
69 static int autotune_ch[17] = { 0 };
70 /* wavefronts per block candidate chosen for the mfma variant */
71 static int autotune_nwf[17] = { 0 };
72
73 const dim3 nthrds_1d(1024, 1, 1);
74 const dim3 nthrds_kstep((*lx), (*lx), 1);
75 const dim3 nblcks((*nel), 1, 1);
76
77#define CASE_1D(LX, C) \
78 hipLaunchKernelGGL( HIP_KERNEL_NAME( \
79 cdtp_kernel_1d<real, LX, NEKO_CHUNKS(LX, C)> ), \
80 nblcks, NEKO_CHUNKS_NTHRDS(LX, C), 0, \
81 (hipStream_t) glb_cmd_queue, \
82 (real *) dtx, (real *) x, \
83 (real *) dr, (real *) ds, (real *) dt, \
84 (real *) dxt, (real *) dyt, (real *) dzt, \
85 (real *) w3); \
86 HIP_CHECK(hipGetLastError());
87
88/* Runtime dispatch onto the tuned chunk candidate */
89#define CASE_1D_SEL(LX, SEL) \
90 switch (SEL) { \
91 case 0: CASE_1D(LX, 0); break; \
92 case 1: CASE_1D(LX, 1); break; \
93 case 2: CASE_1D(LX, 2); break; \
94 default: CASE_1D(LX, 3); break; \
95 }
96
97#define CASE_KSTEP(LX, C) \
98 hipLaunchKernelGGL( HIP_KERNEL_NAME( \
99 cdtp_kernel_kstep<real, LX, NEKO_EB(LX, C)> ), \
100 NEKO_EB_NBLCKS(*nel, LX, C), NEKO_EB_NTHRDS(LX, C), 0, \
101 (hipStream_t) glb_cmd_queue, \
102 (real *) dtx, (real *) x, \
103 (real *) dr, (real *) ds, (real *) dt, \
104 (real *) dxt, (real *) dyt, (real *) dzt, \
105 (real *) w3, *nel); \
106 HIP_CHECK(hipGetLastError());
107
108/* Runtime dispatch onto the tuned candidate */
109#define CASE_KSTEP_SEL(LX, SEL) \
110 switch (SEL) { \
111 case 0: CASE_KSTEP(LX, 0); break; \
112 case 1: CASE_KSTEP(LX, 1); break; \
113 default: CASE_KSTEP(LX, 2); break; \
114 }
115
116#define CASE_MFMA(LX, C) \
117 hipLaunchKernelGGL( HIP_KERNEL_NAME( \
118 cdtp_kernel_mfma<real, LX, NEKO_MFMA_NWF(C)> ), \
119 NEKO_MFMA_NBLCKS(*nel, LX, C), NEKO_MFMA_NTHRDS(C), 0, \
120 (hipStream_t) glb_cmd_queue, \
121 (real *) dtx, (real *) x, \
122 (real *) dr, (real *) ds, (real *) dt, \
123 (real *) dxt, (real *) dyt, (real *) dzt, \
124 (real *) w3, *nel); \
125 HIP_CHECK(hipGetLastError());
126
127/* Runtime dispatch onto the tuned wavefronts per block candidate */
128#define CASE_MFMA_SEL(LX, SEL) \
129 switch (SEL) { \
130 case 0: CASE_MFMA(LX, 0); break; \
131 case 1: CASE_MFMA(LX, 1); break; \
132 case 2: CASE_MFMA(LX, 2); break; \
133 default: CASE_MFMA(LX, 3); break; \
134 }
135
136#define CASE(LX) \
137 case LX: \
138 if(autotune[LX] == 0 ) { \
139 autotune[LX]=tune_cdtp<LX>(dtx, x, \
140 dr, ds, dt, \
141 dxt, dyt, dzt, \
142 w3, nel, lx, &autotune_eb[LX], \
143 &autotune_ch[LX], \
144 &autotune_nwf[LX]); \
145 } else if (autotune[LX] == 1 ) { \
146 CASE_1D_SEL(LX, autotune_ch[LX]); \
147 } else if (autotune[LX] == 2 ) { \
148 CASE_KSTEP_SEL(LX, autotune_eb[LX]); \
149 } else if (autotune[LX] == 3 ) { \
150 CASE_MFMA_SEL(LX, autotune_nwf[LX]); \
151 } \
152 break
153
154#define CASE_LARGE(LX) \
155 case LX: \
156 CASE_KSTEP(LX, 0); \
157 break
158
159
160 if ((*lx) < 13) {
161 switch(*lx) {
162 CASE(2);
163 CASE(3);
164 CASE(4);
165 CASE(5);
166 CASE(6);
167 CASE(7);
168 CASE(8);
169 CASE(9);
170 CASE(10);
171 CASE(11);
172 CASE(12);
173 default:
174 {
175 fprintf(stderr, __FILE__ ": size not supported: %d\n", *lx);
176 exit(1);
177 }
178 }
179 }
180 else {
181 switch(*lx) {
182 CASE_LARGE(13);
183 CASE_LARGE(14);
184 CASE_LARGE(15);
185 CASE_LARGE(16);
186 default:
187 {
188 fprintf(stderr, __FILE__ ": size not supported: %d\n", *lx);
189 exit(1);
190 }
191 }
192 }
193 }
194}
195
196template < const int LX >
197int tune_cdtp(void *dtx, void *x,
198 void *dr, void *ds, void *dt,
199 void *dxt, void *dyt, void *dzt,
200 void *w3, int *nel, int *lx, int *eb_sel, int *ch_sel,
201 int *nwf_sel) {
204 int best1 = 0;
207 int best3 = 0;
208 const int rounds = neko_tune_rounds();
209 const int iters = neko_tune_iters();
210 const int sweep = neko_eb_sweep();
211 const bool mfma = mfma_lx_supported<LX>() && hip_have_mfma();
212 /* Whether the sweep may pick it on its own, see neko_mfma_sweep() */
213 const bool mfma_tune = mfma && neko_mfma_sweep();
214 int best = 0;
215 int retval;
216
217 for (int c = 0; c < NEKO_EB_CANDIDATES; c++) {
219 }
220 for (int c = 0; c < NEKO_CHUNKS_CANDIDATES; c++) {
222 }
223 for (int c = 0; c < NEKO_MFMA_CANDIDATES; c++) {
225 }
226
227 const dim3 nthrds_1d(1024, 1, 1);
228 const dim3 nthrds_kstep((*lx), (*lx), 1);
229 const dim3 nblcks((*nel), 1, 1);
230 const hipStream_t stream = (hipStream_t) glb_cmd_queue;
231
232 char *env_value = NULL;
233 char neko_log_buf[80];
234
235 env_value=getenv("NEKO_AUTOTUNE");
236
237 sprintf(neko_log_buf, "Autotune cdtp (lx: %d)", *lx);
239
240 *eb_sel = 0;
241 *ch_sel = 0;
242 *nwf_sel = 0;
243
244 if(env_value) {
245 if( !strcmp(env_value,"1D") ) {
248 sprintf(neko_log_buf,"Set by env : 1 (1D, %d chunk)",
252 return 1;
253 } else if( !strcmp(env_value,"KSTEP") ) {
254 *eb_sel = neko_eb_env();
256 sprintf(neko_log_buf,"Set by env : 2 (KSTEP, %d elem/block)",
260 return 2;
261 } else if( !strcmp(env_value,"MFMA") ) {
262 if (mfma) {
263 const int c = neko_mfma_env();
264 *nwf_sel = c;
265 CASE_MFMA_SEL(LX, c);
266 sprintf(neko_log_buf,"Set by env : 3 (MFMA, %d wf, %d elem/block)",
270 return 3;
271 } else {
272 sprintf(neko_log_buf, "MFMA strategy not available for this config");
274 }
275 } else {
276 sprintf(neko_log_buf, "Invalid value set for NEKO_AUTOTUNE");
278 }
279 }
280
283
284 /* Warm every variant before timing anything: each specialisation has to be
285 resident and the clocks at steady state, or whichever is timed first is
286 measured on a colder part */
287 for (int i = 0; i < NEKO_TUNE_WARMUP; i++) {
288 CASE_1D(LX, 0);
289 CASE_1D(LX, 1);
290 CASE_1D(LX, 2);
291 CASE_1D(LX, 3);
292 CASE_KSTEP(LX, 0);
293 if (sweep) {
294 CASE_KSTEP(LX, 1);
295 CASE_KSTEP(LX, 2);
296 }
297 if (mfma_tune) {
298 CASE_MFMA(LX, 0);
299 CASE_MFMA(LX, 1);
300 CASE_MFMA(LX, 2);
301 CASE_MFMA(LX, 3);
302 }
303 }
304
305 /* Interleaved rounds, best time per variant */
306 for (int r = 0; r < rounds; r++) {
312 if (sweep) {
315 }
316 if (mfma_tune) {
321 }
322 }
323
326
330 *eb_sel = best;
331 *ch_sel = best1;
332 *nwf_sel = best3;
333
334 if (time1[best1] < time2[best]) {
335 retval = 1;
336 } else {
337 retval = 2;
338 }
339
340 /* The mfma variant joins the comparison only where it exists, its
341 candidates are left at NEKO_TUNE_INIT otherwise */
342 if (time3[best3] < ((retval == 1) ? time1[best1] : time2[best])) {
343 retval = 3;
344 }
345
346 /* Leave the chosen kernel's output in place: the tuner stands in for a real
347 evaluation and the variants do not sum in the same order */
348 if (retval == 1) {
350 } else if (retval == 2) {
352 } else {
354 }
355
356 if (retval == 1) {
357 sprintf(neko_log_buf, "Chose : 1 (1D, %d chunk)",
359 } else if (retval == 2) {
360 sprintf(neko_log_buf, "Chose : 2 (KSTEP, %d elem/block)",
362 } else {
363 sprintf(neko_log_buf, "Chose : 3 (MFMA, %d wf, %d elem/block)",
365 }
368 return retval;
369}
__global__ void ale_add_kinematics_kernel(const int n, T *__restrict__ wx, T *__restrict__ wy, T *__restrict__ wz, const T *__restrict__ x_ref, const T *__restrict__ y_ref, const T *__restrict__ z_ref, const T *__restrict__ phi, const T *__restrict__ x, const T *__restrict__ y, const T *__restrict__ z, const kinematics_params_t kin_params)
const int i
__global__ void const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ dyt
__global__ void const T *__restrict__ const T *__restrict__ const T *__restrict__ ds
__global__ void const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ dzt
__global__ void const T *__restrict__ x
__global__ void const T *__restrict__ const T *__restrict__ dr
__global__ void const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ dt
__global__ void const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ dxt
__global__ void const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ w3
#define NEKO_CHUNKS_CANDIDATES
Definition elem_block.h:125
#define NEKO_EB_CANDIDATES
Definition elem_block.h:63
#define NEKO_EB_SEL(LX, SEL)
Definition elem_block.h:108
#define NEKO_CHUNKS_SEL(LX, SEL)
Definition elem_block.h:147
#define NEKO_TUNE_TIME(T, LAUNCH, LX, C, ITERS)
static int neko_eb_env()
static int neko_tune_rounds()
#define NEKO_TUNE_LOG(LX, T1, T2)
#define NEKO_TUNE_INIT
#define NEKO_TUNE_BEST(T, BEST, N)
static int neko_tune_iters()
static int neko_chunks_env()
static int neko_eb_sweep()
#define NEKO_TUNE_WARMUP
#define HIP_CHECK(err)
Definition check.h:8
#define NEKO_TUNE_LOG_MFMA(LX, T3)
#define NEKO_MFMA_CANDIDATES
static bool hip_have_mfma()
static int neko_mfma_env()
#define NEKO_MFMA_NWF(C)
#define NEKO_MFMA_EB(LX, C)
static int neko_mfma_sweep()
void log_error(char *msg)
void log_message(char *msg)
void log_end_section()
void log_section(char *msg)
int tune_cdtp(void *dtx, void *x, void *dr, void *ds, void *dt, void *dxt, void *dyt, void *dzt, void *w3, int *nel, int *lx, int *eb_sel, int *ch_sel, int *nwf_sel)
Definition opr_cdtp.hip:197
#define CASE_KSTEP_SEL(LX, SEL)
#define CASE(LX)
#define CASE_1D_SEL(LX, SEL)
void hip_cdtp(void *dtx, void *x, void *dr, void *ds, void *dt, void *dxt, void *dyt, void *dzt, void *w3, int *nel, int *lx)
Definition opr_cdtp.hip:60
#define CASE_MFMA_SEL(LX, SEL)
#define CASE_KSTEP(LX, C)
#define CASE_LARGE(LX)
#define CASE_MFMA(LX, C)
#define CASE_1D(LX, C)