73 const dim3 nblcks(((*m)+1024 - 1)/ 1024, 1, 1);
89 const dim3 nblcks(((*m)+1024 - 1)/ 1024, 1, 1);
104 const dim3 nblcks(((*m)+1024 - 1)/ 1024, 1, 1);
119 const dim3 nblcks(((*m)+1024 - 1)/ 1024, 1, 1);
131 void *facet,
int *n1,
int *n2,
int *lx,
132 int *ly,
int *lz,
int *m,
136 const dim3 nblcks(((*m)+1024 - 1)/ 1024, 1, 1);
139 ((
real *)
a, (
real *) b, (
int *)
mask, (
int *) facet, *n1, *n2, *lx,
152 const dim3 nblcks(((*m)+1024 - 1)/ 1024, 1, 1);
166 const dim3 nblcks(((*m)+1024 - 1)/ 1024, 1, 1);
180 const dim3 nblcks(((*m)+1024 - 1)/ 1024, 1, 1);
195 const dim3 nblcks(((*mask_size) + 1024 - 1) / 1024, 1, 1);
215 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
228 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
242 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
255 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
269 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
283 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
299 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
313 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
327 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
341 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
357 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
372 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
387 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
402 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
418 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
436 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
452 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
469 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
486 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
503 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
519 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
532 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
546 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
560 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
573 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
587 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
602 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
616 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
630 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
645 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
660 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
672 void *v1,
void *v2,
void *v3,
int *n,
676 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
689 void *v1,
void *v2,
void *v3,
690 void *w1,
void *w2,
void *
w3,
694 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
738 if (
sizeof(
real) ==
sizeof(
float)) {
743 else if (
sizeof(
real) ==
sizeof(
double)) {
775 if (
sizeof(
real_xp) ==
sizeof(
float)) {
780 else if (
sizeof(
real_xp) ==
sizeof(
double)) {
812 if (
sizeof(
real) ==
sizeof(
float)) {
817 else if (
sizeof(
real) ==
sizeof(
double)) {
849 if (
sizeof(
real) ==
sizeof(
float)) {
854 else if (
sizeof(
real) ==
sizeof(
double)) {
881 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
882 const int nb = ((*n) + 1024 - 1)/ 1024;
909 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
910 const int nb = ((*n) + 1024 - 1)/ 1024;
939 const int nt = 1024/
pow2;
942 const int nb = ((*n) + nt - 1)/nt;
968 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
969 const int nb = ((*n) + 1024 - 1)/ 1024;
997 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
998 const int nb = ((*n) + 1024 - 1)/ 1024;
1025 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
1026 const int nb = ((*n) + 1024 - 1)/ 1024;
1052 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
1053 const int nb = ((*n) + 1024 - 1)/ 1024;
1081 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
1082 const int nb = ((*n) + 1024 - 1)/ 1024;
1111 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
1129 const dim3 nblcks(((*n) + 1024 - 1) / 1024, 1, 1);
1143 const dim3 nblcks(((*n) + 1024 - 1) / 1024, 1, 1);
1157 const dim3 nblcks(((*n) + 1024 - 1) / 1024, 1, 1);
1160 ((
real *)
a, *c, *n);
1171 const dim3 nblcks(((*n) + 1024 - 1) / 1024, 1, 1);
1185 const dim3 nblcks(((*n) + 1024 - 1) / 1024, 1, 1);
1199 const dim3 nblcks(((*n) + 1024 - 1) / 1024, 1, 1);
1213 const dim3 nblcks(((*n) + 1024 - 1) / 1024, 1, 1);
1226 const dim3 nblcks(((*n) + 1024 - 1) / 1024, 1, 1);
1241 const dim3 nblcks(((*n)+1024 - 1)/ 1024, 1, 1);
1244 ((
int *)
a, *c, *n);
void cuda_buffer_reserve(cuda_buffer_t *buf, size_t size)
__global__ void ale_add_kinematics_kernel(const int n, T *__restrict__ wx, T *__restrict__ wy, T *__restrict__ wz, const T *__restrict__ x_ref, const T *__restrict__ y_ref, const T *__restrict__ z_ref, const T *__restrict__ phi, const T *__restrict__ x, const T *__restrict__ y, const T *__restrict__ z, const kinematics_params_t kin_params)
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ w
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ u
__global__ void T *__restrict__ T *__restrict__ const T *__restrict__ const T *__restrict__ v
#define CUDA_BUFFER_INIT_SYMM
__global__ void const T *__restrict__ x
__global__ void const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ const T *__restrict__ w3
void device_mpi_allreduce(void *buf_d, void *buf, int count, int nbytes, int op)
void device_nccl_allreduce(void *sbuf_d, void *rbuf_d, int count, int nbytes, int op, void *stream)
void cuda_global_reduce_min(real *bufred, void *bufred_d, int n, const cudaStream_t stream)
void cuda_pwmax_sca3(void *a, void *b, real *c, int *n, cudaStream_t stream)
void cuda_sqrt_inplace(void *a, int *n, cudaStream_t strm)
void cuda_addcol4(void *a, void *b, void *c, void *d, int *n, cudaStream_t strm)
void cuda_add2s2(void *a, void *b, real *c1, int *n, cudaStream_t strm)
void cuda_vdot3(void *dot, void *u1, void *u2, void *u3, void *v1, void *v2, void *v3, int *n, cudaStream_t strm)
void cuda_add3s2(void *a, void *b, void *c, real *c1, real *c2, int *n, cudaStream_t strm)
void cuda_face_masked_gather_copy(void *a, void *b, void *mask, void *facet, int *n1, int *n2, int *lx, int *ly, int *lz, int *m, cudaStream_t strm)
void cuda_global_reduce_add(real *bufred, void *bufred_d, int n, const cudaStream_t stream)
void cuda_addcol3(void *a, void *b, void *c, int *n, cudaStream_t strm)
void cuda_pwmin_vec2(void *a, void *b, int *n, cudaStream_t stream)
void cuda_pwmax_vec3(void *a, void *b, void *c, int *n, cudaStream_t stream)
void cuda_power(void *ap, void *a, real *p, int *n, cudaStream_t strm)
real cuda_glmin(void *a, real *pinf, int *n, cudaStream_t stream)
void cuda_cmult2(void *a, void *b, real *c, int *n, cudaStream_t strm)
void cuda_add2s1(void *a, void *b, real *c1, int *n, cudaStream_t strm)
void cuda_col3(void *a, void *b, void *c, int *n, cudaStream_t strm)
void cuda_masked_copy_0(void *a, void *b, void *mask, int *n, int *m, cudaStream_t strm)
void cuda_add2s2_many(void *x, void **p, void *alpha, int *j, int *n, cudaStream_t strm)
void cuda_sub3(void *a, void *b, void *c, int *n, cudaStream_t strm)
void cuda_copy(void *a, void *b, int *n, cudaStream_t strm)
void cuda_cfill_mask(void *a, real *c, int *size, int *mask, int *mask_size, cudaStream_t strm)
real_xp cuda_glsc3(void *a, void *b, void *c, int *n, cudaStream_t stream)
void cuda_masked_copy_aligned(void *a, void *b, void *mask, int *n, int *m, cudaStream_t strm)
void cuda_invcol2(void *a, void *b, int *n, cudaStream_t strm)
void cuda_global_reduce_add_xp(real_xp *bufred, void *bufred_d, int n, const cudaStream_t stream)
void cuda_pwmax_sca2(void *a, real *c, int *n, cudaStream_t stream)
void cuda_col2(void *a, void *b, int *n, cudaStream_t strm)
void cuda_masked_scatter_copy(void *a, void *b, void *mask, int *n, int *m, cudaStream_t strm)
void cuda_add2(void *a, void *b, int *n, cudaStream_t strm)
void cuda_redbuf_check_alloc(int nb)
void cuda_cmult(void *a, real *c, int *n, cudaStream_t strm)
void cuda_add4(void *a, void *b, void *c, void *d, int *n, cudaStream_t strm)
void cuda_masked_scatter_copy_aligned(void *a, void *b, void *mask, int *n, int *m, cudaStream_t strm)
void cuda_sub2(void *a, void *b, int *n, cudaStream_t strm)
void cuda_redbuf_check_alloc_xp(int nb)
void cuda_invcol3(void *a, void *b, void *c, int *n, cudaStream_t strm)
void cuda_vcross(void *u1, void *u2, void *u3, void *v1, void *v2, void *v3, void *w1, void *w2, void *w3, int *n, cudaStream_t strm)
real_xp cuda_glsc2(void *a, void *b, int *n, cudaStream_t stream)
void cuda_masked_gather_copy(void *a, void *b, void *mask, int *n, int *m, cudaStream_t strm)
real cuda_vlsc3(void *u, void *v, void *w, int *n, cudaStream_t stream)
void cuda_pwmin_sca3(void *a, void *b, real *c, int *n, cudaStream_t stream)
void cuda_cadd2(void *a, void *b, real *c, int *n, cudaStream_t strm)
void cuda_addsqr2s2(void *a, void *b, real *c1, int *n, cudaStream_t strm)
real_xp cuda_glsum(void *a, int *n, cudaStream_t stream)
void cuda_add3(void *a, void *b, void *c, int *n, cudaStream_t strm)
void cuda_addcol3s2(void *a, void *b, void *c, real *s, int *n, cudaStream_t strm)
void cuda_cwrap(void *a, real *min_val, real *max_val, int *n, cudaStream_t strm)
void cuda_pwmin_sca2(void *a, real *c, int *n, cudaStream_t stream)
void cuda_cfill(void *a, real *c, int *n, cudaStream_t strm)
void cuda_absval(void *a, int *n, cudaStream_t stream)
void cuda_rzero(void *a, int *n, cudaStream_t strm)
void cuda_cdiv2(void *a, void *b, real *c, int *n, cudaStream_t strm)
void cuda_cdiv(void *a, real *c, int *n, cudaStream_t strm)
void cuda_global_reduce_max(real *bufred, void *bufred_d, int n, const cudaStream_t stream)
void cuda_iadd(void *a, int *c, int *n, cudaStream_t stream)
void cuda_masked_gather_copy_aligned(void *a, void *b, void *mask, int *n, int *m, cudaStream_t strm)
void cuda_pwmax_vec2(void *a, void *b, int *n, cudaStream_t stream)
void cuda_glsc3_many(real_xp *h, void *w, void *v, void *mult, int *j, int *n, cudaStream_t stream)
real_xp cuda_glsubnorm2(void *a, void *b, int *n, cudaStream_t stream)
void cuda_masked_atomic_reduction(void *a, void *b, void *mask, int *n, int *m, cudaStream_t strm)
void cuda_add5s4(void *a, void *b, void *c, void *d, void *e, real *c1, real *c2, real *c3, real *c4, int *n, cudaStream_t strm)
void cuda_invcol1(void *a, int *n, cudaStream_t strm)
void cuda_radd(void *a, real *c, int *n, cudaStream_t strm)
void cuda_add4s3(void *a, void *b, void *c, void *d, real *c1, real *c2, real *c3, int *n, cudaStream_t strm)
real cuda_glmax(void *a, real *ninf, int *n, cudaStream_t stream)
void cuda_subcol3(void *a, void *b, void *c, int *n, cudaStream_t strm)
void cuda_pwmin_vec3(void *a, void *b, void *c, int *n, cudaStream_t stream)
Object for handling masks in Neko.