use fast_div to improve performance

bugparty · bugparty · commit 9494833b1176 · 2025-09-14T23:25:39.000-07:00
diff --git a/ggml/src/ggml-cuda/common.cuh b/ggml/src/ggml-cuda/common.cuh
@@ -636,6 +636,14 @@ static __device__ __forceinline__ uint32_t fastmodulo(uint32_t n, const uint3 fa
     return n - fastdiv(n, fastdiv_values) * fastdiv_values.z;
 }
 
+// Calculate both division and modulo at once, returns <n/divisor, n%divisor>
+static __device__ __forceinline__ uint2 fast_div_modulo(uint32_t n, const uint3 fastdiv_values) {
+    // expects  fastdiv_values to contain <mp, L, divisor> in <x, y, z> (see init_fastdiv_values)
+    const uint32_t div_val = fastdiv(n, fastdiv_values);
+    const uint32_t mod_val = n - div_val * fastdiv_values.z;
+    return make_uint2(div_val, mod_val);
+}
+
 typedef void (*dequantize_kernel_t)(const void * vx, const int64_t ib, const int iqs, float2 & v);
 
 static __device__ __forceinline__ float get_alibi_slope(
diff --git a/ggml/src/ggml-cuda/pad_reflect_1d.cu b/ggml/src/ggml-cuda/pad_reflect_1d.cu
@@ -6,6 +6,7 @@ static __global__ void pad_reflect_1d_kernel_f32(
     const int64_t ne0,
     const int64_t ne00,
     const int64_t ne01,
+    const uint3 ne01_packed,
     const int64_t ne02,
     const int64_t ne03,
     const int64_t nb00,
@@ -22,24 +23,25 @@ static __global__ void pad_reflect_1d_kernel_f32(
     const int64_t i3 = blockIdx.z;
     const int64_t i2 = blockIdx.y;
 
-    const int64_t tile1  = blockIdx.x % ne01;     // i1
-    const int64_t tile0  = blockIdx.x / ne01;     // nth i0 tile
+    const uint2 div_mod_packed = fast_div_modulo(blockIdx.x, ne01_packed);
+    const int64_t tile1  = div_mod_packed.y;     // i1
+    const int64_t tile0  = div_mod_packed.x;     // nth i0 tile
     const int64_t i1     = tile1;
     const int64_t i0     = threadIdx.x + tile0 * blockDim.x;
     if ( i0 >= ne0 || i1 >= ne01 || i2 >= ne02 || i3 >= ne03 ) {
         return;
     }
 
     const char * src0_ptr = (const char *)src0 + i3*nb03 + i2*nb02 + i1*nb01;
-    const char * dst_ptr = (const char *)dst + i3*nb3 + i2*nb2 + i1*nb1;
+    char * dst_ptr = (char *)dst + i3*nb3 + i2*nb2 + i1*nb1;
 
     float value;
     const int64_t j = i0 - p0;
 
     if ( j<0 ) {// i0<p0
         // Left padding - reflect
         value = *(const float *)(src0_ptr - j * nb00);
-    } else if (j < ne00) { //i0 < ne0 - p1
+    } else if ( j < ne00 ) { //i0 < ne0 - p1
         // Middle - copy
         value = *(const float *)(src0_ptr + j * nb00);
     } else  {
@@ -63,6 +65,7 @@ void ggml_cuda_op_pad_reflect_1d(ggml_backend_cuda_context & ctx, ggml_tensor *
 
     const int64_t ne00 = src0->ne[0];
     const int64_t ne01 = src0->ne[1];
+    const uint3   ne01_packed = init_fastdiv_values(ne01);
     const int64_t ne02 = src0->ne[2];
     const int64_t ne03 = src0->ne[3];
 
@@ -81,7 +84,7 @@ void ggml_cuda_op_pad_reflect_1d(ggml_backend_cuda_context & ctx, ggml_tensor *
 
     pad_reflect_1d_kernel_f32<<<grid_dims, block_dims, 0, stream>>>(
         src0->data, dst->data,
-        ne0, ne00, ne01, ne02, ne03,
+        ne0, ne00, ne01, ne01_packed, ne02, ne03,
         src0->nb[0], src0->nb[1], src0->nb[2], src0->nb[3],
         dst->nb[0], dst->nb[1], dst->nb[2], dst->nb[3],
         p0, p1