Add missing unrolls

This commit is contained in:
Piotr Wilkin
2026-05-16 15:47:06 +02:00
parent 22bc539f1d
commit cdb48f0f14

View File

@@ -6,6 +6,7 @@
#extension GL_EXT_shader_explicit_arithmetic_types_int16 : require
#extension GL_EXT_shader_explicit_arithmetic_types_int8 : require
#extension GL_EXT_shader_16bit_storage : require
#extension GL_EXT_control_flow_attributes : require
#ifdef USE_OCP_FP4
#extension GL_EXT_float_e2m1 : require
@@ -1768,7 +1769,9 @@ shared FLOAT_TYPE kvalues_iq4nl[16];
void init_iq_shmem(uvec3 wgsize)
{
// copy the table into shared memory and sync
for (uint i = gl_LocalInvocationIndex.x; i < kvalues_iq4nl.length(); i += wgsize.x) {
// Use gl_LocalInvocationIndex (uint scalar) directly - each thread loads its portion
// with stride equal to workgroup size for coalesced shared memory writes
[[unroll]] for (uint i = gl_LocalInvocationIndex; i < kvalues_iq4nl.length(); i += wgsize.x) {
kvalues_iq4nl[i] = FLOAT_TYPE(kvalues_iq4nl_const[i]);
}
barrier();
@@ -1808,11 +1811,12 @@ float ue4m3_to_fp32_build(uint u) {
void init_iq_shmem(uvec3 wgsize)
{
// copy the table into shared memory and sync
for (uint i = gl_LocalInvocationIndex.x; i < kvalues_mxfp4.length(); i += wgsize.x) {
// Use gl_LocalInvocationIndex (uint scalar) directly for thread-strided initialization
[[unroll]] for (uint i = gl_LocalInvocationIndex; i < kvalues_mxfp4.length(); i += wgsize.x) {
kvalues_mxfp4[i] = kvalues_mxfp4_const[i];
}
#if defined(DATA_A_NVFP4)
for (uint i = gl_LocalInvocationIndex.x; i < 128u; i += wgsize.x) {
[[unroll]] for (uint i = gl_LocalInvocationIndex; i < 128u; i += wgsize.x) {
ue4m3_fp32_lut[i] = ue4m3_to_fp32_build(i);
}
#endif