diff --git a/gradle/libs.versions.toml b/gradle/libs.versions.toml index b7eefc99..8dbc9142 100644 --- a/gradle/libs.versions.toml +++ b/gradle/libs.versions.toml @@ -93,6 +93,7 @@ skainet-compile-core = { module = "sk.ainet.core:skainet-compile-core" } skainet-compile-dag = { module = "sk.ainet.core:skainet-compile-dag" } skainet-compile-hlo = { module = "sk.ainet.core:skainet-compile-hlo" } skainet-compile-opt = { module = "sk.ainet.core:skainet-compile-opt" } +skainet-backend-api = { module = "sk.ainet.core:skainet-backend-api" } skainet-backend-cpu = { module = "sk.ainet.core:skainet-backend-cpu" } skainet-backend-nativeCpu = { module = "sk.ainet.core:skainet-backend-native-cpu" } skainet-backend-jniCpu = { module = "sk.ainet.core:skainet-backend-jni-cpu" } diff --git a/llm-core/api/jvm/llm-core.api b/llm-core/api/jvm/llm-core.api index 0a227820..761ab860 100644 --- a/llm-core/api/jvm/llm-core.api +++ b/llm-core/api/jvm/llm-core.api @@ -524,6 +524,11 @@ public final class sk/ainet/apps/llm/weights/BertSafeTensorsNameResolver : sk/ai public fun resolve (Ljava/lang/String;Ljava/lang/String;)Ljava/lang/String; } +public final class sk/ainet/apps/llm/weights/GgmlQuantEncodingsKt { + public static final fun hasPackedMatmulKernel (Lsk/ainet/io/gguf/GGMLQuantizationType;)Z + public static final fun toBlockEncoding (Lsk/ainet/io/gguf/GGMLQuantizationType;)Lsk/ainet/lang/tensor/storage/TensorEncoding; +} + public final class sk/ainet/apps/llm/weights/LlamaGGUFNameResolver : sk/ainet/io/weights/WeightNameResolver { public fun ()V public fun resolve (Ljava/lang/String;Ljava/lang/String;)Ljava/lang/String; diff --git a/llm-core/build.gradle.kts b/llm-core/build.gradle.kts index 3d9feb5a..1780a82b 100644 --- a/llm-core/build.gradle.kts +++ b/llm-core/build.gradle.kts @@ -54,6 +54,10 @@ kotlin { implementation(libs.skainet.compile.opt) implementation(libs.skainet.io.core) implementation(libs.skainet.io.gguf) + // KernelRegistry/KernelProvider capability queries for the + // packed-quant kernel gate (`hasPackedMatmulKernel`, #170). + // implementation-scoped: no backend types leak into the API. + implementation(libs.skainet.backend.api) implementation(libs.kotlinx.io.core) implementation(libs.kotlinx.serialization.json) } diff --git a/llm-core/src/commonMain/kotlin/sk/ainet/apps/llm/weights/GgmlQuantEncodings.kt b/llm-core/src/commonMain/kotlin/sk/ainet/apps/llm/weights/GgmlQuantEncodings.kt new file mode 100644 index 00000000..481ab467 --- /dev/null +++ b/llm-core/src/commonMain/kotlin/sk/ainet/apps/llm/weights/GgmlQuantEncodings.kt @@ -0,0 +1,64 @@ +package sk.ainet.apps.llm.weights + +import sk.ainet.backend.api.kernel.KernelRegistry +import sk.ainet.io.gguf.GGMLQuantizationType +import sk.ainet.lang.tensor.storage.TensorEncoding + +/** + * Map a GGUF [GGMLQuantizationType] to the engine [TensorEncoding] of the + * formats with a packed CPU matmul kernel (+ lazy `ops.transpose` support), + * or `null` for types that must be dequantized. + * + * This is the ggml-keyed front door to the shared + * [sk.ainet.lang.nn.quant.BlockQuantPacking] packer (#184 hoist 2): model + * modules call `qt.toBlockEncoding()?.let { BlockQuantPacking.pack(bytes, it, shape) }` + * and keep only weight *selection* and naming for themselves. + */ +public fun GGMLQuantizationType.toBlockEncoding(): TensorEncoding? = when (this) { + GGMLQuantizationType.Q4_K -> TensorEncoding.Q4_K + GGMLQuantizationType.Q5_K -> TensorEncoding.Q5_K + GGMLQuantizationType.Q6_K -> TensorEncoding.Q6_K + GGMLQuantizationType.Q8_0 -> TensorEncoding.Q8_0 + GGMLQuantizationType.Q4_0 -> TensorEncoding.Q4_0 + GGMLQuantizationType.Q5_0 -> TensorEncoding.Q5_0 + GGMLQuantizationType.Q5_1 -> TensorEncoding.Q5_1 + else -> null +} + +/** + * Runtime kernel gate for the packed-quant conversion path (#170): `true` iff + * some registered, available [sk.ainet.backend.api.kernel.KernelProvider] + * carries an `FP32 activations x packed-` matmul kernel, so a weight + * packed by [sk.ainet.lang.nn.quant.BlockQuantPacking] will actually dispatch + * to a packed kernel instead of the generic elementwise matmul (which reads + * block-major bytes with row-major strides and would be numerically wrong + * after the lazy packed transpose). + * + * Converters gate NEW packed formats on this check and keep the FP32 dequant + * fallback for `false` — availability-based, not engine-version-based: under + * engine 0.39.0 the scalar/Panama Q5_0/Q5_1 kernels already report here, and + * the native FFM/Kotlin-Native/JNI tiers added by SKaiNET#951 (0.40.0) light + * up through the same query without any transformers-side change. + * + * On the JVM this installs `ServiceLoader`-discovered providers first + * (idempotent), mirroring the engine's own lazy `ensureKernelProviders`, so + * the answer is correct even before the first matmul runs. On registry-based + * targets (Kotlin/Native, Android, JS/WASM) providers are registered by the + * platform backend factories; converters run with a live [sk.ainet.context.ExecutionContext], + * which implies that registration has happened. + */ +public fun GGMLQuantizationType.hasPackedMatmulKernel(): Boolean { + val encoding = toBlockEncoding() ?: return false + ensurePlatformKernelProviders() + return KernelRegistry.providers().any { + it.isAvailable() && it.supports("matmul", listOf("Float32", encoding.name)) + } +} + +/** + * Populate [KernelRegistry] with the platform's discoverable providers before + * a capability query. JVM: `KernelServiceLoader.installAll()` (idempotent). + * Registry-based targets: no-op — providers are registered explicitly by the + * backend factories (see `BackendRegistry.registryBased.kt`). + */ +internal expect fun ensurePlatformKernelProviders() diff --git a/llm-core/src/jvmMain/kotlin/sk/ainet/apps/llm/weights/GgmlQuantEncodings.jvm.kt b/llm-core/src/jvmMain/kotlin/sk/ainet/apps/llm/weights/GgmlQuantEncodings.jvm.kt new file mode 100644 index 00000000..7413c172 --- /dev/null +++ b/llm-core/src/jvmMain/kotlin/sk/ainet/apps/llm/weights/GgmlQuantEncodings.jvm.kt @@ -0,0 +1,16 @@ +package sk.ainet.apps.llm.weights + +import sk.ainet.backend.api.kernel.KernelServiceLoader + +/** + * JVM: install every `META-INF/services`-declared + * [sk.ainet.backend.api.kernel.KernelProvider] into the process registry. + * Idempotent ([sk.ainet.backend.api.kernel.KernelRegistry.register] no-ops on + * re-registration), and the same wiring the engine's `DefaultCpuOpsJvm` + * performs lazily on first matmul — done eagerly here so the + * [hasPackedMatmulKernel] gate answers correctly at weight-conversion time, + * which runs before any matmul. + */ +internal actual fun ensurePlatformKernelProviders() { + KernelServiceLoader.installAll() +} diff --git a/llm-core/src/registryBasedMain/kotlin/sk/ainet/apps/llm/weights/GgmlQuantEncodings.registryBased.kt b/llm-core/src/registryBasedMain/kotlin/sk/ainet/apps/llm/weights/GgmlQuantEncodings.registryBased.kt new file mode 100644 index 00000000..a2975e88 --- /dev/null +++ b/llm-core/src/registryBasedMain/kotlin/sk/ainet/apps/llm/weights/GgmlQuantEncodings.registryBased.kt @@ -0,0 +1,12 @@ +package sk.ainet.apps.llm.weights + +/** + * Registry-based targets (Kotlin/Native, Android, JS/WASM): no ServiceLoader. + * Kernel providers are registered explicitly by the platform backend factories + * (e.g. the K/N CPU ops factory pins `ScalarKernelProvider`; Android pins the + * JNI provider) before any converter runs — [hasPackedMatmulKernel] just reads + * the registry as-is. + */ +internal actual fun ensurePlatformKernelProviders() { + // No-op: KernelRegistry is populated by the backend factory on these targets. +} diff --git a/llm-inference/apertus/src/jvmMain/kotlin/sk/ainet/models/apertus/ApertusMemSegConverter.kt b/llm-inference/apertus/src/jvmMain/kotlin/sk/ainet/models/apertus/ApertusMemSegConverter.kt index d1335698..e63ef00c 100644 --- a/llm-inference/apertus/src/jvmMain/kotlin/sk/ainet/models/apertus/ApertusMemSegConverter.kt +++ b/llm-inference/apertus/src/jvmMain/kotlin/sk/ainet/models/apertus/ApertusMemSegConverter.kt @@ -3,6 +3,7 @@ package sk.ainet.models.apertus import sk.ainet.context.ExecutionContext import sk.ainet.io.gguf.GGMLQuantizationType import sk.ainet.io.gguf.dequant.DequantOps +import sk.ainet.lang.nn.quant.BlockQuantPacking import sk.ainet.lang.tensor.Shape import sk.ainet.lang.tensor.Tensor import sk.ainet.lang.tensor.data.Q4_KBlockTensorData @@ -110,13 +111,13 @@ private fun convertOne( } GGMLQuantizationType.Q4_K -> { - val relaid = relayoutKSeriesRowMajorToBlockMajor(bytes, logicalShape, BYTES_PER_Q4_K_BLOCK) + val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, logicalShape, BYTES_PER_Q4_K_BLOCK, K_SERIES_BLOCK_SIZE) val data = Q4_KBlockTensorData.fromRawBytes(logicalShape, relaid) ctx.fromData(data as TensorData, advertisedDtype) as Tensor } GGMLQuantizationType.Q6_K -> { - val relaid = relayoutKSeriesRowMajorToBlockMajor(bytes, logicalShape, BYTES_PER_Q6_K_BLOCK) + val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, logicalShape, BYTES_PER_Q6_K_BLOCK, K_SERIES_BLOCK_SIZE) val data = Q6_KBlockTensorData.fromRawBytes(logicalShape, relaid) ctx.fromData(data as TensorData, advertisedDtype) as Tensor } @@ -139,38 +140,22 @@ private const val BYTES_PER_Q6_K_BLOCK = 210 private const val K_SERIES_BLOCK_SIZE = 256 /** - * Re-layout GGUF K-series bytes from row-major block order - * (block at row `r`, block index `b` within the row → byte offset - * `(r * blocksPerRow + b) * bytesPerBlock`) to the input-block-major layout - * the `matmulQ{K}_Vec` kernels index via `(blockIdx * outDim + r) * bytesPerBlock`. - * - * For a weight of shape `[outDim, inDim]` with `inDim % 256 == 0`, this is - * a 2D block-level transpose of the `[outDim, inDim/256]` block grid. Bytes - * inside a block are untouched. + * Re-layout GGUF K-series bytes from row-major block order to the + * input-block-major layout the `matmulQ{K}_Vec` kernels index. Delegates to + * the shared [sk.ainet.lang.nn.quant.BlockQuantPacking] packer (#184 hoist 2); + * kept as an internal shim for existing call sites and tests. */ +@Deprecated( + "Hoisted to the shared packer (#184): use BlockQuantPacking.relayoutRowMajorToBlockMajor", + ReplaceWith( + "BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, bytesPerBlock, 256)", + "sk.ainet.lang.nn.quant.BlockQuantPacking", + ), +) internal fun relayoutKSeriesRowMajorToBlockMajor( bytes: ByteArray, shape: Shape, bytesPerBlock: Int -): ByteArray { - require(shape.rank == 2) { "K-series weight must be 2D, got rank ${shape.rank}" } - val outDim = shape[0] - val inDim = shape[1] - require(inDim % K_SERIES_BLOCK_SIZE == 0) { - "K-series weight inDim ($inDim) must be a multiple of $K_SERIES_BLOCK_SIZE" - } - val blocksPerRow = inDim / K_SERIES_BLOCK_SIZE - val expected = outDim.toLong() * blocksPerRow.toLong() * bytesPerBlock.toLong() - require(bytes.size.toLong() >= expected) { - "K-series byte buffer size ${bytes.size} < expected $expected for shape [$outDim, $inDim] @ ${bytesPerBlock}B/block" - } - val out = ByteArray(bytes.size) - for (r in 0 until outDim) { - for (b in 0 until blocksPerRow) { - val srcOff = (r * blocksPerRow + b) * bytesPerBlock - val dstOff = (b * outDim + r) * bytesPerBlock - System.arraycopy(bytes, srcOff, out, dstOff, bytesPerBlock) - } - } - return out -} +): ByteArray = BlockQuantPacking.relayoutRowMajorToBlockMajor( + bytes, shape, bytesPerBlock, K_SERIES_BLOCK_SIZE, +) diff --git a/llm-inference/gemma/src/commonMain/kotlin/sk/ainet/models/gemma/GemmaQuantLayout.kt b/llm-inference/gemma/src/commonMain/kotlin/sk/ainet/models/gemma/GemmaQuantLayout.kt index 608f2ab6..7f202ec9 100644 --- a/llm-inference/gemma/src/commonMain/kotlin/sk/ainet/models/gemma/GemmaQuantLayout.kt +++ b/llm-inference/gemma/src/commonMain/kotlin/sk/ainet/models/gemma/GemmaQuantLayout.kt @@ -1,11 +1,10 @@ package sk.ainet.models.gemma +import sk.ainet.apps.llm.weights.hasPackedMatmulKernel +import sk.ainet.apps.llm.weights.toBlockEncoding import sk.ainet.io.gguf.GGMLQuantizationType +import sk.ainet.lang.nn.quant.BlockQuantPacking import sk.ainet.lang.tensor.Shape -import sk.ainet.lang.tensor.data.Q4_KBlockTensorData -import sk.ainet.lang.tensor.data.Q5_KBlockTensorData -import sk.ainet.lang.tensor.data.Q6_KBlockTensorData -import sk.ainet.lang.tensor.data.Q8_0BlockTensorData import sk.ainet.lang.tensor.data.TensorData import sk.ainet.lang.types.DType @@ -55,65 +54,43 @@ internal fun logicalShapeFor(name: String, metadata: Gemma4ModelMetadata): Shape } /** - * Re-layout GGUF K-series bytes from row-major block order - * (`(r * blocksPerRow + b) * bytesPerBlock`) to the input-block-major order the - * `matmulQ{K}` kernels expect (`(b * outDim + r) * bytesPerBlock`). For a - * `[outDim, inDim]` weight with `inDim % 256 == 0`, this is a block-level 2-D - * transpose; bytes inside a block are untouched. + * Re-layout GGUF K-series bytes from row-major block order to the + * input-block-major order the `matmulQ{K}` kernels expect. + * + * Delegates to the shared [BlockQuantPacking.relayoutRowMajorToBlockMajor] + * (#184 hoist 2); kept as an internal shim because the JVM MemSeg converter + * and the gemma tests call it by this name. * * @param bytesPerBlock 144 (Q4_K), 176 (Q5_K), 210 (Q6_K). */ +@Deprecated( + "Hoisted to the shared packer (#184): use BlockQuantPacking.relayoutRowMajorToBlockMajor", + ReplaceWith( + "BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, bytesPerBlock, blockSize)", + "sk.ainet.lang.nn.quant.BlockQuantPacking", + ), +) internal fun relayoutKSeriesRowMajorToBlockMajor( bytes: ByteArray, shape: Shape, bytesPerBlock: Int, blockSize: Int = 256, -): ByteArray { - require(shape.rank == 2) { "K-series weight must be 2D, got rank ${shape.rank}" } - val outDim = shape[0] - val inDim = shape[1] - require(inDim % blockSize == 0) { "K-series weight inDim ($inDim) must be a multiple of $blockSize" } - val blocksPerRow = inDim / blockSize - val expected = outDim.toLong() * blocksPerRow.toLong() * bytesPerBlock.toLong() - require(bytes.size.toLong() >= expected) { - "K-series byte buffer ${bytes.size} < expected $expected for [$outDim, $inDim] @ ${bytesPerBlock}B/block" - } - val out = ByteArray(bytes.size) - for (r in 0 until outDim) { - for (b in 0 until blocksPerRow) { - val srcOff = (r * blocksPerRow + b) * bytesPerBlock - val dstOff = (b * outDim + r) * bytesPerBlock - bytes.copyInto(out, dstOff, srcOff, srcOff + bytesPerBlock) - } - } - return out -} - -/** - * Block geometry `(blockElems, bytesPerBlock)` for the quant types this packer - * handles. The K-series are 256-element super-blocks; Q8_0 is a 32-element block - * (f16 scale + 32 int8). All four have a first-class CPU matmul kernel + a lazy - * transpose in `ops.transpose`, so all four can stay packed instead of FP32. - */ -private fun quantBlockLayout(qt: GGMLQuantizationType): Pair? = when (qt) { - GGMLQuantizationType.Q4_K -> 256 to 144 - GGMLQuantizationType.Q5_K -> 256 to 176 - GGMLQuantizationType.Q6_K -> 256 to 210 - GGMLQuantizationType.Q8_0 -> 32 to 34 - else -> null -} +): ByteArray = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, bytesPerBlock, blockSize) /** * Pack raw GGUF `bytes` of logical `[out, in]` shape into the heap-packed block - * tensor data the matmul kernels read directly (Q4_K / Q5_K / Q6_K / Q8_0). - * Performs the row-major → block-major relayout. Returns `null` for types - * without a packed kernel (caller dequantizes those to FP32). + * tensor data the matmul kernels read directly. Performs the row-major → + * block-major relayout. Returns `null` for types without a packed kernel + * (caller dequantizes those to FP32). + * + * Delegates to the shared [BlockQuantPacking] packer (#184 hoist 2), which + * covers all seven packed-kernel formats — Q4_K / Q5_K / Q6_K / Q8_0 plus + * Q4_0 / Q5_0 / Q5_1 (#170: FunctionGemma "Q5_K_M" checkpoints carry their + * attention q/k and ffn_gate weights as Q5_1, previously dequantized here). * * Q8_0 matters for gemma's tied `output`/lm_head: FunctionGemma's token_embd is * Q8_0, so keeping the lm_head packed (vs ~0.67 GB FP32) is what lets the eager - * decode fit the 1.9 GB board, and it runs on the NEON Q8_0 kernel. (Requires - * the Q8_0 case in `ops.transpose` — engine — so `linearProject` can transpose - * the packed weight; see transformers #178.) + * decode fit the 1.9 GB board, and it runs on the NEON Q8_0 kernel. * * commonMain → works on JVM and Kotlin/Native alike (no MemSeg / Arena). */ @@ -122,14 +99,29 @@ internal fun packGemmaKQuant( qt: GGMLQuantizationType, shape: Shape, ): TensorData? { - val (blockElems, bpb) = quantBlockLayout(qt) ?: return null - val relaid = relayoutKSeriesRowMajorToBlockMajor(bytes, shape, bpb, blockElems) - @Suppress("UNCHECKED_CAST") - return when (qt) { - GGMLQuantizationType.Q4_K -> Q4_KBlockTensorData(shape, relaid) as TensorData - GGMLQuantizationType.Q5_K -> Q5_KBlockTensorData(shape, relaid) as TensorData - GGMLQuantizationType.Q6_K -> Q6_KBlockTensorData(shape, relaid) as TensorData - GGMLQuantizationType.Q8_0 -> Q8_0BlockTensorData(shape, relaid) as TensorData - else -> null - } + val encoding = qt.toBlockEncoding() ?: return null + // The legacy 32-elem formats (Q4_0/Q5_0/Q5_1) are NEW to this packed path + // (#170) — gate them on an actually-registered matmul kernel rather than + // a hard engine version: without a kernel, a packed weight would fall + // through to the generic elementwise matmul, which reads the block-major + // bytes with row-major strides after the lazy transpose (garbage), so the + // correct degradation is `null` → the caller's FP32 dequant fallback + // (#169 behavior). Q4_K/Q5_K/Q6_K/Q8_0 keep their long-standing + // unconditional packing. Under engine 0.39.0 the scalar/Panama Q5_x + // kernels already satisfy the gate; SKaiNET#951 (0.40.0) adds the native + // FFM/K-N/JNI tiers behind the same check. + if (qt in legacyPackedQuantTypes && !qt.hasPackedMatmulKernel()) return null + return BlockQuantPacking.pack(bytes, encoding, shape) } + +/** + * GGUF legacy block formats whose packed-matmul support arrived with the + * 0.39.0 loading + scalar/Panama kernels and completes with the native-tier + * kernels of SKaiNET#951 (0.40.0). Packing these is gated on + * [hasPackedMatmulKernel]; everything else packs as before. + */ +private val legacyPackedQuantTypes: Set = setOf( + GGMLQuantizationType.Q4_0, + GGMLQuantizationType.Q5_0, + GGMLQuantizationType.Q5_1, +) diff --git a/llm-inference/gemma/src/jvmMain/kotlin/sk/ainet/models/gemma/GemmaMemSegConverter.kt b/llm-inference/gemma/src/jvmMain/kotlin/sk/ainet/models/gemma/GemmaMemSegConverter.kt index 61028cda..5569c78f 100644 --- a/llm-inference/gemma/src/jvmMain/kotlin/sk/ainet/models/gemma/GemmaMemSegConverter.kt +++ b/llm-inference/gemma/src/jvmMain/kotlin/sk/ainet/models/gemma/GemmaMemSegConverter.kt @@ -1,14 +1,18 @@ package sk.ainet.models.gemma import java.lang.foreign.Arena +import sk.ainet.apps.llm.weights.hasPackedMatmulKernel import sk.ainet.context.ExecutionContext import sk.ainet.io.gguf.GGMLQuantizationType import sk.ainet.io.gguf.GGML_QUANT_SIZES import sk.ainet.io.gguf.dequant.DequantOps +import sk.ainet.lang.nn.quant.BlockQuantPacking import sk.ainet.lang.tensor.Shape import sk.ainet.lang.tensor.Tensor import sk.ainet.lang.tensor.data.IntArrayTensorData import sk.ainet.lang.tensor.data.Q4_KBlockTensorData +import sk.ainet.lang.tensor.data.Q5_0BlockTensorData +import sk.ainet.lang.tensor.data.Q5_1BlockTensorData import sk.ainet.lang.tensor.data.Q5_KBlockTensorData import sk.ainet.lang.tensor.data.Q6_KBlockTensorData import sk.ainet.lang.tensor.data.Q4MemorySegmentTensorData @@ -159,7 +163,7 @@ private fun convertOne( // layout; combined, a Q4_K_M Gemma 4 E2B checkpoint (3.2 GB on // disk) stays near that footprint in RAM instead of inflating // to ~18 GB FP32. - val relaid = relayoutKSeriesRowMajorToBlockMajor(bytes, shape, 144) + val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 144, 256) val data = Q4_KBlockTensorData.fromRawBytes(shape, relaid) ctx.fromData(data as TensorData, advertisedDtype) as Tensor } @@ -171,7 +175,7 @@ private fun convertOne( // ~12 GB of FP32 bloat (and the corresponding 7.5 GB per-forward // transpose transient). Sanity-checked against FP32 dequant and // Q6_K packed produces identical tokens — kernel math is right. - val relaid = relayoutKSeriesRowMajorToBlockMajor(bytes, shape, 210) + val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 210, 256) val data = Q6_KBlockTensorData.fromRawBytes(shape, relaid) ctx.fromData(data as TensorData, advertisedDtype) as Tensor } @@ -181,17 +185,49 @@ private fun convertOne( // in DefaultCpuOps. FunctionGemma-270M Q5_K_M ships most attn/FFN // weights as Q5_K, so keeping them packed (176 B/block) avoids the // FP32 inflation and runs the in-kernel dequant matmul. - val relaid = relayoutKSeriesRowMajorToBlockMajor(bytes, shape, 176) + val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 176, 256) val data = Q5_KBlockTensorData.fromRawBytes(shape, relaid) ctx.fromData(data as TensorData, advertisedDtype) as Tensor } + GGMLQuantizationType.Q5_1 -> { + // Packed-path treatment for the 32-elem/24-byte legacy blocks + // (#170): FunctionGemma-270M "Q5_K_M" ships attn_q/attn_k and + // ffn_gate/ffn_up as Q5_1 (81 of 236 tensors), which until #170 + // took the FP32 dequant fallback below. The engine has Q5_1 + // kernels (scalar + Panama since 0.39.0; native FFM/K-N/JNI from + // SKaiNET#951 / 0.40.0) + the lazy Q5_1 transpose, so the weights + // stay packed and run the in-kernel dequant matmul. + // + // Gated on kernel AVAILABILITY (not engine version): if no + // registered provider carries a Q5_1 kernel, packing would send + // the weight down the generic elementwise matmul, which misreads + // the block-major bytes after the lazy transpose — so fall back + // to the always-correct #169 FP32 dequant instead. + if (qt.hasPackedMatmulKernel()) { + val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 24, 32) + val data = Q5_1BlockTensorData.fromRawBytes(shape, relaid) + ctx.fromData(data as TensorData, advertisedDtype) as Tensor + } else { + dequantPackedToFp32(bytes, qt, shape, ctx) + } + } + GGMLQuantizationType.Q5_0 -> { + // Same as Q5_1, 22-byte blocks (f16 d + qh + qs, symmetric). + if (qt.hasPackedMatmulKernel()) { + val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 22, 32) + val data = Q5_0BlockTensorData.fromRawBytes(shape, relaid) + ctx.fromData(data as TensorData, advertisedDtype) as Tensor + } else { + dequantPackedToFp32(bytes, qt, shape, ctx) + } + } else -> { - // Any other quant type without a packed SIMD kernel (Q5_0/Q5_1/Q4_1/Q2_K/…) + // Any other quant type without a packed SIMD kernel (Q4_1/Q2_K/…) // would otherwise be left as raw 1-D bytes, which `linearProject` then can't // transpose ("Transpose requires at least 2 dimensions"). Dequantize to a // correct FP32 `[out, in]` weight so the DSL path runs; the supported packed - // types (Q4_0/Q8_0/Q4_K/Q6_K) above keep their fast SIMD form. This trades - // those tensors' memory savings for correctness until a packed kernel exists. + // types above keep their fast SIMD form. This trades those tensors' memory + // savings for correctness until a packed kernel exists. dequantPackedToFp32(bytes, qt, shape, ctx) } } @@ -271,8 +307,15 @@ private fun dequantToFloat( * [relayoutKSeriesRowMajorToBlockMajor] at Q4_K's 144-byte block size. Kept for * any callers outside this file pinned to the old name. */ +@Deprecated( + "Hoisted to the shared packer (#184): use BlockQuantPacking.relayoutRowMajorToBlockMajor", + ReplaceWith( + "BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 144, 256)", + "sk.ainet.lang.nn.quant.BlockQuantPacking", + ), +) internal fun relayoutQ4_KRowMajorToBlockMajor(bytes: ByteArray, shape: sk.ainet.lang.tensor.Shape): ByteArray = - relayoutKSeriesRowMajorToBlockMajor(bytes, shape, 144) + BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 144, 256) private fun extractBytes(data: TensorData<*, *>): ByteArray { if (data is IntArrayTensorData<*>) { diff --git a/llm-inference/gemma/src/jvmTest/kotlin/sk/ainet/models/gemma/GemmaQ5xPackedParityTest.kt b/llm-inference/gemma/src/jvmTest/kotlin/sk/ainet/models/gemma/GemmaQ5xPackedParityTest.kt new file mode 100644 index 00000000..233f2aff --- /dev/null +++ b/llm-inference/gemma/src/jvmTest/kotlin/sk/ainet/models/gemma/GemmaQ5xPackedParityTest.kt @@ -0,0 +1,196 @@ +package sk.ainet.models.gemma + +import java.io.File +import java.lang.foreign.Arena +import kotlin.math.abs +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue +import kotlinx.coroutines.runBlocking +import org.junit.jupiter.api.Assumptions +import org.junit.jupiter.api.Tag +import sk.ainet.context.DirectCpuExecutionContext +import sk.ainet.io.JvmRandomAccessSource +import sk.ainet.io.gguf.GGMLQuantizationType +import sk.ainet.io.gguf.dequant.DequantOps +import sk.ainet.io.model.QuantPolicy +import sk.ainet.lang.nn.transformer.linearProject +import sk.ainet.lang.tensor.Shape +import sk.ainet.lang.tensor.data.Q5_0BlockTensorData +import sk.ainet.lang.tensor.data.Q5_1BlockTensorData +import sk.ainet.lang.tensor.data.Q5_0TensorData +import sk.ainet.lang.tensor.data.Q5_1TensorData +import sk.ainet.lang.tensor.data.TensorData +import sk.ainet.lang.types.FP32 + +/** + * #170: Q5_1 / Q5_0 weights must stay PACKED through the NATIVE_OPTIMIZED + * converter (instead of the #169 dequant-to-FP32 fallback) and their packed + * `linearProject` results must match the FP32-dequant reference. + * + * Two layers of evidence: + * - [synthetic tests] byte-level parity per format: the converter's exact + * packed pipeline (row-major → block-major relayout + `*BlockTensorData` + + * lazy-transpose matmul) vs the converter's exact FP32 fallback pipeline + * (`DequantOps.dequantFromBytes` + identity col→row transpose). Q5_0 is + * covered here only — the FunctionGemma checkpoint carries no Q5_0 tensor. + * - [real checkpoint] FunctionGemma-270M "Q5_K_M" ships 81 of 236 tensors as + * Q5_1 (attn_q / attn_k / ffn_gate / ffn_up): after conversion all of them + * must be `Q5_1TensorData` (packed), none FP32-inflated. Token-for-token + * decode parity of the full model incl. these tensors is asserted by + * [GemmaQ5KPackedParityTest], which decodes NATIVE_OPTIMIZED (now packing + * Q5_1 too) against the DEQUANTIZE_TO_FP32 baseline. + * + * The packed path is availability-gated (`hasPackedMatmulKernel`), so on a + * JVM with scalar/Panama providers (engine >= 0.39.0) these tests exercise + * the packed branch; the FFM native tier from SKaiNET#951 (0.40.0) slots into + * the same dispatch without any change here. + */ +class GemmaQ5xPackedParityTest { + + // --- synthetic byte-level parity ------------------------------------- + + private fun halfBits(f: Float): Int { + val bits = f.toRawBits() + val sign = (bits ushr 16) and 0x8000 + var exp = ((bits ushr 23) and 0xFF) - 127 + 15 + var mant = bits and 0x7FFFFF + if (exp <= 0) return sign + if (exp >= 31) return sign or 0x7C00 + mant += 0x1000 + if (mant and 0x800000 != 0) { + mant = 0 + exp += 1 + if (exp >= 31) return sign or 0x7C00 + } + return sign or (exp shl 10) or (mant ushr 13) + } + + /** One GGUF Q5_1 block (24 B): d f16, m f16, qh[4], qs[16]. */ + private fun q5_1Block(d: Float, m: Float, codes: IntArray): ByteArray { + val out = ByteArray(24) + val db = halfBits(d); val mb = halfBits(m) + out[0] = (db and 0xFF).toByte(); out[1] = ((db ushr 8) and 0xFF).toByte() + out[2] = (mb and 0xFF).toByte(); out[3] = ((mb ushr 8) and 0xFF).toByte() + var qh = 0 + for (j in 0 until 32) if ((codes[j] ushr 4) and 1 == 1) qh = qh or (1 shl j) + for (b in 0 until 4) out[4 + b] = ((qh ushr (8 * b)) and 0xFF).toByte() + for (j in 0 until 16) out[8 + j] = ((codes[j] and 0xF) or ((codes[j + 16] and 0xF) shl 4)).toByte() + return out + } + + /** One GGUF Q5_0 block (22 B): d f16, qh[4], qs[16]. */ + private fun q5_0Block(d: Float, codes: IntArray): ByteArray { + val out = ByteArray(22) + val db = halfBits(d) + out[0] = (db and 0xFF).toByte(); out[1] = ((db ushr 8) and 0xFF).toByte() + var qh = 0 + for (j in 0 until 32) if ((codes[j] ushr 4) and 1 == 1) qh = qh or (1 shl j) + for (b in 0 until 4) out[2 + b] = ((qh ushr (8 * b)) and 0xFF).toByte() + for (j in 0 until 16) out[6 + j] = ((codes[j] and 0xF) or ((codes[j + 16] and 0xF) shl 4)).toByte() + return out + } + + /** + * Packed-vs-dequant parity for one legacy Q5 format over the converter's + * two pipelines, on a multi-block `[out, in]` weight (multi-block in both + * dimensions so a relayout indexing bug cannot cancel out). + */ + private fun assertPackedMatchesDequant(qt: GGMLQuantizationType) { + val outDim = 3 + val inDim = 64 + val blocksPerRow = inDim / 32 + val shape = Shape(outDim, inDim) + val bpb = if (qt == GGMLQuantizationType.Q5_1) 24 else 22 + + val gguf = ByteArray(outDim * blocksPerRow * bpb) + for (r in 0 until outDim) { + for (b in 0 until blocksPerRow) { + val blockIdx = r * blocksPerRow + b + val d = 0.25f * ((blockIdx % 4) + 1) + val codes = IntArray(32) { j -> (j * 11 + r * 7 + b * 3) % 32 } + val block = + if (qt == GGMLQuantizationType.Q5_1) q5_1Block(d, -1.5f + 0.5f * (blockIdx % 3), codes) + else q5_0Block(d, codes) + block.copyInto(gguf, blockIdx * bpb) + } + } + + val ctx = DirectCpuExecutionContext.create() + val x = ctx.fromFloatArray( + Shape(2, inDim), FP32::class, + FloatArray(2 * inDim) { i -> ((i * 13 + 5) % 19 - 9) / 9.0f }, + ) + + // Reference: the converter's FP32 fallback pipeline, verbatim. + val floats = DequantOps.dequantFromBytes(gguf, qt, shape.volume) + val rowMajor = DequantOps.transposeColumnMajorToRowMajor(floats, inDim, outDim) + val wRef = ctx.fromFloatArray(shape, FP32::class, rowMajor) + val ref = linearProject(ctx.ops, x, wRef).data.copyToFloatArray() + + // Packed: the converter's packed pipeline, verbatim. + @Suppress("DEPRECATION") + val relaid = relayoutKSeriesRowMajorToBlockMajor(gguf, shape, bpb, blockSize = 32) + @Suppress("UNCHECKED_CAST") + val data = when (qt) { + GGMLQuantizationType.Q5_1 -> Q5_1BlockTensorData.fromRawBytes(shape, relaid) + else -> Q5_0BlockTensorData.fromRawBytes(shape, relaid) + } as TensorData + val y = linearProject(ctx.ops, x, ctx.fromData(data, FP32::class)).data.copyToFloatArray() + + assertEquals(ref.size, y.size) + for (i in ref.indices) { + assertTrue( + abs(ref[i] - y[i]) <= 1e-3f * maxOf(1.0f, abs(ref[i])), + "$qt packed[$i]=${y[i]} vs FP32-dequant ref ${ref[i]}", + ) + } + } + + @Test + fun q5_1_packed_linearProject_matches_fp32_dequant_reference() = + assertPackedMatchesDequant(GGMLQuantizationType.Q5_1) + + @Test + fun q5_0_packed_linearProject_matches_fp32_dequant_reference() = + assertPackedMatchesDequant(GGMLQuantizationType.Q5_0) + + // --- real checkpoint: Q5_1 tensors stay packed ----------------------- + + @Test + @Tag("integration") + fun functionGemma_q5_1_tensors_stay_packed_after_conversion() = runBlocking { + val gguf = FunctionGemmaFixture.gguf + Assumptions.assumeTrue(File(gguf).exists(), "FunctionGemma GGUF not present — skipping") + + val ctx = DirectCpuExecutionContext.create() + Arena.ofConfined().use { arena -> + val weights = Gemma4WeightLoader( + randomAccessProvider = { JvmRandomAccessSource.open(gguf) }, + quantPolicy = QuantPolicy.NATIVE_OPTIMIZED, + ).loadToMapStreaming(ctx, FP32::class) + val converted = convertGemmaWeightsToMemSeg(weights, ctx, arena) + + val q51 = converted.quantTypes.filterValues { it == GGMLQuantizationType.Q5_1 }.keys + // functiongemma-physical-ai-v10-Q5_K_M carries 81 Q5_1 tensors + // (attn_q/attn_k/ffn_gate/ffn_up); a rename in a future fixture + // still must leave SOME Q5_1 for this test to be meaningful. + assertTrue(q51.isNotEmpty(), "fixture has no Q5_1 tensors — parity claim would be vacuous") + + val notPacked = q51.filter { name -> converted.tensors[name]?.data !is Q5_1TensorData } + assertTrue( + notPacked.isEmpty(), + "Q5_1 tensors not packed after conversion (dequant fallback taken?): $notPacked", + ) + println("Q5_1 packed after conversion: ${q51.size} tensors (e.g. ${q51.take(3)})") + + // No Q5_0 in this checkpoint — the synthetic test above carries + // Q5_0 parity; assert the premise so a fixture change surfaces here. + val q50 = converted.quantTypes.filterValues { it == GGMLQuantizationType.Q5_0 } + if (q50.isNotEmpty()) { + val notPacked50 = q50.keys.filter { converted.tensors[it]?.data !is Q5_0TensorData } + assertTrue(notPacked50.isEmpty(), "Q5_0 tensors not packed: $notPacked50") + } + } + } +} diff --git a/llm-inference/llama/src/commonMain/kotlin/sk/ainet/models/llama/LlamaQuantLayout.kt b/llm-inference/llama/src/commonMain/kotlin/sk/ainet/models/llama/LlamaQuantLayout.kt index 4e0a81d7..d2c2b211 100644 --- a/llm-inference/llama/src/commonMain/kotlin/sk/ainet/models/llama/LlamaQuantLayout.kt +++ b/llm-inference/llama/src/commonMain/kotlin/sk/ainet/models/llama/LlamaQuantLayout.kt @@ -1,11 +1,10 @@ package sk.ainet.models.llama +import sk.ainet.apps.llm.weights.hasPackedMatmulKernel +import sk.ainet.apps.llm.weights.toBlockEncoding import sk.ainet.io.gguf.GGMLQuantizationType +import sk.ainet.lang.nn.quant.BlockQuantPacking import sk.ainet.lang.tensor.Shape -import sk.ainet.lang.tensor.data.Q4_KBlockTensorData -import sk.ainet.lang.tensor.data.Q5_KBlockTensorData -import sk.ainet.lang.tensor.data.Q6_KBlockTensorData -import sk.ainet.lang.tensor.data.Q8_0BlockTensorData import sk.ainet.lang.tensor.data.TensorData import sk.ainet.lang.types.DType @@ -46,61 +45,49 @@ internal fun logicalShapeFor(name: String, metadata: LlamaModelMetadata): Shape? /** * Re-layout GGUF K-series bytes from row-major block order to the input-block-major order the - * `matmulQ{K}` kernels expect. For a `[outDim, inDim]` weight with `inDim % 256 == 0` this is a - * block-level 2-D transpose; bytes inside a block are untouched. (Mirror of GemmaQuantLayout.) + * `matmulQ{K}` kernels expect. Delegates to the shared + * [BlockQuantPacking.relayoutRowMajorToBlockMajor] (#184 hoist 2); kept as an internal shim + * for existing call sites and tests. */ +@Deprecated( + "Hoisted to the shared packer (#184): use BlockQuantPacking.relayoutRowMajorToBlockMajor", + ReplaceWith( + "BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, bytesPerBlock, blockSize)", + "sk.ainet.lang.nn.quant.BlockQuantPacking", + ), +) internal fun relayoutKSeriesRowMajorToBlockMajor( bytes: ByteArray, shape: Shape, bytesPerBlock: Int, blockSize: Int = 256, -): ByteArray { - require(shape.rank == 2) { "K-series weight must be 2D, got rank ${shape.rank}" } - val outDim = shape[0] - val inDim = shape[1] - require(inDim % blockSize == 0) { "K-series weight inDim ($inDim) must be a multiple of $blockSize" } - val blocksPerRow = inDim / blockSize - val expected = outDim.toLong() * blocksPerRow.toLong() * bytesPerBlock.toLong() - require(bytes.size.toLong() >= expected) { - "K-series byte buffer ${bytes.size} < expected $expected for [$outDim, $inDim] @ ${bytesPerBlock}B/block" - } - val out = ByteArray(bytes.size) - for (r in 0 until outDim) { - for (b in 0 until blocksPerRow) { - val srcOff = (r * blocksPerRow + b) * bytesPerBlock - val dstOff = (b * outDim + r) * bytesPerBlock - bytes.copyInto(out, dstOff, srcOff, srcOff + bytesPerBlock) - } - } - return out -} - -private fun quantBlockLayout(qt: GGMLQuantizationType): Pair? = when (qt) { - GGMLQuantizationType.Q4_K -> 256 to 144 - GGMLQuantizationType.Q5_K -> 256 to 176 - GGMLQuantizationType.Q6_K -> 256 to 210 - GGMLQuantizationType.Q8_0 -> 32 to 34 - else -> null -} +): ByteArray = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, bytesPerBlock, blockSize) /** * Pack raw GGUF `bytes` of logical `[out, in]` shape into heap-packed block tensor data the - * matmul kernels read directly (Q4_K / Q5_K / Q6_K / Q8_0), with the row-major → block-major - * relayout. Null for types without a packed kernel (caller dequantizes those to FP32). + * matmul kernels read directly, with the row-major → block-major relayout. Null for types + * without a packed kernel (caller dequantizes those to FP32). Delegates to the shared + * [BlockQuantPacking] packer (#184 hoist 2), which covers Q4_K / Q5_K / Q6_K / Q8_0 plus + * Q4_0 / Q5_0 / Q5_1 (#170). */ internal fun packLlamaKQuant( bytes: ByteArray, qt: GGMLQuantizationType, shape: Shape, ): TensorData? { - val (blockElems, bpb) = quantBlockLayout(qt) ?: return null - val relaid = relayoutKSeriesRowMajorToBlockMajor(bytes, shape, bpb, blockElems) - @Suppress("UNCHECKED_CAST") - return when (qt) { - GGMLQuantizationType.Q4_K -> Q4_KBlockTensorData(shape, relaid) as TensorData - GGMLQuantizationType.Q5_K -> Q5_KBlockTensorData(shape, relaid) as TensorData - GGMLQuantizationType.Q6_K -> Q6_KBlockTensorData(shape, relaid) as TensorData - GGMLQuantizationType.Q8_0 -> Q8_0BlockTensorData(shape, relaid) as TensorData - else -> null - } + val encoding = qt.toBlockEncoding() ?: return null + // Legacy 32-elem formats (Q4_0/Q5_0/Q5_1) are new to this packed path + // (#170): gate on an actually-registered matmul kernel — without one, a + // packed weight would fall through to the generic elementwise matmul, + // which misreads block-major bytes after the lazy transpose. `null` → + // the caller's FP32 dequant fallback. Mirrors `packGemmaKQuant`. + if (qt in legacyPackedQuantTypes && !qt.hasPackedMatmulKernel()) return null + return BlockQuantPacking.pack(bytes, encoding, shape) } + +/** See `GemmaQuantLayout.legacyPackedQuantTypes` — kernel-gated new formats. */ +private val legacyPackedQuantTypes: Set = setOf( + GGMLQuantizationType.Q4_0, + GGMLQuantizationType.Q5_0, + GGMLQuantizationType.Q5_1, +) diff --git a/transformer-core/api/jvm/transformer-core.api b/transformer-core/api/jvm/transformer-core.api index 0715d44a..f9a838e8 100644 --- a/transformer-core/api/jvm/transformer-core.api +++ b/transformer-core/api/jvm/transformer-core.api @@ -113,6 +113,17 @@ public final class sk/ainet/lang/nn/normalization/RMSNormalization : sk/ainet/la public fun getParams ()Ljava/util/List; } +public final class sk/ainet/lang/nn/quant/BlockQuantPacking { + public static final field INSTANCE Lsk/ainet/lang/nn/quant/BlockQuantPacking; + public final fun blockLayoutFor (Lsk/ainet/lang/tensor/storage/TensorEncoding;)Lkotlin/Pair; + public final fun pack ([BLsk/ainet/lang/tensor/storage/TensorEncoding;Lsk/ainet/lang/tensor/Shape;)Lsk/ainet/lang/tensor/data/TensorData; + public final fun packPreTransposed ([BLsk/ainet/lang/tensor/storage/TensorEncoding;Lsk/ainet/lang/tensor/Shape;)Lsk/ainet/lang/tensor/data/TensorData; + public final fun relayoutRowMajorToBlockMajor ([BLsk/ainet/lang/tensor/Shape;II)[B +} + +public abstract interface class sk/ainet/lang/nn/quant/PreTransposedWeight { +} + public final class sk/ainet/lang/nn/transformer/AppendKVCache : sk/ainet/lang/nn/transformer/KVCache { public fun (IIILjava/lang/String;)V public synthetic fun (IIILjava/lang/String;ILkotlin/jvm/internal/DefaultConstructorMarker;)V diff --git a/transformer-core/build.gradle.kts b/transformer-core/build.gradle.kts index feff4d72..10b43c5c 100644 --- a/transformer-core/build.gradle.kts +++ b/transformer-core/build.gradle.kts @@ -42,5 +42,25 @@ kotlin { commonTest.dependencies { implementation(libs.kotlin.test) } + + val jvmTest by getting { + dependencies { + implementation(project.dependencies.platform(project(":llm-bom"))) + // Test-only: a real CPU backend (DirectCpuExecutionContext + + // ServiceLoader-discovered kernel providers) so the + // PreTransposedWeight `linearProject` path is proven against + // the actual packed-quant matmul dispatch. Production code in + // this module stays lang-core-only. + implementation(libs.skainet.backend.cpu) + } + } } } + +// jvmTest runs real packed-quant matmuls (LinearProjectionPreTransposedTest): +// same convention as llm-inference/* — the Vector API module makes the JVM +// backend pick DefaultCpuOpsJvm, which auto-installs the ServiceLoader kernel +// providers the packed dispatch resolves against. +tasks.withType().configureEach { + jvmArgs("--enable-preview", "--add-modules", "jdk.incubator.vector") +} diff --git a/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/BlockQuantPacking.kt b/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/BlockQuantPacking.kt new file mode 100644 index 00000000..997f230a --- /dev/null +++ b/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/BlockQuantPacking.kt @@ -0,0 +1,152 @@ +package sk.ainet.lang.nn.quant + +import sk.ainet.lang.tensor.Shape +import sk.ainet.lang.tensor.data.Q4_0BlockTensorData +import sk.ainet.lang.tensor.data.Q4_KBlockTensorData +import sk.ainet.lang.tensor.data.Q5_0BlockTensorData +import sk.ainet.lang.tensor.data.Q5_1BlockTensorData +import sk.ainet.lang.tensor.data.Q5_KBlockTensorData +import sk.ainet.lang.tensor.data.Q6_KBlockTensorData +import sk.ainet.lang.tensor.data.Q8_0BlockTensorData +import sk.ainet.lang.tensor.data.TensorData +import sk.ainet.lang.tensor.storage.TensorEncoding +import sk.ainet.lang.types.DType + +/** + * Shared GGUF-block → engine-`*BlockTensorData` packing (#184 hoist 2). + * + * Every model that keeps quantized matmul weights packed used to carry its own + * copy of the same two steps (gemma `GemmaQuantLayout.packGemmaKQuant`, llama + * `LlamaQuantLayout.packLlamaKQuant`, apertus' JVM converter): + * + * 1. re-layout the checkpoint's row-major block order to the input-block-major + * order the `matmulQ*` kernels index (`(blockIdx * outDim + r)`), and + * 2. wrap the relaid bytes in the engine block tensor-data type for the format. + * + * This object is that logic, once, keyed by the engine's [TensorEncoding] + * (a `skainet-lang-core` type — so this file needs no GGUF dependency; model + * modules map their `GGMLQuantizationType` to an encoding and keep only weight + * *selection* and naming). Supported: the seven formats with first-class CPU + * matmul kernels + lazy `ops.transpose` support — Q4_K / Q5_K / Q6_K / Q8_0 / + * Q4_0 / Q5_0 / Q5_1. + */ +public object BlockQuantPacking { + + /** + * Block geometry `(blockElems, bytesPerBlock)` for [encoding], or `null` + * when the encoding has no packed matmul kernel (callers dequantize). + */ + public fun blockLayoutFor(encoding: TensorEncoding): Pair? = when (encoding) { + TensorEncoding.Q4_K -> TensorEncoding.Q4_K.BLOCK_SIZE to TensorEncoding.Q4_K.BYTES_PER_BLOCK + TensorEncoding.Q5_K -> TensorEncoding.Q5_K.BLOCK_SIZE to TensorEncoding.Q5_K.BYTES_PER_BLOCK + TensorEncoding.Q6_K -> TensorEncoding.Q6_K.BLOCK_SIZE to TensorEncoding.Q6_K.BYTES_PER_BLOCK + TensorEncoding.Q8_0 -> TensorEncoding.Q8_0.BLOCK_SIZE to TensorEncoding.Q8_0.BYTES_PER_BLOCK + TensorEncoding.Q4_0 -> TensorEncoding.Q4_0.BLOCK_SIZE to TensorEncoding.Q4_0.BYTES_PER_BLOCK + TensorEncoding.Q5_0 -> TensorEncoding.Q5_0.BLOCK_SIZE to TensorEncoding.Q5_0.BYTES_PER_BLOCK + TensorEncoding.Q5_1 -> TensorEncoding.Q5_1.BLOCK_SIZE to TensorEncoding.Q5_1.BYTES_PER_BLOCK + else -> null + } + + /** + * Re-layout packed block bytes of a 2-D `[outDim, inDim]` weight from the + * checkpoint's row-major block order (`(r * blocksPerRow + b) * bytesPerBlock`) + * to the input-block-major order the `matmulQ*` kernels expect + * (`(b * outDim + r) * bytesPerBlock`). A block-level 2-D transpose; bytes + * inside a block are untouched. + */ + public fun relayoutRowMajorToBlockMajor( + bytes: ByteArray, + shape: Shape, + bytesPerBlock: Int, + blockSize: Int, + ): ByteArray { + require(shape.rank == 2) { "packed matmul weight must be 2D, got rank ${shape.rank}" } + val outDim = shape[0] + val inDim = shape[1] + require(inDim % blockSize == 0) { "packed weight inDim ($inDim) must be a multiple of $blockSize" } + val blocksPerRow = inDim / blockSize + val expected = outDim.toLong() * blocksPerRow.toLong() * bytesPerBlock.toLong() + require(bytes.size.toLong() >= expected) { + "packed byte buffer ${bytes.size} < expected $expected for [$outDim, $inDim] @ ${bytesPerBlock}B/block" + } + val out = ByteArray(bytes.size) + for (r in 0 until outDim) { + for (b in 0 until blocksPerRow) { + val srcOff = (r * blocksPerRow + b) * bytesPerBlock + val dstOff = (b * outDim + r) * bytesPerBlock + bytes.copyInto(out, dstOff, srcOff, srcOff + bytesPerBlock) + } + } + return out + } + + /** + * Pack raw checkpoint `bytes` of logical `[out, in]` [shape] into the + * heap-packed block tensor data the matmul kernels read directly, + * performing the row-major → block-major relayout. Returns `null` for + * encodings without a packed kernel (callers dequantize those to FP32). + * + * The result flows through the engine's lazy packed `ops.transpose` + * (pure shape swap) into the quantized matmul kernel dispatch, so a + * weight packed here never round-trips through FP32. + */ + public fun pack( + bytes: ByteArray, + encoding: TensorEncoding, + shape: Shape, + ): TensorData? { + val (blockElems, bpb) = blockLayoutFor(encoding) ?: return null + val relaid = relayoutRowMajorToBlockMajor(bytes, shape, bpb, blockElems) + @Suppress("UNCHECKED_CAST") + return when (encoding) { + TensorEncoding.Q4_K -> Q4_KBlockTensorData(shape, relaid) as TensorData + TensorEncoding.Q5_K -> Q5_KBlockTensorData(shape, relaid) as TensorData + TensorEncoding.Q6_K -> Q6_KBlockTensorData(shape, relaid) as TensorData + TensorEncoding.Q8_0 -> Q8_0BlockTensorData(shape, relaid) as TensorData + TensorEncoding.Q4_0 -> Q4_0BlockTensorData(shape, relaid) as TensorData + TensorEncoding.Q5_0 -> Q5_0BlockTensorData(shape, relaid) as TensorData + TensorEncoding.Q5_1 -> Q5_1BlockTensorData(shape, relaid) as TensorData + else -> null + } + } + + /** + * Like [pack], but returns the weight *already transposed*: logical shape + * `[in, out]` over the same block-major bytes, marked with + * [PreTransposedWeight] so [sk.ainet.lang.nn.transformer.linearProject] + * skips `ops.transpose` and feeds the tensor straight to the packed + * matmul dispatch (#184 hoist 3). This is exactly the tensor data the + * engine's lazy packed `ops.transpose` would produce from [pack]'s result + * — shape swap, zero copy — minus the per-forward wrapper allocation. + * + * [logicalShape] is still the checkpoint's `[out, in]`; the swap happens + * here. Returns `null` for encodings without a packed kernel, same as + * [pack] (callers dequantize those to FP32 and keep the transposing + * `linearProject` path). + * + * Not yet the converters' default: flipping gemma/llama onto this is the + * "enable pre-transposed by default" step gated on the engine 0.40.0 + * kernel train (#951) landing — see #184. + */ + public fun packPreTransposed( + bytes: ByteArray, + encoding: TensorEncoding, + logicalShape: Shape, + ): TensorData? { + val (blockElems, bpb) = blockLayoutFor(encoding) ?: return null + require(logicalShape.rank == 2) { "packed matmul weight must be 2D, got rank ${logicalShape.rank}" } + val relaid = relayoutRowMajorToBlockMajor(bytes, logicalShape, bpb, blockElems) + val transposed = Shape(logicalShape[1], logicalShape[0]) + @Suppress("UNCHECKED_CAST") + return when (encoding) { + TensorEncoding.Q4_K -> PreTransposedQ4_K(Q4_KBlockTensorData(transposed, relaid)) as TensorData + TensorEncoding.Q5_K -> PreTransposedQ5_K(Q5_KBlockTensorData(transposed, relaid)) as TensorData + TensorEncoding.Q6_K -> PreTransposedQ6_K(Q6_KBlockTensorData(transposed, relaid)) as TensorData + TensorEncoding.Q8_0 -> PreTransposedQ8_0(Q8_0BlockTensorData(transposed, relaid)) as TensorData + TensorEncoding.Q4_0 -> PreTransposedQ4_0(Q4_0BlockTensorData(transposed, relaid)) as TensorData + TensorEncoding.Q5_0 -> PreTransposedQ5_0(Q5_0BlockTensorData(transposed, relaid)) as TensorData + TensorEncoding.Q5_1 -> PreTransposedQ5_1(Q5_1BlockTensorData(transposed, relaid)) as TensorData + else -> null + } + } +} diff --git a/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/PreTransposedWeight.kt b/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/PreTransposedWeight.kt new file mode 100644 index 00000000..228ea05f --- /dev/null +++ b/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/PreTransposedWeight.kt @@ -0,0 +1,105 @@ +package sk.ainet.lang.nn.quant + +import sk.ainet.lang.tensor.Shape +import sk.ainet.lang.tensor.data.Q4_0BlockTensorData +import sk.ainet.lang.tensor.data.Q4_0TensorData +import sk.ainet.lang.tensor.data.Q4_KBlockTensorData +import sk.ainet.lang.tensor.data.Q4_KTensorData +import sk.ainet.lang.tensor.data.Q5_0BlockTensorData +import sk.ainet.lang.tensor.data.Q5_0TensorData +import sk.ainet.lang.tensor.data.Q5_1BlockTensorData +import sk.ainet.lang.tensor.data.Q5_1TensorData +import sk.ainet.lang.tensor.data.Q5_KBlockTensorData +import sk.ainet.lang.tensor.data.Q5_KTensorData +import sk.ainet.lang.tensor.data.Q6_KBlockTensorData +import sk.ainet.lang.tensor.data.Q6_KTensorData +import sk.ainet.lang.tensor.data.Q8_0BlockTensorData +import sk.ainet.lang.tensor.data.Q8_0TensorData +import sk.ainet.lang.tensor.storage.PackedBlockStorage + +/** + * Pre-transpose marker (#184 hoist 3, the "Solution C" of the gemma 8B-OOM + * investigation): tensor data implementing this interface declares that its + * *logical shape is already the transposed `[in, out]`* a matmul consumes — + * i.e. the converter that produced it already performed (or absorbed) the + * `W.t()` that [sk.ainet.lang.nn.transformer.linearProject] would otherwise + * apply, so `linearProject` dispatches `ops.matmul(x, W)` directly and skips + * `ops.transpose` entirely. + * + * For GGUF block-quant weights this is free: the row-major → block-major + * relayout ([BlockQuantPacking.relayoutRowMajorToBlockMajor]) already stores + * the bytes in the kernels' input-block-major order, and the engine's lazy + * packed `ops.transpose` is a pure logical-shape swap over those same bytes. + * [BlockQuantPacking.packPreTransposed] performs that shape swap at pack time + * and attaches this marker, cutting the per-forward transpose wrapper + * allocation out of every projection. + * + * The explicit marker exists because a shape heuristic + * (`W.shape[0] == x.shape[-1]`) is ambiguous for square projections — see + * `linearProject`'s kdoc. The engine's own packed `ops.transpose` remains the + * fallback for unmarked weights, so this is opt-in per tensor. + * + * Engine alignment: when the engine grows a first-class pre-transposed flag + * on its tensor data (the other half of #184 (3)), this interface unifies + * with it the same way `RowDequantSource` did in #289 (typealias), keeping + * `linearProject`'s check source-compatible. + */ +public interface PreTransposedWeight + +// Internal marked delegating views over the engine's heap-packed block tensor +// data. Each implements the format's *interface* (what every engine dispatch +// site checks: `chooseQuantizedMatmulHeap`, the JVM matmul intercepts, and the +// lazy packed `ops.transpose` all match on `is Q*TensorData`) plus +// [PackedBlockStorage] (what the compile-path `TensorSpecEncoding` matches on), +// so a marked weight behaves byte-for-byte like the unmarked one everywhere +// except the `linearProject` transpose skip. Members declared by both parents +// are disambiguated explicitly to the same delegate. + +internal class PreTransposedQ4_K(private val d: Q4_KBlockTensorData) : + Q4_KTensorData by d, PackedBlockStorage by d, PreTransposedWeight { + override val shape: Shape get() = d.shape + override val blockCount: Int get() = d.blockCount + override val packedData: ByteArray get() = d.packedData +} + +internal class PreTransposedQ5_K(private val d: Q5_KBlockTensorData) : + Q5_KTensorData by d, PackedBlockStorage by d, PreTransposedWeight { + override val shape: Shape get() = d.shape + override val blockCount: Int get() = d.blockCount + override val packedData: ByteArray get() = d.packedData +} + +internal class PreTransposedQ6_K(private val d: Q6_KBlockTensorData) : + Q6_KTensorData by d, PackedBlockStorage by d, PreTransposedWeight { + override val shape: Shape get() = d.shape + override val blockCount: Int get() = d.blockCount + override val packedData: ByteArray get() = d.packedData +} + +internal class PreTransposedQ8_0(private val d: Q8_0BlockTensorData) : + Q8_0TensorData by d, PackedBlockStorage by d, PreTransposedWeight { + override val shape: Shape get() = d.shape + override val blockCount: Int get() = d.blockCount + override val packedData: ByteArray get() = d.packedData +} + +internal class PreTransposedQ4_0(private val d: Q4_0BlockTensorData) : + Q4_0TensorData by d, PackedBlockStorage by d, PreTransposedWeight { + override val shape: Shape get() = d.shape + override val blockCount: Int get() = d.blockCount + override val packedData: ByteArray get() = d.packedData +} + +internal class PreTransposedQ5_0(private val d: Q5_0BlockTensorData) : + Q5_0TensorData by d, PackedBlockStorage by d, PreTransposedWeight { + override val shape: Shape get() = d.shape + override val blockCount: Int get() = d.blockCount + override val packedData: ByteArray get() = d.packedData +} + +internal class PreTransposedQ5_1(private val d: Q5_1BlockTensorData) : + Q5_1TensorData by d, PackedBlockStorage by d, PreTransposedWeight { + override val shape: Shape get() = d.shape + override val blockCount: Int get() = d.blockCount + override val packedData: ByteArray get() = d.packedData +} diff --git a/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/transformer/LinearProjection.kt b/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/transformer/LinearProjection.kt index e2e01064..e40dd524 100644 --- a/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/transformer/LinearProjection.kt +++ b/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/transformer/LinearProjection.kt @@ -1,5 +1,6 @@ package sk.ainet.lang.nn.transformer +import sk.ainet.lang.nn.quant.PreTransposedWeight import sk.ainet.lang.tensor.Tensor import sk.ainet.lang.tensor.ops.TensorOps import sk.ainet.lang.types.DType @@ -9,26 +10,32 @@ import sk.ainet.lang.types.DType * `[out, in]` checkpoint layout. * * Intended as the single place every transformer DSL module goes through - * when projecting against a weight parameter. Today it just materialises - * the transpose and forwards to `ops.matmul`, so it's a drop-in rename - * for `ops.matmul(x, ops.transpose(W))`. + * when projecting against a weight parameter. * - * The future for this helper is Solution C from `ISSUE-skainet-8b-oom.md`: - * when a MemSeg-style converter pre-transposes a quantized weight (Q4_K / - * Q6_K dequant-and-transpose, or Q4_0 / Q8_0 via a transpose-on-the-fly - * kernel), that conversion step will set an explicit pre-transposed - * marker; this helper will read the marker and skip the transpose in the - * pre-transposed case. A shape-only heuristic (`W.shape[0] == x.shape[-1]`) - * is unsafe for square projections (e.g. `dim == qDim` on GQA-free - * configs), hence the explicit-marker requirement. + * This is Solution C from `ISSUE-skainet-8b-oom.md` (#184 hoist 3): a + * converter that already delivers the weight in the transposed `[in, out]` + * layout marks its tensor data with [PreTransposedWeight] (e.g. via + * [sk.ainet.lang.nn.quant.BlockQuantPacking.packPreTransposed]), and this + * helper skips `ops.transpose` for it, dispatching `ops.matmul(x, W)` + * directly — for packed quant weights that avoids even the engine's lazy + * shape-swap transpose wrapper on every forward. Unmarked weights take the + * classic `ops.matmul(x, ops.transpose(W))` path, where the engine's packed + * `ops.transpose` support remains the fallback. + * + * A shape-only heuristic (`W.shape[0] == x.shape[-1]`) is unsafe for square + * projections (e.g. `dim == qDim` on GQA-free configs), hence the + * explicit-marker requirement. * * @param ops active tensor operations (usually `ctx.ops`) * @param input input tensor of shape `[..., in]` - * @param weight projection weight in `[out, in]` layout + * @param weight projection weight in `[out, in]` layout — or `[in, out]` + * when its data carries the [PreTransposedWeight] marker * @return `input @ W.t()` */ public fun linearProject( ops: TensorOps, input: Tensor, weight: Tensor -): Tensor = ops.matmul(input, ops.transpose(weight)) +): Tensor = + if (weight.data is PreTransposedWeight) ops.matmul(input, weight) + else ops.matmul(input, ops.transpose(weight)) diff --git a/transformer-core/src/commonTest/kotlin/sk/ainet/lang/nn/quant/BlockQuantPackingTest.kt b/transformer-core/src/commonTest/kotlin/sk/ainet/lang/nn/quant/BlockQuantPackingTest.kt new file mode 100644 index 00000000..a08dc85f --- /dev/null +++ b/transformer-core/src/commonTest/kotlin/sk/ainet/lang/nn/quant/BlockQuantPackingTest.kt @@ -0,0 +1,166 @@ +package sk.ainet.lang.nn.quant + +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertFailsWith +import kotlin.test.assertNull +import kotlin.test.assertTrue +import sk.ainet.lang.tensor.Shape +import sk.ainet.lang.tensor.data.Q4_0BlockTensorData +import sk.ainet.lang.tensor.data.Q4_KBlockTensorData +import sk.ainet.lang.tensor.data.Q5_0BlockTensorData +import sk.ainet.lang.tensor.data.Q5_1BlockTensorData +import sk.ainet.lang.tensor.data.Q5_KBlockTensorData +import sk.ainet.lang.tensor.data.Q6_KBlockTensorData +import sk.ainet.lang.tensor.data.Q8_0BlockTensorData +import sk.ainet.lang.tensor.storage.PackedBlockStorage +import sk.ainet.lang.tensor.storage.TensorEncoding +import sk.ainet.lang.types.FP32 + +/** + * Round-trip tests for the shared GGUF-block → `*BlockTensorData` packer + * (#184 hoist 2) — the logic previously duplicated by gemma's + * `GemmaQuantLayout`, llama's `LlamaQuantLayout` and apertus' JVM converter. + * commonTest: runs on JVM and Kotlin/Native alike. + */ +class BlockQuantPackingTest { + + /** All seven packed-kernel encodings with their `(blockElems, bytesPerBlock)`. */ + private val encodings: List> = listOf( + Triple(TensorEncoding.Q4_K, 256, 144), + Triple(TensorEncoding.Q5_K, 256, 176), + Triple(TensorEncoding.Q6_K, 256, 210), + Triple(TensorEncoding.Q8_0, 32, 34), + Triple(TensorEncoding.Q4_0, 32, 18), + Triple(TensorEncoding.Q5_0, 32, 22), + Triple(TensorEncoding.Q5_1, 32, 24), + ) + + @Test + fun block_layout_matches_encoding_constants() { + for ((enc, blockElems, bpb) in encodings) { + val layout = BlockQuantPacking.blockLayoutFor(enc) + assertEquals(blockElems to bpb, layout, "layout for ${enc.name}") + // Geometry must agree with the encoding's own physicalBytes. + assertEquals( + bpb.toLong(), + enc.physicalBytes(blockElems.toLong()), + "physicalBytes(${enc.name}) disagrees with packer geometry", + ) + } + assertNull(BlockQuantPacking.blockLayoutFor(TensorEncoding.Dense(4))) + assertNull(BlockQuantPacking.blockLayoutFor(TensorEncoding.TernaryPacked)) + } + + @Test + fun relayout_is_block_level_transpose_for_every_geometry() { + for ((enc, blockElems, bpb) in encodings) { + val outDim = 3 + val blocksPerRow = 4 + val inDim = blocksPerRow * blockElems + val bytes = ByteArray(outDim * blocksPerRow * bpb) + // Tag each source block with its row-major index in its first byte. + for (i in 0 until outDim * blocksPerRow) bytes[i * bpb] = i.toByte() + + val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor( + bytes, Shape(outDim, inDim), bpb, blockElems, + ) + + // dst block (b*outDim + r) must hold src block (r*blocksPerRow + b). + for (r in 0 until outDim) { + for (b in 0 until blocksPerRow) { + val srcIdx = r * blocksPerRow + b + val dstIdx = b * outDim + r + assertEquals( + srcIdx.toByte(), relaid[dstIdx * bpb], + "${enc.name}: block ($r,$b) misplaced", + ) + } + } + } + } + + @Test + fun pack_produces_the_matching_block_tensor_data_with_relaid_bytes() { + for ((enc, blockElems, bpb) in encodings) { + val outDim = 2 + val blocksPerRow = 2 + val shape = Shape(outDim, blocksPerRow * blockElems) + val bytes = ByteArray(outDim * blocksPerRow * bpb) + for (i in 0 until outDim * blocksPerRow) bytes[i * bpb] = (i + 1).toByte() + + val td = BlockQuantPacking.pack(bytes, enc, shape) + ?: error("${enc.name}: pack unexpectedly returned null") + val expectedRelaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, bpb, blockElems) + val packedData = when (td) { + is Q4_KBlockTensorData -> { assertEquals(TensorEncoding.Q4_K, enc); td.packedData } + is Q5_KBlockTensorData -> { assertEquals(TensorEncoding.Q5_K, enc); td.packedData } + is Q6_KBlockTensorData -> { assertEquals(TensorEncoding.Q6_K, enc); td.packedData } + is Q8_0BlockTensorData -> { assertEquals(TensorEncoding.Q8_0, enc); td.packedData } + is Q4_0BlockTensorData -> { assertEquals(TensorEncoding.Q4_0, enc); td.packedData } + is Q5_0BlockTensorData -> { assertEquals(TensorEncoding.Q5_0, enc); td.packedData } + is Q5_1BlockTensorData -> { assertEquals(TensorEncoding.Q5_1, enc); td.packedData } + else -> error("${enc.name}: unexpected packed type ${td::class.simpleName}") + } + assertTrue( + expectedRelaid.contentEquals(packedData), + "${enc.name}: packedData is not the block-major relayout", + ) + assertEquals(shape, td.shape, "${enc.name}: logical shape must be preserved") + } + } + + @Test + fun packPreTransposed_swaps_shape_and_carries_the_marker() { + for ((enc, blockElems, bpb) in encodings) { + val outDim = 3 + val blocksPerRow = 2 + val shape = Shape(outDim, blocksPerRow * blockElems) + val bytes = ByteArray(outDim * blocksPerRow * bpb) + for (i in 0 until outDim * blocksPerRow) bytes[i * bpb] = (i + 1).toByte() + + val td = BlockQuantPacking.packPreTransposed(bytes, enc, shape) + ?: error("${enc.name}: packPreTransposed unexpectedly returned null") + assertTrue(td is PreTransposedWeight, "${enc.name}: missing PreTransposedWeight marker") + assertEquals( + Shape(shape[1], shape[0]), td.shape, + "${enc.name}: logical shape must be the transposed [in, out]", + ) + // Same block-major bytes as the plain pack — the transpose is a + // pure logical-shape swap, exactly what the engine's lazy packed + // ops.transpose would produce. + val storage = td as PackedBlockStorage + val expectedRelaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, bpb, blockElems) + assertTrue( + expectedRelaid.contentEquals(storage.packedData), + "${enc.name}: pre-transposed packedData must be the block-major relayout", + ) + assertEquals(enc, storage.encoding, "${enc.name}: encoding must survive the marker wrapper") + } + assertNull( + BlockQuantPacking.packPreTransposed(ByteArray(2048), TensorEncoding.Dense(4), Shape(2, 256)), + ) + } + + @Test + fun pack_returns_null_for_unpackable_encodings() { + val shape = Shape(2, 256) + assertNull(BlockQuantPacking.pack(ByteArray(2048), TensorEncoding.Dense(4), shape)) + assertNull(BlockQuantPacking.pack(ByteArray(2048), TensorEncoding.TernaryPacked, shape)) + } + + @Test + fun relayout_rejects_non_2d_and_misaligned_shapes() { + assertFailsWith { + BlockQuantPacking.relayoutRowMajorToBlockMajor(ByteArray(0), Shape(32), 18, 32) + } + assertFailsWith { + // inDim 33 not a multiple of block size 32. + BlockQuantPacking.relayoutRowMajorToBlockMajor(ByteArray(64), Shape(2, 33), 18, 32) + } + assertFailsWith { + // Buffer too small for [2, 64] @ 18 B/block (= 4 blocks = 72 B). + BlockQuantPacking.relayoutRowMajorToBlockMajor(ByteArray(71), Shape(2, 64), 18, 32) + } + } +} diff --git a/transformer-core/src/jvmTest/kotlin/sk/ainet/lang/nn/quant/LinearProjectionPreTransposedTest.kt b/transformer-core/src/jvmTest/kotlin/sk/ainet/lang/nn/quant/LinearProjectionPreTransposedTest.kt new file mode 100644 index 00000000..2df92abc --- /dev/null +++ b/transformer-core/src/jvmTest/kotlin/sk/ainet/lang/nn/quant/LinearProjectionPreTransposedTest.kt @@ -0,0 +1,159 @@ +package sk.ainet.lang.nn.quant + +import kotlin.math.abs +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue +import sk.ainet.context.DirectCpuExecutionContext +import sk.ainet.lang.nn.transformer.linearProject +import sk.ainet.lang.tensor.Shape +import sk.ainet.lang.tensor.data.TensorData +import sk.ainet.lang.tensor.storage.TensorEncoding +import sk.ainet.lang.types.FP32 + +/** + * Proves the #184-hoist-3 pre-transpose marker end to end on a real CPU + * backend: a Q5_1 weight packed with [BlockQuantPacking.packPreTransposed] + * (logical `[in, out]`, [PreTransposedWeight]-marked) must produce, through + * [linearProject]'s transpose-skipping branch, bit-identical output to the + * same bytes packed with [BlockQuantPacking.pack] and run through the classic + * `ops.matmul(x, ops.transpose(W))` path — and both must match an FP32 + * reference built from the analytic Q5_1 dequant formula. + * + * jvmTest because it needs actual packed-matmul kernel dispatch + * (ServiceLoader-discovered scalar/Panama providers); the layout/marker + * mechanics are covered platform-neutrally in [BlockQuantPackingTest]. + */ +class LinearProjectionPreTransposedTest { + + /** IEEE-754 float32 → float16 bits (round-to-nearest-even; small exact values only in tests). */ + private fun halfBits(f: Float): Int { + val bits = f.toRawBits() + val sign = (bits ushr 16) and 0x8000 + var exp = ((bits ushr 23) and 0xFF) - 127 + 15 + var mant = bits and 0x7FFFFF + if (exp <= 0) return sign // flush tiny values; tests only use exact normal halves + if (exp >= 31) return sign or 0x7C00 + // round mantissa 23 -> 10 bits + mant += 0x1000 + if (mant and 0x800000 != 0) { + mant = 0 + exp += 1 + if (exp >= 31) return sign or 0x7C00 + } + return sign or (exp shl 10) or (mant ushr 13) + } + + /** + * Build one 24-byte GGUF Q5_1 block from [d], [m] and 32 5-bit [codes], + * matching the engine layout: `d` f16 LE, `m` f16 LE, `qh` 4 bytes (bit j + * = high bit of code j, LSB-first), `qs` 16 bytes (low nibbles of codes + * 0..15 | low nibbles of codes 16..31 shifted). Dequant: `d * code + m`. + */ + private fun q5_1Block(d: Float, m: Float, codes: IntArray): ByteArray { + require(codes.size == 32) + val out = ByteArray(24) + val db = halfBits(d) + val mb = halfBits(m) + out[0] = (db and 0xFF).toByte(); out[1] = ((db ushr 8) and 0xFF).toByte() + out[2] = (mb and 0xFF).toByte(); out[3] = ((mb ushr 8) and 0xFF).toByte() + var qh = 0 + for (j in 0 until 32) if ((codes[j] ushr 4) and 1 == 1) qh = qh or (1 shl j) + for (b in 0 until 4) out[4 + b] = ((qh ushr (8 * b)) and 0xFF).toByte() + for (j in 0 until 16) out[8 + j] = ((codes[j] and 0xF) or ((codes[j + 16] and 0xF) shl 4)).toByte() + return out + } + + @Test + fun preTransposed_q5_1_matches_classic_path_and_fp32_reference() { + val outDim = 4 + val inDim = 64 // 2 blocks per row -> multi-block both ways, catches relayout mistakes + val blocksPerRow = inDim / 32 + val shape = Shape(outDim, inDim) + + // Deterministic synthetic weight: per-block (d, m) + pseudo-random 5-bit codes. + val expected = FloatArray(outDim * inDim) + val ggufBytes = ByteArray(outDim * blocksPerRow * 24) + for (r in 0 until outDim) { + for (b in 0 until blocksPerRow) { + val blockIdx = r * blocksPerRow + b + val d = 0.25f * ((blockIdx % 4) + 1) // 0.25 / 0.5 / 0.75 / 1.0 — f16-exact + val m = -2.0f + 0.5f * (blockIdx % 3) + val codes = IntArray(32) { j -> (j * 7 + r * 13 + b * 5) % 32 } + q5_1Block(d, m, codes).copyInto(ggufBytes, blockIdx * 24) + for (j in 0 until 32) expected[r * inDim + b * 32 + j] = d * codes[j] + m + } + } + + val ctx = DirectCpuExecutionContext.create() + val x = ctx.fromFloatArray( + Shape(2, inDim), FP32::class, + FloatArray(2 * inDim) { i -> ((i * 31 + 7) % 17 - 8) / 8.0f }, + ) + + // FP32 reference from the analytic dequant. + val wFp32 = ctx.fromFloatArray(shape, FP32::class, expected) + val ref = linearProject(ctx.ops, x, wFp32).data.copyToFloatArray() + + // Classic path: [out, in] packed + lazy transpose inside linearProject. + @Suppress("UNCHECKED_CAST") + val packed = BlockQuantPacking.pack(ggufBytes, TensorEncoding.Q5_1, shape) + as TensorData + val yClassic = linearProject(ctx.ops, x, ctx.fromData(packed, FP32::class)).data.copyToFloatArray() + + // Marked path: [in, out] pre-transposed, linearProject skips ops.transpose. + val pre = BlockQuantPacking.packPreTransposed(ggufBytes, TensorEncoding.Q5_1, shape) + ?: error("packPreTransposed returned null for Q5_1") + assertTrue(pre is PreTransposedWeight) + assertEquals(Shape(inDim, outDim), pre.shape) + @Suppress("UNCHECKED_CAST") + val yMarked = linearProject(ctx.ops, x, ctx.fromData(pre as TensorData, FP32::class)) + .data.copyToFloatArray() + + // Same kernel, same bytes, same dims -> bit-identical between the two packed paths. + assertTrue(yClassic.contentEquals(yMarked), "pre-transposed path diverged from classic packed path") + // And both agree with the analytic FP32 reference (fp accumulation-order tolerance). + for (i in ref.indices) { + assertTrue( + abs(ref[i] - yClassic[i]) <= 1e-3f * maxOf(1.0f, abs(ref[i])), + "packed Q5_1 [$i]: ${yClassic[i]} vs FP32 ref ${ref[i]}", + ) + } + } + + @Test + fun marker_skips_transpose_for_plain_float_data_too() { + val ctx = DirectCpuExecutionContext.create() + val outDim = 3 + val inDim = 5 + val w = FloatArray(outDim * inDim) { (it % 7 - 3).toFloat() } + val x = ctx.fromFloatArray(Shape(2, inDim), FP32::class, FloatArray(2 * inDim) { (it % 5 - 2).toFloat() }) + + val ref = linearProject( + ctx.ops, x, + ctx.fromFloatArray(Shape(outDim, inDim), FP32::class, w), + ).data.copyToFloatArray() + + // Manually transposed [in, out] float weight + marker: linearProject + // must consume it as-is (no second transpose). + val wt = FloatArray(inDim * outDim) + for (r in 0 until outDim) for (c in 0 until inDim) wt[c * outDim + r] = w[r * inDim + c] + val wtTensor = ctx.fromFloatArray(Shape(inDim, outDim), FP32::class, wt) + + class Marked(private val d: TensorData) : + TensorData by d, PreTransposedWeight + + val marked = ctx.fromData(Marked(wtTensor.data), FP32::class) + val y = linearProject(ctx.ops, x, marked).data.copyToFloatArray() + // Tolerance, not bit-equality: the marked weight is not FloatArray-backed, + // so the backend may take a different (elementwise) matmul path than the + // FloatArray fast path of the reference. + assertEquals(ref.size, y.size) + for (i in ref.indices) { + assertTrue( + abs(ref[i] - y[i]) <= 1e-4f * maxOf(1.0f, abs(ref[i])), + "marked FP32 pre-transposed weight [$i]: ${y[i]} vs ${ref[i]}", + ) + } + } +}