diff --git a/gradle/libs.versions.toml b/gradle/libs.versions.toml index 418b3015..2d9aa8c7 100644 --- a/gradle/libs.versions.toml +++ b/gradle/libs.versions.toml @@ -1,5 +1,5 @@ [versions] -skainet = "0.40.0" +skainet = "0.40.1" agp = "9.3.1" jacksonDatabind = "2.22.1" jsonSchemaValidator = "3.0.6" diff --git a/llm-inference/apertus/src/jvmMain/kotlin/sk/ainet/models/apertus/ApertusMemSegConverter.kt b/llm-inference/apertus/src/jvmMain/kotlin/sk/ainet/models/apertus/ApertusMemSegConverter.kt index e63ef00c..43f64818 100644 --- a/llm-inference/apertus/src/jvmMain/kotlin/sk/ainet/models/apertus/ApertusMemSegConverter.kt +++ b/llm-inference/apertus/src/jvmMain/kotlin/sk/ainet/models/apertus/ApertusMemSegConverter.kt @@ -6,11 +6,10 @@ import sk.ainet.io.gguf.dequant.DequantOps import sk.ainet.lang.nn.quant.BlockQuantPacking import sk.ainet.lang.tensor.Shape import sk.ainet.lang.tensor.Tensor -import sk.ainet.lang.tensor.data.Q4_KBlockTensorData import sk.ainet.lang.tensor.data.Q4MemorySegmentTensorData -import sk.ainet.lang.tensor.data.Q6_KBlockTensorData import sk.ainet.lang.tensor.data.Q8MemorySegmentTensorData import sk.ainet.lang.tensor.data.TensorData +import sk.ainet.lang.tensor.storage.TensorEncoding import sk.ainet.lang.types.DType import sk.ainet.lang.types.FP32 import java.lang.foreign.Arena @@ -27,10 +26,14 @@ import java.lang.foreign.Arena * rank-1 byte tensors the loader produces under `NATIVE_OPTIMIZED`. With * this step, each quantized weight ends up as the right wrapper: * - * - `Q4_K` → [Q4_KBlockTensorData] (relayout from GGUF row-major - * `[row, block]` order to the input-block-major `[block, row]` order - * `JvmQuantizedVectorKernels.matmulQ4_KVec` expects). 144-byte blocks. - * - `Q6_K` → [Q6_KBlockTensorData]. 210-byte blocks. + * - `Q4_K` → [BlockQuantPacking.packPreTransposed]: input-block-major relayout + * + `[in, out]` shape + `PreTransposedWeight` marker, so `linearProject` + * skips `ops.transpose` entirely (aligning apertus with the gemma/llama + * converters — the previous inlined relayout under an unmarked `[out, in]` + * shape relied on the pre-0.40.1 shape-swap-only packed transpose and is + * silently corrupted by the physical block-grid permutation the engine + * performs since 0.40.1). 144-byte blocks. + * - `Q6_K` → same pre-transposed packing. 210-byte blocks. * - `Q4_0` → [Q4MemorySegmentTensorData] (arena-allocated, 64-byte aligned). * - `Q8_0` → [Q8MemorySegmentTensorData]. * - `Q5_K` → fallback: dequant to FP32. Apertus-8B-Instruct-2509 Q4_K_S has @@ -111,14 +114,14 @@ private fun convertOne( } GGMLQuantizationType.Q4_K -> { - val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, logicalShape, BYTES_PER_Q4_K_BLOCK, K_SERIES_BLOCK_SIZE) - val data = Q4_KBlockTensorData.fromRawBytes(logicalShape, relaid) + val data = BlockQuantPacking.packPreTransposed(bytes, TensorEncoding.Q4_K, logicalShape) + ?: error("ApertusMemSegConverter: packPreTransposed returned null for Q4_K ('$name')") ctx.fromData(data as TensorData, advertisedDtype) as Tensor } GGMLQuantizationType.Q6_K -> { - val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, logicalShape, BYTES_PER_Q6_K_BLOCK, K_SERIES_BLOCK_SIZE) - val data = Q6_KBlockTensorData.fromRawBytes(logicalShape, relaid) + val data = BlockQuantPacking.packPreTransposed(bytes, TensorEncoding.Q6_K, logicalShape) + ?: error("ApertusMemSegConverter: packPreTransposed returned null for Q6_K ('$name')") ctx.fromData(data as TensorData, advertisedDtype) as Tensor } @@ -135,8 +138,6 @@ private fun convertOne( } } -private const val BYTES_PER_Q4_K_BLOCK = 144 -private const val BYTES_PER_Q6_K_BLOCK = 210 private const val K_SERIES_BLOCK_SIZE = 256 /** diff --git a/llm-inference/apertus/src/jvmTest/kotlin/sk/ainet/models/apertus/ApertusMemSegConverterSyntheticTest.kt b/llm-inference/apertus/src/jvmTest/kotlin/sk/ainet/models/apertus/ApertusMemSegConverterSyntheticTest.kt new file mode 100644 index 00000000..f0d94405 --- /dev/null +++ b/llm-inference/apertus/src/jvmTest/kotlin/sk/ainet/models/apertus/ApertusMemSegConverterSyntheticTest.kt @@ -0,0 +1,123 @@ +package sk.ainet.models.apertus + +import java.lang.foreign.Arena +import kotlin.math.abs +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue +import sk.ainet.context.DirectCpuExecutionContext +import sk.ainet.io.gguf.GGMLQuantizationType +import sk.ainet.lang.nn.quant.PreTransposedWeight +import sk.ainet.lang.nn.transformer.linearProject +import sk.ainet.lang.tensor.Shape +import sk.ainet.lang.tensor.data.Q4_KBlockTensorData +import sk.ainet.lang.tensor.data.Q6_KBlockTensorData +import sk.ainet.lang.tensor.data.TensorData +import sk.ainet.lang.tensor.storage.PackedBlockStorage +import sk.ainet.lang.types.FP32 + +/** + * Synthetic (no real checkpoint) coverage for the Apertus K-quant converter + * path — the gap that let the SKaiNET 0.40.1 layout-contract regression go + * unobserved here: [ApertusRealGgufLoadingTest] needs a real GGUF and skips + * by default, and the previous converter emitted relaid bytes under an + * unmarked `[out, in]` shape, which `linearProject`'s per-forward + * `ops.transpose` silently corrupts on engines >= 0.40.1. + * + * Asserts the converter now emits [PreTransposedWeight]-marked `[in, out]` + * tensors (mirroring gemma/llama) whose `linearProject` output matches an + * FP32 reference built from the canonical dequant of the same bytes. + */ +class ApertusMemSegConverterSyntheticTest { + + /** Deterministic block bytes with FP16 scale fields pinned to 0.25/0.125. */ + private fun buildBlocks(blockCount: Int, bytesPerBlock: Int, f16Offsets: List): ByteArray { + val out = ByteArray(blockCount * bytesPerBlock) + for (b in 0 until blockCount) { + val base = b * bytesPerBlock + for (j in 0 until bytesPerBlock) out[base + j] = ((b * 37 + j * 11 + 5) % 251).toByte() + f16Offsets.forEachIndexed { i, off -> + out[base + off] = 0x00 + out[base + off + 1] = if (i == 0) 0x34 else 0x30 // 0.25f / 0.125f + } + } + return out + } + + private fun metadata() = ApertusModelMetadata( + architecture = "apertus", + embeddingLength = 512, + contextLength = 128, + blockCount = 1, + headCount = 2, + kvHeadCount = 1, + feedForwardLength = 512, + ropeDimensionCount = null, + vocabSize = 100, + ) + + private fun assertConvertedParity(qt: GGMLQuantizationType, bytesPerBlock: Int, f16Offsets: List) { + val outDim = 4 + val inDim = 512 // blocksPerRow = 2: multi-block both grid dimensions + val shape = Shape(outDim, inDim) + val name = "blk.0.attn_q.weight" + val bytes = buildBlocks(outDim * (inDim / 256), bytesPerBlock, f16Offsets) + + val ctx = DirectCpuExecutionContext.create() + // Placeholder for the loader's rank-1 byte tensor; the converter + // replaces it by key from the quantBytes sidecar. + val placeholder = ctx.fromFloatArray(Shape(1), FP32::class, floatArrayOf(0f)) + val weights = ApertusWeights( + metadata = metadata(), + tensors = mapOf(name to placeholder), + quantTypes = mapOf(name to qt), + logicalShapes = mapOf(name to shape), + quantBytes = mapOf(name to bytes), + ) + + Arena.ofConfined().use { arena -> + val converted = convertApertusWeightsToMemSeg(weights, ctx, arena) + val w = converted.tensors[name] ?: error("converted weight missing") + + assertTrue( + w.data is PreTransposedWeight, + "$qt: converter must emit a PreTransposedWeight-marked tensor (unmarked [out,in] + " + + "relaid bytes is corrupted by the >= 0.40.1 physical packed transpose)", + ) + assertEquals(Shape(inDim, outDim), w.shape, "$qt: pre-transposed [in, out] shape") + + // FP32 reference: canonical dequant of the same bytes. + @Suppress("UNCHECKED_CAST") + val canonical = when (qt) { + GGMLQuantizationType.Q4_K -> Q4_KBlockTensorData(shape, bytes) + else -> Q6_KBlockTensorData(shape, bytes) + } as TensorData + val wFlat = (canonical as PackedBlockStorage).toFloatArray() + for (v in wFlat) assertTrue(v.isFinite(), "$qt: non-finite dequant value $v") + val wRef = ctx.fromFloatArray(shape, FP32::class, wFlat) + + val x = ctx.fromFloatArray( + Shape(2, inDim), FP32::class, + FloatArray(2 * inDim) { i -> ((i * 29 + 3) % 23 - 11) / 11.0f }, + ) + val ref = linearProject(ctx.ops, x, wRef).data.copyToFloatArray() + val y = linearProject(ctx.ops, x, w).data.copyToFloatArray() + + assertEquals(ref.size, y.size) + for (i in ref.indices) { + assertTrue( + abs(ref[i] - y[i]) <= 1e-3f * maxOf(1.0f, abs(ref[i])), + "$qt converted[$i]=${y[i]} vs FP32-dequant ref ${ref[i]}", + ) + } + } + } + + @Test + fun q4k_converted_weight_is_marked_and_matches_fp32_reference() = + assertConvertedParity(GGMLQuantizationType.Q4_K, bytesPerBlock = 144, f16Offsets = listOf(0, 2)) + + @Test + fun q6k_converted_weight_is_marked_and_matches_fp32_reference() = + assertConvertedParity(GGMLQuantizationType.Q6_K, bytesPerBlock = 210, f16Offsets = listOf(208)) +} diff --git a/llm-inference/gemma/src/commonTest/kotlin/sk/ainet/models/gemma/GemmaQuantLayoutTest.kt b/llm-inference/gemma/src/commonTest/kotlin/sk/ainet/models/gemma/GemmaQuantLayoutTest.kt index 92502880..f48ce84f 100644 --- a/llm-inference/gemma/src/commonTest/kotlin/sk/ainet/models/gemma/GemmaQuantLayoutTest.kt +++ b/llm-inference/gemma/src/commonTest/kotlin/sk/ainet/models/gemma/GemmaQuantLayoutTest.kt @@ -76,8 +76,10 @@ class GemmaQuantLayoutTest { val td = packGemmaKQuant(bytes, GGMLQuantizationType.Q5_K, shape, preTransposed = false) assertTrue(td is Q5_KBlockTensorData, "Q5_K should pack to the classic Q5_KBlockTensorData when opted out") assertEquals(shape, td.shape, "classic path keeps the checkpoint's [out, in] shape") - val expected = relayoutKSeriesRowMajorToBlockMajor(bytes, shape, 176) - assertTrue(expected.contentEquals(td.packedData)) + // Canonical checkpoint bytes verbatim: the classic path defers the + // block-grid permutation to the engine's physical packed ops.transpose + // (>= 0.40.1) inside linearProject. + assertTrue(bytes.contentEquals(td.packedData)) } @Test diff --git a/llm-inference/gemma/src/jvmTest/kotlin/sk/ainet/models/gemma/GemmaDslQ4KTest.kt b/llm-inference/gemma/src/jvmTest/kotlin/sk/ainet/models/gemma/GemmaDslQ4KTest.kt index 449bde1c..660ee86e 100644 --- a/llm-inference/gemma/src/jvmTest/kotlin/sk/ainet/models/gemma/GemmaDslQ4KTest.kt +++ b/llm-inference/gemma/src/jvmTest/kotlin/sk/ainet/models/gemma/GemmaDslQ4KTest.kt @@ -20,11 +20,11 @@ import sk.ainet.lang.types.FP32 * running on the corresponding dequantised FP32 weights. * * Validates: - * 1. `relayoutQ4_KRowMajorToBlockMajor` correctly reshuffles GGUF - * row-major Q4_K bytes into the input-block-major layout expected by - * `JvmQuantizedVectorKernels.matmulQ4_KVec`. - * 2. The lazy `ops.transpose(Q4_KTensorData)` in `DefaultCpuOpsJvm` - * composes correctly with the Q4_K matmul kernel. + * 1. Canonical (GGUF row-major) Q4_K bytes under an `[out, in]` shape flow + * through the engine's physical packed `ops.transpose` (>= 0.40.1, + * SKaiNET#968) into the input-block-major layout the Q4_K matmul kernel + * expects. + * 2. That transpose composes correctly with the Q4_K matmul kernel dispatch. * 3. Running a tiny Gemma DSL model with Q4_K-backed weights produces * logits close to the FP32 baseline (i.e., no wrong-math bug from * layout or dispatch). @@ -38,9 +38,9 @@ class GemmaDslQ4KTest { // dim and ffnDim chosen so each Q4_K weight has multiple input blocks per // row — `inDim = 512 → blocksPerRow = 2`. With one block per row the - // `relayoutKSeriesRowMajorToBlockMajor` is the identity transform and the - // test silently misses any relayout bug. Two blocks per row exercises the - // real ggml strided codes layout end-to-end. + // canonical and input-block-major orders coincide and the test silently + // misses any block-order bug. Two blocks per row exercises the real ggml + // strided codes layout end-to-end. private val dim = 512 private val nHeads = 2 private val nKvHeads = 1 @@ -176,8 +176,12 @@ class GemmaDslQ4KTest { @Suppress("UNCHECKED_CAST") private fun q4kTensor(rows: Int, cols: Int, bytes: ByteArray): Tensor { - val relaid = relayoutQ4_KRowMajorToBlockMajor(bytes, Shape(rows, cols)) - val data = Q4_KBlockTensorData.fromRawBytes(Shape(rows, cols), relaid) + // Canonical row-major bytes verbatim: the classic [out, in] path defers + // the block-grid permutation to the engine's physical packed + // ops.transpose (>= 0.40.1) inside linearProject. Relayouting here + // (as before the SKaiNET#968 layout-contract fix landed in 0.40.1) + // would get the blocks permuted a second time. + val data = Q4_KBlockTensorData.fromRawBytes(Shape(rows, cols), bytes) return ctx.fromData(data as TensorData, FP32::class) } diff --git a/llm-inference/gemma/src/jvmTest/kotlin/sk/ainet/models/gemma/GemmaQ5xPackedParityTest.kt b/llm-inference/gemma/src/jvmTest/kotlin/sk/ainet/models/gemma/GemmaQ5xPackedParityTest.kt index 233f2aff..42dd4a51 100644 --- a/llm-inference/gemma/src/jvmTest/kotlin/sk/ainet/models/gemma/GemmaQ5xPackedParityTest.kt +++ b/llm-inference/gemma/src/jvmTest/kotlin/sk/ainet/models/gemma/GemmaQ5xPackedParityTest.kt @@ -14,13 +14,13 @@ import sk.ainet.io.JvmRandomAccessSource import sk.ainet.io.gguf.GGMLQuantizationType import sk.ainet.io.gguf.dequant.DequantOps import sk.ainet.io.model.QuantPolicy +import sk.ainet.lang.nn.quant.BlockQuantPacking import sk.ainet.lang.nn.transformer.linearProject import sk.ainet.lang.tensor.Shape -import sk.ainet.lang.tensor.data.Q5_0BlockTensorData -import sk.ainet.lang.tensor.data.Q5_1BlockTensorData import sk.ainet.lang.tensor.data.Q5_0TensorData import sk.ainet.lang.tensor.data.Q5_1TensorData import sk.ainet.lang.tensor.data.TensorData +import sk.ainet.lang.tensor.storage.TensorEncoding import sk.ainet.lang.types.FP32 /** @@ -30,8 +30,8 @@ import sk.ainet.lang.types.FP32 * * Two layers of evidence: * - [synthetic tests] byte-level parity per format: the converter's exact - * packed pipeline (row-major → block-major relayout + `*BlockTensorData` + - * lazy-transpose matmul) vs the converter's exact FP32 fallback pipeline + * packed pipeline (canonical bytes via `BlockQuantPacking.pack` + the + * engine's physical packed transpose matmul) vs the converter's exact FP32 fallback pipeline * (`DequantOps.dequantFromBytes` + identity col→row transpose). Q5_0 is * covered here only — the FunctionGemma checkpoint carries no Q5_0 tensor. * - [real checkpoint] FunctionGemma-270M "Q5_K_M" ships 81 of 236 tensors as @@ -128,14 +128,12 @@ class GemmaQ5xPackedParityTest { val wRef = ctx.fromFloatArray(shape, FP32::class, rowMajor) val ref = linearProject(ctx.ops, x, wRef).data.copyToFloatArray() - // Packed: the converter's packed pipeline, verbatim. - @Suppress("DEPRECATION") - val relaid = relayoutKSeriesRowMajorToBlockMajor(gguf, shape, bpb, blockSize = 32) + // Packed: the converter's classic packed pipeline, verbatim — canonical + // checkpoint bytes, [out, in] shape; the engine's physical packed + // ops.transpose (>= 0.40.1) inside linearProject produces kernel order. + val encoding = if (qt == GGMLQuantizationType.Q5_1) TensorEncoding.Q5_1 else TensorEncoding.Q5_0 @Suppress("UNCHECKED_CAST") - val data = when (qt) { - GGMLQuantizationType.Q5_1 -> Q5_1BlockTensorData.fromRawBytes(shape, relaid) - else -> Q5_0BlockTensorData.fromRawBytes(shape, relaid) - } as TensorData + val data = BlockQuantPacking.pack(gguf, encoding, shape) as TensorData val y = linearProject(ctx.ops, x, ctx.fromData(data, FP32::class)).data.copyToFloatArray() assertEquals(ref.size, y.size) diff --git a/llm-inference/llama/src/commonTest/kotlin/sk/ainet/models/llama/LlamaQuantLayoutTest.kt b/llm-inference/llama/src/commonTest/kotlin/sk/ainet/models/llama/LlamaQuantLayoutTest.kt new file mode 100644 index 00000000..5269ab7f --- /dev/null +++ b/llm-inference/llama/src/commonTest/kotlin/sk/ainet/models/llama/LlamaQuantLayoutTest.kt @@ -0,0 +1,91 @@ +package sk.ainet.models.llama + +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertNull +import kotlin.test.assertTrue +import sk.ainet.io.gguf.GGMLQuantizationType +import sk.ainet.lang.nn.quant.BlockQuantPacking +import sk.ainet.lang.nn.quant.PreTransposedWeight +import sk.ainet.lang.tensor.Shape +import sk.ainet.lang.tensor.data.Q4_KBlockTensorData +import sk.ainet.lang.tensor.data.Q4_KTensorData +import sk.ainet.lang.tensor.storage.PackedBlockStorage +import sk.ainet.lang.types.FP32 + +/** + * Unit tests for the commonMain Llama quant layout helpers — the Llama mirror + * of `GemmaQuantLayoutTest` (which existed since #184; this one was missing + * until the SKaiNET#968/0.40.1 layout-contract regression showed every + * converter needs its own packing coverage). Runs on every target. + */ +class LlamaQuantLayoutTest { + + private val q4kBpb = 144 + private val kBlock = 256 + + @Test + fun pack_q4k_defaults_to_pre_transposed_with_relaid_bytes() { + // [outDim=2, inDim=512] -> blocksPerRow=2, multi-block both ways. + val shape = Shape(2, 512) + val bytes = ByteArray(2 * 2 * q4kBpb) + for (i in 0 until 4) bytes[i * q4kBpb] = (i + 1).toByte() + + val td = packLlamaKQuant(bytes, GGMLQuantizationType.Q4_K, shape) + ?: error("packLlamaKQuant returned null for Q4_K") + assertTrue(td is PreTransposedWeight, "Q4_K should default to the pre-transposed marked path") + assertTrue(td is Q4_KTensorData, "the marked wrapper still satisfies Q4_KTensorData dispatch checks") + assertEquals(Shape(512, 2), td.shape, "pre-transposed result carries the swapped [in, out] shape") + // Bytes are the input-block-major relayout — kernel feed order, + // computed once at load time. + val expected = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, q4kBpb, kBlock) + assertTrue(td is PackedBlockStorage) + assertTrue(expected.contentEquals((td as PackedBlockStorage).packedData)) + } + + @Test + fun pack_q4k_preTransposed_false_keeps_canonical_bytes_verbatim() { + val shape = Shape(2, 512) + val bytes = ByteArray(2 * 2 * q4kBpb) + for (i in 0 until 4) bytes[i * q4kBpb] = (i + 1).toByte() + + val td = packLlamaKQuant(bytes, GGMLQuantizationType.Q4_K, shape, preTransposed = false) + assertTrue(td is Q4_KBlockTensorData, "Q4_K should pack to the classic Q4_KBlockTensorData when opted out") + assertEquals(shape, td.shape, "classic path keeps the checkpoint's [out, in] shape") + // Canonical checkpoint bytes verbatim: the classic path defers the + // block-grid permutation to the engine's physical packed ops.transpose + // (>= 0.40.1) inside linearProject. + assertTrue(bytes.contentEquals(td.packedData)) + } + + @Test + fun pack_unsupported_quant_returns_null() { + assertNull(packLlamaKQuant(ByteArray(20), GGMLQuantizationType.Q4_1, Shape(1, 32))) + } + + @Test + fun logicalShapeFor_maps_the_2d_matmul_weights() { + val md = LlamaModelMetadata( + architecture = "llama", + embeddingLength = 64, + contextLength = 128, + blockCount = 2, + headCount = 4, + kvHeadCount = 2, + feedForwardLength = 256, + ropeDimensionCount = null, + vocabSize = 1000, + ) + assertEquals(Shape(1000, 64), logicalShapeFor(LlamaTensorNames.TOKEN_EMBEDDINGS, md)) + assertEquals(Shape(1000, 64), logicalShapeFor(LlamaTensorNames.OUTPUT_WEIGHT, md)) + assertEquals(Shape(64, 64), logicalShapeFor("blk.0.attn_q.weight", md)) + assertEquals(Shape(32, 64), logicalShapeFor("blk.0.attn_k.weight", md)) + assertEquals(Shape(32, 64), logicalShapeFor("blk.1.attn_v.weight", md)) + assertEquals(Shape(64, 64), logicalShapeFor("blk.0.attn_output.weight", md)) + assertEquals(Shape(256, 64), logicalShapeFor("blk.0.ffn_gate.weight", md)) + assertEquals(Shape(256, 64), logicalShapeFor("blk.1.ffn_up.weight", md)) + assertEquals(Shape(64, 256), logicalShapeFor("blk.0.ffn_down.weight", md)) + assertNull(logicalShapeFor("blk.0.attn_norm.weight", md)) + assertNull(logicalShapeFor("output_norm.weight", md)) + } +} diff --git a/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/BlockQuantPacking.kt b/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/BlockQuantPacking.kt index afa0c4d4..a0b80dd8 100644 --- a/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/BlockQuantPacking.kt +++ b/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/BlockQuantPacking.kt @@ -19,16 +19,33 @@ import sk.ainet.lang.types.DType * copy of the same two steps (gemma `GemmaQuantLayout.packGemmaKQuant`, llama * `LlamaQuantLayout.packLlamaKQuant`, apertus' JVM converter): * - * 1. re-layout the checkpoint's row-major block order to the input-block-major - * order the `matmulQ*` kernels index (`(blockIdx * outDim + r)`), and - * 2. wrap the relaid bytes in the engine block tensor-data type for the format. + * 1. (pre-transposed path only) re-layout the checkpoint's row-major block + * order to the input-block-major order the `matmulQ*` kernels index + * (`(blockIdx * outDim + r)`), and + * 2. wrap the bytes in the engine block tensor-data type for the format. * * This object is that logic, once, keyed by the engine's [TensorEncoding] * (a `skainet-lang-core` type — so this file needs no GGUF dependency; model * modules map their `GGMLQuantizationType` to an encoding and keep only weight * *selection* and naming). Supported: the seven formats with first-class CPU - * matmul kernels + lazy `ops.transpose` support — Q4_K / Q5_K / Q6_K / Q8_0 / - * Q4_0 / Q5_0 / Q5_1. + * matmul kernels — Q4_K / Q5_K / Q6_K / Q8_0 / Q4_0 / Q5_0 / Q5_1. + * + * ## Byte-order contract (engine ≥ 0.40.1) + * + * The engine's `ops.transpose` on packed data performs a **physical** + * canonical→input-block-major block-grid permutation (SKaiNET#968 fix) and + * therefore **requires its input bytes in the checkpoint's canonical row-major + * block order**. Consequently: + * + * - [pack] stores the checkpoint bytes **verbatim** (canonical, `[out, in]`) + * and relies on the engine transpose inside `linearProject`'s classic path + * to produce kernel order — an O(bytes) copy per forward. + * - [packPreTransposed] performs the relayout **once at load time** and marks + * the result [PreTransposedWeight] so `linearProject` never transposes it. + * This is the production path. + * + * Feeding a [pack] result to a pre-0.40.1 engine (shape-swap-only transpose) + * re-creates bug SKaiNET#968; this packer requires the 0.40.1 pin. */ public object BlockQuantPacking { @@ -80,15 +97,34 @@ public object BlockQuantPacking { return out } + /** + * Validates the same preconditions [relayoutRowMajorToBlockMajor] enforces + * (rank 2, block-aligned inDim, sufficient bytes) without copying — so + * [pack]'s no-relayout path fails as loudly as the relayouting path. + */ + private fun requirePackable(bytes: ByteArray, shape: Shape, bytesPerBlock: Int, blockSize: Int) { + require(shape.rank == 2) { "packed matmul weight must be 2D, got rank ${shape.rank}" } + val outDim = shape[0] + val inDim = shape[1] + require(inDim % blockSize == 0) { "packed weight inDim ($inDim) must be a multiple of $blockSize" } + val expected = outDim.toLong() * (inDim / blockSize).toLong() * bytesPerBlock.toLong() + require(bytes.size.toLong() >= expected) { + "packed byte buffer ${bytes.size} < expected $expected for [$outDim, $inDim] @ ${bytesPerBlock}B/block" + } + } + /** * Pack raw checkpoint `bytes` of logical `[out, in]` [shape] into the - * heap-packed block tensor data the matmul kernels read directly, - * performing the row-major → block-major relayout. Returns `null` for - * encodings without a packed kernel (callers dequantize those to FP32). + * heap-packed block tensor data for the format, keeping the checkpoint's + * **canonical row-major block order verbatim** (no relayout). Returns + * `null` for encodings without a packed kernel (callers dequantize those + * to FP32). * - * The result flows through the engine's lazy packed `ops.transpose` - * (pure shape swap) into the quantized matmul kernel dispatch, so a - * weight packed here never round-trips through FP32. + * Canonical order is what the engine's packed `ops.transpose` (≥ 0.40.1) + * requires: `linearProject`'s classic path transposes the weight every forward, + * physically permuting the block grid into the kernels' input-block-major + * order. Prefer [packPreTransposed], which pays that permutation once at + * load time instead. */ public fun pack( bytes: ByteArray, @@ -96,28 +132,29 @@ public object BlockQuantPacking { shape: Shape, ): TensorData? { val (blockElems, bpb) = blockLayoutFor(encoding) ?: return null - val relaid = relayoutRowMajorToBlockMajor(bytes, shape, bpb, blockElems) + requirePackable(bytes, shape, bpb, blockElems) @Suppress("UNCHECKED_CAST") return when (encoding) { - TensorEncoding.Q4_K -> Q4_KBlockTensorData(shape, relaid) as TensorData - TensorEncoding.Q5_K -> Q5_KBlockTensorData(shape, relaid) as TensorData - TensorEncoding.Q6_K -> Q6_KBlockTensorData(shape, relaid) as TensorData - TensorEncoding.Q8_0 -> Q8_0BlockTensorData(shape, relaid) as TensorData - TensorEncoding.Q4_0 -> Q4_0BlockTensorData(shape, relaid) as TensorData - TensorEncoding.Q5_0 -> Q5_0BlockTensorData(shape, relaid) as TensorData - TensorEncoding.Q5_1 -> Q5_1BlockTensorData(shape, relaid) as TensorData + TensorEncoding.Q4_K -> Q4_KBlockTensorData(shape, bytes) as TensorData + TensorEncoding.Q5_K -> Q5_KBlockTensorData(shape, bytes) as TensorData + TensorEncoding.Q6_K -> Q6_KBlockTensorData(shape, bytes) as TensorData + TensorEncoding.Q8_0 -> Q8_0BlockTensorData(shape, bytes) as TensorData + TensorEncoding.Q4_0 -> Q4_0BlockTensorData(shape, bytes) as TensorData + TensorEncoding.Q5_0 -> Q5_0BlockTensorData(shape, bytes) as TensorData + TensorEncoding.Q5_1 -> Q5_1BlockTensorData(shape, bytes) as TensorData else -> null } } /** * Like [pack], but returns the weight *already transposed*: logical shape - * `[in, out]` over the same block-major bytes, marked with - * [PreTransposedWeight] so [sk.ainet.lang.nn.transformer.linearProject] - * skips `ops.transpose` and feeds the tensor straight to the packed - * matmul dispatch (#184 hoist 3). This is exactly the tensor data the - * engine's lazy packed `ops.transpose` would produce from [pack]'s result - * — shape swap, zero copy — minus the per-forward wrapper allocation. + * `[in, out]` over **relaid (input-block-major, kernel-order) bytes**, + * marked with [PreTransposedWeight] so + * [sk.ainet.lang.nn.transformer.linearProject] skips `ops.transpose` and + * feeds the tensor straight to the packed matmul dispatch (#184 hoist 3). + * This is exactly the tensor data the engine's packed `ops.transpose` + * (≥ 0.40.1, physical block-grid permutation) would produce from [pack]'s + * canonical result — computed once at load time instead of every forward. * * [logicalShape] is still the checkpoint's `[out, in]`; the swap happens * here. Returns `null` for encodings without a packed kernel, same as diff --git a/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/PreTransposedWeight.kt b/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/PreTransposedWeight.kt index 228ea05f..fa6ef73b 100644 --- a/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/PreTransposedWeight.kt +++ b/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/PreTransposedWeight.kt @@ -26,13 +26,15 @@ import sk.ainet.lang.tensor.storage.PackedBlockStorage * apply, so `linearProject` dispatches `ops.matmul(x, W)` directly and skips * `ops.transpose` entirely. * - * For GGUF block-quant weights this is free: the row-major → block-major - * relayout ([BlockQuantPacking.relayoutRowMajorToBlockMajor]) already stores - * the bytes in the kernels' input-block-major order, and the engine's lazy - * packed `ops.transpose` is a pure logical-shape swap over those same bytes. - * [BlockQuantPacking.packPreTransposed] performs that shape swap at pack time - * and attaches this marker, cutting the per-forward transpose wrapper - * allocation out of every projection. + * For GGUF block-quant weights this is the production path: the row-major → + * block-major relayout ([BlockQuantPacking.relayoutRowMajorToBlockMajor]) + * stores the bytes in the kernels' input-block-major order once at load time — + * the same physical block-grid permutation the engine's packed `ops.transpose` + * (≥ 0.40.1, SKaiNET#968 fix) would otherwise perform on every forward. + * [BlockQuantPacking.packPreTransposed] performs the relayout plus the + * logical-shape swap at pack time and attaches this marker, cutting both the + * per-forward byte permutation and the wrapper allocation out of every + * projection. * * The explicit marker exists because a shape heuristic * (`W.shape[0] == x.shape[-1]`) is ambiguous for square projections — see diff --git a/transformer-core/src/commonTest/kotlin/sk/ainet/lang/nn/quant/BlockQuantPackingTest.kt b/transformer-core/src/commonTest/kotlin/sk/ainet/lang/nn/quant/BlockQuantPackingTest.kt index a08dc85f..389b4325 100644 --- a/transformer-core/src/commonTest/kotlin/sk/ainet/lang/nn/quant/BlockQuantPackingTest.kt +++ b/transformer-core/src/commonTest/kotlin/sk/ainet/lang/nn/quant/BlockQuantPackingTest.kt @@ -81,7 +81,7 @@ class BlockQuantPackingTest { } @Test - fun pack_produces_the_matching_block_tensor_data_with_relaid_bytes() { + fun pack_produces_the_matching_block_tensor_data_with_canonical_bytes_verbatim() { for ((enc, blockElems, bpb) in encodings) { val outDim = 2 val blocksPerRow = 2 @@ -91,7 +91,6 @@ class BlockQuantPackingTest { val td = BlockQuantPacking.pack(bytes, enc, shape) ?: error("${enc.name}: pack unexpectedly returned null") - val expectedRelaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, bpb, blockElems) val packedData = when (td) { is Q4_KBlockTensorData -> { assertEquals(TensorEncoding.Q4_K, enc); td.packedData } is Q5_KBlockTensorData -> { assertEquals(TensorEncoding.Q5_K, enc); td.packedData } @@ -102,14 +101,32 @@ class BlockQuantPackingTest { is Q5_1BlockTensorData -> { assertEquals(TensorEncoding.Q5_1, enc); td.packedData } else -> error("${enc.name}: unexpected packed type ${td::class.simpleName}") } + // Canonical row-major order is kept verbatim: the engine's packed + // ops.transpose (>= 0.40.1) performs the physical block-grid + // permutation itself and requires this order as its input. assertTrue( - expectedRelaid.contentEquals(packedData), - "${enc.name}: packedData is not the block-major relayout", + bytes.contentEquals(packedData), + "${enc.name}: packedData must be the checkpoint bytes verbatim (canonical order)", ) assertEquals(shape, td.shape, "${enc.name}: logical shape must be preserved") } } + @Test + fun pack_rejects_non_2d_and_misaligned_shapes() { + assertFailsWith { + BlockQuantPacking.pack(ByteArray(64), TensorEncoding.Q4_0, Shape(64)) + } + assertFailsWith { + // inDim 33 not a multiple of block size 32. + BlockQuantPacking.pack(ByteArray(64), TensorEncoding.Q4_0, Shape(2, 33)) + } + assertFailsWith { + // Buffer too small for [2, 64] @ 18 B/block (= 4 blocks = 72 B). + BlockQuantPacking.pack(ByteArray(71), TensorEncoding.Q4_0, Shape(2, 64)) + } + } + @Test fun packPreTransposed_swaps_shape_and_carries_the_marker() { for ((enc, blockElems, bpb) in encodings) { @@ -126,9 +143,9 @@ class BlockQuantPackingTest { Shape(shape[1], shape[0]), td.shape, "${enc.name}: logical shape must be the transposed [in, out]", ) - // Same block-major bytes as the plain pack — the transpose is a - // pure logical-shape swap, exactly what the engine's lazy packed - // ops.transpose would produce. + // Relaid block-major bytes — the load-time equivalent of the + // engine's physical packed ops.transpose (>= 0.40.1) applied to + // pack()'s canonical result. val storage = td as PackedBlockStorage val expectedRelaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, bpb, blockElems) assertTrue( diff --git a/transformer-core/src/jvmTest/kotlin/sk/ainet/lang/nn/quant/LinearProjectionPackedParityMatrixTest.kt b/transformer-core/src/jvmTest/kotlin/sk/ainet/lang/nn/quant/LinearProjectionPackedParityMatrixTest.kt new file mode 100644 index 00000000..e397b556 --- /dev/null +++ b/transformer-core/src/jvmTest/kotlin/sk/ainet/lang/nn/quant/LinearProjectionPackedParityMatrixTest.kt @@ -0,0 +1,139 @@ +package sk.ainet.lang.nn.quant + +import kotlin.math.abs +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue +import sk.ainet.context.DirectCpuExecutionContext +import sk.ainet.lang.nn.transformer.linearProject +import sk.ainet.lang.tensor.Shape +import sk.ainet.lang.tensor.data.TensorData +import sk.ainet.lang.tensor.storage.PackedBlockStorage +import sk.ainet.lang.tensor.storage.TensorEncoding +import sk.ainet.lang.types.FP32 + +/** + * Layout-parity matrix for **all seven** packed matmul formats: for each + * encoding, the classic path ([BlockQuantPacking.pack] → `linearProject`'s + * `ops.matmul(x, ops.transpose(W))`) and the pre-transposed path + * ([BlockQuantPacking.packPreTransposed] → transpose-skipping branch) must + * produce bit-identical output, and both must match an FP32 reference matmul + * over the canonical dequant of the same bytes. + * + * This is the cross-check the 0.40.0→0.40.1 engine regression slipped + * through: the engine's packed `ops.transpose` changed from a shape-swap + * (which required kernel-order input bytes) to a physical block-grid + * permutation (which requires canonical input bytes), and only Q5_1 had a + * test on the classic path. Multi-block in both grid dimensions, so any + * block-order convention mismatch between packer, transpose, and kernel + * shows up as a value error rather than coincidentally passing. + * + * The synthetic blocks carry pseudorandom quant payloads with all FP16 + * scale/min fields pinned to small exact halves, so every format dequantizes + * to finite, well-conditioned values without this test needing to reimplement + * per-format dequant math (the engine's own `toFloatArray()` provides the + * reference weight matrix — it reads canonical block order, matching + * [BlockQuantPacking.pack]'s verbatim bytes). + * + * jvmTest: needs real packed-kernel dispatch (ServiceLoader scalar/Panama + * providers), same as [LinearProjectionPreTransposedTest]. + */ +class LinearProjectionPackedParityMatrixTest { + + private data class Fmt( + val encoding: TensorEncoding, + val blockElems: Int, + val bytesPerBlock: Int, + /** Byte offsets of FP16 fields inside a block to pin to an exact small half. */ + val f16Offsets: List, + ) + + // FP16 field positions per ggml block layout, as mirrored by the engine's + // *TensorData constants (OFFSET_D = 208 for Q6_K; d/dmin at 0/2 for K-quants; + // d (+ m for Q5_1) at 0 (/2) for the 32-element legacy formats). + private val formats = listOf( + Fmt(TensorEncoding.Q4_0, 32, 18, listOf(0)), + Fmt(TensorEncoding.Q8_0, 32, 34, listOf(0)), + Fmt(TensorEncoding.Q5_0, 32, 22, listOf(0)), + Fmt(TensorEncoding.Q5_1, 32, 24, listOf(0, 2)), + Fmt(TensorEncoding.Q4_K, 256, 144, listOf(0, 2)), + Fmt(TensorEncoding.Q5_K, 256, 176, listOf(0, 2)), + Fmt(TensorEncoding.Q6_K, 256, 210, listOf(208)), + ) + + /** 0.25f and 0.125f as FP16 little-endian byte pairs. */ + private val halfQuarter = byteArrayOf(0x00, 0x34) + private val halfEighth = byteArrayOf(0x00, 0x30) + + private fun buildBlocks(fmt: Fmt, blockCount: Int): ByteArray { + val out = ByteArray(blockCount * fmt.bytesPerBlock) + for (b in 0 until blockCount) { + val base = b * fmt.bytesPerBlock + for (j in 0 until fmt.bytesPerBlock) { + // Deterministic pseudorandom payload, distinct per block so a + // misplaced block always changes dequant values. + out[base + j] = ((b * 31 + j * 7 + 13) % 251).toByte() + } + fmt.f16Offsets.forEachIndexed { i, off -> + val half = if (i == 0) halfQuarter else halfEighth + out[base + off] = half[0] + out[base + off + 1] = half[1] + } + } + return out + } + + @Test + fun classic_and_preTransposed_paths_match_fp32_reference_for_all_seven_formats() { + val ctx = DirectCpuExecutionContext.create() + for (fmt in formats) { + val outDim = 4 + val blocksPerRow = 2 // multi-block along the input dim + val inDim = blocksPerRow * fmt.blockElems // and outDim (4) > blocksPerRow (2): non-square block grid + val shape = Shape(outDim, inDim) + val ggufBytes = buildBlocks(fmt, outDim * blocksPerRow) + + val x = ctx.fromFloatArray( + Shape(2, inDim), FP32::class, + FloatArray(2 * inDim) { i -> ((i * 31 + 7) % 17 - 8) / 8.0f }, + ) + + // Canonical packed tensor ([out,in], checkpoint bytes verbatim). + @Suppress("UNCHECKED_CAST") + val packed = BlockQuantPacking.pack(ggufBytes, fmt.encoding, shape) + as TensorData + + // FP32 reference: engine's canonical dequant of the same bytes. + val wFlat = (packed as PackedBlockStorage).toFloatArray() + assertEquals(outDim * inDim, wFlat.size, "${fmt.encoding.name}: dequant size") + for (v in wFlat) assertTrue(v.isFinite(), "${fmt.encoding.name}: non-finite dequant value $v") + val wFp32 = ctx.fromFloatArray(shape, FP32::class, wFlat) + val ref = linearProject(ctx.ops, x, wFp32).data.copyToFloatArray() + + // Classic path: canonical [out,in] + engine ops.transpose every forward. + val yClassic = linearProject(ctx.ops, x, ctx.fromData(packed, FP32::class)).data.copyToFloatArray() + + // Pre-transposed path: [in,out] + marker, transpose skipped. + val pre = BlockQuantPacking.packPreTransposed(ggufBytes, fmt.encoding, shape) + ?: error("${fmt.encoding.name}: packPreTransposed returned null") + assertTrue(pre is PreTransposedWeight, "${fmt.encoding.name}: missing marker") + assertEquals(Shape(inDim, outDim), pre.shape, "${fmt.encoding.name}: pre-transposed shape") + @Suppress("UNCHECKED_CAST") + val yPre = linearProject(ctx.ops, x, ctx.fromData(pre as TensorData, FP32::class)) + .data.copyToFloatArray() + + // Same kernel, same (post-transpose vs pre-relaid) bytes -> bit-identical. + assertTrue( + yClassic.contentEquals(yPre), + "${fmt.encoding.name}: classic (transpose) path diverged from pre-transposed path", + ) + // Both agree with the FP32 reference (accumulation-order tolerance). + for (i in ref.indices) { + assertTrue( + abs(ref[i] - yClassic[i]) <= 1e-3f * maxOf(1.0f, abs(ref[i])), + "${fmt.encoding.name}[$i]: packed ${yClassic[i]} vs FP32 ref ${ref[i]}", + ) + } + } + } +}