diff --git a/gradle/libs.versions.toml b/gradle/libs.versions.toml index 8a43b4d4..418b3015 100644 --- a/gradle/libs.versions.toml +++ b/gradle/libs.versions.toml @@ -1,5 +1,5 @@ [versions] -skainet = "0.39.1" +skainet = "0.40.0" agp = "9.3.1" jacksonDatabind = "2.22.1" jsonSchemaValidator = "3.0.6" diff --git a/llm-inference/gemma/build.gradle.kts b/llm-inference/gemma/build.gradle.kts index 68674ce8..4d2a9c92 100644 --- a/llm-inference/gemma/build.gradle.kts +++ b/llm-inference/gemma/build.gradle.kts @@ -87,6 +87,28 @@ kotlin { // Test-only: write the externalized weights to an IREE .irpa // parameter archive (RealGemmaBakeIrpaTest). implementation(libs.skainet.io.iree.params) + // NOTE (0.40.0 closure train investigation, #170/#184): this + // module deliberately does NOT pull in `skainet.backend. + // nativeCpu` the way `:llm-inference:llama`'s jvmTest does. + // Temporarily adding it while validating this closure train + // confirmed the SKaiNET#951 native (FFM) kernel really is + // faster (real-checkpoint decode: ~53% higher tok/s) and + // still byte-identical via the new pre-transposed path — but + // it also reproducibly zeroed out `GemmaQ5xPackedParityTest`'s + // synthetic Q5_0/Q5_1 byte-level checks, which exercise the + // *classic* (non-pre-transposed) packed weight through + // `linearProject`'s lazy `ops.transpose` branch. Those two + // tests construct `Q5_{0,1}BlockTensorData` directly — no + // code this PR touches — so this is a pre-existing upstream + // engine dispatch gap (native Q5_0/Q5_1 kernel × lazy + // transpose), not something introduced or fixable here. + // Production consumers of Gemma (`:llm-runtime:kgemma`) + // already depend on `skainet.backend.nativeCpu` directly and + // now get the pre-transposed path by default, which sidesteps + // this gap entirely — see `packPreTransposed`, `linearProject`. + // Left un-wired here so this module's own test suite stays + // green without masking that finding; see the closure-train + // PR description for the measurement. } } } diff --git a/llm-inference/gemma/src/commonMain/kotlin/sk/ainet/models/gemma/GemmaQuantLayout.kt b/llm-inference/gemma/src/commonMain/kotlin/sk/ainet/models/gemma/GemmaQuantLayout.kt index 7f202ec9..a3dff488 100644 --- a/llm-inference/gemma/src/commonMain/kotlin/sk/ainet/models/gemma/GemmaQuantLayout.kt +++ b/llm-inference/gemma/src/commonMain/kotlin/sk/ainet/models/gemma/GemmaQuantLayout.kt @@ -93,11 +93,23 @@ internal fun relayoutKSeriesRowMajorToBlockMajor( * decode fit the 1.9 GB board, and it runs on the NEON Q8_0 kernel. * * commonMain → works on JVM and Kotlin/Native alike (no MemSeg / Arena). + * + * @param preTransposed when `true` (the default as of the engine 0.40.0 + * native-kernel closure train, #184 (3)/#170), the result is packed via + * [BlockQuantPacking.packPreTransposed] — logical `[in, out]`, marked + * [sk.ainet.lang.nn.quant.PreTransposedWeight] — so + * [sk.ainet.lang.nn.transformer.linearProject] skips its per-forward + * `ops.transpose` for this weight. Pass `false` to get the classic + * [BlockQuantPacking.pack] `[out, in]` result instead (kept reachable, + * deprecate-don't-delete, for fallback / parity comparison against the + * pre-transposed path — see `GemmaQ5xPackedParityTest` / + * `LinearProjectionPreTransposedTest`). */ internal fun packGemmaKQuant( bytes: ByteArray, qt: GGMLQuantizationType, shape: Shape, + preTransposed: Boolean = true, ): TensorData? { val encoding = qt.toBlockEncoding() ?: return null // The legacy 32-elem formats (Q4_0/Q5_0/Q5_1) are NEW to this packed path @@ -107,11 +119,18 @@ internal fun packGemmaKQuant( // bytes with row-major strides after the lazy transpose (garbage), so the // correct degradation is `null` → the caller's FP32 dequant fallback // (#169 behavior). Q4_K/Q5_K/Q6_K/Q8_0 keep their long-standing - // unconditional packing. Under engine 0.39.0 the scalar/Panama Q5_x - // kernels already satisfy the gate; SKaiNET#951 (0.40.0) adds the native - // FFM/K-N/JNI tiers behind the same check. + // unconditional packing. Engine 0.40.0 (SKaiNET#951) added the native + // FFM/K-N/JNI Q5_0/Q5_1 kernel tiers behind the same + // [hasPackedMatmulKernel] check, which is also what makes pre-transposed + // packing safe to default to here — a weight is only ever marked + // pre-transposed once this same gate has confirmed a real packed kernel + // will consume it. if (qt in legacyPackedQuantTypes && !qt.hasPackedMatmulKernel()) return null - return BlockQuantPacking.pack(bytes, encoding, shape) + return if (preTransposed) { + BlockQuantPacking.packPreTransposed(bytes, encoding, shape) + } else { + BlockQuantPacking.pack(bytes, encoding, shape) + } } /** diff --git a/llm-inference/gemma/src/commonTest/kotlin/sk/ainet/models/gemma/GemmaQuantLayoutTest.kt b/llm-inference/gemma/src/commonTest/kotlin/sk/ainet/models/gemma/GemmaQuantLayoutTest.kt index 82f40d99..92502880 100644 --- a/llm-inference/gemma/src/commonTest/kotlin/sk/ainet/models/gemma/GemmaQuantLayoutTest.kt +++ b/llm-inference/gemma/src/commonTest/kotlin/sk/ainet/models/gemma/GemmaQuantLayoutTest.kt @@ -6,9 +6,13 @@ import kotlin.test.assertNull import kotlin.test.assertTrue import sk.ainet.context.DirectCpuExecutionContext import sk.ainet.io.gguf.GGMLQuantizationType +import sk.ainet.lang.nn.quant.PreTransposedWeight import sk.ainet.lang.tensor.Shape import sk.ainet.lang.tensor.data.Q5_KBlockTensorData +import sk.ainet.lang.tensor.data.Q5_KTensorData import sk.ainet.lang.tensor.data.Q8_0BlockTensorData +import sk.ainet.lang.tensor.data.Q8_0TensorData +import sk.ainet.lang.tensor.storage.PackedBlockStorage import sk.ainet.lang.types.FP32 import sk.ainet.lang.types.Int8 @@ -43,24 +47,53 @@ class GemmaQuantLayoutTest { } @Test - fun pack_q5k_produces_block_tensor_with_relaid_bytes() { + fun pack_q5k_defaults_to_pre_transposed_with_relaid_bytes() { + // #184 (3)/#170, engine 0.40.0 closure train: packGemmaKQuant now + // defaults to the pre-transposed marked path. val shape = Shape(2, 512) val bytes = ByteArray(2 * 2 * 176) for (i in 0 until 4) bytes[i * 176] = (i + 1).toByte() val td = packGemmaKQuant(bytes, GGMLQuantizationType.Q5_K, shape) - assertTrue(td is Q5_KBlockTensorData, "Q5_K should pack to Q5_KBlockTensorData") - // packedData is the block-major relayout of the input. + assertTrue(td is PreTransposedWeight, "Q5_K should default to the pre-transposed marked path") + assertTrue(td is Q5_KTensorData, "the marked wrapper still satisfies Q5_KTensorData dispatch checks") + assertEquals(Shape(512, 2), td.shape, "pre-transposed result carries the swapped [in, out] shape") + // packedData is still the block-major relayout of the input — only the + // logical shape + marker changed, not the bytes. + val expected = relayoutKSeriesRowMajorToBlockMajor(bytes, shape, 176) + assertTrue(td is PackedBlockStorage) + assertTrue(expected.contentEquals((td as PackedBlockStorage).packedData)) + } + + @Test + fun pack_q5k_preTransposed_false_keeps_classic_block_tensor_reachable() { + // Deprecate-don't-delete: the non-transposed packer path stays reachable + // for fallback / parity comparison against the pre-transposed default. + val shape = Shape(2, 512) + val bytes = ByteArray(2 * 2 * 176) + for (i in 0 until 4) bytes[i * 176] = (i + 1).toByte() + + val td = packGemmaKQuant(bytes, GGMLQuantizationType.Q5_K, shape, preTransposed = false) + assertTrue(td is Q5_KBlockTensorData, "Q5_K should pack to the classic Q5_KBlockTensorData when opted out") + assertEquals(shape, td.shape, "classic path keeps the checkpoint's [out, in] shape") val expected = relayoutKSeriesRowMajorToBlockMajor(bytes, shape, 176) assertTrue(expected.contentEquals(td.packedData)) } @Test - fun pack_q8_0_produces_block_tensor() { - // Q8_0 is now packed (32 elems / 34 B per block) so a tied Q8_0 lm_head - // stays packed and runs on the Q8_0 kernel instead of dequanting to FP32. + fun pack_q8_0_defaults_to_pre_transposed_block_tensor() { + // Q8_0 is packed (32 elems / 34 B per block) so a tied Q8_0 lm_head + // stays packed and runs on the Q8_0 kernel instead of dequanting to FP32; + // as of the 0.40.0 closure train it packs pre-transposed by default. val td = packGemmaKQuant(ByteArray(34), GGMLQuantizationType.Q8_0, Shape(1, 32)) - assertTrue(td is Q8_0BlockTensorData, "Q8_0 should pack to Q8_0BlockTensorData") + assertTrue(td is PreTransposedWeight, "Q8_0 should default to the pre-transposed marked path") + assertTrue(td is Q8_0TensorData, "the marked wrapper still satisfies Q8_0TensorData dispatch checks") + } + + @Test + fun pack_q8_0_preTransposed_false_keeps_classic_block_tensor_reachable() { + val td = packGemmaKQuant(ByteArray(34), GGMLQuantizationType.Q8_0, Shape(1, 32), preTransposed = false) + assertTrue(td is Q8_0BlockTensorData, "Q8_0 should pack to the classic Q8_0BlockTensorData when opted out") } @Test diff --git a/llm-inference/gemma/src/jvmMain/kotlin/sk/ainet/models/gemma/GemmaMemSegConverter.kt b/llm-inference/gemma/src/jvmMain/kotlin/sk/ainet/models/gemma/GemmaMemSegConverter.kt index 5569c78f..ba85bdbb 100644 --- a/llm-inference/gemma/src/jvmMain/kotlin/sk/ainet/models/gemma/GemmaMemSegConverter.kt +++ b/llm-inference/gemma/src/jvmMain/kotlin/sk/ainet/models/gemma/GemmaMemSegConverter.kt @@ -10,14 +10,10 @@ import sk.ainet.lang.nn.quant.BlockQuantPacking import sk.ainet.lang.tensor.Shape import sk.ainet.lang.tensor.Tensor import sk.ainet.lang.tensor.data.IntArrayTensorData -import sk.ainet.lang.tensor.data.Q4_KBlockTensorData -import sk.ainet.lang.tensor.data.Q5_0BlockTensorData -import sk.ainet.lang.tensor.data.Q5_1BlockTensorData -import sk.ainet.lang.tensor.data.Q5_KBlockTensorData -import sk.ainet.lang.tensor.data.Q6_KBlockTensorData import sk.ainet.lang.tensor.data.Q4MemorySegmentTensorData import sk.ainet.lang.tensor.data.Q8MemorySegmentTensorData import sk.ainet.lang.tensor.data.TensorData +import sk.ainet.lang.tensor.storage.TensorEncoding import sk.ainet.lang.types.DType import sk.ainet.lang.types.FP32 @@ -31,18 +27,19 @@ import sk.ainet.lang.types.FP32 * MemorySegment-backed Q4 / Q8 tensor data the DSL path can feed to its * SIMD matmul kernels. * - * **Different from `MemSegWeightConverter` (Llama)**: no pre-transpose for - * K-series weights. The Llama-path runtime (`LlamaRuntime.linearProject`) - * picks direct-matmul-vs-transpose-then-matmul based on a shape check, and - * pre-transposing FP32 K-series weights lets it take the direct branch. The - * DSL path's [sk.ainet.lang.nn.transformer.linearProject] always transposes, - * so pre-transposing here would produce double-transposed weights and the - * wrong math. Instead, for K-series we dequant to FP32 and keep the - * canonical `[out, in]` layout — the DSL transposes at runtime like any - * other FP32 weight. That loses the Q4_K / Q6_K memory savings but keeps - * numerical correctness until a quant-aware DSL dispatch (recognising - * `linearProject` on a Q4_K tensor and skipping the transpose) is - * implemented in the backend. + * **K-series and legacy (Q4_0/Q5_0/Q5_1) weights are packed pre-transposed** + * (#184 (3)/#170, engine 0.40.0 native-kernel closure train): each goes + * through [BlockQuantPacking.packPreTransposed], which relays out the GGUF + * blocks and marks the result [sk.ainet.lang.nn.quant.PreTransposedWeight], + * so [sk.ainet.lang.nn.transformer.linearProject] dispatches + * `ops.matmul(x, W)` directly and skips its per-forward `ops.transpose` for + * these weights — no FP32 round-trip, no double-transpose risk, since the + * marker (not a shape heuristic) is what `linearProject` checks. Q4_0/Q5_0/Q5_1 + * stay gated on [hasPackedMatmulKernel] (kernel-availability, not engine + * version) and fall back to the #169 FP32 dequant when no packed kernel is + * registered — see `packPreTransposed`'s call sites below and + * [BlockQuantPacking.pack] (kept reachable for the non-transposed + * fallback / parity-comparison path, deprecate-don't-delete). * * Q4_0 and Q8_0 keep their packed quantized form. The CPU backend's * `ops.transpose` does a lazy shape-swap on those MemSeg tensors (no data @@ -154,49 +151,55 @@ private fun convertOne( ctx.fromData(data as TensorData, advertisedDtype) as Tensor } GGMLQuantizationType.Q4_K -> { - // Keep Q4_K packed, but re-layout the GGUF-stored bytes from - // row-major block order `[row, block]` to the input-block-major - // order `[block, row]` that `JvmQuantizedVectorKernels.matmulQ4_KVec` - // indexes via `(blockIdx * outputDim + o) * bytesPerBlock`. - // - // The lazy Q4_K transpose in `DefaultCpuOpsJvm` expects this - // layout; combined, a Q4_K_M Gemma 4 E2B checkpoint (3.2 GB on - // disk) stays near that footprint in RAM instead of inflating - // to ~18 GB FP32. - val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 144, 256) - val data = Q4_KBlockTensorData.fromRawBytes(shape, relaid) + // Keep Q4_K packed AND pre-transposed (#184 (3), engine 0.40.0 + // closure train): packPreTransposed re-layouts the GGUF-stored + // bytes from row-major block order `[row, block]` to the + // input-block-major order `[block, row]` that + // `JvmQuantizedVectorKernels.matmulQ4_KVec` indexes via + // `(blockIdx * outputDim + o) * bytesPerBlock`, then swaps the + // logical shape to `[in, out]` and marks the result + // `PreTransposedWeight` so `linearProject` skips its per-forward + // `ops.transpose` entirely (previously a lazy shape-swap via + // `DefaultCpuOpsJvm`'s Q4_K transpose; now zero wrapper + // allocation). A Q4_K_M Gemma 4 E2B checkpoint (3.2 GB on disk) + // stays near that footprint in RAM instead of inflating to + // ~18 GB FP32. + val data = BlockQuantPacking.packPreTransposed(bytes, TensorEncoding.Q4_K, shape) + ?: error("packPreTransposed returned null for Q4_K — expected an unconditional kernel") ctx.fromData(data as TensorData, advertisedDtype) as Tensor } GGMLQuantizationType.Q6_K -> { - // Same packed-path treatment as Q4_K, enabled by the - // `matmulQ6_KVec` kernel + lazy transpose in `DefaultCpuOpsJvm`. - // Gemma 4 E2B Q4_K_M uses Q6_K for ffn_gate/up/down, attn_v, - // token_embd, and the tied lm_head — keeping these packed saves - // ~12 GB of FP32 bloat (and the corresponding 7.5 GB per-forward - // transpose transient). Sanity-checked against FP32 dequant and - // Q6_K packed produces identical tokens — kernel math is right. - val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 210, 256) - val data = Q6_KBlockTensorData.fromRawBytes(shape, relaid) + // Same packed-and-pre-transposed treatment as Q4_K, enabled by + // the `matmulQ6_KVec` kernel. Gemma 4 E2B Q4_K_M uses Q6_K for + // ffn_gate/up/down, attn_v, token_embd, and the tied lm_head — + // keeping these packed saves ~12 GB of FP32 bloat (and the + // corresponding 7.5 GB per-forward transpose transient). + // Sanity-checked against FP32 dequant and Q6_K packed produces + // identical tokens — kernel math is right. + val data = BlockQuantPacking.packPreTransposed(bytes, TensorEncoding.Q6_K, shape) + ?: error("packPreTransposed returned null for Q6_K — expected an unconditional kernel") ctx.fromData(data as TensorData, advertisedDtype) as Tensor } GGMLQuantizationType.Q5_K -> { - // Same packed-path treatment as Q4_K/Q6_K, enabled by the Q5_K - // matmul kernel (scalar/Panama/native) + the lazy Q5_K transpose - // in DefaultCpuOps. FunctionGemma-270M Q5_K_M ships most attn/FFN - // weights as Q5_K, so keeping them packed (176 B/block) avoids the - // FP32 inflation and runs the in-kernel dequant matmul. - val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 176, 256) - val data = Q5_KBlockTensorData.fromRawBytes(shape, relaid) + // Same packed-and-pre-transposed treatment as Q4_K/Q6_K, enabled + // by the Q5_K matmul kernel (scalar/Panama/native). + // FunctionGemma-270M Q5_K_M ships most attn/FFN weights as Q5_K, + // so keeping them packed (176 B/block) avoids the FP32 inflation + // and runs the in-kernel dequant matmul. + val data = BlockQuantPacking.packPreTransposed(bytes, TensorEncoding.Q5_K, shape) + ?: error("packPreTransposed returned null for Q5_K — expected an unconditional kernel") ctx.fromData(data as TensorData, advertisedDtype) as Tensor } GGMLQuantizationType.Q5_1 -> { - // Packed-path treatment for the 32-elem/24-byte legacy blocks - // (#170): FunctionGemma-270M "Q5_K_M" ships attn_q/attn_k and + // Packed-and-pre-transposed treatment for the 32-elem/24-byte + // legacy blocks (#170, engine 0.40.0 native-kernel closure + // train): FunctionGemma-270M "Q5_K_M" ships attn_q/attn_k and // ffn_gate/ffn_up as Q5_1 (81 of 236 tensors), which until #170 // took the FP32 dequant fallback below. The engine has Q5_1 // kernels (scalar + Panama since 0.39.0; native FFM/K-N/JNI from - // SKaiNET#951 / 0.40.0) + the lazy Q5_1 transpose, so the weights - // stay packed and run the in-kernel dequant matmul. + // SKaiNET#951 / 0.40.0), so the weights stay packed, marked + // `PreTransposedWeight`, and run the in-kernel dequant matmul + // with no per-forward transpose. // // Gated on kernel AVAILABILITY (not engine version): if no // registered provider carries a Q5_1 kernel, packing would send @@ -204,8 +207,8 @@ private fun convertOne( // the block-major bytes after the lazy transpose — so fall back // to the always-correct #169 FP32 dequant instead. if (qt.hasPackedMatmulKernel()) { - val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 24, 32) - val data = Q5_1BlockTensorData.fromRawBytes(shape, relaid) + val data = BlockQuantPacking.packPreTransposed(bytes, TensorEncoding.Q5_1, shape) + ?: error("packPreTransposed returned null for Q5_1 despite hasPackedMatmulKernel()==true") ctx.fromData(data as TensorData, advertisedDtype) as Tensor } else { dequantPackedToFp32(bytes, qt, shape, ctx) @@ -214,8 +217,8 @@ private fun convertOne( GGMLQuantizationType.Q5_0 -> { // Same as Q5_1, 22-byte blocks (f16 d + qh + qs, symmetric). if (qt.hasPackedMatmulKernel()) { - val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 22, 32) - val data = Q5_0BlockTensorData.fromRawBytes(shape, relaid) + val data = BlockQuantPacking.packPreTransposed(bytes, TensorEncoding.Q5_0, shape) + ?: error("packPreTransposed returned null for Q5_0 despite hasPackedMatmulKernel()==true") ctx.fromData(data as TensorData, advertisedDtype) as Tensor } else { dequantPackedToFp32(bytes, qt, shape, ctx) diff --git a/llm-inference/llama/src/commonMain/kotlin/sk/ainet/models/llama/LlamaQuantLayout.kt b/llm-inference/llama/src/commonMain/kotlin/sk/ainet/models/llama/LlamaQuantLayout.kt index d2c2b211..c5086d5e 100644 --- a/llm-inference/llama/src/commonMain/kotlin/sk/ainet/models/llama/LlamaQuantLayout.kt +++ b/llm-inference/llama/src/commonMain/kotlin/sk/ainet/models/llama/LlamaQuantLayout.kt @@ -69,11 +69,20 @@ internal fun relayoutKSeriesRowMajorToBlockMajor( * without a packed kernel (caller dequantizes those to FP32). Delegates to the shared * [BlockQuantPacking] packer (#184 hoist 2), which covers Q4_K / Q5_K / Q6_K / Q8_0 plus * Q4_0 / Q5_0 / Q5_1 (#170). + * + * @param preTransposed when `true` (the default as of the engine 0.40.0 native-kernel + * closure train, #184 (3)/#170), packs via [BlockQuantPacking.packPreTransposed] — + * logical `[in, out]`, marked [sk.ainet.lang.nn.quant.PreTransposedWeight] — so + * [sk.ainet.lang.nn.transformer.linearProject] skips its per-forward `ops.transpose` + * for this weight. Pass `false` for the classic [BlockQuantPacking.pack] `[out, in]` + * result (kept reachable, deprecate-don't-delete, for fallback / parity comparison). + * Mirrors `packGemmaKQuant`. */ internal fun packLlamaKQuant( bytes: ByteArray, qt: GGMLQuantizationType, shape: Shape, + preTransposed: Boolean = true, ): TensorData? { val encoding = qt.toBlockEncoding() ?: return null // Legacy 32-elem formats (Q4_0/Q5_0/Q5_1) are new to this packed path @@ -82,7 +91,11 @@ internal fun packLlamaKQuant( // which misreads block-major bytes after the lazy transpose. `null` → // the caller's FP32 dequant fallback. Mirrors `packGemmaKQuant`. if (qt in legacyPackedQuantTypes && !qt.hasPackedMatmulKernel()) return null - return BlockQuantPacking.pack(bytes, encoding, shape) + return if (preTransposed) { + BlockQuantPacking.packPreTransposed(bytes, encoding, shape) + } else { + BlockQuantPacking.pack(bytes, encoding, shape) + } } /** See `GemmaQuantLayout.legacyPackedQuantTypes` — kernel-gated new formats. */ diff --git a/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/BlockQuantPacking.kt b/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/BlockQuantPacking.kt index 997f230a..afa0c4d4 100644 --- a/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/BlockQuantPacking.kt +++ b/transformer-core/src/commonMain/kotlin/sk/ainet/lang/nn/quant/BlockQuantPacking.kt @@ -124,9 +124,15 @@ public object BlockQuantPacking { * [pack] (callers dequantize those to FP32 and keep the transposing * `linearProject` path). * - * Not yet the converters' default: flipping gemma/llama onto this is the - * "enable pre-transposed by default" step gated on the engine 0.40.0 - * kernel train (#951) landing — see #184. + * The converters' default as of the engine 0.40.0 native-kernel closure + * train (SKaiNET#951 landed, gate confirmed live): `GemmaQuantLayout + * .packGemmaKQuant` / `LlamaQuantLayout.packLlamaKQuant` and + * `GemmaMemSegConverter`'s JVM MemSeg path call this instead of [pack] + * whenever the kernel-availability gate ([hasPackedMatmulKernel]) has + * confirmed a packed kernel exists — see #184, #170. [pack] itself is + * kept reachable (deprecate-don't-delete) as the non-transposed fallback + * / parity-comparison path, and both `pack*` functions still expose a + * `preTransposed` parameter to opt back into it. */ public fun packPreTransposed( bytes: ByteArray,