Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion gradle/libs.versions.toml
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
[versions]
skainet = "0.39.1"
skainet = "0.40.0"
agp = "9.3.1"
jacksonDatabind = "2.22.1"
jsonSchemaValidator = "3.0.6"
Expand Down
22 changes: 22 additions & 0 deletions llm-inference/gemma/build.gradle.kts
Original file line number Diff line number Diff line change
Expand Up @@ -87,6 +87,28 @@ kotlin {
// Test-only: write the externalized weights to an IREE .irpa
// parameter archive (RealGemmaBakeIrpaTest).
implementation(libs.skainet.io.iree.params)
// NOTE (0.40.0 closure train investigation, #170/#184): this
// module deliberately does NOT pull in `skainet.backend.
// nativeCpu` the way `:llm-inference:llama`'s jvmTest does.
// Temporarily adding it while validating this closure train
// confirmed the SKaiNET#951 native (FFM) kernel really is
// faster (real-checkpoint decode: ~53% higher tok/s) and
// still byte-identical via the new pre-transposed path — but
// it also reproducibly zeroed out `GemmaQ5xPackedParityTest`'s
// synthetic Q5_0/Q5_1 byte-level checks, which exercise the
// *classic* (non-pre-transposed) packed weight through
// `linearProject`'s lazy `ops.transpose` branch. Those two
// tests construct `Q5_{0,1}BlockTensorData` directly — no
// code this PR touches — so this is a pre-existing upstream
// engine dispatch gap (native Q5_0/Q5_1 kernel × lazy
// transpose), not something introduced or fixable here.
// Production consumers of Gemma (`:llm-runtime:kgemma`)
// already depend on `skainet.backend.nativeCpu` directly and
// now get the pre-transposed path by default, which sidesteps
// this gap entirely — see `packPreTransposed`, `linearProject`.
// Left un-wired here so this module's own test suite stays
// green without masking that finding; see the closure-train
// PR description for the measurement.
}
}
}
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -93,11 +93,23 @@ internal fun relayoutKSeriesRowMajorToBlockMajor(
* decode fit the 1.9 GB board, and it runs on the NEON Q8_0 kernel.
*
* commonMain → works on JVM and Kotlin/Native alike (no MemSeg / Arena).
*
* @param preTransposed when `true` (the default as of the engine 0.40.0
* native-kernel closure train, #184 (3)/#170), the result is packed via
* [BlockQuantPacking.packPreTransposed] — logical `[in, out]`, marked
* [sk.ainet.lang.nn.quant.PreTransposedWeight] — so
* [sk.ainet.lang.nn.transformer.linearProject] skips its per-forward
* `ops.transpose` for this weight. Pass `false` to get the classic
* [BlockQuantPacking.pack] `[out, in]` result instead (kept reachable,
* deprecate-don't-delete, for fallback / parity comparison against the
* pre-transposed path — see `GemmaQ5xPackedParityTest` /
* `LinearProjectionPreTransposedTest`).
*/
internal fun <T : DType> packGemmaKQuant(
bytes: ByteArray,
qt: GGMLQuantizationType,
shape: Shape,
preTransposed: Boolean = true,
): TensorData<T, *>? {
val encoding = qt.toBlockEncoding() ?: return null
// The legacy 32-elem formats (Q4_0/Q5_0/Q5_1) are NEW to this packed path
Expand All @@ -107,11 +119,18 @@ internal fun <T : DType> packGemmaKQuant(
// bytes with row-major strides after the lazy transpose (garbage), so the
// correct degradation is `null` → the caller's FP32 dequant fallback
// (#169 behavior). Q4_K/Q5_K/Q6_K/Q8_0 keep their long-standing
// unconditional packing. Under engine 0.39.0 the scalar/Panama Q5_x
// kernels already satisfy the gate; SKaiNET#951 (0.40.0) adds the native
// FFM/K-N/JNI tiers behind the same check.
// unconditional packing. Engine 0.40.0 (SKaiNET#951) added the native
// FFM/K-N/JNI Q5_0/Q5_1 kernel tiers behind the same
// [hasPackedMatmulKernel] check, which is also what makes pre-transposed
// packing safe to default to here — a weight is only ever marked
// pre-transposed once this same gate has confirmed a real packed kernel
// will consume it.
if (qt in legacyPackedQuantTypes && !qt.hasPackedMatmulKernel()) return null
return BlockQuantPacking.pack(bytes, encoding, shape)
return if (preTransposed) {
BlockQuantPacking.packPreTransposed(bytes, encoding, shape)
} else {
BlockQuantPacking.pack(bytes, encoding, shape)
}
}

/**
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -6,9 +6,13 @@ import kotlin.test.assertNull
import kotlin.test.assertTrue
import sk.ainet.context.DirectCpuExecutionContext
import sk.ainet.io.gguf.GGMLQuantizationType
import sk.ainet.lang.nn.quant.PreTransposedWeight
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.data.Q5_KBlockTensorData
import sk.ainet.lang.tensor.data.Q5_KTensorData
import sk.ainet.lang.tensor.data.Q8_0BlockTensorData
import sk.ainet.lang.tensor.data.Q8_0TensorData
import sk.ainet.lang.tensor.storage.PackedBlockStorage
import sk.ainet.lang.types.FP32
import sk.ainet.lang.types.Int8

Expand Down Expand Up @@ -43,24 +47,53 @@ class GemmaQuantLayoutTest {
}

@Test
fun pack_q5k_produces_block_tensor_with_relaid_bytes() {
fun pack_q5k_defaults_to_pre_transposed_with_relaid_bytes() {
// #184 (3)/#170, engine 0.40.0 closure train: packGemmaKQuant now
// defaults to the pre-transposed marked path.
val shape = Shape(2, 512)
val bytes = ByteArray(2 * 2 * 176)
for (i in 0 until 4) bytes[i * 176] = (i + 1).toByte()

val td = packGemmaKQuant<FP32>(bytes, GGMLQuantizationType.Q5_K, shape)
assertTrue(td is Q5_KBlockTensorData, "Q5_K should pack to Q5_KBlockTensorData")
// packedData is the block-major relayout of the input.
assertTrue(td is PreTransposedWeight, "Q5_K should default to the pre-transposed marked path")
assertTrue(td is Q5_KTensorData, "the marked wrapper still satisfies Q5_KTensorData dispatch checks")
assertEquals(Shape(512, 2), td.shape, "pre-transposed result carries the swapped [in, out] shape")
// packedData is still the block-major relayout of the input — only the
// logical shape + marker changed, not the bytes.
val expected = relayoutKSeriesRowMajorToBlockMajor(bytes, shape, 176)
assertTrue(td is PackedBlockStorage)
assertTrue(expected.contentEquals((td as PackedBlockStorage).packedData))
}

@Test
fun pack_q5k_preTransposed_false_keeps_classic_block_tensor_reachable() {
// Deprecate-don't-delete: the non-transposed packer path stays reachable
// for fallback / parity comparison against the pre-transposed default.
val shape = Shape(2, 512)
val bytes = ByteArray(2 * 2 * 176)
for (i in 0 until 4) bytes[i * 176] = (i + 1).toByte()

val td = packGemmaKQuant<FP32>(bytes, GGMLQuantizationType.Q5_K, shape, preTransposed = false)
assertTrue(td is Q5_KBlockTensorData, "Q5_K should pack to the classic Q5_KBlockTensorData when opted out")
assertEquals(shape, td.shape, "classic path keeps the checkpoint's [out, in] shape")
val expected = relayoutKSeriesRowMajorToBlockMajor(bytes, shape, 176)
assertTrue(expected.contentEquals(td.packedData))
}

@Test
fun pack_q8_0_produces_block_tensor() {
// Q8_0 is now packed (32 elems / 34 B per block) so a tied Q8_0 lm_head
// stays packed and runs on the Q8_0 kernel instead of dequanting to FP32.
fun pack_q8_0_defaults_to_pre_transposed_block_tensor() {
// Q8_0 is packed (32 elems / 34 B per block) so a tied Q8_0 lm_head
// stays packed and runs on the Q8_0 kernel instead of dequanting to FP32;
// as of the 0.40.0 closure train it packs pre-transposed by default.
val td = packGemmaKQuant<FP32>(ByteArray(34), GGMLQuantizationType.Q8_0, Shape(1, 32))
assertTrue(td is Q8_0BlockTensorData, "Q8_0 should pack to Q8_0BlockTensorData")
assertTrue(td is PreTransposedWeight, "Q8_0 should default to the pre-transposed marked path")
assertTrue(td is Q8_0TensorData, "the marked wrapper still satisfies Q8_0TensorData dispatch checks")
}

@Test
fun pack_q8_0_preTransposed_false_keeps_classic_block_tensor_reachable() {
val td = packGemmaKQuant<FP32>(ByteArray(34), GGMLQuantizationType.Q8_0, Shape(1, 32), preTransposed = false)
assertTrue(td is Q8_0BlockTensorData, "Q8_0 should pack to the classic Q8_0BlockTensorData when opted out")
}

@Test
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -10,14 +10,10 @@ import sk.ainet.lang.nn.quant.BlockQuantPacking
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.Tensor
import sk.ainet.lang.tensor.data.IntArrayTensorData
import sk.ainet.lang.tensor.data.Q4_KBlockTensorData
import sk.ainet.lang.tensor.data.Q5_0BlockTensorData
import sk.ainet.lang.tensor.data.Q5_1BlockTensorData
import sk.ainet.lang.tensor.data.Q5_KBlockTensorData
import sk.ainet.lang.tensor.data.Q6_KBlockTensorData
import sk.ainet.lang.tensor.data.Q4MemorySegmentTensorData
import sk.ainet.lang.tensor.data.Q8MemorySegmentTensorData
import sk.ainet.lang.tensor.data.TensorData
import sk.ainet.lang.tensor.storage.TensorEncoding
import sk.ainet.lang.types.DType
import sk.ainet.lang.types.FP32

Expand All @@ -31,18 +27,19 @@ import sk.ainet.lang.types.FP32
* MemorySegment-backed Q4 / Q8 tensor data the DSL path can feed to its
* SIMD matmul kernels.
*
* **Different from `MemSegWeightConverter` (Llama)**: no pre-transpose for
* K-series weights. The Llama-path runtime (`LlamaRuntime.linearProject`)
* picks direct-matmul-vs-transpose-then-matmul based on a shape check, and
* pre-transposing FP32 K-series weights lets it take the direct branch. The
* DSL path's [sk.ainet.lang.nn.transformer.linearProject] always transposes,
* so pre-transposing here would produce double-transposed weights and the
* wrong math. Instead, for K-series we dequant to FP32 and keep the
* canonical `[out, in]` layout — the DSL transposes at runtime like any
* other FP32 weight. That loses the Q4_K / Q6_K memory savings but keeps
* numerical correctness until a quant-aware DSL dispatch (recognising
* `linearProject` on a Q4_K tensor and skipping the transpose) is
* implemented in the backend.
* **K-series and legacy (Q4_0/Q5_0/Q5_1) weights are packed pre-transposed**
* (#184 (3)/#170, engine 0.40.0 native-kernel closure train): each goes
* through [BlockQuantPacking.packPreTransposed], which relays out the GGUF
* blocks and marks the result [sk.ainet.lang.nn.quant.PreTransposedWeight],
* so [sk.ainet.lang.nn.transformer.linearProject] dispatches
* `ops.matmul(x, W)` directly and skips its per-forward `ops.transpose` for
* these weights — no FP32 round-trip, no double-transpose risk, since the
* marker (not a shape heuristic) is what `linearProject` checks. Q4_0/Q5_0/Q5_1
* stay gated on [hasPackedMatmulKernel] (kernel-availability, not engine
* version) and fall back to the #169 FP32 dequant when no packed kernel is
* registered — see `packPreTransposed`'s call sites below and
* [BlockQuantPacking.pack] (kept reachable for the non-transposed
* fallback / parity-comparison path, deprecate-don't-delete).
*
* Q4_0 and Q8_0 keep their packed quantized form. The CPU backend's
* `ops.transpose` does a lazy shape-swap on those MemSeg tensors (no data
Expand Down Expand Up @@ -154,58 +151,64 @@ private fun <T : DType, V> convertOne(
ctx.fromData(data as TensorData<FP32, Float>, advertisedDtype) as Tensor<T, V>
}
GGMLQuantizationType.Q4_K -> {
// Keep Q4_K packed, but re-layout the GGUF-stored bytes from
// row-major block order `[row, block]` to the input-block-major
// order `[block, row]` that `JvmQuantizedVectorKernels.matmulQ4_KVec`
// indexes via `(blockIdx * outputDim + o) * bytesPerBlock`.
//
// The lazy Q4_K transpose in `DefaultCpuOpsJvm` expects this
// layout; combined, a Q4_K_M Gemma 4 E2B checkpoint (3.2 GB on
// disk) stays near that footprint in RAM instead of inflating
// to ~18 GB FP32.
val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 144, 256)
val data = Q4_KBlockTensorData.fromRawBytes(shape, relaid)
// Keep Q4_K packed AND pre-transposed (#184 (3), engine 0.40.0
// closure train): packPreTransposed re-layouts the GGUF-stored
// bytes from row-major block order `[row, block]` to the
// input-block-major order `[block, row]` that
// `JvmQuantizedVectorKernels.matmulQ4_KVec` indexes via
// `(blockIdx * outputDim + o) * bytesPerBlock`, then swaps the
// logical shape to `[in, out]` and marks the result
// `PreTransposedWeight` so `linearProject` skips its per-forward
// `ops.transpose` entirely (previously a lazy shape-swap via
// `DefaultCpuOpsJvm`'s Q4_K transpose; now zero wrapper
// allocation). A Q4_K_M Gemma 4 E2B checkpoint (3.2 GB on disk)
// stays near that footprint in RAM instead of inflating to
// ~18 GB FP32.
val data = BlockQuantPacking.packPreTransposed<FP32>(bytes, TensorEncoding.Q4_K, shape)
?: error("packPreTransposed returned null for Q4_K — expected an unconditional kernel")
ctx.fromData(data as TensorData<FP32, Float>, advertisedDtype) as Tensor<T, V>
}
GGMLQuantizationType.Q6_K -> {
// Same packed-path treatment as Q4_K, enabled by the
// `matmulQ6_KVec` kernel + lazy transpose in `DefaultCpuOpsJvm`.
// Gemma 4 E2B Q4_K_M uses Q6_K for ffn_gate/up/down, attn_v,
// token_embd, and the tied lm_head — keeping these packed saves
// ~12 GB of FP32 bloat (and the corresponding 7.5 GB per-forward
// transpose transient). Sanity-checked against FP32 dequant and
// Q6_K packed produces identical tokens — kernel math is right.
val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 210, 256)
val data = Q6_KBlockTensorData.fromRawBytes(shape, relaid)
// Same packed-and-pre-transposed treatment as Q4_K, enabled by
// the `matmulQ6_KVec` kernel. Gemma 4 E2B Q4_K_M uses Q6_K for
// ffn_gate/up/down, attn_v, token_embd, and the tied lm_head —
// keeping these packed saves ~12 GB of FP32 bloat (and the
// corresponding 7.5 GB per-forward transpose transient).
// Sanity-checked against FP32 dequant and Q6_K packed produces
// identical tokens — kernel math is right.
val data = BlockQuantPacking.packPreTransposed<FP32>(bytes, TensorEncoding.Q6_K, shape)
?: error("packPreTransposed returned null for Q6_K — expected an unconditional kernel")
ctx.fromData(data as TensorData<FP32, Float>, advertisedDtype) as Tensor<T, V>
}
GGMLQuantizationType.Q5_K -> {
// Same packed-path treatment as Q4_K/Q6_K, enabled by the Q5_K
// matmul kernel (scalar/Panama/native) + the lazy Q5_K transpose
// in DefaultCpuOps. FunctionGemma-270M Q5_K_M ships most attn/FFN
// weights as Q5_K, so keeping them packed (176 B/block) avoids the
// FP32 inflation and runs the in-kernel dequant matmul.
val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 176, 256)
val data = Q5_KBlockTensorData.fromRawBytes(shape, relaid)
// Same packed-and-pre-transposed treatment as Q4_K/Q6_K, enabled
// by the Q5_K matmul kernel (scalar/Panama/native).
// FunctionGemma-270M Q5_K_M ships most attn/FFN weights as Q5_K,
// so keeping them packed (176 B/block) avoids the FP32 inflation
// and runs the in-kernel dequant matmul.
val data = BlockQuantPacking.packPreTransposed<FP32>(bytes, TensorEncoding.Q5_K, shape)
?: error("packPreTransposed returned null for Q5_K — expected an unconditional kernel")
ctx.fromData(data as TensorData<FP32, Float>, advertisedDtype) as Tensor<T, V>
}
GGMLQuantizationType.Q5_1 -> {
// Packed-path treatment for the 32-elem/24-byte legacy blocks
// (#170): FunctionGemma-270M "Q5_K_M" ships attn_q/attn_k and
// Packed-and-pre-transposed treatment for the 32-elem/24-byte
// legacy blocks (#170, engine 0.40.0 native-kernel closure
// train): FunctionGemma-270M "Q5_K_M" ships attn_q/attn_k and
// ffn_gate/ffn_up as Q5_1 (81 of 236 tensors), which until #170
// took the FP32 dequant fallback below. The engine has Q5_1
// kernels (scalar + Panama since 0.39.0; native FFM/K-N/JNI from
// SKaiNET#951 / 0.40.0) + the lazy Q5_1 transpose, so the weights
// stay packed and run the in-kernel dequant matmul.
// SKaiNET#951 / 0.40.0), so the weights stay packed, marked
// `PreTransposedWeight`, and run the in-kernel dequant matmul
// with no per-forward transpose.
//
// Gated on kernel AVAILABILITY (not engine version): if no
// registered provider carries a Q5_1 kernel, packing would send
// the weight down the generic elementwise matmul, which misreads
// the block-major bytes after the lazy transpose — so fall back
// to the always-correct #169 FP32 dequant instead.
if (qt.hasPackedMatmulKernel()) {
val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 24, 32)
val data = Q5_1BlockTensorData.fromRawBytes(shape, relaid)
val data = BlockQuantPacking.packPreTransposed<FP32>(bytes, TensorEncoding.Q5_1, shape)
?: error("packPreTransposed returned null for Q5_1 despite hasPackedMatmulKernel()==true")
ctx.fromData(data as TensorData<FP32, Float>, advertisedDtype) as Tensor<T, V>
} else {
dequantPackedToFp32<T, V>(bytes, qt, shape, ctx)
Expand All @@ -214,8 +217,8 @@ private fun <T : DType, V> convertOne(
GGMLQuantizationType.Q5_0 -> {
// Same as Q5_1, 22-byte blocks (f16 d + qh + qs, symmetric).
if (qt.hasPackedMatmulKernel()) {
val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, 22, 32)
val data = Q5_0BlockTensorData.fromRawBytes(shape, relaid)
val data = BlockQuantPacking.packPreTransposed<FP32>(bytes, TensorEncoding.Q5_0, shape)
?: error("packPreTransposed returned null for Q5_0 despite hasPackedMatmulKernel()==true")
ctx.fromData(data as TensorData<FP32, Float>, advertisedDtype) as Tensor<T, V>
} else {
dequantPackedToFp32<T, V>(bytes, qt, shape, ctx)
Expand Down
Loading