Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion gradle/libs.versions.toml
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
[versions]
skainet = "0.40.0"
skainet = "0.40.1"
agp = "9.3.1"
jacksonDatabind = "2.22.1"
jsonSchemaValidator = "3.0.6"
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -6,11 +6,10 @@ import sk.ainet.io.gguf.dequant.DequantOps
import sk.ainet.lang.nn.quant.BlockQuantPacking
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.Tensor
import sk.ainet.lang.tensor.data.Q4_KBlockTensorData
import sk.ainet.lang.tensor.data.Q4MemorySegmentTensorData
import sk.ainet.lang.tensor.data.Q6_KBlockTensorData
import sk.ainet.lang.tensor.data.Q8MemorySegmentTensorData
import sk.ainet.lang.tensor.data.TensorData
import sk.ainet.lang.tensor.storage.TensorEncoding
import sk.ainet.lang.types.DType
import sk.ainet.lang.types.FP32
import java.lang.foreign.Arena
Expand All @@ -27,10 +26,14 @@ import java.lang.foreign.Arena
* rank-1 byte tensors the loader produces under `NATIVE_OPTIMIZED`. With
* this step, each quantized weight ends up as the right wrapper:
*
* - `Q4_K` → [Q4_KBlockTensorData] (relayout from GGUF row-major
* `[row, block]` order to the input-block-major `[block, row]` order
* `JvmQuantizedVectorKernels.matmulQ4_KVec` expects). 144-byte blocks.
* - `Q6_K` → [Q6_KBlockTensorData]. 210-byte blocks.
* - `Q4_K` → [BlockQuantPacking.packPreTransposed]: input-block-major relayout
* + `[in, out]` shape + `PreTransposedWeight` marker, so `linearProject`
* skips `ops.transpose` entirely (aligning apertus with the gemma/llama
* converters — the previous inlined relayout under an unmarked `[out, in]`
* shape relied on the pre-0.40.1 shape-swap-only packed transpose and is
* silently corrupted by the physical block-grid permutation the engine
* performs since 0.40.1). 144-byte blocks.
* - `Q6_K` → same pre-transposed packing. 210-byte blocks.
* - `Q4_0` → [Q4MemorySegmentTensorData] (arena-allocated, 64-byte aligned).
* - `Q8_0` → [Q8MemorySegmentTensorData].
* - `Q5_K` → fallback: dequant to FP32. Apertus-8B-Instruct-2509 Q4_K_S has
Expand Down Expand Up @@ -111,14 +114,14 @@ private fun <T : DType, V> convertOne(
}

GGMLQuantizationType.Q4_K -> {
val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, logicalShape, BYTES_PER_Q4_K_BLOCK, K_SERIES_BLOCK_SIZE)
val data = Q4_KBlockTensorData.fromRawBytes(logicalShape, relaid)
val data = BlockQuantPacking.packPreTransposed<FP32>(bytes, TensorEncoding.Q4_K, logicalShape)
?: error("ApertusMemSegConverter: packPreTransposed returned null for Q4_K ('$name')")
ctx.fromData(data as TensorData<FP32, Float>, advertisedDtype) as Tensor<T, V>
}

GGMLQuantizationType.Q6_K -> {
val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, logicalShape, BYTES_PER_Q6_K_BLOCK, K_SERIES_BLOCK_SIZE)
val data = Q6_KBlockTensorData.fromRawBytes(logicalShape, relaid)
val data = BlockQuantPacking.packPreTransposed<FP32>(bytes, TensorEncoding.Q6_K, logicalShape)
?: error("ApertusMemSegConverter: packPreTransposed returned null for Q6_K ('$name')")
ctx.fromData(data as TensorData<FP32, Float>, advertisedDtype) as Tensor<T, V>
}

Expand All @@ -135,8 +138,6 @@ private fun <T : DType, V> convertOne(
}
}

private const val BYTES_PER_Q4_K_BLOCK = 144
private const val BYTES_PER_Q6_K_BLOCK = 210
private const val K_SERIES_BLOCK_SIZE = 256

/**
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,123 @@
package sk.ainet.models.apertus

import java.lang.foreign.Arena
import kotlin.math.abs
import kotlin.test.Test
import kotlin.test.assertEquals
import kotlin.test.assertTrue
import sk.ainet.context.DirectCpuExecutionContext
import sk.ainet.io.gguf.GGMLQuantizationType
import sk.ainet.lang.nn.quant.PreTransposedWeight
import sk.ainet.lang.nn.transformer.linearProject
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.data.Q4_KBlockTensorData
import sk.ainet.lang.tensor.data.Q6_KBlockTensorData
import sk.ainet.lang.tensor.data.TensorData
import sk.ainet.lang.tensor.storage.PackedBlockStorage
import sk.ainet.lang.types.FP32

/**
* Synthetic (no real checkpoint) coverage for the Apertus K-quant converter
* path — the gap that let the SKaiNET 0.40.1 layout-contract regression go
* unobserved here: [ApertusRealGgufLoadingTest] needs a real GGUF and skips
* by default, and the previous converter emitted relaid bytes under an
* unmarked `[out, in]` shape, which `linearProject`'s per-forward
* `ops.transpose` silently corrupts on engines >= 0.40.1.
*
* Asserts the converter now emits [PreTransposedWeight]-marked `[in, out]`
* tensors (mirroring gemma/llama) whose `linearProject` output matches an
* FP32 reference built from the canonical dequant of the same bytes.
*/
class ApertusMemSegConverterSyntheticTest {

/** Deterministic block bytes with FP16 scale fields pinned to 0.25/0.125. */
private fun buildBlocks(blockCount: Int, bytesPerBlock: Int, f16Offsets: List<Int>): ByteArray {
val out = ByteArray(blockCount * bytesPerBlock)
for (b in 0 until blockCount) {
val base = b * bytesPerBlock
for (j in 0 until bytesPerBlock) out[base + j] = ((b * 37 + j * 11 + 5) % 251).toByte()
f16Offsets.forEachIndexed { i, off ->
out[base + off] = 0x00
out[base + off + 1] = if (i == 0) 0x34 else 0x30 // 0.25f / 0.125f
}
}
return out
}

private fun metadata() = ApertusModelMetadata(
architecture = "apertus",
embeddingLength = 512,
contextLength = 128,
blockCount = 1,
headCount = 2,
kvHeadCount = 1,
feedForwardLength = 512,
ropeDimensionCount = null,
vocabSize = 100,
)

private fun assertConvertedParity(qt: GGMLQuantizationType, bytesPerBlock: Int, f16Offsets: List<Int>) {
val outDim = 4
val inDim = 512 // blocksPerRow = 2: multi-block both grid dimensions
val shape = Shape(outDim, inDim)
val name = "blk.0.attn_q.weight"
val bytes = buildBlocks(outDim * (inDim / 256), bytesPerBlock, f16Offsets)

val ctx = DirectCpuExecutionContext.create()
// Placeholder for the loader's rank-1 byte tensor; the converter
// replaces it by key from the quantBytes sidecar.
val placeholder = ctx.fromFloatArray<FP32, Float>(Shape(1), FP32::class, floatArrayOf(0f))
val weights = ApertusWeights<FP32, Float>(
metadata = metadata(),
tensors = mapOf(name to placeholder),
quantTypes = mapOf(name to qt),
logicalShapes = mapOf(name to shape),
quantBytes = mapOf(name to bytes),
)

Arena.ofConfined().use { arena ->
val converted = convertApertusWeightsToMemSeg(weights, ctx, arena)
val w = converted.tensors[name] ?: error("converted weight missing")

assertTrue(
w.data is PreTransposedWeight,
"$qt: converter must emit a PreTransposedWeight-marked tensor (unmarked [out,in] + " +
"relaid bytes is corrupted by the >= 0.40.1 physical packed transpose)",
)
assertEquals(Shape(inDim, outDim), w.shape, "$qt: pre-transposed [in, out] shape")

// FP32 reference: canonical dequant of the same bytes.
@Suppress("UNCHECKED_CAST")
val canonical = when (qt) {
GGMLQuantizationType.Q4_K -> Q4_KBlockTensorData(shape, bytes)
else -> Q6_KBlockTensorData(shape, bytes)
} as TensorData<FP32, Float>
val wFlat = (canonical as PackedBlockStorage).toFloatArray()
for (v in wFlat) assertTrue(v.isFinite(), "$qt: non-finite dequant value $v")
val wRef = ctx.fromFloatArray<FP32, Float>(shape, FP32::class, wFlat)

val x = ctx.fromFloatArray<FP32, Float>(
Shape(2, inDim), FP32::class,
FloatArray(2 * inDim) { i -> ((i * 29 + 3) % 23 - 11) / 11.0f },
)
val ref = linearProject(ctx.ops, x, wRef).data.copyToFloatArray()
val y = linearProject(ctx.ops, x, w).data.copyToFloatArray()

assertEquals(ref.size, y.size)
for (i in ref.indices) {
assertTrue(
abs(ref[i] - y[i]) <= 1e-3f * maxOf(1.0f, abs(ref[i])),
"$qt converted[$i]=${y[i]} vs FP32-dequant ref ${ref[i]}",
)
}
}
}

@Test
fun q4k_converted_weight_is_marked_and_matches_fp32_reference() =
assertConvertedParity(GGMLQuantizationType.Q4_K, bytesPerBlock = 144, f16Offsets = listOf(0, 2))

@Test
fun q6k_converted_weight_is_marked_and_matches_fp32_reference() =
assertConvertedParity(GGMLQuantizationType.Q6_K, bytesPerBlock = 210, f16Offsets = listOf(208))
}
Original file line number Diff line number Diff line change
Expand Up @@ -76,8 +76,10 @@ class GemmaQuantLayoutTest {
val td = packGemmaKQuant<FP32>(bytes, GGMLQuantizationType.Q5_K, shape, preTransposed = false)
assertTrue(td is Q5_KBlockTensorData, "Q5_K should pack to the classic Q5_KBlockTensorData when opted out")
assertEquals(shape, td.shape, "classic path keeps the checkpoint's [out, in] shape")
val expected = relayoutKSeriesRowMajorToBlockMajor(bytes, shape, 176)
assertTrue(expected.contentEquals(td.packedData))
// Canonical checkpoint bytes verbatim: the classic path defers the
// block-grid permutation to the engine's physical packed ops.transpose
// (>= 0.40.1) inside linearProject.
assertTrue(bytes.contentEquals(td.packedData))
}

@Test
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -20,11 +20,11 @@ import sk.ainet.lang.types.FP32
* running on the corresponding dequantised FP32 weights.
*
* Validates:
* 1. `relayoutQ4_KRowMajorToBlockMajor` correctly reshuffles GGUF
* row-major Q4_K bytes into the input-block-major layout expected by
* `JvmQuantizedVectorKernels.matmulQ4_KVec`.
* 2. The lazy `ops.transpose(Q4_KTensorData)` in `DefaultCpuOpsJvm`
* composes correctly with the Q4_K matmul kernel.
* 1. Canonical (GGUF row-major) Q4_K bytes under an `[out, in]` shape flow
* through the engine's physical packed `ops.transpose` (>= 0.40.1,
* SKaiNET#968) into the input-block-major layout the Q4_K matmul kernel
* expects.
* 2. That transpose composes correctly with the Q4_K matmul kernel dispatch.
* 3. Running a tiny Gemma DSL model with Q4_K-backed weights produces
* logits close to the FP32 baseline (i.e., no wrong-math bug from
* layout or dispatch).
Expand All @@ -38,9 +38,9 @@ class GemmaDslQ4KTest {

// dim and ffnDim chosen so each Q4_K weight has multiple input blocks per
// row — `inDim = 512 → blocksPerRow = 2`. With one block per row the
// `relayoutKSeriesRowMajorToBlockMajor` is the identity transform and the
// test silently misses any relayout bug. Two blocks per row exercises the
// real ggml strided codes layout end-to-end.
// canonical and input-block-major orders coincide and the test silently
// misses any block-order bug. Two blocks per row exercises the real ggml
// strided codes layout end-to-end.
private val dim = 512
private val nHeads = 2
private val nKvHeads = 1
Expand Down Expand Up @@ -176,8 +176,12 @@ class GemmaDslQ4KTest {

@Suppress("UNCHECKED_CAST")
private fun q4kTensor(rows: Int, cols: Int, bytes: ByteArray): Tensor<FP32, Float> {
val relaid = relayoutQ4_KRowMajorToBlockMajor(bytes, Shape(rows, cols))
val data = Q4_KBlockTensorData.fromRawBytes(Shape(rows, cols), relaid)
// Canonical row-major bytes verbatim: the classic [out, in] path defers
// the block-grid permutation to the engine's physical packed
// ops.transpose (>= 0.40.1) inside linearProject. Relayouting here
// (as before the SKaiNET#968 layout-contract fix landed in 0.40.1)
// would get the blocks permuted a second time.
val data = Q4_KBlockTensorData.fromRawBytes(Shape(rows, cols), bytes)
return ctx.fromData(data as TensorData<FP32, Float>, FP32::class)
}

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -14,13 +14,13 @@ import sk.ainet.io.JvmRandomAccessSource
import sk.ainet.io.gguf.GGMLQuantizationType
import sk.ainet.io.gguf.dequant.DequantOps
import sk.ainet.io.model.QuantPolicy
import sk.ainet.lang.nn.quant.BlockQuantPacking
import sk.ainet.lang.nn.transformer.linearProject
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.data.Q5_0BlockTensorData
import sk.ainet.lang.tensor.data.Q5_1BlockTensorData
import sk.ainet.lang.tensor.data.Q5_0TensorData
import sk.ainet.lang.tensor.data.Q5_1TensorData
import sk.ainet.lang.tensor.data.TensorData
import sk.ainet.lang.tensor.storage.TensorEncoding
import sk.ainet.lang.types.FP32

/**
Expand All @@ -30,8 +30,8 @@ import sk.ainet.lang.types.FP32
*
* Two layers of evidence:
* - [synthetic tests] byte-level parity per format: the converter's exact
* packed pipeline (row-major → block-major relayout + `*BlockTensorData` +
* lazy-transpose matmul) vs the converter's exact FP32 fallback pipeline
* packed pipeline (canonical bytes via `BlockQuantPacking.pack` + the
* engine's physical packed transpose matmul) vs the converter's exact FP32 fallback pipeline
* (`DequantOps.dequantFromBytes` + identity col→row transpose). Q5_0 is
* covered here only — the FunctionGemma checkpoint carries no Q5_0 tensor.
* - [real checkpoint] FunctionGemma-270M "Q5_K_M" ships 81 of 236 tensors as
Expand Down Expand Up @@ -128,14 +128,12 @@ class GemmaQ5xPackedParityTest {
val wRef = ctx.fromFloatArray<FP32, Float>(shape, FP32::class, rowMajor)
val ref = linearProject(ctx.ops, x, wRef).data.copyToFloatArray()

// Packed: the converter's packed pipeline, verbatim.
@Suppress("DEPRECATION")
val relaid = relayoutKSeriesRowMajorToBlockMajor(gguf, shape, bpb, blockSize = 32)
// Packed: the converter's classic packed pipeline, verbatim — canonical
// checkpoint bytes, [out, in] shape; the engine's physical packed
// ops.transpose (>= 0.40.1) inside linearProject produces kernel order.
val encoding = if (qt == GGMLQuantizationType.Q5_1) TensorEncoding.Q5_1 else TensorEncoding.Q5_0
@Suppress("UNCHECKED_CAST")
val data = when (qt) {
GGMLQuantizationType.Q5_1 -> Q5_1BlockTensorData.fromRawBytes(shape, relaid)
else -> Q5_0BlockTensorData.fromRawBytes(shape, relaid)
} as TensorData<FP32, Float>
val data = BlockQuantPacking.pack<FP32>(gguf, encoding, shape) as TensorData<FP32, Float>
val y = linearProject(ctx.ops, x, ctx.fromData(data, FP32::class)).data.copyToFloatArray()

assertEquals(ref.size, y.size)
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,91 @@
package sk.ainet.models.llama

import kotlin.test.Test
import kotlin.test.assertEquals
import kotlin.test.assertNull
import kotlin.test.assertTrue
import sk.ainet.io.gguf.GGMLQuantizationType
import sk.ainet.lang.nn.quant.BlockQuantPacking
import sk.ainet.lang.nn.quant.PreTransposedWeight
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.data.Q4_KBlockTensorData
import sk.ainet.lang.tensor.data.Q4_KTensorData
import sk.ainet.lang.tensor.storage.PackedBlockStorage
import sk.ainet.lang.types.FP32

/**
* Unit tests for the commonMain Llama quant layout helpers — the Llama mirror
* of `GemmaQuantLayoutTest` (which existed since #184; this one was missing
* until the SKaiNET#968/0.40.1 layout-contract regression showed every
* converter needs its own packing coverage). Runs on every target.
*/
class LlamaQuantLayoutTest {

private val q4kBpb = 144
private val kBlock = 256

@Test
fun pack_q4k_defaults_to_pre_transposed_with_relaid_bytes() {
// [outDim=2, inDim=512] -> blocksPerRow=2, multi-block both ways.
val shape = Shape(2, 512)
val bytes = ByteArray(2 * 2 * q4kBpb)
for (i in 0 until 4) bytes[i * q4kBpb] = (i + 1).toByte()

val td = packLlamaKQuant<FP32>(bytes, GGMLQuantizationType.Q4_K, shape)
?: error("packLlamaKQuant returned null for Q4_K")
assertTrue(td is PreTransposedWeight, "Q4_K should default to the pre-transposed marked path")
assertTrue(td is Q4_KTensorData, "the marked wrapper still satisfies Q4_KTensorData dispatch checks")
assertEquals(Shape(512, 2), td.shape, "pre-transposed result carries the swapped [in, out] shape")
// Bytes are the input-block-major relayout — kernel feed order,
// computed once at load time.
val expected = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, q4kBpb, kBlock)
assertTrue(td is PackedBlockStorage)
assertTrue(expected.contentEquals((td as PackedBlockStorage).packedData))
}

@Test
fun pack_q4k_preTransposed_false_keeps_canonical_bytes_verbatim() {
val shape = Shape(2, 512)
val bytes = ByteArray(2 * 2 * q4kBpb)
for (i in 0 until 4) bytes[i * q4kBpb] = (i + 1).toByte()

val td = packLlamaKQuant<FP32>(bytes, GGMLQuantizationType.Q4_K, shape, preTransposed = false)
assertTrue(td is Q4_KBlockTensorData, "Q4_K should pack to the classic Q4_KBlockTensorData when opted out")
assertEquals(shape, td.shape, "classic path keeps the checkpoint's [out, in] shape")
// Canonical checkpoint bytes verbatim: the classic path defers the
// block-grid permutation to the engine's physical packed ops.transpose
// (>= 0.40.1) inside linearProject.
assertTrue(bytes.contentEquals(td.packedData))
}

@Test
fun pack_unsupported_quant_returns_null() {
assertNull(packLlamaKQuant<FP32>(ByteArray(20), GGMLQuantizationType.Q4_1, Shape(1, 32)))
}

@Test
fun logicalShapeFor_maps_the_2d_matmul_weights() {
val md = LlamaModelMetadata(
architecture = "llama",
embeddingLength = 64,
contextLength = 128,
blockCount = 2,
headCount = 4,
kvHeadCount = 2,
feedForwardLength = 256,
ropeDimensionCount = null,
vocabSize = 1000,
)
assertEquals(Shape(1000, 64), logicalShapeFor(LlamaTensorNames.TOKEN_EMBEDDINGS, md))
assertEquals(Shape(1000, 64), logicalShapeFor(LlamaTensorNames.OUTPUT_WEIGHT, md))
assertEquals(Shape(64, 64), logicalShapeFor("blk.0.attn_q.weight", md))
assertEquals(Shape(32, 64), logicalShapeFor("blk.0.attn_k.weight", md))
assertEquals(Shape(32, 64), logicalShapeFor("blk.1.attn_v.weight", md))
assertEquals(Shape(64, 64), logicalShapeFor("blk.0.attn_output.weight", md))
assertEquals(Shape(256, 64), logicalShapeFor("blk.0.ffn_gate.weight", md))
assertEquals(Shape(256, 64), logicalShapeFor("blk.1.ffn_up.weight", md))
assertEquals(Shape(64, 256), logicalShapeFor("blk.0.ffn_down.weight", md))
assertNull(logicalShapeFor("blk.0.attn_norm.weight", md))
assertNull(logicalShapeFor("output_norm.weight", md))
}
}
Loading