Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions gradle/libs.versions.toml
Original file line number Diff line number Diff line change
Expand Up @@ -93,6 +93,7 @@ skainet-compile-core = { module = "sk.ainet.core:skainet-compile-core" }
skainet-compile-dag = { module = "sk.ainet.core:skainet-compile-dag" }
skainet-compile-hlo = { module = "sk.ainet.core:skainet-compile-hlo" }
skainet-compile-opt = { module = "sk.ainet.core:skainet-compile-opt" }
skainet-backend-api = { module = "sk.ainet.core:skainet-backend-api" }
skainet-backend-cpu = { module = "sk.ainet.core:skainet-backend-cpu" }
skainet-backend-nativeCpu = { module = "sk.ainet.core:skainet-backend-native-cpu" }
skainet-backend-jniCpu = { module = "sk.ainet.core:skainet-backend-jni-cpu" }
Expand Down
5 changes: 5 additions & 0 deletions llm-core/api/jvm/llm-core.api
Original file line number Diff line number Diff line change
Expand Up @@ -524,6 +524,11 @@ public final class sk/ainet/apps/llm/weights/BertSafeTensorsNameResolver : sk/ai
public fun resolve (Ljava/lang/String;Ljava/lang/String;)Ljava/lang/String;
}

public final class sk/ainet/apps/llm/weights/GgmlQuantEncodingsKt {
public static final fun hasPackedMatmulKernel (Lsk/ainet/io/gguf/GGMLQuantizationType;)Z
public static final fun toBlockEncoding (Lsk/ainet/io/gguf/GGMLQuantizationType;)Lsk/ainet/lang/tensor/storage/TensorEncoding;
}

public final class sk/ainet/apps/llm/weights/LlamaGGUFNameResolver : sk/ainet/io/weights/WeightNameResolver {
public fun <init> ()V
public fun resolve (Ljava/lang/String;Ljava/lang/String;)Ljava/lang/String;
Expand Down
4 changes: 4 additions & 0 deletions llm-core/build.gradle.kts
Original file line number Diff line number Diff line change
Expand Up @@ -54,6 +54,10 @@ kotlin {
implementation(libs.skainet.compile.opt)
implementation(libs.skainet.io.core)
implementation(libs.skainet.io.gguf)
// KernelRegistry/KernelProvider capability queries for the
// packed-quant kernel gate (`hasPackedMatmulKernel`, #170).
// implementation-scoped: no backend types leak into the API.
implementation(libs.skainet.backend.api)
implementation(libs.kotlinx.io.core)
implementation(libs.kotlinx.serialization.json)
}
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,64 @@
package sk.ainet.apps.llm.weights

import sk.ainet.backend.api.kernel.KernelRegistry
import sk.ainet.io.gguf.GGMLQuantizationType
import sk.ainet.lang.tensor.storage.TensorEncoding

/**
* Map a GGUF [GGMLQuantizationType] to the engine [TensorEncoding] of the
* formats with a packed CPU matmul kernel (+ lazy `ops.transpose` support),
* or `null` for types that must be dequantized.
*
* This is the ggml-keyed front door to the shared
* [sk.ainet.lang.nn.quant.BlockQuantPacking] packer (#184 hoist 2): model
* modules call `qt.toBlockEncoding()?.let { BlockQuantPacking.pack(bytes, it, shape) }`
* and keep only weight *selection* and naming for themselves.
*/
public fun GGMLQuantizationType.toBlockEncoding(): TensorEncoding? = when (this) {
GGMLQuantizationType.Q4_K -> TensorEncoding.Q4_K
GGMLQuantizationType.Q5_K -> TensorEncoding.Q5_K
GGMLQuantizationType.Q6_K -> TensorEncoding.Q6_K
GGMLQuantizationType.Q8_0 -> TensorEncoding.Q8_0
GGMLQuantizationType.Q4_0 -> TensorEncoding.Q4_0
GGMLQuantizationType.Q5_0 -> TensorEncoding.Q5_0
GGMLQuantizationType.Q5_1 -> TensorEncoding.Q5_1
else -> null
}

/**
* Runtime kernel gate for the packed-quant conversion path (#170): `true` iff
* some registered, available [sk.ainet.backend.api.kernel.KernelProvider]
* carries an `FP32 activations x packed-<this>` matmul kernel, so a weight
* packed by [sk.ainet.lang.nn.quant.BlockQuantPacking] will actually dispatch
* to a packed kernel instead of the generic elementwise matmul (which reads
* block-major bytes with row-major strides and would be numerically wrong
* after the lazy packed transpose).
*
* Converters gate NEW packed formats on this check and keep the FP32 dequant
* fallback for `false` — availability-based, not engine-version-based: under
* engine 0.39.0 the scalar/Panama Q5_0/Q5_1 kernels already report here, and
* the native FFM/Kotlin-Native/JNI tiers added by SKaiNET#951 (0.40.0) light
* up through the same query without any transformers-side change.
*
* On the JVM this installs `ServiceLoader`-discovered providers first
* (idempotent), mirroring the engine's own lazy `ensureKernelProviders`, so
* the answer is correct even before the first matmul runs. On registry-based
* targets (Kotlin/Native, Android, JS/WASM) providers are registered by the
* platform backend factories; converters run with a live [sk.ainet.context.ExecutionContext],
* which implies that registration has happened.
*/
public fun GGMLQuantizationType.hasPackedMatmulKernel(): Boolean {
val encoding = toBlockEncoding() ?: return false
ensurePlatformKernelProviders()
return KernelRegistry.providers().any {
it.isAvailable() && it.supports("matmul", listOf("Float32", encoding.name))
}
}

/**
* Populate [KernelRegistry] with the platform's discoverable providers before
* a capability query. JVM: `KernelServiceLoader.installAll()` (idempotent).
* Registry-based targets: no-op — providers are registered explicitly by the
* backend factories (see `BackendRegistry.registryBased.kt`).
*/
internal expect fun ensurePlatformKernelProviders()
Original file line number Diff line number Diff line change
@@ -0,0 +1,16 @@
package sk.ainet.apps.llm.weights

import sk.ainet.backend.api.kernel.KernelServiceLoader

/**
* JVM: install every `META-INF/services`-declared
* [sk.ainet.backend.api.kernel.KernelProvider] into the process registry.
* Idempotent ([sk.ainet.backend.api.kernel.KernelRegistry.register] no-ops on
* re-registration), and the same wiring the engine's `DefaultCpuOpsJvm`
* performs lazily on first matmul — done eagerly here so the
* [hasPackedMatmulKernel] gate answers correctly at weight-conversion time,
* which runs before any matmul.
*/
internal actual fun ensurePlatformKernelProviders() {
KernelServiceLoader.installAll()
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,12 @@
package sk.ainet.apps.llm.weights

/**
* Registry-based targets (Kotlin/Native, Android, JS/WASM): no ServiceLoader.
* Kernel providers are registered explicitly by the platform backend factories
* (e.g. the K/N CPU ops factory pins `ScalarKernelProvider`; Android pins the
* JNI provider) before any converter runs — [hasPackedMatmulKernel] just reads
* the registry as-is.
*/
internal actual fun ensurePlatformKernelProviders() {
// No-op: KernelRegistry is populated by the backend factory on these targets.
}
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@ package sk.ainet.models.apertus
import sk.ainet.context.ExecutionContext
import sk.ainet.io.gguf.GGMLQuantizationType
import sk.ainet.io.gguf.dequant.DequantOps
import sk.ainet.lang.nn.quant.BlockQuantPacking
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.Tensor
import sk.ainet.lang.tensor.data.Q4_KBlockTensorData
Expand Down Expand Up @@ -110,13 +111,13 @@ private fun <T : DType, V> convertOne(
}

GGMLQuantizationType.Q4_K -> {
val relaid = relayoutKSeriesRowMajorToBlockMajor(bytes, logicalShape, BYTES_PER_Q4_K_BLOCK)
val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, logicalShape, BYTES_PER_Q4_K_BLOCK, K_SERIES_BLOCK_SIZE)
val data = Q4_KBlockTensorData.fromRawBytes(logicalShape, relaid)
ctx.fromData(data as TensorData<FP32, Float>, advertisedDtype) as Tensor<T, V>
}

GGMLQuantizationType.Q6_K -> {
val relaid = relayoutKSeriesRowMajorToBlockMajor(bytes, logicalShape, BYTES_PER_Q6_K_BLOCK)
val relaid = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, logicalShape, BYTES_PER_Q6_K_BLOCK, K_SERIES_BLOCK_SIZE)
val data = Q6_KBlockTensorData.fromRawBytes(logicalShape, relaid)
ctx.fromData(data as TensorData<FP32, Float>, advertisedDtype) as Tensor<T, V>
}
Expand All @@ -139,38 +140,22 @@ private const val BYTES_PER_Q6_K_BLOCK = 210
private const val K_SERIES_BLOCK_SIZE = 256

/**
* Re-layout GGUF K-series bytes from row-major block order
* (block at row `r`, block index `b` within the row → byte offset
* `(r * blocksPerRow + b) * bytesPerBlock`) to the input-block-major layout
* the `matmulQ{K}_Vec` kernels index via `(blockIdx * outDim + r) * bytesPerBlock`.
*
* For a weight of shape `[outDim, inDim]` with `inDim % 256 == 0`, this is
* a 2D block-level transpose of the `[outDim, inDim/256]` block grid. Bytes
* inside a block are untouched.
* Re-layout GGUF K-series bytes from row-major block order to the
* input-block-major layout the `matmulQ{K}_Vec` kernels index. Delegates to
* the shared [sk.ainet.lang.nn.quant.BlockQuantPacking] packer (#184 hoist 2);
* kept as an internal shim for existing call sites and tests.
*/
@Deprecated(
"Hoisted to the shared packer (#184): use BlockQuantPacking.relayoutRowMajorToBlockMajor",
ReplaceWith(
"BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, bytesPerBlock, 256)",
"sk.ainet.lang.nn.quant.BlockQuantPacking",
),
)
internal fun relayoutKSeriesRowMajorToBlockMajor(
bytes: ByteArray,
shape: Shape,
bytesPerBlock: Int
): ByteArray {
require(shape.rank == 2) { "K-series weight must be 2D, got rank ${shape.rank}" }
val outDim = shape[0]
val inDim = shape[1]
require(inDim % K_SERIES_BLOCK_SIZE == 0) {
"K-series weight inDim ($inDim) must be a multiple of $K_SERIES_BLOCK_SIZE"
}
val blocksPerRow = inDim / K_SERIES_BLOCK_SIZE
val expected = outDim.toLong() * blocksPerRow.toLong() * bytesPerBlock.toLong()
require(bytes.size.toLong() >= expected) {
"K-series byte buffer size ${bytes.size} < expected $expected for shape [$outDim, $inDim] @ ${bytesPerBlock}B/block"
}
val out = ByteArray(bytes.size)
for (r in 0 until outDim) {
for (b in 0 until blocksPerRow) {
val srcOff = (r * blocksPerRow + b) * bytesPerBlock
val dstOff = (b * outDim + r) * bytesPerBlock
System.arraycopy(bytes, srcOff, out, dstOff, bytesPerBlock)
}
}
return out
}
): ByteArray = BlockQuantPacking.relayoutRowMajorToBlockMajor(
bytes, shape, bytesPerBlock, K_SERIES_BLOCK_SIZE,
)
Original file line number Diff line number Diff line change
@@ -1,11 +1,10 @@
package sk.ainet.models.gemma

import sk.ainet.apps.llm.weights.hasPackedMatmulKernel
import sk.ainet.apps.llm.weights.toBlockEncoding
import sk.ainet.io.gguf.GGMLQuantizationType
import sk.ainet.lang.nn.quant.BlockQuantPacking
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.data.Q4_KBlockTensorData
import sk.ainet.lang.tensor.data.Q5_KBlockTensorData
import sk.ainet.lang.tensor.data.Q6_KBlockTensorData
import sk.ainet.lang.tensor.data.Q8_0BlockTensorData
import sk.ainet.lang.tensor.data.TensorData
import sk.ainet.lang.types.DType

Expand Down Expand Up @@ -55,65 +54,43 @@ internal fun logicalShapeFor(name: String, metadata: Gemma4ModelMetadata): Shape
}

/**
* Re-layout GGUF K-series bytes from row-major block order
* (`(r * blocksPerRow + b) * bytesPerBlock`) to the input-block-major order the
* `matmulQ{K}` kernels expect (`(b * outDim + r) * bytesPerBlock`). For a
* `[outDim, inDim]` weight with `inDim % 256 == 0`, this is a block-level 2-D
* transpose; bytes inside a block are untouched.
* Re-layout GGUF K-series bytes from row-major block order to the
* input-block-major order the `matmulQ{K}` kernels expect.
*
* Delegates to the shared [BlockQuantPacking.relayoutRowMajorToBlockMajor]
* (#184 hoist 2); kept as an internal shim because the JVM MemSeg converter
* and the gemma tests call it by this name.
*
* @param bytesPerBlock 144 (Q4_K), 176 (Q5_K), 210 (Q6_K).
*/
@Deprecated(
"Hoisted to the shared packer (#184): use BlockQuantPacking.relayoutRowMajorToBlockMajor",
ReplaceWith(
"BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, bytesPerBlock, blockSize)",
"sk.ainet.lang.nn.quant.BlockQuantPacking",
),
)
internal fun relayoutKSeriesRowMajorToBlockMajor(
bytes: ByteArray,
shape: Shape,
bytesPerBlock: Int,
blockSize: Int = 256,
): ByteArray {
require(shape.rank == 2) { "K-series weight must be 2D, got rank ${shape.rank}" }
val outDim = shape[0]
val inDim = shape[1]
require(inDim % blockSize == 0) { "K-series weight inDim ($inDim) must be a multiple of $blockSize" }
val blocksPerRow = inDim / blockSize
val expected = outDim.toLong() * blocksPerRow.toLong() * bytesPerBlock.toLong()
require(bytes.size.toLong() >= expected) {
"K-series byte buffer ${bytes.size} < expected $expected for [$outDim, $inDim] @ ${bytesPerBlock}B/block"
}
val out = ByteArray(bytes.size)
for (r in 0 until outDim) {
for (b in 0 until blocksPerRow) {
val srcOff = (r * blocksPerRow + b) * bytesPerBlock
val dstOff = (b * outDim + r) * bytesPerBlock
bytes.copyInto(out, dstOff, srcOff, srcOff + bytesPerBlock)
}
}
return out
}

/**
* Block geometry `(blockElems, bytesPerBlock)` for the quant types this packer
* handles. The K-series are 256-element super-blocks; Q8_0 is a 32-element block
* (f16 scale + 32 int8). All four have a first-class CPU matmul kernel + a lazy
* transpose in `ops.transpose`, so all four can stay packed instead of FP32.
*/
private fun quantBlockLayout(qt: GGMLQuantizationType): Pair<Int, Int>? = when (qt) {
GGMLQuantizationType.Q4_K -> 256 to 144
GGMLQuantizationType.Q5_K -> 256 to 176
GGMLQuantizationType.Q6_K -> 256 to 210
GGMLQuantizationType.Q8_0 -> 32 to 34
else -> null
}
): ByteArray = BlockQuantPacking.relayoutRowMajorToBlockMajor(bytes, shape, bytesPerBlock, blockSize)

/**
* Pack raw GGUF `bytes` of logical `[out, in]` shape into the heap-packed block
* tensor data the matmul kernels read directly (Q4_K / Q5_K / Q6_K / Q8_0).
* Performs the row-major → block-major relayout. Returns `null` for types
* without a packed kernel (caller dequantizes those to FP32).
* tensor data the matmul kernels read directly. Performs the row-major →
* block-major relayout. Returns `null` for types without a packed kernel
* (caller dequantizes those to FP32).
*
* Delegates to the shared [BlockQuantPacking] packer (#184 hoist 2), which
* covers all seven packed-kernel formats — Q4_K / Q5_K / Q6_K / Q8_0 plus
* Q4_0 / Q5_0 / Q5_1 (#170: FunctionGemma "Q5_K_M" checkpoints carry their
* attention q/k and ffn_gate weights as Q5_1, previously dequantized here).
*
* Q8_0 matters for gemma's tied `output`/lm_head: FunctionGemma's token_embd is
* Q8_0, so keeping the lm_head packed (vs ~0.67 GB FP32) is what lets the eager
* decode fit the 1.9 GB board, and it runs on the NEON Q8_0 kernel. (Requires
* the Q8_0 case in `ops.transpose` — engine — so `linearProject` can transpose
* the packed weight; see transformers #178.)
* decode fit the 1.9 GB board, and it runs on the NEON Q8_0 kernel.
*
* commonMain → works on JVM and Kotlin/Native alike (no MemSeg / Arena).
*/
Expand All @@ -122,14 +99,29 @@ internal fun <T : DType> packGemmaKQuant(
qt: GGMLQuantizationType,
shape: Shape,
): TensorData<T, *>? {
val (blockElems, bpb) = quantBlockLayout(qt) ?: return null
val relaid = relayoutKSeriesRowMajorToBlockMajor(bytes, shape, bpb, blockElems)
@Suppress("UNCHECKED_CAST")
return when (qt) {
GGMLQuantizationType.Q4_K -> Q4_KBlockTensorData(shape, relaid) as TensorData<T, *>
GGMLQuantizationType.Q5_K -> Q5_KBlockTensorData(shape, relaid) as TensorData<T, *>
GGMLQuantizationType.Q6_K -> Q6_KBlockTensorData(shape, relaid) as TensorData<T, *>
GGMLQuantizationType.Q8_0 -> Q8_0BlockTensorData(shape, relaid) as TensorData<T, *>
else -> null
}
val encoding = qt.toBlockEncoding() ?: return null
// The legacy 32-elem formats (Q4_0/Q5_0/Q5_1) are NEW to this packed path
// (#170) — gate them on an actually-registered matmul kernel rather than
// a hard engine version: without a kernel, a packed weight would fall
// through to the generic elementwise matmul, which reads the block-major
// bytes with row-major strides after the lazy transpose (garbage), so the
// correct degradation is `null` → the caller's FP32 dequant fallback
// (#169 behavior). Q4_K/Q5_K/Q6_K/Q8_0 keep their long-standing
// unconditional packing. Under engine 0.39.0 the scalar/Panama Q5_x
// kernels already satisfy the gate; SKaiNET#951 (0.40.0) adds the native
// FFM/K-N/JNI tiers behind the same check.
if (qt in legacyPackedQuantTypes && !qt.hasPackedMatmulKernel()) return null
return BlockQuantPacking.pack(bytes, encoding, shape)
}

/**
* GGUF legacy block formats whose packed-matmul support arrived with the
* 0.39.0 loading + scalar/Panama kernels and completes with the native-tier
* kernels of SKaiNET#951 (0.40.0). Packing these is gated on
* [hasPackedMatmulKernel]; everything else packs as before.
*/
private val legacyPackedQuantTypes: Set<GGMLQuantizationType> = setOf(
GGMLQuantizationType.Q4_0,
GGMLQuantizationType.Q5_0,
GGMLQuantizationType.Q5_1,
)
Loading