Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -272,6 +272,7 @@ public class sk/ainet/exec/tensor/ops/DefaultCpuOpsBase : sk/ainet/lang/tensor/o
public fun lt (Lsk/ainet/lang/tensor/Tensor;F)Lsk/ainet/lang/tensor/Tensor;
protected final fun mapIndex ([ILsk/ainet/lang/tensor/Shape;)[I
public fun matmul (Lsk/ainet/lang/tensor/Tensor;Lsk/ainet/lang/tensor/Tensor;)Lsk/ainet/lang/tensor/Tensor;
public fun matmulWeightTransposed (Lsk/ainet/lang/tensor/Tensor;Lsk/ainet/lang/tensor/Tensor;)Lsk/ainet/lang/tensor/Tensor;
public fun maxPool2d (Lsk/ainet/lang/tensor/Tensor;Lkotlin/Pair;Lkotlin/Pair;Lkotlin/Pair;)Lsk/ainet/lang/tensor/Tensor;
public fun mean (Lsk/ainet/lang/tensor/Tensor;Ljava/lang/Integer;)Lsk/ainet/lang/tensor/Tensor;
public fun mulScalar (Lsk/ainet/lang/tensor/Tensor;Ljava/lang/Number;)Lsk/ainet/lang/tensor/Tensor;
Expand All @@ -283,6 +284,7 @@ public class sk/ainet/exec/tensor/ops/DefaultCpuOpsBase : sk/ainet/lang/tensor/o
public fun pow (Lsk/ainet/lang/tensor/Tensor;Lsk/ainet/lang/tensor/Tensor;)Lsk/ainet/lang/tensor/Tensor;
public fun powScalar (Lsk/ainet/lang/tensor/Tensor;Ljava/lang/Number;)Lsk/ainet/lang/tensor/Tensor;
public fun rdivScalar (Ljava/lang/Number;Lsk/ainet/lang/tensor/Tensor;)Lsk/ainet/lang/tensor/Tensor;
public fun relayoutPackedWeightForKernels (Lsk/ainet/lang/tensor/Tensor;)Lsk/ainet/lang/tensor/Tensor;
public fun relu (Lsk/ainet/lang/tensor/Tensor;)Lsk/ainet/lang/tensor/Tensor;
protected final fun requireSameDType (Lsk/ainet/lang/tensor/Tensor;Lsk/ainet/lang/tensor/Tensor;)V
public fun reshape (Lsk/ainet/lang/tensor/Tensor;Lsk/ainet/lang/tensor/Shape;)Lsk/ainet/lang/tensor/Tensor;
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -872,32 +872,82 @@ public open class DefaultCpuOpsBase(protected val dataFactory: TensorDataFactory
return newTensor(outData, a.dtype, a, b)
}

@TensorOp()
override fun <T : DType, V> transpose(tensor: Tensor<T, V>): Tensor<T, V> {
/**
* Weights already relayouted into kernel feed order, keyed by the identity of the packed bytes
* they came from (#973/#1096).
*
* The relayout is O(bytes). Doing it inside [matmulWeightTransposed] once per weight instead of
* once per call is what removes the per-forward copy `Linear.onForward` used to pay: a model's
* weights are stable, so the first forward pass converts and every later one reuses. Bounded,
* because a cache that grows without limit on a 2 GB device is its own bug; a model with more
* than [PREPACK_CACHE_LIMIT] distinct packed weights simply converts the overflow each time,
* which is exactly the old behaviour.
*/
private val prepackedWeights: MutableList<Pair<ByteArray, Tensor<*, *>>> = mutableListOf()

/** How many relayouted weights to keep; beyond this the oldest is dropped and reconverted on demand. */
private val PREPACK_CACHE_LIMIT: Int = 64

/**
* `x · Wᵀ` with the weight as `[out, in]` — the primitive, and the way out of the per-forward
* copy (#973 "the deeper semantic problem", #1096).
*
* For a block-quantized weight this relayouts **once** and reuses the result; for anything else
* it is the ordinary `matmul(x, transpose(w))`, which for dense data is a free shape swap.
*/
@Suppress("UNCHECKED_CAST")
override fun <T : DType, V> matmulWeightTransposed(x: Tensor<T, V>, weight: Tensor<T, V>): Tensor<T, V> {
if (weight.shape.rank != 2 || !isHeapPackedWeight(weight.data)) return matmul(x, transpose(weight))
val packed = weight.data as sk.ainet.lang.tensor.storage.PackedBlockStorage
val source = packed.packedData
val cached = prepackedWeights.firstOrNull { it.first === source }?.second
val kernelOrder = cached ?: relayoutPackedWeightForKernels(weight).also { relayouted ->
if (prepackedWeights.size >= PREPACK_CACHE_LIMIT) prepackedWeights.removeAt(0)
prepackedWeights.add(Pair(source, relayouted as Tensor<*, *>))
}
return matmul(x, kernelOrder as Tensor<T, V>)
}


/**
* The block-grid permutation that used to live in `transpose` (#973/#1096).
*
* The packed matmul kernels read `packedData` as **input-block-major** —
* `(blockIdx * outputDim + o)`, every output row's block for one input block contiguous —
* whatever shape the tensor declares. Canonical packed storage, as loaded from a GGUF or built
* by any row-major producer, is the other order. The two coincide only at one block per row, so
* for a real weight a bare shape relabel hands the kernel bytes in the wrong physical order and
* it reads garbage without failing (#968, and downstream SKaiNET-transformers#307).
*
* So this is a real O(bytes) permutation, and [matmulWeightTransposed] runs it **once** per
* weight rather than once per call — which is the difference #1096 exists to make.
*
* @return the relayouted weight, or `null` for a data type with no packed relayout
*/
/** The heap packed data types whose kernels read input-block-major bytes. */
private fun isHeapPackedWeight(data: sk.ainet.lang.tensor.data.TensorData<*, *>): Boolean =
data is Q4_KTensorData || data is Q5_KTensorData || data is Q6_KTensorData ||
data is Q5_1TensorData || data is Q5_0TensorData || data is Q8_0TensorData || data is Q4_0TensorData

/**
* The block relayout by its own name (#973/#1096) — what `transpose` used to do to a packed
* weight, for the callers that genuinely want the permuted bytes rather than a product.
*
* Prefer [matmulWeightTransposed], which does this once per weight instead of once per call.
*
* @throws UnsupportedOperationException for a data type with no packed relayout
*/
override fun <T : DType, V> relayoutPackedWeightForKernels(weight: Tensor<T, V>): Tensor<T, V> =
transposePackedWeight(weight)
?: throw UnsupportedOperationException(
"no packed relayout for ${weight.data::class.simpleName}",
)

@Suppress("UNCHECKED_CAST")
private fun <T : DType, V> transposePackedWeight(tensor: Tensor<T, V>): Tensor<T, V>? {
val rank = tensor.shape.rank
require(rank >= 2) { "Transpose requires at least 2 dimensions" }
val rows = tensor.shape[rank - 2]
val cols = tensor.shape[rank - 1]

// Transpose for heap-packed quant weights (Q4_K/Q5_K/Q6_K/Q5_1/Q5_0/Q8_0/Q4_0):
// the packed-quant matmul kernels (chooseQuantizedMatmulHeap, native/Panama/
// scalar alike) always read `packedData` as *input-block-major* —
// `(blockIdx * outputDim + o)`, all `outputDim` rows' blocks for a fixed
// input block contiguous (see e.g. q5_0_matmul.c) — regardless of the
// tensor's declared shape. Canonical packed storage (as loaded verbatim from
// GGUF, or as built by any other row-major producer) is instead *row-major*:
// for output row `o`, its `blocksPerInputDim` blocks are contiguous, then row
// `o + 1`. Those two orderings coincide only when there's a single block per
// row (`blocksPerInputDim == 1`); for any weight wider than one block — i.e.
// virtually every real model, since inputDim is almost always >> the 32/256-
// element block size — a bare shape relabel hands the kernel bytes in the
// WRONG physical order and it silently reads garbage (SKaiNET#968,
// surfaced downstream via SKaiNET-transformers#307's all-zero Q5_0/Q5_1
// matmul). So this performs the actual O(bytes) block-grid permutation
// (`transposePackedBlocks`) instead of a free shape swap. Still avoids the
// FP32 dequant round-trip `ops.matmul(x, ops.transpose(W))` was written to
// dodge — just not for free anymore.
if (rank == 2) {
@Suppress("UNCHECKED_CAST")
when (val d = tensor.data) {
is Q4_KTensorData -> {
Expand Down Expand Up @@ -941,19 +991,48 @@ public open class DefaultCpuOpsBase(protected val dataFactory: TensorDataFactory
val reordered = transposePackedBlocks(d.packedData, rows, blocksPerInputDim, Q4_0TensorData.BYTES_PER_BLOCK)
return newTensor(Q4_0BlockTensorData(Shape(cols, rows), reordered) as TensorData<T, V>, tensor.dtype, tensor)
}
// Narrow floats (FP16/BF16) relaid input-major at load: the transpose is the
// same buffer read with the other shape's strides, so hand back an ordinary
// dense narrow tensor over it. This is what lets a KEEP_NATIVE weight survive
// `Linear.onForward`'s `weight.t()` and reach the narrow matmul kernel — see
// issue #888. Note the asymmetry with the block-quant arms above: only the
// *input-major* type is safe to reinterpret. A row-major narrow buffer falls
// through to the generic path on purpose, because swapping its shape would
// silently yield a different matrix rather than the transpose.
// Lives only here: DefaultCpuOpsJvm.transpose intercepts nothing that would
// shadow this case, so the JVM falls through to this arm too.
is NarrowFloatInputMajorTensorData -> return newTensor(d.transposedView() as TensorData<T, V>, tensor.dtype, tensor)
else -> {}
}

return null
}

@TensorOp()
override fun <T : DType, V> transpose(tensor: Tensor<T, V>): Tensor<T, V> {
val rank = tensor.shape.rank
require(rank >= 2) { "Transpose requires at least 2 dimensions" }
val rows = tensor.shape[rank - 2]
val cols = tensor.shape[rank - 1]

// Only the *heap* packed types, whose kernels read input-block-major bytes. The
// MemorySegment tier reads canonical bytes, so for those a shape swap is genuinely correct
// and stays where it is — census contradiction #3, now stated instead of implied.
val heapPacked = rank == 2 && isHeapPackedWeight(tensor.data)
if (heapPacked) {
val packedData = tensor.data as sk.ainet.lang.tensor.storage.PackedBlockStorage
// Transposing block-quantized data is not a representable operation (#973): blocks
// quantize runs along the input dimension, so a real transpose needs requantization.
// What used to happen here was a layout conversion wearing transpose's name — an
// O(bytes) copy per call, not an involution, and a lie about what the result means.
throw UnsupportedOperationException(
"transpose() is not defined for a ${packedData.encoding.name} weight: blocks quantize runs " +
"along the input dimension, so transposing them needs requantization, and what this used to " +
"do was a per-call layout copy that is not its own inverse (#973). Use " +
"ops.matmulWeightTransposed(x, weight) with the weight as [out, in], or relayout explicitly " +
"with PackedWeights.prepackForMatmul. See docs/design/memory/packed-weight-layout.md.",
)
}

// Narrow floats (FP16/BF16) relaid input-major at load are a *view* rewrap, not a packed
// relayout: the transpose is the same buffer read with the other shape's strides, which is
// what lets a KEEP_NATIVE weight survive Linear's weight handling and reach the narrow
// matmul kernel (#888). Only the block-quantized types are refused above (#973).
if (rank == 2) {
val narrow = tensor.data as? NarrowFloatInputMajorTensorData
if (narrow != null) {
@Suppress("UNCHECKED_CAST")
return newTensor(narrow.transposedView() as TensorData<T, V>, tensor.dtype, tensor)
}
}

// Fast path: 2D float tensor — direct buffer swap
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,7 @@ import kotlin.test.assertTrue

/**
* SKEEP-003 golden gate, dispatch half (runs on every target): for all seven GGML packed
* encodings, `ops.matmul(x, ops.transpose(w))` on a canonical row-major packed weight must agree
* encodings, `ops.matmulWeightTransposed(x, w)` on a canonical row-major packed weight must agree
* with an FP32 reference computed from the decoded weight. Tolerance-based because the JVM may
* pick SIMD / native / Q8-activation kernel tiers (#944), which are not bit-identical to the
* scalar reference; the bit-identical guarantees live in the goldenTest source set.
Expand Down Expand Up @@ -74,7 +74,7 @@ class PackedMatmulDispatchParityTest {
val xf = FloatArray(batch * inDim) { rng.nextFloat() * 2f - 1f }
val x = ctx.fromFloatArray<FP32, Float>(Shape(batch, inDim), FP32::class, xf)

val actual = ctx.ops.matmul(x, ctx.ops.transpose(w)).data.copyToFloatArray()
val actual = ctx.ops.matmulWeightTransposed(x, w).data.copyToFloatArray()

// FP32 reference from the decoded weight
val wf = FloatArray(outDim * inDim); val tmp = FloatArray(f.blockSize)
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,95 @@
package sk.ainet.exec.tensor.ops

import sk.ainet.context.DirectCpuExecutionContext
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.Tensor
import sk.ainet.lang.tensor.data.Q8_0BlockTensorData
import sk.ainet.lang.tensor.data.TensorData
import sk.ainet.lang.types.FP32
import kotlin.math.abs
import kotlin.test.Test
import kotlin.test.assertContentEquals
import kotlin.test.assertFailsWith
import kotlin.test.assertTrue

/**
* #1096 (#973): `x · Wᵀ` as a primitive, and `transpose` on a packed weight as an error.
*
* The old shape — `matmul(x, transpose(w))` — copied the whole weight on every call, and
* `transpose(transpose(w))` was not `w`. Both are gone: the product is asked for directly, the
* relayout happens once per weight, and `transpose` refuses rather than lying.
*/
class MatmulWeightTransposedTest {

private val ctx = DirectCpuExecutionContext()
private val outDim = 4
private val inDim = 96 // three Q8_0 blocks per row: the case where block order matters

@Suppress("UNCHECKED_CAST")
private fun weight(): Tensor<FP32, Float> {
val blocks = outDim * (inDim / 32)
val bytes = ByteArray(blocks * 34)
var seed = 11
for (b in 0 until blocks) {
val base = b * 34
bytes[base] = 0x00; bytes[base + 1] = 0x3C // fp16 scale 1.0
for (i in 0 until 32) {
seed = seed * 1103515245 + 12345
bytes[base + 2 + i] = ((seed ushr 16) % 17 - 8).toByte()
}
}
val data = Q8_0BlockTensorData(Shape(outDim, inDim), bytes)
return ctx.fromData(data as TensorData<FP32, Float>, FP32::class)
}

private fun activation(): Tensor<FP32, Float> =
ctx.fromFloatArray<FP32, Float>(Shape(1, inDim), FP32::class, FloatArray(inDim) { (it % 7) * 0.125f })

@Test
fun `the primitive agrees with the relayout then matmul it replaces`() {
val w = weight()
val x = activation()
val viaPrimitive = ctx.ops.matmulWeightTransposed(x, w).data.copyToFloatArray()
val viaRelayout = ctx.ops.matmul(x, ctx.ops.relayoutPackedWeightForKernels(w)).data.copyToFloatArray()
assertContentEquals(viaRelayout, viaPrimitive, "the primitive must compute exactly what the old path did")
}

@Test
fun `a weight is relayouted once however many times it is used`() {
val w = weight()
val x = activation()
val first = ctx.ops.matmulWeightTransposed(x, w).data.copyToFloatArray()
repeat(5) {
assertContentEquals(first, ctx.ops.matmulWeightTransposed(x, w).data.copyToFloatArray())
}
// and the answer keeps matching the explicit relayout, so the cache is not stale
assertContentEquals(
ctx.ops.matmul(x, ctx.ops.relayoutPackedWeightForKernels(w)).data.copyToFloatArray(),
first,
)
}

@Test
fun `transpose refuses a packed weight and says what to use instead`() {
val failure = assertFailsWith<UnsupportedOperationException> { ctx.ops.transpose(weight()) }
val message = failure.message!!
assertTrue(message.contains("not defined for a Q8_0 weight"), message)
assertTrue(message.contains("matmulWeightTransposed"), "it names the primitive: $message")
assertTrue(message.contains("prepackForMatmul"), "and the explicit relayout: $message")
assertTrue(message.contains("requantization"), "and why: $message")
}

@Test
fun `a dense weight transposes as it always did`() {
val dense: Tensor<FP32, Float> =
ctx.fromFloatArray<FP32, Float>(Shape(outDim, inDim), FP32::class, FloatArray(outDim * inDim) { it * 0.01f })
val t = ctx.ops.transpose(dense)
assertTrue(t.shape == Shape(inDim, outDim))
val x = activation()
val viaPrimitive = ctx.ops.matmulWeightTransposed(x, dense).data.copyToFloatArray()
val viaTranspose = ctx.ops.matmul(x, t).data.copyToFloatArray()
for (i in viaPrimitive.indices) {
assertTrue(abs(viaPrimitive[i] - viaTranspose[i]) < 1e-4f, "[$i]: ${viaPrimitive[i]} vs ${viaTranspose[i]}")
}
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,7 @@ import sk.ainet.lang.tensor.data.TensorData
import sk.ainet.lang.types.FP32

/**
* End-to-end proof that packed-quant weights flow through `ctx.ops.matmul(x, ops.transpose(W))`
* End-to-end proof that packed-quant weights flow through `ctx.ops.matmulWeightTransposed(x, W)`
* on EVERY platform — exercising the lazy-transpose shape-swap + `chooseQuantizedMatmulHeap` in
* DefaultCpuOpsBase, resolving the registered kernel (scalar on Native/JS/WASM, Panama/FFM on JVM).
* Runs on jvmTest AND linuxX64Test; a green linuxX64 run is the headline "Native packed matmul works".
Expand All @@ -45,7 +45,7 @@ class PackedMatmulDispatchTest {
* .transpose` is responsible for the canonical → kernel-native block-grid
* permutation (`DefaultCpuOpsBase.transposePackedBlocks`); this generator
* must hand it genuinely canonical input or it isn't testing the real
* `ops.matmul(x, ops.transpose(W))` path — see SKaiNET-transformers#307.
* `ops.matmulWeightTransposed(x, W)` path — see SKaiNET-transformers#307.
*/
private fun q5_1(inDim: Int, outDim: Int, rng: Random): Pair<ByteArray, FloatArray> {
val blocks = inDim / 32; val bytes = ByteArray(outDim * blocks * 24); val wf = FloatArray(outDim * inDim)
Expand Down Expand Up @@ -135,7 +135,7 @@ class PackedMatmulDispatchTest {
)
val xf = FloatArray(inDim) { rng.nextFloat() - 0.5f }
val x = ctx.fromFloatArray<FP32, Float>(Shape(1, inDim), FP32::class, xf)
val out = ctx.ops.matmul(x, ctx.ops.transpose(w)).data.copyToFloatArray()
val out = ctx.ops.matmulWeightTransposed(x, w).data.copyToFloatArray()
val expected = FloatArray(outDim) { o -> var s = 0f; for (i in 0 until inDim) s += xf[i] * wf[o * inDim + i]; s }
var maxErr = 0f; var maxAbs = 1f
for (o in 0 until outDim) { maxErr = maxOf(maxErr, abs(expected[o] - out[o])); maxAbs = maxOf(maxAbs, abs(expected[o])) }
Expand Down Expand Up @@ -173,7 +173,7 @@ class PackedMatmulDispatchTest {
val bytes = ByteArray(outDim * (inDim / blockElems) * bpb)
val w = ctx.fromData(build(Shape(outDim, inDim), bytes), FP32::class)
// The bug threw here for unhandled packed types.
val t = ctx.ops.transpose(w)
val t = ctx.ops.relayoutPackedWeightForKernels(w)
assertEquals(Shape(inDim, outDim), t.shape, "$name: transpose did not flip shape")
assertTrue(
t.data::class.simpleName?.contains("Block") == true,
Expand Down
Loading
Loading