Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,16 @@ package sk.ainet.exec.golden
*/
internal object Goldens {
val expected: Map<String, String> = mapOf(
// #1034 — the zero-copy packed transpose: a TensorView whose block axis moved, decoding
// the same matrix the block-grid permutation in DefaultCpuOps.transpose produces.
"transpose/Q4_0" to "n=384 fnv=272715de48d49929 head=3c442000,3df90000,be8ed800,be2e0000",
"transpose/Q4_K" to "n=3072 fnv=894ff6387d5c031f head=41ca6280,4129ccc0,414a2844,41d82e00",
"transpose/Q5_0" to "n=384 fnv=73c3dedcac1b0471 head=bd4ce000,3ddf7000,3e857000,3c2e0000",
"transpose/Q5_1" to "n=384 fnv=c7a3a894ff2b8d10 head=3ec3a800,be569800,3ef3fe00,bdc0f000",
"transpose/Q5_K" to "n=3072 fnv=9646306126643f2d head=41cc33a0,40480970,421ca022,41b5e2b8",
"transpose/Q6_K" to "n=3072 fnv=fee2e34031fb8e86 head=411c7680,c21b1e00,416739c0,4140f8c8",
"transpose/Q8_0" to "n=384 fnv=9c5a086e8bc0d528 head=40117580,bf912800,4014f800,3fb2db00",

// #1033 — the ternary encodings, encoded and decoded by TernaryCodec (the GGML layout,
// interleave included). TQ1_0 and TQ2_0 share a decode digest on purpose: the same ternary
// values in two different byte layouts must come back identical.
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,129 @@
package sk.ainet.exec.golden

import sk.ainet.context.DirectCpuExecutionContext
import sk.ainet.exec.golden.GoldenSupport.Packed
import sk.ainet.lang.memory.ExperimentalMemoryApi
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.Tensor
import sk.ainet.lang.tensor.data.Q4_0BlockTensorData
import sk.ainet.lang.tensor.data.Q4_KBlockTensorData
import sk.ainet.lang.tensor.data.Q5_0BlockTensorData
import sk.ainet.lang.tensor.data.Q5_1BlockTensorData
import sk.ainet.lang.tensor.data.Q5_KBlockTensorData
import sk.ainet.lang.tensor.data.Q6_KBlockTensorData
import sk.ainet.lang.tensor.data.Q8_0BlockTensorData
import sk.ainet.lang.tensor.data.TensorData
import sk.ainet.lang.tensor.storage.PackedBlockStorage
import sk.ainet.lang.tensor.t
import sk.ainet.lang.types.FP32
import kotlin.test.Test
import kotlin.test.assertEquals
import kotlin.test.assertTrue

/**
* #1034: a packed transpose is metadata.
*
* `DefaultCpuOps.transpose` permutes the block grid byte by byte, because the packed matmul
* kernels read their weight as input-block-major regardless of its declared shape — the contract
* that made #968/#971 read garbage from a bare shape swap, and that #973 exists to write down.
* A `TensorView` needs no such permutation: transposing swaps two strides and moves the block axis
* ([sk.ainet.lang.memory.Layout.blockAxis]) with it.
*
* Both must describe the *same matrix*. This asserts exactly that, for every packed encoding, on
* JVM and Kotlin/Native alike — and records the transposed values as goldens.
*/
@OptIn(ExperimentalMemoryApi::class)
class PackedTransposeGoldenTest {

private companion object {
const val ROWS = 4
const val BLOCKS_PER_ROW = 3
const val SEED = 0x5EED_0003L
}

private val ctx = DirectCpuExecutionContext()

private fun build(p: Packed, shape: Shape, bytes: ByteArray): PackedBlockStorage = when (p) {
Packed.Q4_0 -> Q4_0BlockTensorData(shape, bytes)
Packed.Q5_0 -> Q5_0BlockTensorData(shape, bytes)
Packed.Q5_1 -> Q5_1BlockTensorData(shape, bytes)
Packed.Q8_0 -> Q8_0BlockTensorData(shape, bytes)
Packed.Q4_K -> Q4_KBlockTensorData(shape, bytes)
Packed.Q5_K -> Q5_KBlockTensorData(shape, bytes)
Packed.Q6_K -> Q6_KBlockTensorData(shape, bytes)
}

@Suppress("UNCHECKED_CAST")
private fun transposeAgrees(p: Packed) {
val shape = Shape(ROWS, BLOCKS_PER_ROW * p.blockSize)
val bytes = GoldenSupport.rowMajor(GoldenSupport.weightBlocks(p, ROWS, BLOCKS_PER_ROW, SEED))
val data = build(p, shape, bytes)
val tensor: Tensor<FP32, Float> = ctx.fromData(data as TensorData<FP32, Float>, FP32::class)

val view = data.packedView
val transposed = view.transpose()
assertEquals(Shape(shape[1], shape[0]), transposed.shape, "${p.name}: shape")
assertEquals(view.storage.id, transposed.storage.id, "${p.name}: the transpose must not touch the bytes")

// 1. the view transposes what it decodes
for (r in 0 until shape[0]) {
for (c in 0 until shape[1]) {
assertEquals(view.get(r, c), transposed.get(c, r), "${p.name}: element ($r,$c)")
}
}

// 2. and it describes the same matrix as the physical block-grid permutation the kernels
// still need. `DefaultCpuOps.transpose` reorders the blocks to *input-block-major* —
// block (bI, o) at index `bI * rows + o` — because that is how the packed kernels read a
// weight, whatever shape it declares (#968/#971; the contract #973 exists to write down).
// So the permuted bytes are not the row-major encoding of the transposed matrix, and the
// two paths are not interchangeable until #973 lands: decoded *as block-major*, they carry
// exactly the values the zero-copy view exposes.
val physical = tensor.t()
assertEquals(Shape(shape[1], shape[0]), physical.shape, "${p.name}: ops.transpose shape")
val permutedBytes = (physical.data as PackedBlockStorage).packedData
assertTrue(
permutedBytes.contentEquals(GoldenSupport.blockMajor(GoldenSupport.weightBlocks(p, ROWS, BLOCKS_PER_ROW, SEED))),
"${p.name}: ops.transpose must produce input-block-major bytes",
)
val permuted = build(p, shape, permutedBytes)
val block = FloatArray(p.blockSize)
val fromView = FloatArray(shape.volume)
var i = 0
for (c in 0 until shape[1]) {
for (r in 0 until shape[0]) {
permuted.dequantizeBlock((c / p.blockSize) * ROWS + r, block, 0)
assertEquals(view.get(r, c), block[c % p.blockSize], "${p.name}: block-major element ($r,$c)")
fromView[i++] = transposed.get(c, r)
}
}
GoldenSupport.check("transpose/${p.name}", GoldenSupport.digest(fromView))
}

@Test fun q4_0() = transposeAgrees(Packed.Q4_0)
@Test fun q5_0() = transposeAgrees(Packed.Q5_0)
@Test fun q5_1() = transposeAgrees(Packed.Q5_1)
@Test fun q8_0() = transposeAgrees(Packed.Q8_0)
@Test fun q4_K() = transposeAgrees(Packed.Q4_K)
@Test fun q5_K() = transposeAgrees(Packed.Q5_K)
@Test fun q6_K() = transposeAgrees(Packed.Q6_K)

/** Narrowing a transposed packed view still addresses whole blocks — on the axis that carries them. */
@Test
fun aTransposedPackedViewStillSlicesByBlock() {
val p = Packed.Q8_0
val shape = Shape(ROWS, BLOCKS_PER_ROW * p.blockSize)
val bytes = GoldenSupport.rowMajor(GoldenSupport.weightBlocks(p, ROWS, BLOCKS_PER_ROW, SEED))
val data = build(p, shape, bytes)
val view = data.packedView
val transposed = view.transpose()

val head = transposed.narrow(0, 0, p.blockSize) // one block along the (now leading) block axis
assertEquals(Shape(p.blockSize, ROWS), head.shape)
for (c in 0 until p.blockSize) for (r in 0 until ROWS) assertEquals(view.get(r, c), head.get(c, r))

val rows = transposed.narrow(1, 1, 2) // the plain axis narrows freely
assertEquals(Shape(shape[1], 2), rows.shape)
assertEquals(view.get(1, 0), rows.get(0, 0))
}
}
14 changes: 12 additions & 2 deletions skainet-lang/skainet-lang-core/api/jvm/skainet-lang-core.api
Original file line number Diff line number Diff line change
Expand Up @@ -797,10 +797,11 @@ public final class sk/ainet/lang/memory/ForwardScope : sk/ainet/lang/memory/Scop

public final class sk/ainet/lang/memory/Layout {
public static final field Companion Lsk/ainet/lang/memory/Layout$Companion;
public fun <init> (Lsk/ainet/lang/tensor/Shape;[IJIZ)V
public synthetic fun <init> (Lsk/ainet/lang/tensor/Shape;[IJIZILkotlin/jvm/internal/DefaultConstructorMarker;)V
public fun <init> (Lsk/ainet/lang/tensor/Shape;[IJIZI)V
public synthetic fun <init> (Lsk/ainet/lang/tensor/Shape;[IJIZIILkotlin/jvm/internal/DefaultConstructorMarker;)V
public final fun byteOffsetOf ([I)J
public fun equals (Ljava/lang/Object;)Z
public final fun getBlockAxis ()I
public final fun getBlocked ()Z
public final fun getElementBytes ()I
public final fun getElementCount ()J
Expand All @@ -814,6 +815,7 @@ public final class sk/ainet/lang/memory/Layout {
public final fun isRowMajor ()Z
public final fun narrow (III)Lsk/ainet/lang/memory/Layout;
public final fun squeeze (I)Lsk/ainet/lang/memory/Layout;
public final fun step (II)Lsk/ainet/lang/memory/Layout;
public fun toString ()Ljava/lang/String;
public final fun transpose (II)Lsk/ainet/lang/memory/Layout;
public static synthetic fun transpose$default (Lsk/ainet/lang/memory/Layout;IIILjava/lang/Object;)Lsk/ainet/lang/memory/Layout;
Expand Down Expand Up @@ -1190,6 +1192,7 @@ public final class sk/ainet/lang/memory/TensorView {
public final fun narrow (III)Lsk/ainet/lang/memory/TensorView;
public final fun set ([IF)V
public final fun squeeze (I)Lsk/ainet/lang/memory/TensorView;
public final fun step (II)Lsk/ainet/lang/memory/TensorView;
public final fun toFloatArray ()[F
public fun toString ()Ljava/lang/String;
public final fun transpose (II)Lsk/ainet/lang/memory/TensorView;
Expand Down Expand Up @@ -1236,6 +1239,13 @@ public final class sk/ainet/lang/memory/TernaryCodec {
public final fun encodeTq2_0 ([F)[B
}

public final class sk/ainet/lang/memory/ViewsKt {
public static final fun slice (Lsk/ainet/lang/memory/TensorView;Ljava/util/List;)Lsk/ainet/lang/memory/TensorView;
public static final fun slice (Lsk/ainet/lang/memory/TensorView;[Lsk/ainet/lang/tensor/Slice;)Lsk/ainet/lang/memory/TensorView;
public static final fun view (Lsk/ainet/lang/tensor/Tensor;)Lsk/ainet/lang/memory/TensorView;
public static final fun viewOrNull (Lsk/ainet/lang/tensor/Tensor;)Lsk/ainet/lang/memory/TensorView;
}

public final class sk/ainet/lang/memory/plan/ActualMemory {
public static final field Companion Lsk/ainet/lang/memory/plan/ActualMemory$Companion;
public fun <init> (Ljava/util/Map;Ljava/util/Map;Ljava/util/Map;Ljava/util/Map;)V
Expand Down
Original file line number Diff line number Diff line change
@@ -1,3 +1,5 @@
@file:Suppress("DEPRECATION") // the pre-#1034 view mechanism: kept working until the next major

package sk.ainet.lang.memory

import sk.ainet.lang.tensor.Tensor
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,9 @@ import sk.ainet.lang.tensor.Shape
* @property strides one stride per dimension, in elements (or blocks for a blocked layout)
* @property offsetElements element (or block) offset of the first element inside the storage
* @property elementBytes bytes per element (dense) or per block (blocked)
* @property blockAxis for a [blocked] layout, the axis whose extent is measured in blocks. It is
* the last axis as loaded, and it *moves* with [transpose]/[unsqueeze]/[squeeze] — which is what
* lets a packed weight be transposed as metadata and still decode correctly (#1034).
*/
@ExperimentalMemoryApi
public class Layout(
Expand All @@ -24,11 +27,13 @@ public class Layout(
public val offsetElements: Long = 0L,
public val elementBytes: Int = 4,
public val blocked: Boolean = false,
public val blockAxis: Int = shape.rank - 1,
) {
init {
require(strides.size == shape.rank) { "strides (${strides.size}) must match rank (${shape.rank})" }
require(offsetElements >= 0) { "offsetElements must be >= 0" }
require(elementBytes > 0) { "elementBytes must be > 0" }
if (blocked) require(blockAxis in 0 until shape.rank) { "blockAxis $blockAxis out of range for rank ${shape.rank}" }
}

/** Number of elements addressed (the shape's volume). */
Expand Down Expand Up @@ -74,7 +79,7 @@ public class Layout(
require(axis in 0 until shape.rank) { "axis $axis out of range for rank ${shape.rank}" }
require(from >= 0 && size >= 0 && from + size <= shape[axis]) { "narrow($axis, $from, $size) outside extent ${shape[axis]}" }
val dims = shape.dimensions.copyOf(); dims[axis] = size
return Layout(Shape(dims), strides.copyOf(), offsetElements + from.toLong() * strides[axis], elementBytes, blocked)
return Layout(Shape(dims), strides.copyOf(), offsetElements + from.toLong() * strides[axis], elementBytes, blocked, blockAxis)
}

/** Swap two axes — metadata only; the bytes are untouched (the packed-transpose trick, rule 5). */
Expand All @@ -84,7 +89,26 @@ public class Layout(
val dims = shape.dimensions.copyOf(); val st = strides.copyOf()
val d = dims[axis0]; dims[axis0] = dims[axis1]; dims[axis1] = d
val s = st[axis0]; st[axis0] = st[axis1]; st[axis1] = s
return Layout(Shape(dims), st, offsetElements, elementBytes, blocked)
// The block axis travels with its extent: a transposed packed view still knows which index
// splits into (block, offset-within-block).
val movedBlockAxis = when (blockAxis) { axis0 -> axis1; axis1 -> axis0; else -> blockAxis }
return Layout(Shape(dims), st, offsetElements, elementBytes, blocked, movedBlockAxis)
}

/**
* Every [step]-th element along [axis] — a stride multiply, zero-copy. This is what makes the
* strided half of the old `Slice.Step` a layout operation rather than an index remapper
* (#1034): the bytes are untouched and the result is an ordinary [Layout].
*/
public fun step(axis: Int, step: Int): Layout {
require(axis in 0 until shape.rank) { "axis $axis out of range for rank ${shape.rank}" }
require(step > 0) { "step must be positive, got $step" }
if (step == 1) return this
val dims = shape.dimensions.copyOf()
dims[axis] = (shape[axis] + step - 1) / step
val st = strides.copyOf()
st[axis] = strides[axis] * step
return Layout(Shape(dims), st, offsetElements, elementBytes, blocked, blockAxis)
}

/** Insert a unit axis at [axis] (stride 0 — it is never stepped). */
Expand All @@ -95,29 +119,30 @@ public class Layout(
for (i in 0..shape.rank) {
if (i == axis) { dims[i] = 1; st[i] = 0 } else { dims[i] = shape[j]; st[i] = strides[j]; j++ }
}
return Layout(Shape(dims), st, offsetElements, elementBytes, blocked)
return Layout(Shape(dims), st, offsetElements, elementBytes, blocked, if (axis <= blockAxis) blockAxis + 1 else blockAxis)
}

/** Drop the unit axis at [axis]. */
public fun squeeze(axis: Int): Layout {
require(axis in 0 until shape.rank) { "axis $axis out of range for rank ${shape.rank}" }
require(shape[axis] == 1) { "axis $axis has extent ${shape[axis]}, not 1" }
val dims = ArrayList<Int>(shape.rank - 1); val st = ArrayList<Int>(shape.rank - 1)
require(!(blocked && axis == blockAxis)) { "cannot drop the block axis of a blocked layout" }
for (i in 0 until shape.rank) if (i != axis) { dims += shape[i]; st += strides[i] }
return Layout(Shape(dims.toIntArray()), st.toIntArray(), offsetElements, elementBytes, blocked)
return Layout(Shape(dims.toIntArray()), st.toIntArray(), offsetElements, elementBytes, blocked, if (axis < blockAxis) blockAxis - 1 else blockAxis)
}

override fun toString(): String =
"Layout($shape, strides=${strides.joinToString(",", "[", "]")}, offset=$offsetElements${if (blocked) " blocks" else ""}, ${if (isContiguous) "contiguous" else "strided"})"

override fun equals(other: Any?): Boolean = other is Layout && other.shape == shape &&
other.strides.contentEquals(strides) && other.offsetElements == offsetElements &&
other.elementBytes == elementBytes && other.blocked == blocked
other.elementBytes == elementBytes && other.blocked == blocked && other.blockAxis == blockAxis

override fun hashCode(): Int {
var h = shape.hashCode()
h = 31 * h + strides.contentHashCode(); h = 31 * h + offsetElements.hashCode()
h = 31 * h + elementBytes; h = 31 * h + blocked.hashCode()
h = 31 * h + elementBytes; h = 31 * h + blocked.hashCode(); h = 31 * h + blockAxis
return h
}

Expand Down
Loading
Loading