Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,91 @@
package sk.ainet.io.gguf

import sk.ainet.io.RandomAccessSource

/**
* I2_S scale resolution (#1140), shared by [StreamingGgufParametersLoader] and any other reader
* of an I2_S GGUF — notably an AOT converter (#1207) that needs the exact same answer the
* streaming loader would give, without duplicating the decision.
*/

/** The `<name>_scale` companion tensor for an I2_S weight, if the converter wrote one (#1140). */
internal fun i2sCompanionScaleTensor(
tensorInfo: StreamingTensorInfo,
tensors: List<StreamingTensorInfo>,
): StreamingTensorInfo? = tensors.firstOrNull {
it.tensorType == GGMLQuantizationType.F32 && it.name == "${tensorInfo.name}_scale"
}

private fun le32(bytes: ByteArray, offset: Int = 0): Float = Float.fromBits(
(bytes[offset].toInt() and 0xFF) or
((bytes[offset + 1].toInt() and 0xFF) shl 8) or
((bytes[offset + 2].toInt() and 0xFF) shl 16) or
((bytes[offset + 3].toInt() and 0xFF) shl 24)
)

/**
* The little-endian FP32 immediately after [tensorInfo]'s payload — BitNet.cpp's trailer
* convention — or `null` if unreadable, non-finite, or zero. [StreamingTensorInfo.nBytes]
* deliberately sizes the payload only, so `absoluteDataOffset + nBytes` is exactly where a
* trailer would start.
*/
internal fun i2sTrailerScale(tensorInfo: StreamingTensorInfo, source: RandomAccessSource): Float? =
runCatching { le32(source.readAt(tensorInfo.absoluteDataOffset + tensorInfo.nBytes, 4)) }
.getOrNull()?.takeIf { it.isFinite() && it != 0f }

/**
* The per-tensor FP32 scale of an I2_S weight, from wherever its converter put it (#1140):
*
* - **BitNet.cpp** writes it as a trailer after the payload (a 32-byte-aligned region whose
* first 4 bytes are the LE FP32 scale; `w = (code − 1) · scale`). Read directly from the
* source at `absoluteDataOffset + payload` — [StreamingTensorInfo.nBytes] deliberately sizes
* the payload only.
* - **NeoGPU's converter** writes a companion `<name>_scale` F32 scalar, defined as "divide the
* projection output by it" — so the stored multiplier is its inverse.
* - Neither present (or unreadable/non-finite/zero): `1.0`, i.e. the raw codes. Loud in the
* trace via the repack conversion's byte counts, never a crash.
*
* [layout] decides which source is tried first; both are accepted either way, because a
* sequential file with a trailer or a group file with a companion costs nothing to honour.
*/
internal fun resolveI2sScale(
tensorInfo: StreamingTensorInfo,
tensors: List<StreamingTensorInfo>,
reader: StreamingGGUFReader,
source: RandomAccessSource,
layout: I2sGgufLayout,
): Float {
fun companionInverse(): Float? {
val companion = i2sCompanionScaleTensor(tensorInfo, tensors) ?: return null
val value = runCatching { le32(reader.loadTensorData(companion)) }.getOrNull() ?: return null
if (!value.isFinite() || value == 0f) return null
return 1f / value
}

return when (layout) {
I2sGgufLayout.GROUP_128, I2sGgufLayout.GROUP_64 -> i2sTrailerScale(tensorInfo, source) ?: companionInverse() ?: 1f
I2sGgufLayout.SEQUENTIAL -> companionInverse() ?: i2sTrailerScale(tensorInfo, source) ?: 1f
}
}

/**
* Whether an I2_S tensor's on-disk bytes are, as-is, a complete kernel-ready `BITNET_B1_58`
* buffer — payload immediately followed by its own trailing FP32 scale — so mapping
* `[absoluteDataOffset, absoluteDataOffset + nBytes + 4)` directly gives exactly what
* [resolveI2sScale] would have computed anyway (#1203).
*
* Only ever true for [I2sGgufLayout.SEQUENTIAL]: the payload itself is already in
* `BITNET_B1_58` order there, whereas `GROUP_128`/`GROUP_64` payloads still need permuting
* regardless of where the scale lives. False whenever a companion `<name>_scale` tensor exists
* — [resolveI2sScale]'s `SEQUENTIAL` order prefers it over a trailer — or the trailer bytes
* don't parse to a finite, nonzero float.
*/
internal fun i2sTrailerScaleIsMappable(
tensorInfo: StreamingTensorInfo,
tensors: List<StreamingTensorInfo>,
source: RandomAccessSource,
layout: I2sGgufLayout,
): Boolean =
layout == I2sGgufLayout.SEQUENTIAL &&
i2sCompanionScaleTensor(tensorInfo, tensors) == null &&
i2sTrailerScale(tensorInfo, source) != null
Original file line number Diff line number Diff line change
Expand Up @@ -317,7 +317,7 @@ public class StreamingGgufParametersLoader(
// tensor overrides them) -- otherwise fall through to i2sTensor's
// repack, unchanged.
GGMLQuantizationType.I2_S ->
if (i2sTrailerScaleIsMappable(tensorInfo, tensors, source)) {
if (i2sTrailerScaleIsMappable(tensorInfo, tensors, source, i2sLayout)) {
sk.ainet.lang.tensor.storage.TensorEncoding.BITNET_B1_58
} else {
null
Expand Down Expand Up @@ -420,7 +420,7 @@ public class StreamingGgufParametersLoader(

GGMLQuantizationType.I2_S -> i2sTensor(
ctx, dtype, shape, tensorInfo, rawBytes, tensorForm,
scale = resolveI2sScale(tensorInfo, tensors, reader, source),
scale = resolveI2sScale(tensorInfo, tensors, reader, source, i2sLayout),
)

else -> throw IllegalStateException(
Expand Down Expand Up @@ -540,85 +540,6 @@ public class StreamingGgufParametersLoader(
"${it.name}_scale" == tensorInfo.name
}

/**
* The per-tensor FP32 scale of an I2_S weight, from wherever its converter put it (#1140):
*
* - **BitNet.cpp** writes it as a trailer after the payload (a 32-byte-aligned region whose
* first 4 bytes are the LE FP32 scale; `w = (code − 1) · scale`). Read directly from the
* source at `absoluteDataOffset + payload` — [StreamingTensorInfo.nBytes] deliberately
* sizes the payload only.
* - **NeoGPU's converter** writes a companion `<name>_scale` F32 scalar, defined as "divide
* the projection output by it" — so the stored multiplier is its inverse.
* - Neither present (or unreadable/non-finite/zero): `1.0`, i.e. the raw codes. Loud in the
* trace via the repack conversion's byte counts, never a crash.
*
* The flavor decides which source is tried first; both are accepted either way, because a
* sequential file with a trailer or a group file with a companion costs nothing to honour.
*/
private fun resolveI2sScale(
tensorInfo: StreamingTensorInfo,
tensors: List<StreamingTensorInfo>,
reader: StreamingGGUFReader,
source: RandomAccessSource,
): Float {
fun companionInverse(): Float? {
val companion = i2sCompanionScaleTensor(tensorInfo, tensors) ?: return null
val value = runCatching { bytesToFloatArray(reader.loadTensorData(companion)).firstOrNull() }
.getOrNull() ?: return null
if (!value.isFinite() || value == 0f) return null
return 1f / value
}

return when (i2sLayout) {
I2sGgufLayout.GROUP_128, I2sGgufLayout.GROUP_64 -> i2sTrailerScale(tensorInfo, source) ?: companionInverse() ?: 1f
I2sGgufLayout.SEQUENTIAL -> companionInverse() ?: i2sTrailerScale(tensorInfo, source) ?: 1f
}
}

/** The `<name>_scale` companion tensor for an I2_S weight, if the converter wrote one (#1140). */
private fun i2sCompanionScaleTensor(
tensorInfo: StreamingTensorInfo,
tensors: List<StreamingTensorInfo>,
): StreamingTensorInfo? = tensors.firstOrNull {
it.tensorType == GGMLQuantizationType.F32 && it.name == "${tensorInfo.name}_scale"
}

/**
* The little-endian FP32 immediately after [tensorInfo]'s payload — BitNet.cpp's trailer
* convention — or `null` if unreadable, non-finite, or zero. [StreamingTensorInfo.nBytes]
* deliberately sizes the payload only, so `absoluteDataOffset + nBytes` is exactly where a
* trailer would start.
*/
private fun i2sTrailerScale(tensorInfo: StreamingTensorInfo, source: RandomAccessSource): Float? = runCatching {
val bytes = source.readAt(tensorInfo.absoluteDataOffset + tensorInfo.nBytes, 4)
val bits = (bytes[0].toInt() and 0xFF) or
((bytes[1].toInt() and 0xFF) shl 8) or
((bytes[2].toInt() and 0xFF) shl 16) or
((bytes[3].toInt() and 0xFF) shl 24)
Float.fromBits(bits)
}.getOrNull()?.takeIf { it.isFinite() && it != 0f }

/**
* Whether an I2_S tensor's on-disk bytes are, as-is, a complete kernel-ready `BITNET_B1_58`
* buffer — payload immediately followed by its own trailing FP32 scale — so mapping
* `[absoluteDataOffset, absoluteDataOffset + nBytes + 4)` directly gives exactly what
* [resolveI2sScale] would have computed anyway (#1203).
*
* Only ever true for [I2sGgufLayout.SEQUENTIAL]: the payload itself is already in
* `BITNET_B1_58` order there, whereas `GROUP_128`/`GROUP_64` payloads still need permuting
* regardless of where the scale lives. False whenever a companion `<name>_scale` tensor
* exists — [resolveI2sScale]'s `SEQUENTIAL` order prefers it over a trailer — or the trailer
* bytes don't parse to a finite, nonzero float.
*/
private fun i2sTrailerScaleIsMappable(
tensorInfo: StreamingTensorInfo,
tensors: List<StreamingTensorInfo>,
source: RandomAccessSource,
): Boolean =
i2sLayout == I2sGgufLayout.SEQUENTIAL &&
i2sCompanionScaleTensor(tensorInfo, tensors) == null &&
i2sTrailerScale(tensorInfo, source) != null

/**
* Materialize an I2_S tensor (#1140): repack the payload into the sequential `BITNET_B1_58`
* order under [i2sLayout] (code 3 fails fast in [I2sRepack]), fold [scale] into the trailer,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -125,7 +125,8 @@ public object GGUFWriter {
require(entry.shape.all { it > 0 }) {
"Tensor ${entry.ggufName} has non-positive dimensions ${entry.shape}"
}
require(GGML_QUANT_SIZES.containsKey(entry.quantization)) {
// A rawBytes entry declares its own size and needs no block-size metadata at all.
require(entry.rawBytes != null || GGML_QUANT_SIZES.containsKey(entry.quantization)) {
"Quantization ${entry.quantization} missing size metadata"
}
}
Expand Down Expand Up @@ -209,82 +210,88 @@ public object GGUFWriter {
}

private fun materializeTensor(entry: GgufTensorEntry, expectedSize: Int): ByteArray {
val bytes = when (entry.quantization) {
GGMLQuantizationType.F32 -> materializeF32(entry)
GGMLQuantizationType.F16 -> materializeF16(entry)
GGMLQuantizationType.BF16 -> materializeBF16(entry)
GGMLQuantizationType.F64 -> materializeF64(entry)
GGMLQuantizationType.I8 -> materializeI8(entry)
GGMLQuantizationType.I16 -> materializeI16(entry)
GGMLQuantizationType.I32 -> materializeI32(entry)
GGMLQuantizationType.I64 -> materializeI64(entry)
else -> materializeRaw(entry)
val bytes = entry.rawBytes ?: run {
val tensor = checkNotNull(entry.tensor) {
"GgufTensorEntry '${entry.ggufName}' has neither tensor nor rawBytes"
}
when (entry.quantization) {
GGMLQuantizationType.F32 -> materializeF32(tensor)
GGMLQuantizationType.F16 -> materializeF16(tensor)
GGMLQuantizationType.BF16 -> materializeBF16(tensor)
GGMLQuantizationType.F64 -> materializeF64(tensor)
GGMLQuantizationType.I8 -> materializeI8(tensor)
GGMLQuantizationType.I16 -> materializeI16(tensor)
GGMLQuantizationType.I32 -> materializeI32(tensor)
GGMLQuantizationType.I64 -> materializeI64(tensor)
else -> materializeRaw(tensor)
}
}
require(bytes.size == expectedSize) {
"Tensor ${entry.ggufName} size mismatch: expected $expectedSize, got ${bytes.size}"
}
return bytes
}

private fun materializeF32(entry: GgufTensorEntry): ByteArray {
val floatData = TensorFlatten.flattenFloats(entry.tensor)
private fun materializeF32(tensor: Tensor<*, *>): ByteArray {
val floatData = TensorFlatten.flattenFloats(tensor)
val out = ByteWriter()
floatData.forEach { out.writeFloat32(it) }
return out.toByteArray()
}

private fun materializeF16(entry: GgufTensorEntry): ByteArray {
val floatData = TensorFlatten.flattenFloats(entry.tensor)
private fun materializeF16(tensor: Tensor<*, *>): ByteArray {
val floatData = TensorFlatten.flattenFloats(tensor)
val out = ByteWriter()
floatData.forEach { out.writeUInt16(floatToHalfBits(it).toUShort()) }
return out.toByteArray()
}

private fun materializeBF16(entry: GgufTensorEntry): ByteArray {
val floatData = TensorFlatten.flattenFloats(entry.tensor)
private fun materializeBF16(tensor: Tensor<*, *>): ByteArray {
val floatData = TensorFlatten.flattenFloats(tensor)
val out = ByteWriter()
floatData.forEach { out.writeUInt16(bfloat16Bits(it).toUShort()) }
return out.toByteArray()
}

private fun materializeF64(entry: GgufTensorEntry): ByteArray {
val doubleData = TensorFlatten.flattenDoubles(entry.tensor)
private fun materializeF64(tensor: Tensor<*, *>): ByteArray {
val doubleData = TensorFlatten.flattenDoubles(tensor)
val out = ByteWriter()
doubleData.forEach { out.writeFloat64(it) }
return out.toByteArray()
}

private fun materializeI8(entry: GgufTensorEntry): ByteArray {
val bytes = TensorFlatten.flattenBytes(entry.tensor)
private fun materializeI8(tensor: Tensor<*, *>): ByteArray {
val bytes = TensorFlatten.flattenBytes(tensor)
return bytes
}

private fun materializeI16(entry: GgufTensorEntry): ByteArray {
val values = TensorFlatten.flattenShorts(entry.tensor)
private fun materializeI16(tensor: Tensor<*, *>): ByteArray {
val values = TensorFlatten.flattenShorts(tensor)
val out = ByteWriter()
values.forEach { out.writeUInt16(it.toUShort()) }
return out.toByteArray()
}

private fun materializeI32(entry: GgufTensorEntry): ByteArray {
val ints = TensorFlatten.flattenInts(entry.tensor)
private fun materializeI32(tensor: Tensor<*, *>): ByteArray {
val ints = TensorFlatten.flattenInts(tensor)
val out = ByteWriter()
ints.forEach { out.writeInt32(it) }
return out.toByteArray()
}

private fun materializeI64(entry: GgufTensorEntry): ByteArray {
val longs = TensorFlatten.flattenLongs(entry.tensor)
private fun materializeI64(tensor: Tensor<*, *>): ByteArray {
val longs = TensorFlatten.flattenLongs(tensor)
val out = ByteWriter()
longs.forEach { out.writeInt64(it) }
return out.toByteArray()
}

private fun materializeRaw(entry: GgufTensorEntry): ByteArray {
return TensorFlatten.flattenBytes(entry.tensor)
private fun materializeRaw(tensor: Tensor<*, *>): ByteArray {
return TensorFlatten.flattenBytes(tensor)
}

private fun expectedTensorSize(entry: GgufTensorEntry): Int {
entry.rawBytes?.let { return it.size }
val (blockSize, typeSize) = GGML_QUANT_SIZES[entry.quantization]
?: error("Quantization ${entry.quantization} missing size metadata")
val volume = entry.shape.fold(1L) { acc, d -> acc * max(1, d).toLong() }
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -36,13 +36,33 @@ public data class GgufExportOptions(
val provenance: Map<String, Any> = emptyMap()
)

/** Tensor entry to be consumed by a future GGUF writer implementation. */
/**
* Tensor entry to be consumed by [sk.ainet.io.gguf.export.GGUFWriter].
*
* Exactly one of [tensor] or [rawBytes] must be set. [tensor] is the original path: a SKaiNET
* tensor whose elements the writer encodes itself (dense float widths, or [rawBytes]-free raw
* passthrough via `TensorFlatten.flattenBytes`, which needs a real element-indexed `Tensor`).
*
* [rawBytes] (#1207) is for a converter that already has the exact on-disk bytes for this entry
* — a passthrough of a quantized blob this writer doesn't need to interpret, or a buffer whose
* true size exceeds its type's formal [sk.ainet.io.gguf.GGML_QUANT_SIZES] block math entirely
* (e.g. an I2_S `BITNET_B1_58` buffer with its trailing FP32 scale). When set, [tensor] is
* ignored, [TensorFlatten] is bypassed, and the entry's written size is exactly
* `rawBytes.size` — not derived from [quantization]/[shape] at all.
*/
public data class GgufTensorEntry(
val ggufName: String,
val tensor: Tensor<*, *>,
val tensor: Tensor<*, *>? = null,
val quantization: GGMLQuantizationType,
val shape: List<Int>
)
val shape: List<Int>,
val rawBytes: ByteArray? = null,
) {
init {
require((tensor == null) != (rawBytes == null)) {
"GgufTensorEntry '$ggufName' needs exactly one of tensor or rawBytes"
}
}
}

/** Aggregate export payload prepared by the facade; writer will consume this. */
public data class GgufWriteRequest(
Expand Down
Loading
Loading