Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -48,7 +48,18 @@ enum class GGMLQuantizationType(val value: Int) {
BF16(30),
TQ1_0(34),
TQ2_0(35),
// Note: types 31-33 and 36-38 have been removed in llama.cpp

/**
* BitNet.cpp's ternary type in the slot llama.cpp vacated (types 31-33 and
* 37-38 remain removed there). Not block-regular: the payload is 2-bit
* codes, 4 per byte, and the per-*tensor* FP32 scale sits in a trailer
* after the whole payload (BitNet.cpp) or in a companion `<name>_scale`
* tensor (NeoGPU's converter) — and the payload's bit order additionally
* depends on which converter wrote the file (see `I2sGgufLayout`). The
* loader normalizes all of it to `TensorEncoding.BITNET_B1_58` at load;
* I2_S never becomes a `TensorEncoding` of its own.
*/
I2_S(36),
MXFP4(39),

/**
Expand Down Expand Up @@ -113,6 +124,11 @@ val GGML_QUANT_SIZES: Map<GGMLQuantizationType, Pair<Int, Int>> = mapOf(
GGMLQuantizationType.BF16 to (1 to 2),
GGMLQuantizationType.TQ1_0 to (256 to 2 + 4 * 13),
GGMLQuantizationType.TQ2_0 to (256 to 2 + 64),
// I2_S: 4 codes per byte. This sizes the PAYLOAD only — the per-tensor
// scale trailer (BitNet.cpp writes 32 bytes after the payload; NeoGPU
// writes none) is deliberately outside the block math, read separately
// by the loader. See GGMLQuantizationType.I2_S.
GGMLQuantizationType.I2_S to (4 to 1),
// MXFP4: Microscaling FP4 format - 32 elements per block, 17 bytes (16 for data + 1 for scale)
GGMLQuantizationType.MXFP4 to (32 to 17)
)
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,96 @@
package sk.ainet.io.gguf

/**
* Which bit order an I2_S GGUF's payload is in — a property of the *converter that wrote the
* file*, not recoverable from the bytes, so the caller has to say (#1140).
*
* BitNet.cpp's `quantize_i2_s` packs 2-bit codes into fixed-size blocks, element `j` of a block
* going to byte `j % (QK/4)`, bit-pair `6 − 2·(j / (QK/4))` (high bits first) — and `QK_I2_S`
* is **128 when the file was quantized on x86 (AVX) and 64 on ARM (NEON)**, an architecture-
* dependent file format. NeoGPU's `convert_bitnet_to_gguf.py` packs the same codes sequentially,
* four consecutive elements per byte, low bit-pair first — byte-identical to SKaiNET's
* `BITNET_B1_58` payload. All three agree on the code mapping `{0,1,2} → {-1,0,+1}`.
*/
public enum class I2sGgufLayout(internal val blockElements: Int) {
/** BitNet.cpp file quantized with the x86/AVX pipeline (`QK_I2_S = 128`, 32-byte blocks). The common case for published GGUFs. */
GROUP_128(128),

/** BitNet.cpp file quantized with the ARM/NEON pipeline (`QK_I2_S = 64`, 16-byte blocks). */
GROUP_64(64),

/** NeoGPU's converter: sequential 4-per-byte, low bit-pair first — already the `BITNET_B1_58` payload order. */
SEQUENTIAL(4),
}

/**
* Repacks an I2_S payload into the sequential `BITNET_B1_58` payload order (#1140).
*
* Byte code 3 is **rejected here, at import** — it has no ternary meaning (the kernels decode it
* as +2 by LUT arithmetic, deliberately unvalidated), so a file that contains it is corrupt and
* the load fails fast instead of silently producing garbage logits.
*/
public object I2sRepack {

/**
* The sequential `BITNET_B1_58` payload of [elementCount] codes read from [bytes] under
* [layout]. For [I2sGgufLayout.SEQUENTIAL] the payload is validated and copied as-is.
*
* @throws IllegalArgumentException on byte code 3, or when [elementCount] does not fill
* [layout]'s blocks exactly
*/
public fun toSequentialPayload(bytes: ByteArray, elementCount: Int, layout: I2sGgufLayout): ByteArray {
require(elementCount % 4 == 0) { "I2_S element count must be a multiple of 4; got $elementCount" }
val payloadBytes = elementCount / 4
require(bytes.size >= payloadBytes) {
"I2_S payload needs $payloadBytes bytes for $elementCount elements; got ${bytes.size}"
}
if (layout == I2sGgufLayout.SEQUENTIAL) {
for (i in 0 until payloadBytes) {
val b = bytes[i].toInt() and 0xFF
for (lane in 0 until 4) {
val element = i * 4 + lane
if (element >= elementCount) break
requireValidCode((b shr (lane * 2)) and 3, element)
}
}
return bytes.copyOf(payloadBytes)
}
val qk = layout.blockElements
require(elementCount % qk == 0) {
"a ${layout.name} I2_S tensor must be a multiple of $qk elements; got $elementCount " +
"(the file may be the other BitNet.cpp flavor — try ${otherGroup(layout).name})"
}
val bytesPerBlock = qk / 4
val out = ByteArray(payloadBytes)
for (element in 0 until elementCount) {
val jb = element % qk
val src = bytes[(element / qk) * bytesPerBlock + jb % bytesPerBlock].toInt() and 0xFF
val code = (src shr (6 - 2 * (jb / bytesPerBlock))) and 3
requireValidCode(code, element)
val shift = (element % 4) * 2
out[element / 4] = (out[element / 4].toInt() or (code shl shift)).toByte()
}
return out
}

/** [payload] with [scale] appended as the little-endian FP32 trailer — a complete `BITNET_B1_58` buffer. */
public fun withScale(payload: ByteArray, scale: Float): ByteArray {
val out = payload.copyOf(payload.size + 4)
val bits = scale.toRawBits()
out[payload.size] = (bits and 0xFF).toByte()
out[payload.size + 1] = ((bits shr 8) and 0xFF).toByte()
out[payload.size + 2] = ((bits shr 16) and 0xFF).toByte()
out[payload.size + 3] = ((bits shr 24) and 0xFF).toByte()
return out
}

private fun requireValidCode(code: Int, element: Int) {
require(code != 3) {
"I2_S element $element holds byte code 3, which is not a ternary value — the file is " +
"corrupt, or it was read under the wrong I2sGgufLayout"
}
}

private fun otherGroup(layout: I2sGgufLayout): I2sGgufLayout =
if (layout == I2sGgufLayout.GROUP_128) I2sGgufLayout.GROUP_64 else I2sGgufLayout.GROUP_128
}
Original file line number Diff line number Diff line change
Expand Up @@ -103,6 +103,14 @@ public class StreamingGgufParametersLoader(
* Defaults to [NoopTraceSink]: nothing is recorded and nothing is allocated.
*/
private val traceSink: TraceSink = NoopTraceSink,
/**
* The bit order I2_S (type 36) payloads in this file are in — a property of the converter
* that wrote the file, not recoverable from the bytes (#1140, see [I2sGgufLayout]). Defaults
* to [I2sGgufLayout.GROUP_128], the BitNet.cpp x86 pipeline behind the commonly published
* GGUFs. Wrong-layout loads fail fast on code 3 where possible, but a misdeclared layout can
* also decode silently wrong — this knob is the caller's responsibility.
*/
private val i2sLayout: I2sGgufLayout = I2sGgufLayout.GROUP_128,
) : ParametersLoader {

/**
Expand Down Expand Up @@ -205,6 +213,14 @@ public class StreamingGgufParametersLoader(
var current = 0L

for (tensorInfo in tensors) {
// NeoGPU's I2_S converter emits a companion `<name>_scale` F32 scalar per ternary
// weight; it is consumed by the I2_S branch below (folded into the BITNET_B1_58
// trailer), never delivered as a parameter of its own.
if (isI2sCompanionScale(tensorInfo, tensors)) {
current += 1
onProgress(current, total, tensorInfo.name)
continue
}
val tensorForm = formFor(tensorInfo.name)
val shape = shapeOf(tensorInfo, tensorForm)
// A dense F32 tensor under MAPPED staging never reaches the heap: it is a view over
Expand Down Expand Up @@ -293,6 +309,11 @@ public class StreamingGgufParametersLoader(
GGMLQuantizationType.TQ1_0,
GGMLQuantizationType.TQ2_0 -> quantizedTensor(ctx, dtype, shape, tensorInfo, rawBytes, tensorForm)

GGMLQuantizationType.I2_S -> i2sTensor(
ctx, dtype, shape, tensorInfo, rawBytes, tensorForm,
scale = resolveI2sScale(tensorInfo, tensors, reader, source),
)

else -> throw IllegalStateException(
"StreamingGgufParametersLoader: tensor '${tensorInfo.name}' of type " +
"${tensorInfo.tensorType} passed the load-time pre-scan but has no load " +
Expand Down Expand Up @@ -387,6 +408,115 @@ public class StreamingGgufParametersLoader(
return ctx.fromData(delivered as sk.ainet.lang.tensor.data.TensorData<T, V>, dtype)
}

/**
* Whether [tensorInfo] is a NeoGPU-converter companion scale — an F32 scalar named
* `<weight>_scale` next to an I2_S tensor of that name. Consumed by [resolveI2sScale],
* skipped as a parameter.
*/
private fun isI2sCompanionScale(
tensorInfo: StreamingTensorInfo,
tensors: List<StreamingTensorInfo>,
): Boolean =
tensorInfo.tensorType == GGMLQuantizationType.F32 &&
tensorInfo.name.endsWith("_scale") &&
tensors.any {
it.tensorType == GGMLQuantizationType.I2_S &&
"${it.name}_scale" == tensorInfo.name
}

/**
* The per-tensor FP32 scale of an I2_S weight, from wherever its converter put it (#1140):
*
* - **BitNet.cpp** writes it as a trailer after the payload (a 32-byte-aligned region whose
* first 4 bytes are the LE FP32 scale; `w = (code − 1) · scale`). Read directly from the
* source at `absoluteDataOffset + payload` — [StreamingTensorInfo.nBytes] deliberately
* sizes the payload only.
* - **NeoGPU's converter** writes a companion `<name>_scale` F32 scalar, defined as "divide
* the projection output by it" — so the stored multiplier is its inverse.
* - Neither present (or unreadable/non-finite/zero): `1.0`, i.e. the raw codes. Loud in the
* trace via the repack conversion's byte counts, never a crash.
*
* The flavor decides which source is tried first; both are accepted either way, because a
* sequential file with a trailer or a group file with a companion costs nothing to honour.
*/
private fun resolveI2sScale(
tensorInfo: StreamingTensorInfo,
tensors: List<StreamingTensorInfo>,
reader: StreamingGGUFReader,
source: RandomAccessSource,
): Float {
fun trailer(): Float? = runCatching {
val bytes = source.readAt(tensorInfo.absoluteDataOffset + tensorInfo.nBytes, 4)
val bits = (bytes[0].toInt() and 0xFF) or
((bytes[1].toInt() and 0xFF) shl 8) or
((bytes[2].toInt() and 0xFF) shl 16) or
((bytes[3].toInt() and 0xFF) shl 24)
Float.fromBits(bits)
}.getOrNull()?.takeIf { it.isFinite() && it != 0f }

fun companionInverse(): Float? {
val companion = tensors.firstOrNull {
it.tensorType == GGMLQuantizationType.F32 && it.name == "${tensorInfo.name}_scale"
} ?: return null
val value = runCatching { bytesToFloatArray(reader.loadTensorData(companion)).firstOrNull() }
.getOrNull() ?: return null
if (!value.isFinite() || value == 0f) return null
return 1f / value
}

return when (i2sLayout) {
I2sGgufLayout.GROUP_128, I2sGgufLayout.GROUP_64 -> trailer() ?: companionInverse() ?: 1f
I2sGgufLayout.SEQUENTIAL -> companionInverse() ?: trailer() ?: 1f
}
}

/**
* Materialize an I2_S tensor (#1140): repack the payload into the sequential `BITNET_B1_58`
* order under [i2sLayout] (code 3 fails fast in [I2sRepack]), fold [scale] into the trailer,
* and keep it packed — 0.25 bytes per weight instead of the #1033 FP32 widening. A
* `DequantizeTo` form still gets dense FP32, decoded through the same codec the kernels are
* defined against.
*/
@Suppress("UNCHECKED_CAST")
private fun <T : DType, V> i2sTensor(
ctx: ExecutionContext,
dtype: KClass<T>,
shape: Shape,
tensorInfo: StreamingTensorInfo,
rawBytes: ByteArray,
tensorForm: WeightForm,
scale: Float,
): Tensor<T, V> {
require(dtype == FP32::class || dtype == FP16::class) {
"tensor '${tensorInfo.name}' is I2_S; ternary weights are logically FP32, so the " +
"requested dtype $dtype is not supported"
}
val payload = I2sRepack.toSequentialPayload(rawBytes, tensorInfo.nElements.toInt(), i2sLayout)
val packedBytes = I2sRepack.withScale(payload, scale)
traceConversion(
kind = "repack-i2s",
tensorName = tensorInfo.name,
from = ggufFormat(GGMLQuantizationType.I2_S, rawBytes.size.toLong()),
to = Format(FP32, sk.ainet.lang.tensor.storage.TensorEncoding.BITNET_B1_58),
bytesBefore = rawBytes.size.toLong(),
bytesAfter = packedBytes.size.toLong(),
)
if (tensorForm.encoding is EncodingRequest.DequantizeTo) {
val dest = sk.ainet.lang.memory.TernaryCodec.decodeBitNet(packedBytes, tensorInfo.nElements.toInt())
traceConversion(
kind = "widen-i2s",
tensorName = tensorInfo.name,
from = Format(FP32, sk.ainet.lang.tensor.storage.TensorEncoding.BITNET_B1_58),
to = Format.dense(FP32),
bytesBefore = packedBytes.size.toLong(),
bytesAfter = denseFp32Bytes(tensorInfo.nElements),
)
return ctx.wrapFloatArray<T, Float>(shape, dtype, dest) as Tensor<T, V>
}
val packed = sk.ainet.lang.tensor.data.BitNetB158TensorData(shape, packedBytes)
return ctx.fromData(packed as sk.ainet.lang.tensor.data.TensorData<T, V>, dtype)
}

/**
* [packed] with its blocks permuted into the order the packed matmul kernels read, and *saying
* so* (#1120).
Expand Down Expand Up @@ -489,6 +619,10 @@ public class StreamingGgufParametersLoader(
// cannot read the same bytes differently.
GGMLQuantizationType.TQ1_0,
GGMLQuantizationType.TQ2_0,
// #1140: BitNet.cpp / NeoGPU ternary. Repacked at load into the
// sequential BITNET_B1_58 layout and kept packed (0.25 B/weight)
// — the first ternary type that does NOT widen to FP32 (#1033).
GGMLQuantizationType.I2_S,
)

private const val MAX_LISTED_TENSORS = 8
Expand Down
Loading
Loading