Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 21 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,27 @@

## [Unreleased]

### Fixed

- **GGUF `DEQUANTIZE_TO_FP32` no longer over-allocates**
([#782](https://github.com/SKaiNET-developers/SKaiNET/issues/782)): loading a 1.1B Q4_K_M
transiently needed >12 GB heap against a ~4.4 GB dense-FP32 floor. Three compounding causes,
all in `skainet-io-gguf`: (1) the legacy `GGUFReader` eagerly materialized **every** tensor
payload as a boxed `List<Any>` at parse time — measured at **41x** the payload size in
allocations (~26 GB for a 637 MB file); payloads are now constant-space lazy views that
decode elements on access (same `List<Any>` API, same contents). (2) every dense tensor paid
a full-size defensive `copyOf` in the tensor factory on top of the dequant intermediate —
`StreamingGgufParametersLoader` now wraps its loader-owned arrays zero-copy
(`ctx.wrapFloatArray`). (3) the K-quant kernels allocated per-block `copyOfRange` scratch —
they now index the source buffer directly, so a full-tensor dequant allocates exactly the
destination `FloatArray`. `StreamingGgufParametersLoader` also gains an optional
`quantPolicy` parameter: `DEQUANTIZE_TO_FP32` streams each quantized tensor block-by-block
straight into its destination array (peak transient per tensor = the packed source bytes;
measured: eager FP32 load of a synthetic multi-tensor model allocates 1.38x the FP32 total
vs 2.1-2.3x for the historical copy chain, with peak live ≈ 1.05x). The default
(`NATIVE_OPTIMIZED`) keeps the loader's historical packed-block behavior bit-for-bit; a
parity test pins the dequant path to the packed accessors bit-exactly across all seven
supported quant formats.
## [0.39.1] - 2026-08-11

Headline: **eager overhead off the JVM is gone.** The eager CPU ops gain
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -122,30 +122,75 @@ class GGUFReader(
function()
}

// Internal: materialize raw tensor payload based on ggml type
// Internal: materialize raw tensor payload based on ggml type.
//
// Payloads are returned as *lazy views* over the file buffer (#782): elements
// are decoded on access instead of being boxed into an eagerly-built list.
// The previous eager materialization stored one boxed object per element for
// every tensor in the file at parse time — for a 1.1B-parameter Q4_K_M GGUF
// that alone was >10 GB of transient heap. The returned lists have identical
// size, contents and element types; only the storage strategy changed.
private fun materializeTensorData(
ggmlType: GGMLQuantizationType,
dataOffs: Int,
nElems: Int,
nBytes: Int
): List<Any> {
return when (ggmlType) {
GGMLQuantizationType.F16 -> data.readDataByType<UShort>(dataOffs, nElems).let { halfs ->
if (decodeF16ToFloat) halfs.map { halfToFloat(it) } else halfs
}
GGMLQuantizationType.BF16 -> data.readDataByType<UShort>(dataOffs, nElems).let { bf16s ->
if (decodeBF16ToFloat) bf16s.map { bfloat16ToFloat(it) } else bf16s
GGMLQuantizationType.F16 ->
if (decodeF16ToFloat) LazyPayloadList(nElems) { halfToFloat(readUShortLE(dataOffs + it * 2)) }
else LazyPayloadList(nElems) { readUShortLE(dataOffs + it * 2) }
GGMLQuantizationType.BF16 ->
if (decodeBF16ToFloat) LazyPayloadList(nElems) { bfloat16ToFloat(readUShortLE(dataOffs + it * 2)) }
else LazyPayloadList(nElems) { readUShortLE(dataOffs + it * 2) }
GGMLQuantizationType.F32 -> LazyPayloadList(nElems) { Float.fromBits(readIntLE(dataOffs + it * 4)) }
GGMLQuantizationType.F64 -> LazyPayloadList(nElems) { Double.fromBits(readLongLE(dataOffs + it * 8)) }
GGMLQuantizationType.I8 -> LazyPayloadList(nElems) { data[dataOffs + it] }
GGMLQuantizationType.I16 -> LazyPayloadList(nElems) { readUShortLE(dataOffs + it * 2).toShort() }
GGMLQuantizationType.I32 -> LazyPayloadList(nElems) { readIntLE(dataOffs + it * 4) }
GGMLQuantizationType.I64 -> LazyPayloadList(nElems) { readLongLE(dataOffs + it * 8) }
else -> LazyPayloadList(nBytes) { data[dataOffs + it].toUByte() }
}
}

/**
* Constant-space `List<Any>` view over the file buffer: decodes one element
* per [get] call instead of storing boxed elements. Equality/hashCode follow
* the [AbstractList] contract, so it compares equal to an eagerly-built list
* with the same contents.
*/
private class LazyPayloadList(
override val size: Int,
private val element: (Int) -> Any,
) : AbstractList<Any>() {
override fun get(index: Int): Any {
if (index < 0 || index >= size) {
throw IndexOutOfBoundsException("index: $index, size: $size")
}
GGMLQuantizationType.F32 -> data.readDataByType<Float>(dataOffs, nElems)
GGMLQuantizationType.F64 -> data.readDataByType<Double>(dataOffs, nElems)
GGMLQuantizationType.I8 -> data.readDataByType<Byte>(dataOffs, nElems)
GGMLQuantizationType.I16 -> data.readDataByType<Short>(dataOffs, nElems)
GGMLQuantizationType.I32 -> data.readDataByType<Int>(dataOffs, nElems)
GGMLQuantizationType.I64 -> data.readDataByType<Long>(dataOffs, nElems)
else -> data.readDataByType<UByte>(dataOffs, nBytes)
return element(index)
}
}

private fun readUShortLE(offset: Int): UShort =
(((data[offset].toInt() and 0xFF)) or
((data[offset + 1].toInt() and 0xFF) shl 8)).toUShort()

private fun readIntLE(offset: Int): Int =
(data[offset].toInt() and 0xFF) or
((data[offset + 1].toInt() and 0xFF) shl 8) or
((data[offset + 2].toInt() and 0xFF) shl 16) or
((data[offset + 3].toInt() and 0xFF) shl 24)

private fun readLongLE(offset: Int): Long =
(data[offset].toLong() and 0xFF) or
((data[offset + 1].toLong() and 0xFF) shl 8) or
((data[offset + 2].toLong() and 0xFF) shl 16) or
((data[offset + 3].toLong() and 0xFF) shl 24) or
((data[offset + 4].toLong() and 0xFF) shl 32) or
((data[offset + 5].toLong() and 0xFF) shl 40) or
((data[offset + 6].toLong() and 0xFF) shl 48) or
((data[offset + 7].toLong() and 0xFF) shl 56)

private fun buildTensorInfoFields() {
// Build tensor info fields
val (newOffs, tensorFields) = buildTensorInfo(offs, tensorCount.toInt())
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,8 @@ package sk.ainet.io.gguf
import sk.ainet.context.ExecutionContext
import sk.ainet.io.ParametersLoader
import sk.ainet.io.RandomAccessSource
import sk.ainet.io.gguf.dequant.DequantOps
import sk.ainet.io.model.QuantPolicy
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.Tensor
import sk.ainet.lang.tensor.data.Bf16DenseTensorData
Expand Down Expand Up @@ -54,8 +56,30 @@ public class StreamingGgufParametersLoader(
* Keep `BF16` source tensors packed. Off by default — flip via `withPolicy(Require(BF16))`.
*/
private val keepBf16Native: Boolean = false,
/**
* How quantized tensors are materialized (#782).
*
* - [QuantPolicy.NATIVE_OPTIMIZED] (default — the loader's historical behavior):
* quantized tensors are delivered as packed block [TensorData]; F32/F16/BF16
* are dense FP32 (subject to [keepF16Native]/[keepBf16Native]).
* - [QuantPolicy.DEQUANTIZE_TO_FP32]: quantized tensors are dequantized
* *streaming, per tensor, block-by-block into the destination `FloatArray`*,
* which is then wrapped zero-copy. Peak transient memory per tensor is the
* packed source bytes only — there is no full-size intermediate copy.
* - [QuantPolicy.RAW_BYTES] is not supported by this loader (it preserves
* packed block storage instead) and is rejected eagerly.
*/
private val quantPolicy: QuantPolicy = QuantPolicy.NATIVE_OPTIMIZED,
) : ParametersLoader {

init {
require(quantPolicy != QuantPolicy.RAW_BYTES) {
"StreamingGgufParametersLoader does not support QuantPolicy.RAW_BYTES — quantized " +
"tensors are preserved as packed block TensorData (NATIVE_OPTIMIZED) or " +
"dequantized to dense FP32 (DEQUANTIZE_TO_FP32)."
}
}

@Suppress("UNCHECKED_CAST")
override suspend fun <T : DType, V> load(
ctx: ExecutionContext,
Expand All @@ -74,9 +98,10 @@ public class StreamingGgufParametersLoader(

val tensor: Tensor<T, V>? = when (tensorInfo.tensorType) {
GGMLQuantizationType.F32 -> {
val floats = bytesToFloatArray(rawBytes)
when (dtype) {
FP32::class -> ctx.fromFloatArray<T, Float>(shape, dtype, floats) as Tensor<T, V>
// The freshly decoded array is loader-owned — wrap it zero-copy
// instead of paying the factory's defensive copy (#782).
FP32::class -> ctx.wrapFloatArray<T, Float>(shape, dtype, bytesToFloatArray(rawBytes)) as Tensor<T, V>
else -> null
}
}
Expand All @@ -97,7 +122,8 @@ public class StreamingGgufParametersLoader(
val packed = Fp16DenseTensorData(shape, rawBytes)
ctx.fromData<T, V>(packed as sk.ainet.lang.tensor.data.TensorData<T, V>, dtype)
} else {
ctx.fromFloatArray<T, Float>(shape, dtype, dequantF16(rawBytes)) as Tensor<T, V>
// Loader-owned widened array — zero-copy wrap (#782).
ctx.wrapFloatArray<T, Float>(shape, dtype, dequantF16(rawBytes)) as Tensor<T, V>
}
else -> null
}
Expand All @@ -108,52 +134,19 @@ public class StreamingGgufParametersLoader(
val packed = Bf16DenseTensorData(shape, rawBytes)
ctx.fromData<T, V>(packed as sk.ainet.lang.tensor.data.TensorData<T, V>, dtype)
} else {
ctx.fromFloatArray<T, Float>(shape, dtype, dequantBF16(rawBytes)) as Tensor<T, V>
// Loader-owned widened array — zero-copy wrap (#782).
ctx.wrapFloatArray<T, Float>(shape, dtype, dequantBF16(rawBytes)) as Tensor<T, V>
}
else -> null
}

GGMLQuantizationType.Q4_K -> {
@Suppress("UNCHECKED_CAST")
val packed = Q4_KBlockTensorData.fromRawBytes(shape, rawBytes)
ctx.fromData<T, V>(packed as sk.ainet.lang.tensor.data.TensorData<T, V>, dtype)
}

GGMLQuantizationType.Q5_K -> {
@Suppress("UNCHECKED_CAST")
val packed = Q5_KBlockTensorData.fromRawBytes(shape, rawBytes)
ctx.fromData<T, V>(packed as sk.ainet.lang.tensor.data.TensorData<T, V>, dtype)
}

GGMLQuantizationType.Q6_K -> {
@Suppress("UNCHECKED_CAST")
val packed = Q6_KBlockTensorData.fromRawBytes(shape, rawBytes)
ctx.fromData<T, V>(packed as sk.ainet.lang.tensor.data.TensorData<T, V>, dtype)
}

GGMLQuantizationType.Q8_0 -> {
@Suppress("UNCHECKED_CAST")
val packed = Q8_0BlockTensorData.fromRawBytes(shape, rawBytes)
ctx.fromData<T, V>(packed as sk.ainet.lang.tensor.data.TensorData<T, V>, dtype)
}

GGMLQuantizationType.Q4_0 -> {
@Suppress("UNCHECKED_CAST")
val packed = Q4_0BlockTensorData.fromRawBytes(shape, rawBytes)
ctx.fromData<T, V>(packed as sk.ainet.lang.tensor.data.TensorData<T, V>, dtype)
}

GGMLQuantizationType.Q5_0 -> {
@Suppress("UNCHECKED_CAST")
val packed = Q5_0BlockTensorData.fromRawBytes(shape, rawBytes)
ctx.fromData<T, V>(packed as sk.ainet.lang.tensor.data.TensorData<T, V>, dtype)
}

GGMLQuantizationType.Q5_1 -> {
@Suppress("UNCHECKED_CAST")
val packed = Q5_1BlockTensorData.fromRawBytes(shape, rawBytes)
ctx.fromData<T, V>(packed as sk.ainet.lang.tensor.data.TensorData<T, V>, dtype)
}
GGMLQuantizationType.Q4_K,
GGMLQuantizationType.Q5_K,
GGMLQuantizationType.Q6_K,
GGMLQuantizationType.Q8_0,
GGMLQuantizationType.Q4_0,
GGMLQuantizationType.Q5_0,
GGMLQuantizationType.Q5_1 -> quantizedTensor(ctx, dtype, shape, tensorInfo, rawBytes)

else -> throw IllegalStateException(
"StreamingGgufParametersLoader: tensor '${tensorInfo.name}' of type " +
Expand All @@ -173,6 +166,47 @@ public class StreamingGgufParametersLoader(
}
}

/**
* Materialize a quantized tensor according to [quantPolicy].
*
* DEQUANTIZE_TO_FP32 (#782): the packed bytes are unpacked block-by-block
* straight into one destination `FloatArray` (the shared [DequantOps]
* kernels write each block into the single output array — no boxed values,
* no per-tensor intermediate), and the destination is wrapped zero-copy.
* Peak transient allocation per tensor is the packed source bytes.
*
* Any other policy (or a non-float destination dtype) preserves the packed
* block storage exactly as before.
*/
@Suppress("UNCHECKED_CAST")
private fun <T : DType, V> quantizedTensor(
ctx: ExecutionContext,
dtype: KClass<T>,
shape: Shape,
tensorInfo: StreamingTensorInfo,
rawBytes: ByteArray,
): Tensor<T, V> {
if (quantPolicy == QuantPolicy.DEQUANTIZE_TO_FP32 &&
(dtype == FP32::class || dtype == FP16::class)
) {
val dest = DequantOps.dequantFromBytes(rawBytes, tensorInfo.tensorType, tensorInfo.nElements.toInt())
return ctx.wrapFloatArray<T, Float>(shape, dtype, dest) as Tensor<T, V>
}
val packed = when (tensorInfo.tensorType) {
GGMLQuantizationType.Q4_K -> Q4_KBlockTensorData.fromRawBytes(shape, rawBytes)
GGMLQuantizationType.Q5_K -> Q5_KBlockTensorData.fromRawBytes(shape, rawBytes)
GGMLQuantizationType.Q6_K -> Q6_KBlockTensorData.fromRawBytes(shape, rawBytes)
GGMLQuantizationType.Q8_0 -> Q8_0BlockTensorData.fromRawBytes(shape, rawBytes)
GGMLQuantizationType.Q4_0 -> Q4_0BlockTensorData.fromRawBytes(shape, rawBytes)
GGMLQuantizationType.Q5_0 -> Q5_0BlockTensorData.fromRawBytes(shape, rawBytes)
GGMLQuantizationType.Q5_1 -> Q5_1BlockTensorData.fromRawBytes(shape, rawBytes)
else -> throw IllegalStateException(
"quantizedTensor called for non-quantized type ${tensorInfo.tensorType}"
)
}
return ctx.fromData(packed as sk.ainet.lang.tensor.data.TensorData<T, V>, dtype)
}

private fun bytesToFloatArray(bytes: ByteArray): FloatArray {
val count = bytes.size / 4
return FloatArray(count) { i ->
Expand Down
Loading
Loading