From 64ca0d4a2bb6318767b319ff75bfd03977c745a7 Mon Sep 17 00:00:00 2001 From: Michal Harakal Date: Mon, 10 Aug 2026 10:48:55 +0200 Subject: [PATCH] fix(io): map all seven GGUF quant formats to their TensorEncodings; real byte counts for Opaque MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit StreamingGGUFReader.ggmlTypeToEncoding mapped only Q4_K and Q8_0 to dedicated encodings. Q4_0, Q5_0, Q5_1, Q5_K and Q6_K — all of which have TensorEncoding objects with correct block sizes and CPU kernels — fell through to Opaque(name, 0), with a comment promising 'size computed from tensor info' that nothing ever computed. Since Opaque.physicalBytes returns rawBytes non-null, TensorStorage.physicalBytes never fell back to buffer.sizeInBytes: every such tensor reported 0 physical bytes, memory reports were wrong for five of seven supported formats, and compressionRatio silently returned 1.0. Map all seven quant formats to their encodings, and pass the tensor's real nBytes into the Opaque fallback for genuinely unknown types, so physicalBytes stays truthful either way. Both loadTensorStorage and loadTensorStorageMapped call sites updated. Tests: synthesized GGUF with a Q4_0 tensor (dedicated encoding, 18-byte block) and a Q8_1 tensor (no dedicated encoding -> Opaque with real 40-byte count), covering heap and mapped storage paths. Closes #928 --- CHANGELOG.md | 10 ++ .../sk/ainet/io/gguf/StreamingGGUFReader.kt | 28 ++-- .../gguf/StreamingGGUFReaderEncodingTest.kt | 125 ++++++++++++++++++ 3 files changed, 150 insertions(+), 13 deletions(-) create mode 100644 skainet-io/skainet-io-gguf/src/jvmTest/kotlin/sk/ainet/io/gguf/StreamingGGUFReaderEncodingTest.kt diff --git a/CHANGELOG.md b/CHANGELOG.md index fca59a3c2..c2a39b8b4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,16 @@ ## [Unreleased] +### Fixed + +- **GGUF tensors report their real encodings and sizes in `TensorStorage`.** + `StreamingGGUFReader` mapped only Q4_K/Q8_0 to dedicated `TensorEncoding`s; Q4_0, Q5_0, + Q5_1, Q5_K and Q6_K — all of which have encoding objects and CPU kernels — fell through to + `Opaque(name, 0)`, whose zero byte count short-circuits `TensorStorage.physicalBytes` and + silently corrupted every memory report and compression ratio. All seven quant formats now + map to their encodings, and genuinely unknown types carry the tensor's real byte count in + `Opaque`. (#928) + ## [0.38.0] - 2026-07-30 ### Added diff --git a/skainet-io/skainet-io-gguf/src/commonMain/kotlin/sk/ainet/io/gguf/StreamingGGUFReader.kt b/skainet-io/skainet-io-gguf/src/commonMain/kotlin/sk/ainet/io/gguf/StreamingGGUFReader.kt index df21211fe..4af9fd3d3 100644 --- a/skainet-io/skainet-io-gguf/src/commonMain/kotlin/sk/ainet/io/gguf/StreamingGGUFReader.kt +++ b/skainet-io/skainet-io-gguf/src/commonMain/kotlin/sk/ainet/io/gguf/StreamingGGUFReader.kt @@ -119,7 +119,7 @@ public class StreamingGGUFReader private constructor( return TensorStorage( shape = shape, logicalType = ggmlTypeToLogical(tensor.tensorType), - encoding = ggmlTypeToEncoding(tensor.tensorType), + encoding = ggmlTypeToEncoding(tensor.tensorType, tensor.nBytes), buffer = BufferHandle.Borrowed(bytes, isMutable = false), placement = Placement.CPU_HEAP ) @@ -149,7 +149,7 @@ public class StreamingGGUFReader private constructor( return TensorStorage( shape = shape, logicalType = ggmlTypeToLogical(tensor.tensorType), - encoding = ggmlTypeToEncoding(tensor.tensorType), + encoding = ggmlTypeToEncoding(tensor.tensorType, tensor.nBytes), buffer = BufferHandle.FileBacked( path = filePath, fileOffset = tensor.absoluteDataOffset, @@ -172,7 +172,7 @@ public class StreamingGGUFReader private constructor( else -> LogicalDType.FLOAT32 } - private fun ggmlTypeToEncoding(type: GGMLQuantizationType): TensorEncoding = when (type) { + private fun ggmlTypeToEncoding(type: GGMLQuantizationType, nBytes: Long): TensorEncoding = when (type) { GGMLQuantizationType.F32 -> TensorEncoding.Dense(4) GGMLQuantizationType.F16 -> TensorEncoding.Dense(2) GGMLQuantizationType.BF16 -> TensorEncoding.Dense(2) @@ -181,17 +181,19 @@ public class StreamingGGUFReader private constructor( GGMLQuantizationType.I16 -> TensorEncoding.Dense(2) GGMLQuantizationType.I32 -> TensorEncoding.Dense(4) GGMLQuantizationType.I64 -> TensorEncoding.Dense(8) - GGMLQuantizationType.Q4_K -> TensorEncoding.Q4_K + GGMLQuantizationType.Q4_0 -> TensorEncoding.Q4_0 + GGMLQuantizationType.Q5_0 -> TensorEncoding.Q5_0 + GGMLQuantizationType.Q5_1 -> TensorEncoding.Q5_1 GGMLQuantizationType.Q8_0 -> TensorEncoding.Q8_0 - else -> { - // For other quantized types, use Opaque with raw byte count - val quantInfo = GGML_QUANT_SIZES[type] - if (quantInfo != null) { - TensorEncoding.Opaque(type.name, 0) // size computed from tensor info - } else { - TensorEncoding.Opaque(type.name, 0) - } - } + GGMLQuantizationType.Q4_K -> TensorEncoding.Q4_K + GGMLQuantizationType.Q5_K -> TensorEncoding.Q5_K + GGMLQuantizationType.Q6_K -> TensorEncoding.Q6_K + // Types without a dedicated TensorEncoding carry their real byte count, + // so TensorStorage.physicalBytes (which prefers encoding.physicalBytes + // over buffer.sizeInBytes) stays truthful. The previous Opaque(name, 0) + // made every such tensor report 0 physical bytes and silently corrupted + // memory reports and compression ratios. + else -> TensorEncoding.Opaque(type.name, nBytes) } // ========== Parsing Implementation ========== diff --git a/skainet-io/skainet-io-gguf/src/jvmTest/kotlin/sk/ainet/io/gguf/StreamingGGUFReaderEncodingTest.kt b/skainet-io/skainet-io-gguf/src/jvmTest/kotlin/sk/ainet/io/gguf/StreamingGGUFReaderEncodingTest.kt new file mode 100644 index 000000000..75af1f16a --- /dev/null +++ b/skainet-io/skainet-io-gguf/src/jvmTest/kotlin/sk/ainet/io/gguf/StreamingGGUFReaderEncodingTest.kt @@ -0,0 +1,125 @@ +package sk.ainet.io.gguf + +import org.junit.Test +import sk.ainet.io.JvmRandomAccessSource +import sk.ainet.lang.tensor.storage.TensorEncoding +import java.io.File +import java.io.RandomAccessFile +import java.nio.ByteBuffer +import java.nio.ByteOrder +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +/** + * Regression tests for [StreamingGGUFReader]'s GGML-type -> [TensorEncoding] + * mapping (#928): every quant format with a dedicated encoding must map to it, + * and formats without one must carry their real byte count in + * [TensorEncoding.Opaque] — the previous `Opaque(name, 0)` made + * `TensorStorage.physicalBytes` report 0 and silently corrupted memory + * reports and compression ratios. + */ +class StreamingGGUFReaderEncodingTest { + + /** One 32-element block per tensor: Q4_0 = 18 bytes, Q8_1 = 40 bytes. */ + private fun createGgufFile(): File { + val file = File.createTempFile("encoding_test_", ".gguf") + RandomAccessFile(file, "rw").use { raf -> + val buf = ByteBuffer.allocate(4096).order(ByteOrder.LITTLE_ENDIAN) + + buf.putInt(0x46554747.toInt()) // Magic + buf.putInt(3) // Version + buf.putLong(2) // Tensor count + buf.putLong(1) // KV count + + val key = "general.architecture".encodeToByteArray() + buf.putLong(key.size.toLong()) + buf.put(key) + buf.putInt(GGUFValueType.STRING.value) + val value = "test".encodeToByteArray() + buf.putLong(value.size.toLong()) + buf.put(value) + + // Tensor 1: "w_q40", Q4_0, shape [32], offset 0 (18 bytes) + val name1 = "w_q40".encodeToByteArray() + buf.putLong(name1.size.toLong()) + buf.put(name1) + buf.putInt(1) + buf.putLong(32) + buf.putInt(GGMLQuantizationType.Q4_0.value) + buf.putLong(0) + + // Tensor 2: "w_q81", Q8_1 (no dedicated TensorEncoding), shape [32], offset 18 -> padded + val name2 = "w_q81".encodeToByteArray() + buf.putLong(name2.size.toLong()) + buf.put(name2) + buf.putInt(1) + buf.putLong(32) + buf.putInt(GGMLQuantizationType.Q8_1.value) + buf.putLong(32) // relative offset, aligned + + val padding = (32 - (buf.position() % 32)) % 32 + repeat(padding) { buf.put(0) } + + // Q4_0 data: 18 bytes + pad to 32 for the second tensor's alignment + repeat(18) { buf.put(1) } + repeat(14) { buf.put(0) } + // Q8_1 data: 40 bytes + repeat(40) { buf.put(2) } + + buf.flip() + val bytes = ByteArray(buf.remaining()) + buf.get(bytes) + raf.write(bytes) + } + return file + } + + @Test + fun `quant formats with dedicated encodings map to them`() { + val file = createGgufFile() + try { + StreamingGGUFReader.open(JvmRandomAccessSource.open(file)).use { reader -> + val storage = reader.loadTensorStorage("w_q40") + assertEquals(TensorEncoding.Q4_0, storage.encoding) + assertEquals(18L, storage.physicalBytes) + assertEquals(128L, storage.logicalBytes) // 32 * 4 (logical FP32) + } + } finally { + file.delete() + } + } + + @Test + fun `formats without dedicated encoding carry their real byte count in Opaque`() { + val file = createGgufFile() + try { + StreamingGGUFReader.open(JvmRandomAccessSource.open(file)).use { reader -> + val storage = reader.loadTensorStorage("w_q81") + val encoding = storage.encoding + assertTrue(encoding is TensorEncoding.Opaque, "Q8_1 should map to Opaque, got $encoding") + assertEquals("Q8_1", encoding.name) + assertEquals(40L, encoding.rawBytes) // 32 elems: 4+4+32 bytes per block + // The point of #928: physicalBytes must not report 0. + assertEquals(40L, storage.physicalBytes) + } + } finally { + file.delete() + } + } + + @Test + fun `mapped storage reports the same encodings`() { + val file = createGgufFile() + try { + StreamingGGUFReader.open(JvmRandomAccessSource.open(file)).use { reader -> + val q40 = reader.loadTensorStorageMapped( + reader.tensors.first { it.name == "w_q40" }, file.absolutePath + ) + assertEquals(TensorEncoding.Q4_0, q40.encoding) + assertEquals(18L, q40.physicalBytes) + } + } finally { + file.delete() + } + } +}