diff --git a/CHANGELOG.md b/CHANGELOG.md index 4523b860c..d62631748 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -30,6 +30,13 @@ ### Fixed +- **GGUF tensors report their real encodings and sizes in `TensorStorage`.** + `StreamingGGUFReader` mapped only Q4_K/Q8_0 to dedicated `TensorEncoding`s; Q4_0, Q5_0, + Q5_1, Q5_K and Q6_K — all of which have encoding objects and CPU kernels — fell through to + `Opaque(name, 0)`, whose zero byte count short-circuits `TensorStorage.physicalBytes` and + silently corrupted every memory report and compression ratio. All seven quant formats now + map to their encodings, and genuinely unknown types carry the tensor's real byte count in + `Opaque`. (#928) - **Memory-copy diagnostics attribute copies to their source.** `MemoryTracker.recordCopy` discarded the `sourceName` every instrumented call site passes; reports now carry a per-source breakdown (`copiesBySource: Map`, included in the diff --git a/skainet-io/skainet-io-gguf/src/commonMain/kotlin/sk/ainet/io/gguf/StreamingGGUFReader.kt b/skainet-io/skainet-io-gguf/src/commonMain/kotlin/sk/ainet/io/gguf/StreamingGGUFReader.kt index df21211fe..4af9fd3d3 100644 --- a/skainet-io/skainet-io-gguf/src/commonMain/kotlin/sk/ainet/io/gguf/StreamingGGUFReader.kt +++ b/skainet-io/skainet-io-gguf/src/commonMain/kotlin/sk/ainet/io/gguf/StreamingGGUFReader.kt @@ -119,7 +119,7 @@ public class StreamingGGUFReader private constructor( return TensorStorage( shape = shape, logicalType = ggmlTypeToLogical(tensor.tensorType), - encoding = ggmlTypeToEncoding(tensor.tensorType), + encoding = ggmlTypeToEncoding(tensor.tensorType, tensor.nBytes), buffer = BufferHandle.Borrowed(bytes, isMutable = false), placement = Placement.CPU_HEAP ) @@ -149,7 +149,7 @@ public class StreamingGGUFReader private constructor( return TensorStorage( shape = shape, logicalType = ggmlTypeToLogical(tensor.tensorType), - encoding = ggmlTypeToEncoding(tensor.tensorType), + encoding = ggmlTypeToEncoding(tensor.tensorType, tensor.nBytes), buffer = BufferHandle.FileBacked( path = filePath, fileOffset = tensor.absoluteDataOffset, @@ -172,7 +172,7 @@ public class StreamingGGUFReader private constructor( else -> LogicalDType.FLOAT32 } - private fun ggmlTypeToEncoding(type: GGMLQuantizationType): TensorEncoding = when (type) { + private fun ggmlTypeToEncoding(type: GGMLQuantizationType, nBytes: Long): TensorEncoding = when (type) { GGMLQuantizationType.F32 -> TensorEncoding.Dense(4) GGMLQuantizationType.F16 -> TensorEncoding.Dense(2) GGMLQuantizationType.BF16 -> TensorEncoding.Dense(2) @@ -181,17 +181,19 @@ public class StreamingGGUFReader private constructor( GGMLQuantizationType.I16 -> TensorEncoding.Dense(2) GGMLQuantizationType.I32 -> TensorEncoding.Dense(4) GGMLQuantizationType.I64 -> TensorEncoding.Dense(8) - GGMLQuantizationType.Q4_K -> TensorEncoding.Q4_K + GGMLQuantizationType.Q4_0 -> TensorEncoding.Q4_0 + GGMLQuantizationType.Q5_0 -> TensorEncoding.Q5_0 + GGMLQuantizationType.Q5_1 -> TensorEncoding.Q5_1 GGMLQuantizationType.Q8_0 -> TensorEncoding.Q8_0 - else -> { - // For other quantized types, use Opaque with raw byte count - val quantInfo = GGML_QUANT_SIZES[type] - if (quantInfo != null) { - TensorEncoding.Opaque(type.name, 0) // size computed from tensor info - } else { - TensorEncoding.Opaque(type.name, 0) - } - } + GGMLQuantizationType.Q4_K -> TensorEncoding.Q4_K + GGMLQuantizationType.Q5_K -> TensorEncoding.Q5_K + GGMLQuantizationType.Q6_K -> TensorEncoding.Q6_K + // Types without a dedicated TensorEncoding carry their real byte count, + // so TensorStorage.physicalBytes (which prefers encoding.physicalBytes + // over buffer.sizeInBytes) stays truthful. The previous Opaque(name, 0) + // made every such tensor report 0 physical bytes and silently corrupted + // memory reports and compression ratios. + else -> TensorEncoding.Opaque(type.name, nBytes) } // ========== Parsing Implementation ========== diff --git a/skainet-io/skainet-io-gguf/src/jvmTest/kotlin/sk/ainet/io/gguf/StreamingGGUFReaderEncodingTest.kt b/skainet-io/skainet-io-gguf/src/jvmTest/kotlin/sk/ainet/io/gguf/StreamingGGUFReaderEncodingTest.kt new file mode 100644 index 000000000..75af1f16a --- /dev/null +++ b/skainet-io/skainet-io-gguf/src/jvmTest/kotlin/sk/ainet/io/gguf/StreamingGGUFReaderEncodingTest.kt @@ -0,0 +1,125 @@ +package sk.ainet.io.gguf + +import org.junit.Test +import sk.ainet.io.JvmRandomAccessSource +import sk.ainet.lang.tensor.storage.TensorEncoding +import java.io.File +import java.io.RandomAccessFile +import java.nio.ByteBuffer +import java.nio.ByteOrder +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +/** + * Regression tests for [StreamingGGUFReader]'s GGML-type -> [TensorEncoding] + * mapping (#928): every quant format with a dedicated encoding must map to it, + * and formats without one must carry their real byte count in + * [TensorEncoding.Opaque] — the previous `Opaque(name, 0)` made + * `TensorStorage.physicalBytes` report 0 and silently corrupted memory + * reports and compression ratios. + */ +class StreamingGGUFReaderEncodingTest { + + /** One 32-element block per tensor: Q4_0 = 18 bytes, Q8_1 = 40 bytes. */ + private fun createGgufFile(): File { + val file = File.createTempFile("encoding_test_", ".gguf") + RandomAccessFile(file, "rw").use { raf -> + val buf = ByteBuffer.allocate(4096).order(ByteOrder.LITTLE_ENDIAN) + + buf.putInt(0x46554747.toInt()) // Magic + buf.putInt(3) // Version + buf.putLong(2) // Tensor count + buf.putLong(1) // KV count + + val key = "general.architecture".encodeToByteArray() + buf.putLong(key.size.toLong()) + buf.put(key) + buf.putInt(GGUFValueType.STRING.value) + val value = "test".encodeToByteArray() + buf.putLong(value.size.toLong()) + buf.put(value) + + // Tensor 1: "w_q40", Q4_0, shape [32], offset 0 (18 bytes) + val name1 = "w_q40".encodeToByteArray() + buf.putLong(name1.size.toLong()) + buf.put(name1) + buf.putInt(1) + buf.putLong(32) + buf.putInt(GGMLQuantizationType.Q4_0.value) + buf.putLong(0) + + // Tensor 2: "w_q81", Q8_1 (no dedicated TensorEncoding), shape [32], offset 18 -> padded + val name2 = "w_q81".encodeToByteArray() + buf.putLong(name2.size.toLong()) + buf.put(name2) + buf.putInt(1) + buf.putLong(32) + buf.putInt(GGMLQuantizationType.Q8_1.value) + buf.putLong(32) // relative offset, aligned + + val padding = (32 - (buf.position() % 32)) % 32 + repeat(padding) { buf.put(0) } + + // Q4_0 data: 18 bytes + pad to 32 for the second tensor's alignment + repeat(18) { buf.put(1) } + repeat(14) { buf.put(0) } + // Q8_1 data: 40 bytes + repeat(40) { buf.put(2) } + + buf.flip() + val bytes = ByteArray(buf.remaining()) + buf.get(bytes) + raf.write(bytes) + } + return file + } + + @Test + fun `quant formats with dedicated encodings map to them`() { + val file = createGgufFile() + try { + StreamingGGUFReader.open(JvmRandomAccessSource.open(file)).use { reader -> + val storage = reader.loadTensorStorage("w_q40") + assertEquals(TensorEncoding.Q4_0, storage.encoding) + assertEquals(18L, storage.physicalBytes) + assertEquals(128L, storage.logicalBytes) // 32 * 4 (logical FP32) + } + } finally { + file.delete() + } + } + + @Test + fun `formats without dedicated encoding carry their real byte count in Opaque`() { + val file = createGgufFile() + try { + StreamingGGUFReader.open(JvmRandomAccessSource.open(file)).use { reader -> + val storage = reader.loadTensorStorage("w_q81") + val encoding = storage.encoding + assertTrue(encoding is TensorEncoding.Opaque, "Q8_1 should map to Opaque, got $encoding") + assertEquals("Q8_1", encoding.name) + assertEquals(40L, encoding.rawBytes) // 32 elems: 4+4+32 bytes per block + // The point of #928: physicalBytes must not report 0. + assertEquals(40L, storage.physicalBytes) + } + } finally { + file.delete() + } + } + + @Test + fun `mapped storage reports the same encodings`() { + val file = createGgufFile() + try { + StreamingGGUFReader.open(JvmRandomAccessSource.open(file)).use { reader -> + val q40 = reader.loadTensorStorageMapped( + reader.tensors.first { it.name == "w_q40" }, file.absolutePath + ) + assertEquals(TensorEncoding.Q4_0, q40.encoding) + assertEquals(18L, q40.physicalBytes) + } + } finally { + file.delete() + } + } +}