Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,13 @@

### Fixed

- **GGUF tensors report their real encodings and sizes in `TensorStorage`.**
`StreamingGGUFReader` mapped only Q4_K/Q8_0 to dedicated `TensorEncoding`s; Q4_0, Q5_0,
Q5_1, Q5_K and Q6_K — all of which have encoding objects and CPU kernels — fell through to
`Opaque(name, 0)`, whose zero byte count short-circuits `TensorStorage.physicalBytes` and
silently corrupted every memory report and compression ratio. All seven quant formats now
map to their encodings, and genuinely unknown types carry the tensor's real byte count in
`Opaque`. (#928)
- **Memory-copy diagnostics attribute copies to their source.** `MemoryTracker.recordCopy`
discarded the `sourceName` every instrumented call site passes; reports now carry a
per-source breakdown (`copiesBySource: Map<String, CopySourceStat>`, included in the
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -119,7 +119,7 @@ public class StreamingGGUFReader private constructor(
return TensorStorage(
shape = shape,
logicalType = ggmlTypeToLogical(tensor.tensorType),
encoding = ggmlTypeToEncoding(tensor.tensorType),
encoding = ggmlTypeToEncoding(tensor.tensorType, tensor.nBytes),
buffer = BufferHandle.Borrowed(bytes, isMutable = false),
placement = Placement.CPU_HEAP
)
Expand Down Expand Up @@ -149,7 +149,7 @@ public class StreamingGGUFReader private constructor(
return TensorStorage(
shape = shape,
logicalType = ggmlTypeToLogical(tensor.tensorType),
encoding = ggmlTypeToEncoding(tensor.tensorType),
encoding = ggmlTypeToEncoding(tensor.tensorType, tensor.nBytes),
buffer = BufferHandle.FileBacked(
path = filePath,
fileOffset = tensor.absoluteDataOffset,
Expand All @@ -172,7 +172,7 @@ public class StreamingGGUFReader private constructor(
else -> LogicalDType.FLOAT32
}

private fun ggmlTypeToEncoding(type: GGMLQuantizationType): TensorEncoding = when (type) {
private fun ggmlTypeToEncoding(type: GGMLQuantizationType, nBytes: Long): TensorEncoding = when (type) {
GGMLQuantizationType.F32 -> TensorEncoding.Dense(4)
GGMLQuantizationType.F16 -> TensorEncoding.Dense(2)
GGMLQuantizationType.BF16 -> TensorEncoding.Dense(2)
Expand All @@ -181,17 +181,19 @@ public class StreamingGGUFReader private constructor(
GGMLQuantizationType.I16 -> TensorEncoding.Dense(2)
GGMLQuantizationType.I32 -> TensorEncoding.Dense(4)
GGMLQuantizationType.I64 -> TensorEncoding.Dense(8)
GGMLQuantizationType.Q4_K -> TensorEncoding.Q4_K
GGMLQuantizationType.Q4_0 -> TensorEncoding.Q4_0
GGMLQuantizationType.Q5_0 -> TensorEncoding.Q5_0
GGMLQuantizationType.Q5_1 -> TensorEncoding.Q5_1
GGMLQuantizationType.Q8_0 -> TensorEncoding.Q8_0
else -> {
// For other quantized types, use Opaque with raw byte count
val quantInfo = GGML_QUANT_SIZES[type]
if (quantInfo != null) {
TensorEncoding.Opaque(type.name, 0) // size computed from tensor info
} else {
TensorEncoding.Opaque(type.name, 0)
}
}
GGMLQuantizationType.Q4_K -> TensorEncoding.Q4_K
GGMLQuantizationType.Q5_K -> TensorEncoding.Q5_K
GGMLQuantizationType.Q6_K -> TensorEncoding.Q6_K
// Types without a dedicated TensorEncoding carry their real byte count,
// so TensorStorage.physicalBytes (which prefers encoding.physicalBytes
// over buffer.sizeInBytes) stays truthful. The previous Opaque(name, 0)
// made every such tensor report 0 physical bytes and silently corrupted
// memory reports and compression ratios.
else -> TensorEncoding.Opaque(type.name, nBytes)
}

// ========== Parsing Implementation ==========
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,125 @@
package sk.ainet.io.gguf

import org.junit.Test
import sk.ainet.io.JvmRandomAccessSource
import sk.ainet.lang.tensor.storage.TensorEncoding
import java.io.File
import java.io.RandomAccessFile
import java.nio.ByteBuffer
import java.nio.ByteOrder
import kotlin.test.assertEquals
import kotlin.test.assertTrue

/**
* Regression tests for [StreamingGGUFReader]'s GGML-type -> [TensorEncoding]
* mapping (#928): every quant format with a dedicated encoding must map to it,
* and formats without one must carry their real byte count in
* [TensorEncoding.Opaque] — the previous `Opaque(name, 0)` made
* `TensorStorage.physicalBytes` report 0 and silently corrupted memory
* reports and compression ratios.
*/
class StreamingGGUFReaderEncodingTest {

/** One 32-element block per tensor: Q4_0 = 18 bytes, Q8_1 = 40 bytes. */
private fun createGgufFile(): File {
val file = File.createTempFile("encoding_test_", ".gguf")
RandomAccessFile(file, "rw").use { raf ->
val buf = ByteBuffer.allocate(4096).order(ByteOrder.LITTLE_ENDIAN)

buf.putInt(0x46554747.toInt()) // Magic
buf.putInt(3) // Version
buf.putLong(2) // Tensor count
buf.putLong(1) // KV count

val key = "general.architecture".encodeToByteArray()
buf.putLong(key.size.toLong())
buf.put(key)
buf.putInt(GGUFValueType.STRING.value)
val value = "test".encodeToByteArray()
buf.putLong(value.size.toLong())
buf.put(value)

// Tensor 1: "w_q40", Q4_0, shape [32], offset 0 (18 bytes)
val name1 = "w_q40".encodeToByteArray()
buf.putLong(name1.size.toLong())
buf.put(name1)
buf.putInt(1)
buf.putLong(32)
buf.putInt(GGMLQuantizationType.Q4_0.value)
buf.putLong(0)

// Tensor 2: "w_q81", Q8_1 (no dedicated TensorEncoding), shape [32], offset 18 -> padded
val name2 = "w_q81".encodeToByteArray()
buf.putLong(name2.size.toLong())
buf.put(name2)
buf.putInt(1)
buf.putLong(32)
buf.putInt(GGMLQuantizationType.Q8_1.value)
buf.putLong(32) // relative offset, aligned

val padding = (32 - (buf.position() % 32)) % 32
repeat(padding) { buf.put(0) }

// Q4_0 data: 18 bytes + pad to 32 for the second tensor's alignment
repeat(18) { buf.put(1) }
repeat(14) { buf.put(0) }
// Q8_1 data: 40 bytes
repeat(40) { buf.put(2) }

buf.flip()
val bytes = ByteArray(buf.remaining())
buf.get(bytes)
raf.write(bytes)
}
return file
}

@Test
fun `quant formats with dedicated encodings map to them`() {
val file = createGgufFile()
try {
StreamingGGUFReader.open(JvmRandomAccessSource.open(file)).use { reader ->
val storage = reader.loadTensorStorage("w_q40")
assertEquals(TensorEncoding.Q4_0, storage.encoding)
assertEquals(18L, storage.physicalBytes)
assertEquals(128L, storage.logicalBytes) // 32 * 4 (logical FP32)
}
} finally {
file.delete()
}
}

@Test
fun `formats without dedicated encoding carry their real byte count in Opaque`() {
val file = createGgufFile()
try {
StreamingGGUFReader.open(JvmRandomAccessSource.open(file)).use { reader ->
val storage = reader.loadTensorStorage("w_q81")
val encoding = storage.encoding
assertTrue(encoding is TensorEncoding.Opaque, "Q8_1 should map to Opaque, got $encoding")
assertEquals("Q8_1", encoding.name)
assertEquals(40L, encoding.rawBytes) // 32 elems: 4+4+32 bytes per block
// The point of #928: physicalBytes must not report 0.
assertEquals(40L, storage.physicalBytes)
}
} finally {
file.delete()
}
}

@Test
fun `mapped storage reports the same encodings`() {
val file = createGgufFile()
try {
StreamingGGUFReader.open(JvmRandomAccessSource.open(file)).use { reader ->
val q40 = reader.loadTensorStorageMapped(
reader.tensors.first { it.name == "w_q40" }, file.absolutePath
)
assertEquals(TensorEncoding.Q4_0, q40.encoding)
assertEquals(18L, q40.physicalBytes)
}
} finally {
file.delete()
}
}
}
Loading