diff --git a/CHANGELOG.md b/CHANGELOG.md index d545c1760..613b43e15 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,26 @@ ## [Unreleased] +### Added + +- **Off-heap / mmap tensor storage on Android** + ([#921](https://github.com/SKaiNET-developers/SKaiNET/issues/921), SKEEP-002/SKEEP-003 + slice): the java.nio memory-mapped storage that existed jvmMain-only is now shared + source between the JVM and Android compilations (`FileChannel.map` is API 1 — no JNI): + `MmapFloatTensorData`/`MmapTensorSource` (skainet-lang-core), `JvmMappedMemoryChunk`, + `MappedRandomAccessSource` and the `BufferHandle.FileBacked` resolver + `JvmFileBackedResolver` (skainet-io-core). New `MappedGgufWeights` (skainet-io-gguf, + JVM+Android) opens a GGUF once, maps it read-only, and serves dense F32 tensors as + zero-heap mapped views, any tensor as a `FileBacked` `TensorStorage` descriptor, and + packed payloads as heap bytes for the existing kernels. Weight bytes live in file-backed + pages the OS pages in/out — outside the hard ART heap cap that limited practical model + size on Android. Verified host-side (no device required): a 640 MB dense model loads and + reads through mapped views with **1.4 MB** of managed-heap allocation (0.0022x of the + dense size; per-thread allocation counters), and the Android compilation is exercised by + new `androidHostTest` suites (96 MB payload, ~240 KB used-heap growth). Files over 2 GB + are rejected fast (single-region mapping); windowed mapping is a follow-up under + SKEEP-003's IO pipeline improvement. + ### Fixed - **GGUF `DEQUANTIZE_TO_FP32` no longer over-allocates** diff --git a/build-logic/convention/src/main/kotlin/sk.ainet.dokka.gradle.kts b/build-logic/convention/src/main/kotlin/sk.ainet.dokka.gradle.kts index ffdeb01dd..e7f241035 100644 --- a/build-logic/convention/src/main/kotlin/sk.ainet.dokka.gradle.kts +++ b/build-logic/convention/src/main/kotlin/sk.ainet.dokka.gradle.kts @@ -28,6 +28,17 @@ extensions.configure { suppress.set(true) } + // Modules with a shared src/jvmAndroidMain source *directory* (#966) + // compile the same files into BOTH the jvm and android compilations. + // Dokka refuses files that belong to two source sets (dokka#3701), and + // the android pages would duplicate the jvm ones anyway — document the + // shared API once, via jvm, by suppressing the android source set. + if ((name == "androidMain" || name == "android") && + projectDir.resolve("src/jvmAndroidMain").exists() + ) { + suppress.set(true) + } + sourceLink { localDirectory.set(projectDir.resolve("src")) remoteUrl("https://github.com/SKaiNET-developers/skainet/tree/main/${project.path.replace(":", "/").removePrefix("/")}/src") diff --git a/skainet-io/skainet-io-core/build.gradle.kts b/skainet-io/skainet-io-core/build.gradle.kts index 70404d0b3..5c3b96787 100644 --- a/skainet-io/skainet-io-core/build.gradle.kts +++ b/skainet-io/skainet-io-core/build.gradle.kts @@ -90,6 +90,19 @@ kotlin { } } + // Memory-mapped IO (JvmMappedMemoryChunk, JvmFileBackedResolver, + // MappedRandomAccessSource), shared source between the JVM and Android + // compilations (#921): pure java.nio (FileChannel.map — API 1 on + // Android), so model weights can live in file-backed pages outside the + // ART heap. Shared directory rather than an intermediate source set — + // each compilation builds against its full platform classpath. + getByName("jvmMain") { + kotlin.srcDir("src/jvmAndroidMain/kotlin") + } + getByName("androidMain") { + kotlin.srcDir("src/jvmAndroidMain/kotlin") + } + val commonTest by getting { dependencies { implementation(libs.kotlin.test) diff --git a/skainet-io/skainet-io-core/src/androidHostTest/kotlin/sk/ainet/io/MappedMemoryChunkAndroidHostTest.kt b/skainet-io/skainet-io-core/src/androidHostTest/kotlin/sk/ainet/io/MappedMemoryChunkAndroidHostTest.kt new file mode 100644 index 000000000..422937db0 --- /dev/null +++ b/skainet-io/skainet-io-core/src/androidHostTest/kotlin/sk/ainet/io/MappedMemoryChunkAndroidHostTest.kt @@ -0,0 +1,75 @@ +package sk.ainet.io + +import sk.ainet.lang.tensor.storage.BufferHandle +import java.io.File +import kotlin.test.AfterTest +import kotlin.test.BeforeTest +import kotlin.test.Test +import kotlin.test.assertContentEquals +import kotlin.test.assertEquals + +/** + * Host-side test of the *Android compilation* of the shared mmap IO (#921): + * [JvmMappedMemoryChunk], [MappedRandomAccessSource] and + * [JvmFileBackedResolver] are java.nio-only (FileChannel.map is API 1) and + * are compiled into androidMain from the shared `jvmAndroidMain` source + * directory. This test compiles against the android variant and runs on the + * host JVM — no device/emulator required. + */ +class MappedMemoryChunkAndroidHostTest { + + private val payload = ByteArray(64 * 1024) { ((it * 31) and 0xFF).toByte() } + private lateinit var file: File + + @BeforeTest + fun setUp() { + file = File.createTempFile("android-mmap-chunk-", ".bin") + file.writeBytes(payload) + } + + @AfterTest + fun tearDown() { + file.delete() + } + + @Test + fun `mapped chunk reads, slices and offsets match the file`() { + JvmMappedMemoryChunk.open(file).use { chunk -> + assertEquals(payload.size.toLong(), chunk.size) + assertEquals(payload[0], chunk.readByte(0)) + assertEquals(payload[12345], chunk.readByte(12345)) + assertContentEquals(payload.copyOfRange(1000, 1256), chunk.readBytes(1000, 256)) + + val slice = chunk.slice(4096, 512) + assertContentEquals(payload.copyOfRange(4096, 4096 + 512), slice.readBytes(0, 512)) + } + } + + @Test + fun `mapped random access source serves positional reads`() { + MappedRandomAccessSource.open(file).use { source -> + assertEquals(payload.size.toLong(), source.size) + assertContentEquals(payload.copyOfRange(777, 777 + 64), source.readAt(777, 64)) + val buf = ByteArray(128) + assertEquals(128, source.readAt(2048, buf, 0, 128)) + assertContentEquals(payload.copyOfRange(2048, 2048 + 128), buf) + } + } + + @Test + fun `FileBacked handles resolve to mmap-backed accessors`() { + val handle = BufferHandle.FileBacked( + path = file.absolutePath, + fileOffset = 8192, + sizeInBytes = 1024, + ) + val accessor = JvmFileBackedResolver.resolveFileBacked(handle) + try { + assertEquals(1024, accessor.sizeInBytes) + assertEquals(payload[8192], accessor.readByte(0)) + assertContentEquals(payload.copyOfRange(8192, 8192 + 1024), accessor.readBytes(0, 1024)) + } finally { + accessor.close() + } + } +} diff --git a/skainet-io/skainet-io-core/src/jvmMain/kotlin/sk/ainet/io/JvmFileBackedResolver.kt b/skainet-io/skainet-io-core/src/jvmAndroidMain/kotlin/sk/ainet/io/JvmFileBackedResolver.kt similarity index 100% rename from skainet-io/skainet-io-core/src/jvmMain/kotlin/sk/ainet/io/JvmFileBackedResolver.kt rename to skainet-io/skainet-io-core/src/jvmAndroidMain/kotlin/sk/ainet/io/JvmFileBackedResolver.kt diff --git a/skainet-io/skainet-io-core/src/jvmMain/kotlin/sk/ainet/io/JvmMappedMemoryChunk.kt b/skainet-io/skainet-io-core/src/jvmAndroidMain/kotlin/sk/ainet/io/JvmMappedMemoryChunk.kt similarity index 100% rename from skainet-io/skainet-io-core/src/jvmMain/kotlin/sk/ainet/io/JvmMappedMemoryChunk.kt rename to skainet-io/skainet-io-core/src/jvmAndroidMain/kotlin/sk/ainet/io/JvmMappedMemoryChunk.kt diff --git a/skainet-io/skainet-io-core/src/jvmMain/kotlin/sk/ainet/io/MappedRandomAccessSource.kt b/skainet-io/skainet-io-core/src/jvmAndroidMain/kotlin/sk/ainet/io/MappedRandomAccessSource.kt similarity index 100% rename from skainet-io/skainet-io-core/src/jvmMain/kotlin/sk/ainet/io/MappedRandomAccessSource.kt rename to skainet-io/skainet-io-core/src/jvmAndroidMain/kotlin/sk/ainet/io/MappedRandomAccessSource.kt diff --git a/skainet-io/skainet-io-gguf/build.gradle.kts b/skainet-io/skainet-io-gguf/build.gradle.kts index 0d6b25956..0936f979c 100644 --- a/skainet-io/skainet-io-gguf/build.gradle.kts +++ b/skainet-io/skainet-io-gguf/build.gradle.kts @@ -25,6 +25,9 @@ kotlin { compilerOptions { jvmTarget.set(JvmTarget.JVM_1_8) } + // Host-side (JVM) unit tests for the Android compilation — exercises + // the mmap-backed weight path (MappedGgufWeights) without a device (#921). + withHostTest {} } iosArm64() @@ -63,6 +66,19 @@ kotlin { implementation(libs.kotlin.test) } } + + // Memory-mapped GGUF weight access (MappedGgufWeights), shared source + // between the JVM and Android compilations (#921): pure java.nio + // (FileChannel.map — API 1 on Android). Shared directory rather than an + // intermediate source set — each compilation builds against its full + // platform classpath. + getByName("jvmMain") { + kotlin.srcDir("src/jvmAndroidMain/kotlin") + } + getByName("androidMain") { + kotlin.srcDir("src/jvmAndroidMain/kotlin") + } + val jvmTest by getting { dependencies { implementation(libs.junit) diff --git a/skainet-io/skainet-io-gguf/src/androidHostTest/kotlin/sk/ainet/io/gguf/MappedGgufWeightsAndroidHostTest.kt b/skainet-io/skainet-io-gguf/src/androidHostTest/kotlin/sk/ainet/io/gguf/MappedGgufWeightsAndroidHostTest.kt new file mode 100644 index 000000000..fb452c54e --- /dev/null +++ b/skainet-io/skainet-io-gguf/src/androidHostTest/kotlin/sk/ainet/io/gguf/MappedGgufWeightsAndroidHostTest.kt @@ -0,0 +1,135 @@ +package sk.ainet.io.gguf + +import sk.ainet.io.JvmFileBackedResolver +import sk.ainet.lang.tensor.storage.BufferHandle +import sk.ainet.lang.types.FP32 +import java.io.File +import java.io.RandomAccessFile +import java.nio.ByteBuffer +import java.nio.ByteOrder +import kotlin.test.Test +import kotlin.test.assertContentEquals +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +/** + * Host-side test of the *Android compilation* of the mmap weight path (#921): + * this source set compiles against androidMain (android.jar nio API), so it + * proves `MappedGgufWeights`, `MmapFloatTensorData` and the shared + * `JvmFileBackedResolver` all build and behave on the Android variant — + * without a device or emulator (the precise allocation-counter budget gate + * lives in jvmTest's `MappedGgufHeapBudgetTest`, which exercises the + * byte-identical shared source). + * + * Includes a coarse heap check on a sparse 96 MB model: after load + reads, + * used-heap growth stays far under the payload size (lenient bound — host GC + * is not deterministic; the strict gate is the jvmTest allocation counter). + */ +class MappedGgufWeightsAndroidHostTest { + + private fun writeSparseF32Gguf(file: File, tensorName: String, elements: Int, sentinels: Map): File { + val head = ByteBuffer.allocate(4096).order(ByteOrder.LITTLE_ENDIAN) + head.putInt(0x46554747) + head.putInt(3) + head.putLong(1) + head.putLong(1) + val key = "general.architecture".encodeToByteArray() + head.putLong(key.size.toLong()) + head.put(key) + head.putInt(GGUFValueType.STRING.value) + val value = "test".encodeToByteArray() + head.putLong(value.size.toLong()) + head.put(value) + val nameBytes = tensorName.encodeToByteArray() + head.putLong(nameBytes.size.toLong()) + head.put(nameBytes) + head.putInt(1) + head.putLong(elements.toLong()) + head.putInt(GGMLQuantizationType.F32.value) + head.putLong(0) + val padding = (32 - (head.position() % 32)) % 32 + repeat(padding) { head.put(0) } + val dataStart = head.position().toLong() + + RandomAccessFile(file, "rw").use { raf -> + raf.write(head.array(), 0, head.position()) + raf.setLength(dataStart + elements.toLong() * 4) + for ((idx, v) in sentinels) { + raf.seek(dataStart + idx.toLong() * 4) + val bits = v.toRawBits() + raf.write( + byteArrayOf( + (bits and 0xFF).toByte(), + ((bits shr 8) and 0xFF).toByte(), + ((bits shr 16) and 0xFF).toByte(), + ((bits shr 24) and 0xFF).toByte(), + ) + ) + } + } + return file + } + + @Test + fun `android-compiled mapped weight path loads a 96 MB tensor without heap-sized allocation`() { + val elements = 24 * 1024 * 1024 // 96 MB dense F32 + val sentinels = mapOf(0 to 1.5f, 12_345_678 to -2.25f, elements - 1 to 3.125f) + val file = File.createTempFile("android_mmap_", ".gguf") + file.deleteOnExit() + writeSparseF32Gguf(file, "big.weight", elements, sentinels) + try { + val rt = Runtime.getRuntime() + System.gc() + val usedBefore = rt.totalMemory() - rt.freeMemory() + + var checksum = 0.0 + MappedGgufWeights.open(file.absolutePath).use { weights -> + val tensor = weights.mappedFloatTensor("big.weight") + assertEquals(elements, tensor.shape.volume) + var i = 0 + while (i < elements) { + checksum += tensor[i] + i += 262_144 + } + for ((idx, v) in sentinels) { + assertEquals(v, tensor[idx], "big.weight[$idx]") + } + + // FileBacked descriptor + the shared resolver, on the android variant. + val storage = weights.mappedStorage("big.weight") + assertTrue(storage.isFileBacked) + val accessor = JvmFileBackedResolver.resolveFileBacked(storage.buffer as BufferHandle.FileBacked) + try { + // First 4 sentinel bytes come back identical through the mmap accessor. + val viaAccessor = accessor.readBytes(0, 4) + val expectedBits = 1.5f.toRawBits() + assertContentEquals( + byteArrayOf( + (expectedBits and 0xFF).toByte(), + ((expectedBits shr 8) and 0xFF).toByte(), + ((expectedBits shr 16) and 0xFF).toByte(), + ((expectedBits shr 24) and 0xFF).toByte(), + ), + viaAccessor, + ) + } finally { + accessor.close() + } + } + + System.gc() + val usedAfter = rt.totalMemory() - rt.freeMemory() + val growth = usedAfter - usedBefore + println("android host mapped load: payload=96 MB, used-heap growth=${growth / 1024} KB, checksum=$checksum") + // Lenient (GC nondeterminism): far below the 96 MB payload — a heap + // materialization would add >= 96 MB here. + assertTrue( + growth < 48L * 1024 * 1024, + "used heap grew by ${growth / (1024 * 1024)} MB for a 96 MB mapped payload — " + + "weights are being materialized on the managed heap (#921)", + ) + } finally { + file.delete() + } + } +} diff --git a/skainet-io/skainet-io-gguf/src/jvmAndroidMain/kotlin/sk/ainet/io/gguf/MappedGgufWeights.kt b/skainet-io/skainet-io-gguf/src/jvmAndroidMain/kotlin/sk/ainet/io/gguf/MappedGgufWeights.kt new file mode 100644 index 000000000..fa508f831 --- /dev/null +++ b/skainet-io/skainet-io-gguf/src/jvmAndroidMain/kotlin/sk/ainet/io/gguf/MappedGgufWeights.kt @@ -0,0 +1,135 @@ +package sk.ainet.io.gguf + +import sk.ainet.lang.tensor.Shape +import sk.ainet.lang.tensor.data.MmapFloatTensorData +import sk.ainet.lang.tensor.data.MmapTensorSource +import sk.ainet.lang.tensor.storage.TensorStorage +import sk.ainet.lang.types.DType +import java.io.RandomAccessFile +import java.nio.channels.FileChannel + +/** + * Memory-mapped GGUF weight access for the JVM and Android (#921). + * + * On Android every heap array counts against the hard ART cap (256 MB + * default, 512 MB with `largeHeap`), which limits practical model size no + * matter how good the kernels are. This helper keeps weight bytes in + * *file-backed mapped pages* instead: the file is mapped once with + * [FileChannel.map] (available since API 1 — no JNI), the OS pages tensor + * data in on demand and evicts it under memory pressure, and dense F32 + * tensors are exposed as zero-copy [MmapFloatTensorData] views whose bytes + * never touch the managed heap. + * + * Per-tensor access: + * - [mappedFloatTensor] — dense F32 tensors as zero-heap mapped views + * (wrap into a `Tensor` with `ctx.fromData`). + * - [mappedStorage] — any tensor as a [TensorStorage] descriptor with + * [sk.ainet.lang.tensor.storage.BufferHandle.FileBacked] bytes; resolve + * with `JvmFileBackedResolver.createResolver()` (shared with Android) or + * materialize via `copyMaterialize(resolver)`. + * - [packedBytes] — the raw packed payload on the heap, for quantized + * tensors that existing kernels consume as `ByteArray` (their transient + * load cost is already streamed per tensor, see #782). + * + * The whole file is mapped in one region, so files larger than 2 GB are + * rejected ([open] fails fast); windowed mapping for >2 GB files is a + * follow-up under SKEEP-003's IO pipeline improvement. + * + * Not thread-safe for concurrent [close]; tensor views stay valid only + * while this object is open. + */ +public class MappedGgufWeights private constructor( + public val filePath: String, + private val reader: StreamingGGUFReader, + private val raf: RandomAccessFile, + private val mmap: MmapTensorSource, +) : AutoCloseable { + + /** Tensor directory of the file (metadata only — no payload on heap). */ + public val tensors: List get() = reader.tensors + + /** Parsed GGUF metadata key/value fields. */ + public val fields: Map get() = reader.fields + + /** Look up a tensor's metadata by name. */ + public fun info(name: String): StreamingTensorInfo = + reader.tensors.firstOrNull { it.name == name } + ?: throw IllegalArgumentException( + "Tensor not found: $name (file has ${reader.tensors.size} tensors)" + ) + + /** + * A dense F32 tensor as a zero-copy view over the mapped file region. + * The returned data reads directly from file-backed pages — nothing is + * allocated on the managed heap beyond the small view object. + * + * @throws IllegalArgumentException if the tensor is not F32; use + * [packedBytes] (quantized) or [mappedStorage] (descriptor) instead. + */ + public fun mappedFloatTensor(name: String): MmapFloatTensorData { + val t = info(name) + require(t.tensorType == GGMLQuantizationType.F32) { + "Tensor '$name' is ${t.tensorType}, not F32 — mappedFloatTensor serves dense " + + "float tensors only. Use packedBytes() for quantized payloads or " + + "mappedStorage() for a FileBacked descriptor." + } + val shape = Shape(*t.shape.map { it.toInt() }.toIntArray()) + return mmap.floatTensorAt(t.absoluteDataOffset, shape) + } + + /** + * Any tensor as a [TensorStorage] descriptor whose buffer is + * [sk.ainet.lang.tensor.storage.BufferHandle.FileBacked] — the + * placement-aware entry point (see SKEEP-003). Bytes are read only when + * the handle is resolved. + */ + public fun mappedStorage(name: String): TensorStorage = + reader.loadTensorStorageMapped(info(name), filePath) + + /** + * The raw packed payload of a tensor as a heap `ByteArray` — for + * quantized tensors whose kernels consume packed byte arrays. + */ + public fun packedBytes(name: String): ByteArray = + reader.loadTensorData(info(name)) + + override fun close() { + try { + reader.close() + } finally { + try { + mmap.close() + } finally { + raf.close() + } + } + } + + public companion object { + + /** + * Open a GGUF file for memory-mapped weight access. + * + * Parses the metadata through a positional-read source (heap cost is + * O(metadata)), then maps the whole file read-only. + */ + public fun open(filePath: String): MappedGgufWeights { + val source = createRandomAccessSource(filePath) + ?: throw IllegalArgumentException("Cannot open for random access: $filePath") + val reader = StreamingGGUFReader.open(source) + val raf = RandomAccessFile(filePath, "r") + try { + require(raf.length() <= Int.MAX_VALUE) { + "File is ${raf.length()} bytes (> 2 GB) — single-region mapping uses " + + "int offsets. Windowed mapping is a follow-up (SKEEP-003, improvement 4)." + } + val mmap = MmapTensorSource.fromChannel(raf.channel) + return MappedGgufWeights(filePath, reader, raf, mmap) + } catch (t: Throwable) { + raf.close() + reader.close() + throw t + } + } + } +} diff --git a/skainet-io/skainet-io-gguf/src/jvmTest/kotlin/sk/ainet/io/gguf/MappedGgufHeapBudgetTest.kt b/skainet-io/skainet-io-gguf/src/jvmTest/kotlin/sk/ainet/io/gguf/MappedGgufHeapBudgetTest.kt new file mode 100644 index 000000000..0b13482ee --- /dev/null +++ b/skainet-io/skainet-io-gguf/src/jvmTest/kotlin/sk/ainet/io/gguf/MappedGgufHeapBudgetTest.kt @@ -0,0 +1,163 @@ +package sk.ainet.io.gguf + +import sk.ainet.lang.types.FP32 +import java.io.File +import java.io.RandomAccessFile +import java.lang.management.ManagementFactory +import java.nio.ByteBuffer +import java.nio.ByteOrder +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +/** + * The #921 verification gate, host-simulated: a model whose dense weights are + * **larger than the 512 MB ART large-heap cap** must load and be readable + * through [MappedGgufWeights] with only O(metadata) managed-heap allocation — + * the weight bytes live in file-backed mapped pages, exactly as they would on + * an Android device (`FileChannel.map` is API 1; the identical shared source + * compiles into the androidMain variant, exercised by + * `MappedGgufWeightsAndroidHostTest`). + * + * Measured with the JVM's per-thread allocation counter — deterministic, + * unlike sampling heap peaks around GC. No device/emulator is involved; this + * is the JVM-simulated harness variant of the tracker's verification row. + * + * The file is written *sparsely* (header + a few sentinel floats + a + * `setLength` tail), so the test creates a 640 MB model in milliseconds while + * the mapped reads still go through real file pages. + */ +class MappedGgufHeapBudgetTest { + + private val threadMx = ManagementFactory.getThreadMXBean() as com.sun.management.ThreadMXBean + + private fun allocatedBytes(): Long = threadMx.getThreadAllocatedBytes(Thread.currentThread().id) + + /** name -> (elements, sentinel flat indices) */ + private val model = listOf( + Triple("blk0.weight", 64 * 1024 * 1024, intArrayOf(0, 1_000_000, 64 * 1024 * 1024 - 1)), + Triple("blk1.weight", 64 * 1024 * 1024, intArrayOf(7, 33_554_431)), + Triple("output.weight", 32 * 1024 * 1024, intArrayOf(12_345, 32 * 1024 * 1024 - 1)), + ) + + private fun sentinelValue(tensor: String, flatIndex: Int): Float = + (tensor.hashCode() xor flatIndex) * 1e-3f + + /** Sparse GGUF: real header, sentinel floats patched, zero tail via setLength. */ + private fun writeSparseModel(): File { + val file = File.createTempFile("sparse_640mb_", ".gguf") + file.deleteOnExit() + + val head = ByteBuffer.allocate(16 * 1024).order(ByteOrder.LITTLE_ENDIAN) + head.putInt(0x46554747) + head.putInt(3) + head.putLong(model.size.toLong()) + head.putLong(1) + val key = "general.architecture".encodeToByteArray() + head.putLong(key.size.toLong()) + head.put(key) + head.putInt(GGUFValueType.STRING.value) + val value = "test".encodeToByteArray() + head.putLong(value.size.toLong()) + head.put(value) + var rel = 0L + for ((name, elements, _) in model) { + val nameBytes = name.encodeToByteArray() + head.putLong(nameBytes.size.toLong()) + head.put(nameBytes) + head.putInt(1) + head.putLong(elements.toLong()) + head.putInt(GGMLQuantizationType.F32.value) + head.putLong(rel) + rel += elements.toLong() * 4 + } + val padding = (32 - (head.position() % 32)) % 32 + repeat(padding) { head.put(0) } + val dataStart = head.position().toLong() + + RandomAccessFile(file, "rw").use { raf -> + raf.write(head.array(), 0, head.position()) + raf.setLength(dataStart + rel) // sparse zero payload + // Patch sentinel floats at known flat indices. + var tensorBase = dataStart + for ((name, elements, sentinels) in model) { + for (idx in sentinels) { + raf.seek(tensorBase + idx.toLong() * 4) + val bits = sentinelValue(name, idx).toRawBits() + raf.write( + byteArrayOf( + (bits and 0xFF).toByte(), + ((bits shr 8) and 0xFF).toByte(), + ((bits shr 16) and 0xFF).toByte(), + ((bits shr 24) and 0xFF).toByte(), + ) + ) + } + tensorBase += elements.toLong() * 4 + } + } + return file + } + + @Test + fun `640 MB dense model loads and reads within an O(metadata) heap budget`() { + val file = writeSparseModel() + val totalDenseBytes = model.sumOf { it.second.toLong() * 4 } + assertTrue(totalDenseBytes > 512L * 1024 * 1024, "model must exceed the 512 MB ART budget") + try { + // Warm-up on a tiny file: classloading and JIT. + MappedGgufWeightsTest.writeGguf( + listOf( + MappedGgufWeightsTest.Companion.GgufTestTensor( + "w", GGMLQuantizationType.F32, 8, + MappedGgufWeightsTest.f32ToBytes(FloatArray(8) { it.toFloat() }), + ) + ) + ).let { warm -> + MappedGgufWeights.open(warm.absolutePath).use { it.mappedFloatTensor("w")[3] } + warm.delete() + } + + val before = allocatedBytes() + + var checksum = 0.0 + MappedGgufWeights.open(file.absolutePath).use { weights -> + for ((name, elements, sentinels) in model) { + val tensor = weights.mappedFloatTensor(name) + assertEquals(elements, tensor.shape.volume, name) + // Sample-read across the whole tensor (touches mapped pages, + // allocates nothing on the heap) … + var i = 0 + while (i < elements) { + checksum += tensor[i] + i += 1_048_576 + } + // … and verify the sentinels round-trip through the mapping. + for (idx in sentinels) { + assertEquals(sentinelValue(name, idx), tensor[idx], "$name[$idx]") + } + } + } + + val allocated = allocatedBytes() - before + println( + "mapped GGUF load: dense=${totalDenseBytes / (1024 * 1024)} MB, " + + "heap allocated=${allocated / 1024} KB " + + "(${"%.4f".format(allocated / totalDenseBytes.toDouble())}x of dense size), checksum=$checksum", + ) + + // O(metadata): parsing + view objects. 8 MB is ~1.2% of the dense + // size and orders of magnitude under the 512 MB ART budget; a heap + // materialization of even one tensor (256 MB) trips this instantly. + val budget = 8L * 1024 * 1024 + assertTrue( + allocated in 0..budget, + "loading a ${totalDenseBytes / (1024 * 1024)} MB dense model allocated " + + "${allocated / (1024 * 1024)} MB on the managed heap — weight bytes are " + + "no longer staying in mapped pages (#921)", + ) + } finally { + file.delete() + } + } +} diff --git a/skainet-io/skainet-io-gguf/src/jvmTest/kotlin/sk/ainet/io/gguf/MappedGgufWeightsTest.kt b/skainet-io/skainet-io-gguf/src/jvmTest/kotlin/sk/ainet/io/gguf/MappedGgufWeightsTest.kt new file mode 100644 index 000000000..709861aa7 --- /dev/null +++ b/skainet-io/skainet-io-gguf/src/jvmTest/kotlin/sk/ainet/io/gguf/MappedGgufWeightsTest.kt @@ -0,0 +1,128 @@ +package sk.ainet.io.gguf + +import sk.ainet.io.JvmFileBackedResolver +import sk.ainet.lang.tensor.storage.BufferHandle +import sk.ainet.lang.tensor.storage.Placement +import sk.ainet.lang.types.FP32 +import java.io.File +import java.io.RandomAccessFile +import java.nio.ByteBuffer +import java.nio.ByteOrder +import kotlin.test.Test +import kotlin.test.assertContentEquals +import kotlin.test.assertEquals +import kotlin.test.assertFailsWith +import kotlin.test.assertTrue + +/** + * Functional coverage for [MappedGgufWeights] (#921): mapped F32 views, + * FileBacked descriptors resolved through the (JVM+Android shared) + * [JvmFileBackedResolver], packed-byte fallback, and the fail-fast guards. + */ +class MappedGgufWeightsTest { + + @Test + fun `mapped F32 view, FileBacked storage and packed bytes agree with the file`() { + val f32Values = FloatArray(1024) { (it - 512) * 0.25f } + val q80Payload = ByteArray(34 * 4) { (it * 7 + 3).toByte() } // 4 blocks of Q8_0 + val file = writeGguf( + listOf( + GgufTestTensor("dense.f32", GGMLQuantizationType.F32, 1024, f32ToBytes(f32Values)), + GgufTestTensor("packed.q80", GGMLQuantizationType.Q8_0, 128, q80Payload), + ) + ) + try { + MappedGgufWeights.open(file.absolutePath).use { weights -> + assertEquals(2, weights.tensors.size) + + // Zero-heap mapped view: values read straight from mapped pages. + val dense = weights.mappedFloatTensor("dense.f32") + assertEquals(1024, dense.shape.volume) + for (i in intArrayOf(0, 1, 511, 512, 1023)) { + assertEquals(f32Values[i], dense[i], "flat index $i") + } + + // FileBacked descriptor + shared resolver: bytes match the payload. + val storage = weights.mappedStorage("packed.q80") + assertTrue(storage.isFileBacked, "storage should be FileBacked") + assertEquals(Placement.MMAP_WEIGHTS, storage.placement) + val handle = storage.buffer as BufferHandle.FileBacked + val accessor = JvmFileBackedResolver.resolveFileBacked(handle) + try { + assertContentEquals(q80Payload, accessor.readBytes(0, q80Payload.size)) + } finally { + accessor.close() + } + + // copyMaterialize with the resolver turns FileBacked into Owned bytes. + val materialized = storage.copyMaterialize(JvmFileBackedResolver.createResolver()) + val owned = materialized.buffer as BufferHandle.Owned + assertContentEquals(q80Payload, owned.data) + + // Heap fallback for packed kernels. + assertContentEquals(q80Payload, weights.packedBytes("packed.q80")) + + // Guards. + assertFailsWith { weights.mappedFloatTensor("packed.q80") } + assertFailsWith { weights.info("missing") } + } + } finally { + file.delete() + } + } + + companion object { + fun f32ToBytes(values: FloatArray): ByteArray { + val buf = ByteBuffer.allocate(values.size * 4).order(ByteOrder.LITTLE_ENDIAN) + values.forEach { buf.putFloat(it) } + return buf.array() + } + + data class GgufTestTensor( + val name: String, + val type: GGMLQuantizationType, + val elementCount: Long, + val data: ByteArray, + ) + + /** Write a minimal GGUF v3 file (32-byte aligned data section). */ + fun writeGguf(tensors: List, file: File = File.createTempFile("mapped_gguf_", ".gguf")): File { + file.deleteOnExit() + val head = ByteBuffer.allocate(16 * 1024).order(ByteOrder.LITTLE_ENDIAN) + head.putInt(0x46554747) // "GGUF" + head.putInt(3) + head.putLong(tensors.size.toLong()) + head.putLong(1) + val key = "general.architecture".encodeToByteArray() + head.putLong(key.size.toLong()) + head.put(key) + head.putInt(GGUFValueType.STRING.value) + val value = "test".encodeToByteArray() + head.putLong(value.size.toLong()) + head.put(value) + var dataOffset = 0L + for (t in tensors) { + val name = t.name.encodeToByteArray() + head.putLong(name.size.toLong()) + head.put(name) + head.putInt(1) + head.putLong(t.elementCount) + head.putInt(t.type.value) + head.putLong(dataOffset) + dataOffset += padded(t.data.size) + } + val padding = (32 - (head.position() % 32)) % 32 + repeat(padding) { head.put(0) } + RandomAccessFile(file, "rw").use { raf -> + raf.write(head.array(), 0, head.position()) + for (t in tensors) { + raf.write(t.data) + repeat(padded(t.data.size) - t.data.size) { raf.write(0) } + } + } + return file + } + + private fun padded(size: Int): Int = ((size + 31) / 32) * 32 + } +} diff --git a/skainet-lang/skainet-lang-core/api/jvm/skainet-lang-core.api b/skainet-lang/skainet-lang-core/api/jvm/skainet-lang-core.api index d31b22737..97b8cffdc 100644 --- a/skainet-lang/skainet-lang-core/api/jvm/skainet-lang-core.api +++ b/skainet-lang/skainet-lang-core/api/jvm/skainet-lang-core.api @@ -2385,6 +2385,21 @@ public class sk/ainet/lang/tensor/DequantizationBenchmark { public final fun setup ()V } +public final class sk/ainet/lang/tensor/Dim { + public static final field DYNAMIC I + public static final field INSTANCE Lsk/ainet/lang/tensor/Dim; + public final fun compatible (II)Z + public final fun concat (Ljava/util/List;)I + public final fun isDynamic (I)Z + public final fun isStatic (I)Z + public final fun render (I)Ljava/lang/String; +} + +public final class sk/ainet/lang/tensor/DimKt { + public static final fun hasDynamic (Ljava/util/List;)Z + public static final fun hasDynamic ([I)Z +} + public final class sk/ainet/lang/tensor/GradState { public fun ()V public fun (ZLsk/ainet/lang/tensor/Tensor;)V @@ -2521,10 +2536,13 @@ public final class sk/ainet/lang/tensor/Shape { public fun equals (Ljava/lang/Object;)Z public final fun get (I)I public final fun getDimensions ()[I + public final fun getDynamicAxes ()Ljava/util/List; public final fun getRank ()I public final fun getVolume ()I + public final fun hasDynamic ()Z public fun hashCode ()I public final fun index ([I)I + public final fun isDynamic (I)Z public fun toString ()Ljava/lang/String; } @@ -4929,10 +4947,12 @@ public final class sk/ainet/lang/tensor/storage/ActiveMemoryTracker { } public final class sk/ainet/lang/tensor/storage/AggregateMemoryReport { - public fun (IJJJIIIIJJLjava/util/List;)V + public fun (IJJJIIIIJJLjava/util/List;Ljava/util/Map;)V + public synthetic fun (IJJJIIIIJJLjava/util/List;Ljava/util/Map;ILkotlin/jvm/internal/DefaultConstructorMarker;)V public final fun component1 ()I public final fun component10 ()J public final fun component11 ()Ljava/util/List; + public final fun component12 ()Ljava/util/Map; public final fun component2 ()J public final fun component3 ()J public final fun component4 ()J @@ -4941,11 +4961,12 @@ public final class sk/ainet/lang/tensor/storage/AggregateMemoryReport { public final fun component7 ()I public final fun component8 ()I public final fun component9 ()J - public final fun copy (IJJJIIIIJJLjava/util/List;)Lsk/ainet/lang/tensor/storage/AggregateMemoryReport; - public static synthetic fun copy$default (Lsk/ainet/lang/tensor/storage/AggregateMemoryReport;IJJJIIIIJJLjava/util/List;ILjava/lang/Object;)Lsk/ainet/lang/tensor/storage/AggregateMemoryReport; + public final fun copy (IJJJIIIIJJLjava/util/List;Ljava/util/Map;)Lsk/ainet/lang/tensor/storage/AggregateMemoryReport; + public static synthetic fun copy$default (Lsk/ainet/lang/tensor/storage/AggregateMemoryReport;IJJJIIIIJJLjava/util/List;Ljava/util/Map;ILjava/lang/Object;)Lsk/ainet/lang/tensor/storage/AggregateMemoryReport; public fun equals (Ljava/lang/Object;)Z public final fun getAliasedCount ()I public final fun getBorrowedCount ()I + public final fun getCopiesBySource ()Ljava/util/Map; public final fun getCopyBytes ()J public final fun getCopyCount ()J public final fun getEntries ()Ljava/util/List; @@ -5076,6 +5097,19 @@ public final class sk/ainet/lang/tensor/storage/CompressedKvAttention$DequantStr public static fun values ()[Lsk/ainet/lang/tensor/storage/CompressedKvAttention$DequantStrategy; } +public final class sk/ainet/lang/tensor/storage/CopySourceStat { + public fun (JJ)V + public final fun component1 ()J + public final fun component2 ()J + public final fun copy (JJ)Lsk/ainet/lang/tensor/storage/CopySourceStat; + public static synthetic fun copy$default (Lsk/ainet/lang/tensor/storage/CopySourceStat;JJILjava/lang/Object;)Lsk/ainet/lang/tensor/storage/CopySourceStat; + public fun equals (Ljava/lang/Object;)Z + public final fun getBytes ()J + public final fun getCount ()J + public fun hashCode ()I + public fun toString ()Ljava/lang/String; +} + public final class sk/ainet/lang/tensor/storage/DefaultBufferResolver : sk/ainet/lang/tensor/storage/BufferResolver { public fun ()V public fun (Lkotlin/jvm/functions/Function1;)V @@ -5625,8 +5659,10 @@ public final class sk/ainet/lang/tensor/storage/TensorStorage { public final fun copy (Lsk/ainet/lang/tensor/Shape;Lsk/ainet/lang/tensor/storage/LogicalDType;Lsk/ainet/lang/tensor/storage/TensorEncoding;Lsk/ainet/lang/tensor/storage/BufferHandle;Lsk/ainet/lang/tensor/storage/Placement;J[JZ)Lsk/ainet/lang/tensor/storage/TensorStorage; public static synthetic fun copy$default (Lsk/ainet/lang/tensor/storage/TensorStorage;Lsk/ainet/lang/tensor/Shape;Lsk/ainet/lang/tensor/storage/LogicalDType;Lsk/ainet/lang/tensor/storage/TensorEncoding;Lsk/ainet/lang/tensor/storage/BufferHandle;Lsk/ainet/lang/tensor/storage/Placement;J[JZILjava/lang/Object;)Lsk/ainet/lang/tensor/storage/TensorStorage; public final fun copyMaterialize ()Lsk/ainet/lang/tensor/storage/TensorStorage; + public final fun copyMaterialize (Lsk/ainet/lang/tensor/storage/BufferResolver;)Lsk/ainet/lang/tensor/storage/TensorStorage; public final fun copyToDevice (Lsk/ainet/lang/tensor/storage/DeviceKind;)Lsk/ainet/lang/tensor/storage/TensorStorage; public final fun copyToHost ()Lsk/ainet/lang/tensor/storage/TensorStorage; + public final fun copyToHost (Lsk/ainet/lang/tensor/storage/BufferResolver;)Lsk/ainet/lang/tensor/storage/TensorStorage; public fun equals (Ljava/lang/Object;)Z public final fun getBuffer ()Lsk/ainet/lang/tensor/storage/BufferHandle; public final fun getByteOffset ()J diff --git a/skainet-lang/skainet-lang-core/build.gradle.kts b/skainet-lang/skainet-lang-core/build.gradle.kts index 10a8f9437..b06dc9f56 100644 --- a/skainet-lang/skainet-lang-core/build.gradle.kts +++ b/skainet-lang/skainet-lang-core/build.gradle.kts @@ -62,6 +62,19 @@ kotlin { implementation(libs.kotlinx.benchmark.runtime) } + // java.nio-based mmap tensor storage (MmapTensorData.kt), shared source + // between the JVM and Android compilations (#921): FileChannel.map / + // MappedByteBuffer are available on Android since API 1, so the same + // implementation serves both targets. A plain shared directory (not an + // intermediate source set) keeps each compilation against its full + // platform classpath. + jvmMain { + kotlin.srcDir("src/jvmAndroidMain/kotlin") + } + getByName("androidMain") { + kotlin.srcDir("src/jvmAndroidMain/kotlin") + } + commonTest.dependencies { implementation(libs.kotlin.test) } diff --git a/skainet-lang/skainet-lang-core/src/jvmMain/kotlin/sk/ainet/lang/tensor/data/MmapTensorData.kt b/skainet-lang/skainet-lang-core/src/jvmAndroidMain/kotlin/sk/ainet/lang/tensor/data/MmapTensorData.kt similarity index 93% rename from skainet-lang/skainet-lang-core/src/jvmMain/kotlin/sk/ainet/lang/tensor/data/MmapTensorData.kt rename to skainet-lang/skainet-lang-core/src/jvmAndroidMain/kotlin/sk/ainet/lang/tensor/data/MmapTensorData.kt index 91402668c..d3c609fa5 100644 --- a/skainet-lang/skainet-lang-core/src/jvmMain/kotlin/sk/ainet/lang/tensor/data/MmapTensorData.kt +++ b/skainet-lang/skainet-lang-core/src/jvmAndroidMain/kotlin/sk/ainet/lang/tensor/data/MmapTensorData.kt @@ -124,11 +124,13 @@ public class MmapTensorSource( "Tensor region [${byteOffset}, ${byteOffset + byteSize}) exceeds buffer capacity ${mappedBuffer.capacity()}" } - // Create a slice view of the mapped buffer - val slice = mappedBuffer.duplicate() - .position(byteOffset.toInt()) - .limit((byteOffset + byteSize).toInt()) - .slice() + // Create a slice view of the mapped buffer. Deliberately not chained: + // on the Android SDK (pre-Java-9 nio API) position/limit return + // Buffer, not ByteBuffer, so chaining does not compile there. + val dup = mappedBuffer.duplicate() + dup.position(byteOffset.toInt()) + dup.limit((byteOffset + byteSize).toInt()) + val slice = dup.slice() return MmapFloatTensorData(shape, slice) }