Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -115,7 +115,7 @@ public object PackedWeights {
require(inputAligned || !rowsAligned) {
"weight [$rows, $inputDim] looks like [in, out]: ${encoding.name} tiles the *input* dimension in " +
"blocks of $blockSize, and $inputDim is not a multiple of it while $rows is. A GGUF's ne order " +
"produces exactly this — load with WeightOrientation.OUT_IN, or transpose the label before " +
"produces exactly this — load with WeightShapeOrientation.OUT_IN, or transpose the label before " +
"relayouting (#973, the Packed weight layout page in the docs site (explanation/packed-weight-layout))."
}
require(inputAligned) {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -116,7 +116,7 @@ class PackedWeightsTest {
PackedWeights.requireOutIn(rows = 128, inputDim = 3, encoding = TensorEncoding.Q8_0)
}
assertTrue(failure.message!!.contains("looks like [in, out]"), failure.message!!)
assertTrue(failure.message!!.contains("WeightOrientation.OUT_IN"), "and says how to fix it")
assertTrue(failure.message!!.contains("WeightShapeOrientation.OUT_IN"), "and says how to fix it")

PackedWeights.requireOutIn(rows = 3, inputDim = 128, encoding = TensorEncoding.Q8_0)
}
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -27,7 +27,7 @@ public class AndroidRandomAccessSource private constructor(
private val channel: FileChannel,
private val raf: RandomAccessFile,
override val size: Long,
/** The file these bytes come from — what `StagingPolicy.MAPPED` maps (#1037). */
/** The file these bytes come from — what `WeightResidency.MAPPED` maps (#1037, #1159). */
override val filePath: String? = null,
) : RandomAccessSource {

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -62,7 +62,7 @@ public interface RandomAccessSource : AutoCloseable {
* The path these bytes came from, when they came from a file — `null` for a Blob, a network
* stream or an in-memory source.
*
* This is what lets a loader honour `StagingPolicy.MAPPED` (#1037): the same source that reads
* This is what lets a loader honour `WeightResidency.MAPPED` (#1037, #1159): the same source that reads
* the header positionally can name the file to map for the tensor payloads. Defaulted, so no
* existing implementation has to change.
*/
Expand Down

This file was deleted.

This file was deleted.

This file was deleted.

Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,7 @@ public class JvmRandomAccessSource private constructor(
private val channel: FileChannel,
private val raf: RandomAccessFile,
override val size: Long,
/** The file these bytes come from — what `StagingPolicy.MAPPED` maps (#1037). */
/** The file these bytes come from — what `WeightResidency.MAPPED` maps (#1037, #1159). */
override val filePath: String? = null,
) : RandomAccessSource {

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -31,7 +31,7 @@ import platform.posix.strerror
public class PosixPreadRandomAccessSource private constructor(
private val fd: Int,
override val size: Long,
/** The file these bytes come from — what `StagingPolicy.MAPPED` would map (#1037). */
/** The file these bytes come from — what `WeightResidency.MAPPED` would map (#1037, #1159). */
override val filePath: String? = null,
) : RandomAccessSource {

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -2,8 +2,7 @@ package sk.ainet.io.gguf

import kotlinx.coroutines.runBlocking
import sk.ainet.context.DefaultDataExecutionContext
import sk.ainet.io.model.QuantPolicy
import sk.ainet.io.model.StagingPolicy
import sk.ainet.lang.memory.plan.WeightForm
import sk.ainet.lang.memory.ExperimentalMemoryApi
import sk.ainet.lang.memory.plan.DeviceMemory
import sk.ainet.lang.memory.plan.PlannerProfile
Expand Down Expand Up @@ -87,7 +86,7 @@ class AndroidGgufLoadingHostTest {
"dense F32 must come from file-backed pages on Android, got ${mapped.getValue("w_f32").data::class.simpleName}",
)
// and the heap path is still reachable, producing the same numbers
val onHeap = load(AndroidGguf.loader(f.absolutePath, staging = StagingPolicy.HEAP))
val onHeap = load(AndroidGguf.loader(f.absolutePath, weightForm = WeightForm()))
assertTrue(onHeap.getValue("w_f32").data is FloatArrayTensorData<*>)
assertContentEquals(
onHeap.getValue("w_f32").data.copyToFloatArray(),
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -4,8 +4,8 @@ import android.app.ActivityManager
import android.content.Context
import sk.ainet.io.RandomAccessSource
import sk.ainet.io.openRandomAccessSource
import sk.ainet.io.model.QuantPolicy
import sk.ainet.io.model.StagingPolicy
import sk.ainet.lang.memory.plan.WeightForm
import sk.ainet.lang.memory.plan.WeightResidency
import sk.ainet.lang.memory.ExperimentalMemoryApi
import sk.ainet.lang.memory.plan.Budget
import sk.ainet.lang.memory.plan.DeviceFit
Expand All @@ -22,7 +22,7 @@ import sk.ainet.lang.memory.plan.fitOn
*
* The managed heap is the binding constraint on a phone — hard-capped at 256 MB (512 MB with
* `largeHeap`) no matter how much RAM the device has — so the Android configuration of the loader
* is `staging = MAPPED`: weights come from file-backed pages the OS pages in on demand and evicts
* asks for `WeightResidency.MAPPED`: weights come from file-backed pages the OS pages in on demand and evicts
* under pressure, and never count against the cap.
*
* What is *not* solved yet: packed (quantized) tensors still arrive as heap arrays, because the
Expand All @@ -36,20 +36,18 @@ public object AndroidGguf {

/**
* The loader Android should use: positional reads for the metadata, mapped pages for tensor
* payloads. [quantPolicy] is the caller's choice as usual; [staging] defaults to
* [StagingPolicy.MAPPED] and is a parameter only so a test or a benchmark can ask for the
* heap path explicitly.
* payloads. [weightForm] defaults to mapped residency — the managed heap is the binding
* constraint on a phone — and is a parameter only so a test or a benchmark can ask for the
* heap path (or a dequantizing form) explicitly.
*/
public fun loader(
filePath: String,
quantPolicy: QuantPolicy = QuantPolicy.NATIVE_OPTIMIZED,
staging: StagingPolicy = StagingPolicy.MAPPED,
weightForm: WeightForm = WeightForm(residency = WeightResidency.MAPPED),
onProgress: (current: Long, total: Long, message: String?) -> Unit = { _, _, _ -> },
): StreamingGgufParametersLoader = StreamingGgufParametersLoader(
sourceProvider = { openSource(filePath) },
onProgress = onProgress,
quantPolicy = quantPolicy,
staging = staging,
weightForm = weightForm,
)

/**
Expand Down Expand Up @@ -88,7 +86,7 @@ public object AndroidGguf {
* Will this model load on this device? Checks the header-derived plan against both pools —
* managed heap and physical RAM — before a byte of payload is read.
*
* @param weightsMapped whether the load will use [StagingPolicy.MAPPED] (what [loader] does)
* @param weightsMapped whether the load maps the file (`WeightResidency.MAPPED`, what [loader] does)
*/
public fun fits(context: Context, filePath: String, ctx: Int, weightsMapped: Boolean = true): DeviceFit =
fits(deviceMemory(context), filePath, ctx, weightsMapped)
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -15,9 +15,6 @@ import sk.ainet.lang.memory.plan.WeightByteOrder
import sk.ainet.lang.memory.plan.WeightForm
import sk.ainet.lang.memory.plan.WeightResidency
import sk.ainet.lang.memory.plan.WeightShapeOrientation
import sk.ainet.io.model.QuantPolicy
import sk.ainet.io.model.StagingPolicy
import sk.ainet.io.model.WeightOrientation
import sk.ainet.io.openMappedFile
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.Tensor
Expand Down Expand Up @@ -71,64 +68,23 @@ public class StreamingGgufParametersLoader(
*/
private val keepBf16Native: Boolean = false,
/**
* How quantized tensors are materialized (#782).
* The form weights should take in memory — encoding × byte order × shape × residency as one
* decision (#1109, #1115, #1159).
*
* - [QuantPolicy.NATIVE_OPTIMIZED] (default — the loader's historical behavior):
* quantized tensors are delivered as packed block [TensorData]; F32/F16/BF16
* are dense FP32 (subject to [keepF16Native]/[keepBf16Native]).
* - [QuantPolicy.DEQUANTIZE_TO_FP32]: quantized tensors are dequantized
* *streaming, per tensor, block-by-block into the destination `FloatArray`*,
* which is then wrapped zero-copy. Peak transient memory per tensor is the
* packed source bytes only — there is no full-size intermediate copy.
* - [QuantPolicy.RAW_BYTES] is not supported by this loader (it preserves
* packed block storage instead) and is rejected eagerly.
*/
private val quantPolicy: QuantPolicy = QuantPolicy.NATIVE_OPTIMIZED,
/**
* Where tensor bytes live on their way into a tensor (#1037). [QuantPolicy] says *what* the
* values are; this says *where the bytes are* — the two axes of one loader.
*
* - [StagingPolicy.HEAP] (default — today's behaviour): every tensor is read onto the heap.
* - [StagingPolicy.MAPPED]: the file is mapped once and dense F32 tensors are served as
* zero-heap views over its pages (what `MappedGgufWeights` did as a separate helper). Packed
* tensors still come through as heap arrays, because that is what their kernels take until
* #973; and the whole thing falls back to heap staging when the platform cannot map or the
* source is not a file, so a browser build behaves exactly as before.
*/
private val staging: StagingPolicy = StagingPolicy.HEAP,
/**
* Which way round a 2-D weight's shape comes out (#1098, #973 census contradiction #6).
*
* GGUF writes dimensions in `ne` order, so a weight the rest of the engine calls `[out, in]`
* arrives labelled `[in, out]` — while its *bytes* are already `[out, in]` row-major. Nothing
* about the data changes here; only the label. [WeightOrientation.OUT_IN] fixes the label,
* which is what the packed block relayout needs to compute the right permutation.
*
* Defaults to [WeightOrientation.AS_STORED], today's behaviour, because reversing shapes
* changes what every consumer sees. New code should ask for `OUT_IN`.
*/
private val weightOrientation: WeightOrientation = WeightOrientation.AS_STORED,
/**
* The form weights should take in memory, as one decision instead of three (#1109, #1115).
*
* [quantPolicy], [staging] and [weightOrientation] are the same three axes asked separately,
* and asked of the *caller* — who has to answer for a device they may not be building for.
* `WeightFormResolver.resolve(stored, profile, capabilities)` answers instead, from what the
* file holds, what the device is, and what the backend's kernels can feed.
*
* `null` (the default) means "use the three parameters", so every existing caller is
* byte-identical. Passing a form while also setting any of the three is rejected rather than
* silently resolved, so nobody loses a setting they thought they had.
* `null` (the default) means [WeightForm.AS_STORED_ON_HEAP]: the file's bytes, its order, on
* the heap — the loader's historical behaviour. Callers who know their device pass a form;
* callers who don't let `WeightFormResolver`/[ResolvedGguf] resolve one from what the file
* holds, what the device is, and what the backend's kernels can feed.
*/
private val weightForm: WeightForm? = null,
/**
* Per-tensor forms — the *user wins* channel (#1144).
*
* Precedence, explicit and documented: **this function > [weightForm] > the three legacy
* parameters > nothing**. Whatever you return for a tensor outranks every resolver and every
* profile — including the deliberately blunt `WeightForm(DequantizeTo(FP32), residency = HEAP)`
* Precedence, explicit and documented: **this function > [weightForm] > the as-stored-on-heap
* default**. Whatever you return for a tensor outranks every resolver and every profile —
* including the deliberately blunt `WeightForm(DequantizeTo(FP32), residency = HEAP)`
* ("everything dense, on the managed heap"). Return `null` for tensors you have no opinion on;
* they fall through to [weightForm] (or the legacy axes).
* they fall through to [weightForm].
*
* The intended producer is `WeightFormResolver`/`resolveWeightForms` via [ResolvedGguf], which
* resolves per tensor from the file × profile × kernel capability — but the contract is the
Expand Down Expand Up @@ -181,22 +137,8 @@ public class StreamingGgufParametersLoader(
/** The dense FP32 size of [elements] — what every widening in this loader converts to. */
private fun denseFp32Bytes(elements: Long): Long = elements * 4

/** The three axes as one value: [weightForm] if given, otherwise what the three parameters say. */
private val form: WeightForm = weightForm ?: WeightForm(
encoding = when (quantPolicy) {
QuantPolicy.DEQUANTIZE_TO_FP32 -> EncodingRequest.DequantizeTo(FP32)
else -> EncodingRequest.KeepAsStored
},
order = WeightByteOrder.AS_STORED,
shape = when (weightOrientation) {
WeightOrientation.OUT_IN -> WeightShapeOrientation.OUT_IN
else -> WeightShapeOrientation.AS_STORED
},
residency = when (staging) {
StagingPolicy.MAPPED -> WeightResidency.MAPPED
else -> WeightResidency.HEAP
},
)
/** The uniform form: [weightForm] if given, else the historical as-stored-on-heap default. */
private val form: WeightForm = weightForm ?: WeightForm.AS_STORED_ON_HEAP

/**
* The form [tensorName] loads under — the precedence order of [weightFormFor], validated the
Expand All @@ -209,22 +151,6 @@ public class StreamingGgufParametersLoader(
}

init {
require(quantPolicy != QuantPolicy.RAW_BYTES) {
"StreamingGgufParametersLoader does not support QuantPolicy.RAW_BYTES — quantized " +
"tensors are preserved as packed block TensorData (NATIVE_OPTIMIZED) or " +
"dequantized to dense FP32 (DEQUANTIZE_TO_FP32)."
}
if (weightForm != null || weightFormFor != null) {
require(
quantPolicy == QuantPolicy.NATIVE_OPTIMIZED &&
staging == StagingPolicy.HEAP &&
weightOrientation == WeightOrientation.AS_STORED,
) {
"a WeightForm (or weightFormFor) and the quantPolicy/staging/weightOrientation " +
"parameters were both set. They are the same three axes, so one of them would " +
"have been silently ignored; pass only the form."
}
}
validateForm(form, "weightForm")
}

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -11,8 +11,8 @@ import java.nio.channels.FileChannel
/**
* Memory-mapped GGUF weight access for the JVM and Android (#921).
*
* Since #1037 this is the **per-tensor** face of `StagingPolicy.MAPPED`: to load a whole model
* from mapped pages, pass `staging = StagingPolicy.MAPPED` to [StreamingGgufParametersLoader] and
* Since #1037 this is the **per-tensor** face of `WeightResidency.MAPPED`: to load a whole model
* from mapped pages, pass `WeightForm(residency = WeightResidency.MAPPED)` to [StreamingGgufParametersLoader] and
* get the same file-backed tensors through the ordinary loader. This class stays for callers that
* want to reach individual tensors (or their `TensorStorage` descriptors) without loading a model.
*
Expand Down
Loading
Loading