Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,6 @@ public final class sk/ainet/context/DirectCpuExecutionContext : sk/ainet/context
public fun getHooks ()Lsk/ainet/lang/nn/hooks/ForwardHooks;
public fun getInTraining ()Z
public fun getMemoryInfo ()Lsk/ainet/context/MemoryInfo;
public fun getMemoryPlanner ()Lsk/ainet/lang/tensor/storage/MemoryPlanner;
public fun getMemoryScope ()Lsk/ainet/lang/memory/Scope;
public fun getMemoryTracker ()Lsk/ainet/lang/tensor/storage/MemoryTracker;
public fun getObservers ()Lsk/ainet/context/ExecutionObserverRegistry;
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -172,7 +172,6 @@ public final class sk/ainet/lang/graph/DefaultGraphExecutionContext : sk/ainet/l
public fun getHooks ()Lsk/ainet/lang/nn/hooks/ForwardHooks;
public fun getInTraining ()Z
public fun getMemoryInfo ()Lsk/ainet/context/MemoryInfo;
public fun getMemoryPlanner ()Lsk/ainet/lang/tensor/storage/MemoryPlanner;
public fun getMemoryScope ()Lsk/ainet/lang/memory/Scope;
public fun getMemoryTracker ()Lsk/ainet/lang/tensor/storage/MemoryTracker;
public fun getObservers ()Lsk/ainet/context/ExecutionObserverRegistry;
Expand Down Expand Up @@ -449,7 +448,6 @@ public final class sk/ainet/lang/graph/exec/GraphExecutionContext$DefaultImpls {
public static fun full (Lsk/ainet/lang/graph/exec/GraphExecutionContext;Lsk/ainet/lang/tensor/Shape;Lkotlin/reflect/KClass;Ljava/lang/Number;)Lsk/ainet/lang/tensor/Tensor;
public static fun getHooks (Lsk/ainet/lang/graph/exec/GraphExecutionContext;)Lsk/ainet/lang/nn/hooks/ForwardHooks;
public static fun getInTraining (Lsk/ainet/lang/graph/exec/GraphExecutionContext;)Z
public static fun getMemoryPlanner (Lsk/ainet/lang/graph/exec/GraphExecutionContext;)Lsk/ainet/lang/tensor/storage/MemoryPlanner;
public static fun getMemoryScope (Lsk/ainet/lang/graph/exec/GraphExecutionContext;)Lsk/ainet/lang/memory/Scope;
public static fun getMemoryTracker (Lsk/ainet/lang/graph/exec/GraphExecutionContext;)Lsk/ainet/lang/tensor/storage/MemoryTracker;
public static fun getScratch (Lsk/ainet/lang/graph/exec/GraphExecutionContext;)Lsk/ainet/lang/tensor/scratch/ScratchPool;
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,6 @@ import sk.ainet.io.RandomAccessSource
import sk.ainet.lang.tensor.storage.LogicalDType
import sk.ainet.lang.tensor.storage.MemoryDomain
import sk.ainet.lang.tensor.storage.Ownership
import sk.ainet.lang.tensor.storage.Residency
import kotlin.test.Test
import kotlin.test.assertEquals
import kotlin.test.assertFalse
Expand Down Expand Up @@ -127,7 +126,6 @@ class StorageAwareSafeTensorsLoaderTest {
assertTrue(storage.isFileBacked)
assertEquals(Ownership.FILE_BACKED, storage.ownership)
assertEquals(MemoryDomain.MMAP_FILE, storage.placement.domain)
assertEquals(Residency.PERSISTENT, storage.placement.residency)
assertFalse(storage.isMutable)
}

Expand Down
96 changes: 6 additions & 90 deletions skainet-lang/skainet-lang-core/api/jvm/skainet-lang-core.api

Large diffs are not rendered by default.

Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,6 @@ import sk.ainet.lang.tensor.operators.OpsBoundTensor
import sk.ainet.lang.tensor.ops.TensorOps
import sk.ainet.lang.tensor.scratch.NoopScratchPool
import sk.ainet.lang.tensor.scratch.ScratchPool
import sk.ainet.lang.tensor.storage.MemoryPlanner
import sk.ainet.lang.tensor.storage.MemoryTracker
import sk.ainet.lang.types.DType
import kotlin.reflect.KClass
Expand Down Expand Up @@ -187,9 +186,6 @@ public interface ExecutionContext {
public val memoryInfo: MemoryInfo
public val executionStats: ExecutionStats

/** Memory planner for resolving placement intents. Default: CPU-only. */
public val memoryPlanner: MemoryPlanner get() = MemoryPlanner()

/** Memory tracker for observability and copy tracing. Default: no-op (not tracking). */
public val memoryTracker: MemoryTracker? get() = null
}
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,6 @@ package sk.ainet.lang.tensor.ops.turboquant

import sk.ainet.lang.tensor.storage.KvCacheConfig
import sk.ainet.lang.tensor.storage.Placement
import sk.ainet.lang.tensor.storage.Residency
import sk.ainet.lang.tensor.storage.TensorEncoding

/**
Expand Down Expand Up @@ -39,7 +38,7 @@ public object TurboQuantPresets {
maxSeqLen = maxSeqLen,
keyEncoding = TensorEncoding.Q8_0,
valueEncoding = TensorEncoding.TurboQuantPolar(bitsPerElement = 4),
placement = Placement.CPU_HEAP.copy(residency = Residency.PERSISTENT)
placement = Placement.CPU_HEAP
),
keyQuantConfig = null, // Q8_0 uses standard quantization, not TurboQuant
valueQuantConfig = TurboQuantConfig.polarOnly(bits = 4)
Expand All @@ -65,7 +64,7 @@ public object TurboQuantPresets {
maxSeqLen = maxSeqLen,
keyEncoding = TensorEncoding.TurboQuantPolar(bitsPerElement = 4),
valueEncoding = TensorEncoding.TurboQuantPolar(bitsPerElement = 4),
placement = Placement.CPU_HEAP.copy(residency = Residency.PERSISTENT)
placement = Placement.CPU_HEAP
),
keyQuantConfig = TurboQuantConfig.polarOnly(bits = 4),
valueQuantConfig = TurboQuantConfig.polarOnly(bits = 4)
Expand All @@ -92,7 +91,7 @@ public object TurboQuantPresets {
maxSeqLen = maxSeqLen,
keyEncoding = TensorEncoding.TurboQuantPolar(bitsPerElement = 3),
valueEncoding = TensorEncoding.TurboQuantPolar(bitsPerElement = 3),
placement = Placement.CPU_HEAP.copy(residency = Residency.PERSISTENT)
placement = Placement.CPU_HEAP
),
keyQuantConfig = TurboQuantConfig.polarOnly(bits = 3),
valueQuantConfig = TurboQuantConfig.polarOnly(bits = 3)
Expand Down
Original file line number Diff line number Diff line change
@@ -1,51 +1,5 @@
package sk.ainet.lang.tensor.storage

/**
* Declares placement intent for a tensor parameter or property.
*
* The [MemoryPlanner] reads these annotations (via reflection or codegen)
* to decide where tensors should be allocated. This expresses *intent*,
* not a hard guarantee — the planner may fall back if the target is
* unavailable and [requirement] is [Requirement.PREFERRED].
*
* Example:
* ```kotlin
* @Place(device = DeviceKind.GPU, memory = MemoryDomain.DEVICE_LOCAL)
* val projectionWeight: Tensor<FP32, Float>
* ```
*/
@Target(AnnotationTarget.PROPERTY, AnnotationTarget.VALUE_PARAMETER, AnnotationTarget.FIELD)
@Retention(AnnotationRetention.RUNTIME)
public annotation class Place(
val device: DeviceKind = DeviceKind.AUTO,
val memory: MemoryDomain = MemoryDomain.HOST_HEAP,
val requirement: Requirement = Requirement.PREFERRED
)

/**
* Marks a tensor as an immutable weight that should be file-backed
* (memory-mapped) when possible.
*
* Equivalent to `@Place(device = CPU, memory = MMAP_FILE)` with
* [Residency.PERSISTENT]. The planner treats these tensors as
* read-only and long-lived, preferring OS-paged file access over
* heap allocation.
*
* Example:
* ```kotlin
* @Weights
* val embeddings: Tensor<FP32, Float>
*
* @Weights(memory = MemoryDomain.HOST_HEAP) // force heap for small weights
* val biasVector: Tensor<FP32, Float>
* ```
*/
@Target(AnnotationTarget.PROPERTY, AnnotationTarget.VALUE_PARAMETER, AnnotationTarget.FIELD)
@Retention(AnnotationRetention.RUNTIME)
public annotation class Weights(
val memory: MemoryDomain = MemoryDomain.MMAP_FILE
)

/**
* Configures TurboQuant KV-cache compression for an attention layer.
*
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -300,7 +300,7 @@ public data class KvCacheConfig(
val maxSeqLen: Int,
val keyEncoding: TensorEncoding = TensorEncoding.Dense(4),
val valueEncoding: TensorEncoding = TensorEncoding.Dense(4),
val placement: Placement = Placement.CPU_HEAP.copy(residency = Residency.PERSISTENT),
val placement: Placement = Placement.CPU_HEAP,
/**
* The dtype the key ring stores; with [keyEncoding] it forms the store's `keyFormat` (#1077).
* `FP32` is what the dense store has always held; `BF16`/`FP16` halve the ring.
Expand Down

This file was deleted.

Original file line number Diff line number Diff line change
Expand Up @@ -4,15 +4,16 @@ package sk.ainet.lang.tensor.storage
* High-level placement descriptor: where a tensor lives and how the runtime
* should manage it.
*
* Placement is *intent* — it tells the planner what to aim for but does not
* encode backend scratch-memory details. The planner resolves placement to
* a concrete [BufferHandle] and falls back if the preferred target is
* unavailable.
* Placement is *intent* — it tells the runtime what to aim for but does not
* encode backend scratch-memory details.
*
* Lifetime is deliberately not part of placement: how long bytes live is a
* scope decision (`sk.ainet.lang.memory.ScopeKind`), and how a weight is
* staged at load is a resolver decision (`WeightForm.WeightResidency`).
*/
public data class Placement(
val device: DeviceKind = DeviceKind.CPU,
val domain: MemoryDomain = MemoryDomain.HOST_HEAP,
val residency: Residency = Residency.PERSISTENT,
val requirement: Requirement = Requirement.PREFERRED,
val fallback: DeviceKind = DeviceKind.CPU
) {
Expand All @@ -21,23 +22,20 @@ public data class Placement(
public val CPU_HEAP: Placement = Placement(
device = DeviceKind.CPU,
domain = MemoryDomain.HOST_HEAP,
residency = Residency.TRANSIENT,
requirement = Requirement.PREFERRED
)

/** File-backed placement for immutable model weights. */
public val MMAP_WEIGHTS: Placement = Placement(
device = DeviceKind.CPU,
domain = MemoryDomain.MMAP_FILE,
residency = Residency.PERSISTENT,
requirement = Requirement.PREFERRED
)

/** GPU-preferred placement with CPU fallback. */
public val GPU_PREFERRED: Placement = Placement(
device = DeviceKind.GPU,
domain = MemoryDomain.DEVICE_LOCAL,
residency = Residency.PERSISTENT,
requirement = Requirement.PREFERRED,
fallback = DeviceKind.CPU
)
Expand Down Expand Up @@ -66,13 +64,6 @@ public enum class MemoryDomain {
DEVICE_LOCAL
}

public enum class Residency {
/** Short-lived: activations, temporaries, intermediate results. */
TRANSIENT,
/** Long-lived: model weights, embeddings, caches. */
PERSISTENT
}

public enum class Requirement {
/** Best-effort: fall back to [Placement.fallback] if unavailable. */
PREFERRED,
Expand Down
Loading
Loading