Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
96 changes: 96 additions & 0 deletions skainet-lang/skainet-lang-core/api/jvm/skainet-lang-core.api

Large diffs are not rendered by default.

Original file line number Diff line number Diff line change
@@ -0,0 +1,74 @@
package sk.ainet.lang.memory

import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.storage.MemoryDomain

/**
* The lifetime class an allocation belongs to (SKEEP-003 §4.5). `Scope` itself — the object that
* allocates and frees — arrives with milestone M1; this enum is its `kind`, declared now so that
* memory plans and allocation specs can name the lifetime without depending on the allocator.
*/
@ExperimentalMemoryApi
public enum class ScopeKind {
/** Lives until the model is closed: weights, KV-cache backing, embedding tables. */
MODEL,
/** Recycled every forward pass: activations, attention scratch, adapter outputs. */
FORWARD,
/** Garbage-collected, no explicit lifetime — the default for notebooks, tests and ad-hoc tensors. */
AMBIENT,
}

/**
* What an allocation needs: the [Format] of the elements, how many, and where / for how long the
* bytes should live. Pure description — owns nothing. The single input of the memory plan
* (milestone M0) and, from M1, of `Storage.allocate(spec, scope)`.
*
* Replaces the never-consumed `StorageSpec` (decision recorded in SKEEP-003: "StorageSpec becomes
* the allocation spec").
*
* @property format dtype + encoding of the elements
* @property elementCount logical number of elements
* @property domain where the bytes should be (heap, off-heap/pinned, mapped file, device …)
* @property scope the lifetime class the allocation belongs to
* @property mutable whether the bytes may be written after allocation
* @property alignment required byte alignment of the start of the buffer (SIMD kernels want 16–64)
*/
@ExperimentalMemoryApi
public data class AllocationSpec(
val format: Format,
val elementCount: Long,
val domain: MemoryDomain = MemoryDomain.HOST_HEAP,
val scope: ScopeKind = ScopeKind.AMBIENT,
val mutable: Boolean = true,
val alignment: Int = DEFAULT_ALIGNMENT,
) {
init {
require(elementCount >= 0) { "elementCount must be >= 0, was $elementCount" }
require(alignment > 0 && (alignment and (alignment - 1)) == 0) { "alignment must be a power of two, was $alignment" }
}

/** Physical bytes this allocation needs, or `null` when the encoding cannot tell (opaque payloads). */
val bytesOrNull: Long? get() = format.physicalBytes(elementCount)

/**
* Physical bytes this allocation needs.
* @throws IllegalStateException for an encoding that cannot compute its size (see [bytesOrNull])
*/
val bytes: Long
get() = bytesOrNull ?: throw IllegalStateException("Encoding ${format.encoding.name} cannot compute a byte size for $elementCount elements")

public companion object {
/** 64 bytes: satisfies AVX-512 / NEON / cache-line alignment for every current kernel. */
public const val DEFAULT_ALIGNMENT: Int = 64

/** Spec for a tensor of [shape] in [format]. */
public fun of(
format: Format,
shape: Shape,
domain: MemoryDomain = MemoryDomain.HOST_HEAP,
scope: ScopeKind = ScopeKind.AMBIENT,
mutable: Boolean = true,
alignment: Int = DEFAULT_ALIGNMENT,
): AllocationSpec = AllocationSpec(format, shape.volume.toLong(), domain, scope, mutable, alignment)
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
package sk.ainet.lang.memory

/**
* Marks the SKEEP-003 memory-architecture API (`sk.ainet.lang.memory`) that is still being
* shaped through milestones M0–M1 (`Format`, `AllocationSpec`, later `Storage`, `Scope`,
* `TensorView`, …). The types are usable and tested, but their shape may still change before the
* compatibility promise applies; opt in explicitly with `@OptIn(ExperimentalMemoryApi::class)`.
*/
@RequiresOptIn(
message = "SKEEP-003 memory-architecture API: usable, but may change until milestone M1 is complete.",
level = RequiresOptIn.Level.WARNING,
)
@Retention(AnnotationRetention.BINARY)
@Target(
AnnotationTarget.CLASS, AnnotationTarget.FUNCTION, AnnotationTarget.PROPERTY,
AnnotationTarget.TYPEALIAS, AnnotationTarget.CONSTRUCTOR,
)
public annotation class ExperimentalMemoryApi
Original file line number Diff line number Diff line change
@@ -0,0 +1,58 @@
package sk.ainet.lang.memory

import sk.ainet.lang.tensor.Tensor
import sk.ainet.lang.tensor.storage.TensorEncoding
import sk.ainet.lang.tensor.storage.TensorStorage
import sk.ainet.lang.types.DType

/**
* The pair `(dtype, encoding)` — **what** a value means and **how** its bytes are laid out.
*
* A Q4_K weight is `Format(FP32, TensorEncoding.Q4_K)`: logically FP32, stored as Q4_K blocks. A
* plain float tensor is `Format(FP32, TensorEncoding.Dense(4))`. Kernel dispatch keys on formats
* (SKEEP-003 §0, §5); rule 3 — the logical dtype is never erased by a packed encoding.
*
* `Format` is pure metadata: it owns no bytes and carries no shape.
*/
@ExperimentalMemoryApi
public data class Format(val dtype: DType, val encoding: TensorEncoding) {

/** True when the bytes are the dtype's own dense representation (no block packing). */
val isDense: Boolean get() = encoding is TensorEncoding.Dense

/** Physical bytes for [elementCount] elements under this format, or `null` if the encoding cannot tell. */
public fun physicalBytes(elementCount: Long): Long? = encoding.physicalBytes(elementCount)

/** `Float32/Q4_K`, `Float32/Dense(4B)` — the form the `toString()` renderer prints. */
override fun toString(): String = "${dtype.name}/${encoding.name}"

public companion object {
/** The dense format of [dtype] at its own width (`Dense(dtype.sizeInBytes)`). */
public fun dense(dtype: DType): Format = Format(dtype, TensorEncoding.Dense(dtype.sizeInBytes))
}
}

/**
* The [Format] of this tensor: its [Tensor.dtype] witness mapped back to a [DType] plus the
* encoding its data reports ([sk.ainet.lang.tensor.data.TensorData.encoding], dense at the
* dtype's width when the data reports none).
*
* @throws IllegalStateException if [Tensor.dtype] is not a concrete dtype class (e.g. `DType::class`)
*/
@ExperimentalMemoryApi
public val Tensor<*, *>.format: Format
get() = formatOrNull
?: throw IllegalStateException("Tensor.dtype ${this.dtype} is not a concrete DType witness; cannot derive a Format")

/** The [Format] of this tensor, or `null` if its dtype witness is not a concrete dtype class. */
@ExperimentalMemoryApi
public val Tensor<*, *>.formatOrNull: Format?
get() {
val dt = DType.fromWitnessOrNull(this.dtype) ?: return null
return Format(dt, this.data.encoding ?: TensorEncoding.Dense(dt.sizeInBytes))
}

/** The [Format] of this storage descriptor: `(dtype, encoding)`. */
@ExperimentalMemoryApi
public val TensorStorage.format: Format
get() = Format(dtype, encoding)
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
package sk.ainet.lang.tensor.data

import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.storage.TensorEncoding
import sk.ainet.lang.types.DType
import sk.ainet.lang.types.Fp16Codec
import sk.ainet.lang.types.NarrowFloatCodec
Expand Down Expand Up @@ -58,6 +59,9 @@ public open class NarrowFloatDenseTensorData(
private val strides: IntArray = shape.computeStrides()
override val packedData: ByteArray get() = data

/** Physically two bytes per element whatever the declared dtype witness. */
override val encoding: TensorEncoding get() = TensorEncoding.Dense(NarrowFloatTensorData.BYTES_PER_ELEMENT)

init {
val requiredBytes = shape.volume * NarrowFloatTensorData.BYTES_PER_ELEMENT
require(data.size >= requiredBytes) {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -68,6 +68,18 @@ public interface TensorData<T : DType, V> : ItemsAccessor<V> {
*/
public val shape: Shape

/**
* How this data's bytes are laid out, or `null` when they are the plain dense representation
* of the tensor's logical dtype (the default for array-backed data).
*
* Packed implementations (Q4_0 … Q8_0, Q4_K/Q5_K/Q6_K, ternary) report their block encoding;
* narrow-float data reports `Dense(2)`. Together with the tensor's dtype witness this yields
* the tensor's `Format` (SKEEP-003 rule 3: a packed weight is *logically* FP32 — the encoding
* says how it is stored, it never replaces the dtype). A default member, not an abstract one,
* because `TensorData` is implemented outside this module.
*/
public val encoding: sk.ainet.lang.tensor.storage.TensorEncoding? get() = null

/**
* Copies all tensor data to a FloatArray.
*
Expand Down
Original file line number Diff line number Diff line change
@@ -1,5 +1,9 @@
package sk.ainet.lang.tensor.storage

import sk.ainet.lang.memory.AllocationSpec
import sk.ainet.lang.memory.ExperimentalMemoryApi
import sk.ainet.lang.memory.Format
import sk.ainet.lang.memory.ScopeKind
import sk.ainet.lang.types.DType

/**
Expand All @@ -11,7 +15,15 @@ import sk.ainet.lang.types.DType
* [sk.ainet.lang.tensor.data.TensorFactoryRegistry]). Existing dtype-based
* lookups remain as a convenience — they build a default [StorageSpec]
* with [TensorEncoding.Dense] and [Ownership.OWNED].
*
* Deprecated (SKEEP-003 Phase 0): never consumed by any factory; the allocation
* description is [sk.ainet.lang.memory.AllocationSpec] (`Format` + element count +
* domain + scope). Use [toAllocationSpec] to convert. Removed at the next major release.
*/
@Deprecated(
message = "StorageSpec was never consumed; describe allocations with sk.ainet.lang.memory.AllocationSpec (SKEEP-003).",
replaceWith = ReplaceWith("AllocationSpec", "sk.ainet.lang.memory.AllocationSpec"),
)
public data class StorageSpec(
val logicalType: LogicalDType,
val encoding: TensorEncoding = TensorEncoding.Dense(logicalType.sizeInBytes),
Expand All @@ -21,8 +33,24 @@ public data class StorageSpec(
/** The [DType] of [logicalType] (SKEEP-003 Phase 0 bridge; see [LogicalDType.toDType]). */
val dtype: DType get() = logicalType.toDType()

/**
* The [AllocationSpec] equivalent of this spec for [elementCount] elements: `Format(dtype,
* encoding)`, the placement's memory domain, `MODEL` scope for persistent placements and
* `AMBIENT` otherwise, mutable only when owned.
*/
@OptIn(ExperimentalMemoryApi::class)
public fun toAllocationSpec(elementCount: Long): AllocationSpec = AllocationSpec(
format = Format(dtype, encoding),
elementCount = elementCount,
domain = placement.domain,
scope = if (placement.residency == Residency.PERSISTENT) ScopeKind.MODEL else ScopeKind.AMBIENT,
mutable = ownership == Ownership.OWNED,
)

@Suppress("DEPRECATION") // the factories build the deprecated type on purpose
public companion object {
/** Build a default spec from a legacy DType (dense, owned, CPU heap). */
@Deprecated("StorageSpec is deprecated; build an AllocationSpec (sk.ainet.lang.memory).")
public fun fromDType(dtype: DType): StorageSpec {
val logical = dtype.toLogicalDType()
return StorageSpec(
Expand All @@ -34,6 +62,7 @@ public data class StorageSpec(
}

/** Spec for borrowed dense data. */
@Deprecated("StorageSpec is deprecated; build an AllocationSpec (sk.ainet.lang.memory).")
public fun borrowed(dtype: DType): StorageSpec {
val logical = dtype.toLogicalDType()
return StorageSpec(
Expand All @@ -45,6 +74,7 @@ public data class StorageSpec(
}

/** Spec for Q4_K packed data. */
@Deprecated("StorageSpec is deprecated; build an AllocationSpec (sk.ainet.lang.memory).")
public fun q4k(placement: Placement = Placement.CPU_HEAP): StorageSpec = StorageSpec(
logicalType = LogicalDType.FLOAT32,
encoding = TensorEncoding.Q4_K,
Expand All @@ -53,6 +83,7 @@ public data class StorageSpec(
)

/** Spec for Q8_0 packed data. */
@Deprecated("StorageSpec is deprecated; build an AllocationSpec (sk.ainet.lang.memory).")
public fun q80(placement: Placement = Placement.CPU_HEAP): StorageSpec = StorageSpec(
logicalType = LogicalDType.FLOAT32,
encoding = TensorEncoding.Q8_0,
Expand All @@ -61,6 +92,7 @@ public data class StorageSpec(
)

/** Spec for file-backed weights. */
@Deprecated("StorageSpec is deprecated; build an AllocationSpec (sk.ainet.lang.memory).")
public fun mmapWeights(dtype: DType): StorageSpec {
val logical = dtype.toLogicalDType()
return StorageSpec(
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,86 @@
package sk.ainet.lang.memory

import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.storage.MemoryDomain
import sk.ainet.lang.tensor.storage.Ownership
import sk.ainet.lang.tensor.storage.Placement
import sk.ainet.lang.tensor.storage.StorageSpec
import sk.ainet.lang.tensor.storage.TensorEncoding
import sk.ainet.lang.types.BF16
import sk.ainet.lang.types.FP32
import sk.ainet.lang.types.Int8
import kotlin.test.Test
import kotlin.test.assertEquals
import kotlin.test.assertFailsWith
import kotlin.test.assertFalse
import kotlin.test.assertNull
import kotlin.test.assertTrue

/** SKEEP-003 Phase 0: `AllocationSpec` replaces the never-consumed `StorageSpec`. */
@OptIn(ExperimentalMemoryApi::class)
@Suppress("DEPRECATION") // StorageSpec.toAllocationSpec is the migration path under test
class AllocationSpecTest {

@Test
fun bytesFollowTheEncoding() {
assertEquals(4L * 1000, AllocationSpec(Format.dense(FP32), 1000).bytes)
assertEquals(2L * 1000, AllocationSpec(Format.dense(BF16), 1000).bytes)
assertEquals(1000L, AllocationSpec(Format.dense(Int8), 1000).bytes)
// Q4_K: 144 bytes per 256 elements
assertEquals(144L * 4, AllocationSpec(Format(FP32, TensorEncoding.Q4_K), 1024).bytes)
assertEquals(144L, AllocationSpec(Format(FP32, TensorEncoding.Q4_K), 1).bytes) // partial block rounds up
// Q8_0: 34 bytes per 32 elements
assertEquals(34L * 2, AllocationSpec.of(Format(FP32, TensorEncoding.Q8_0), Shape(2, 32)).bytes)
// TurboQuant 4-bit, block 128: seed(4) + 4 groups × 2 B scales + 64 B codes = 76 per block
val tq = TensorEncoding.TurboQuantPolar(bitsPerElement = 4, blockSize = 128)
assertEquals(tq.physicalBytes(256), AllocationSpec(Format(FP32, tq), 256).bytes)
}

@Test
fun opaqueEncodingHasNoComputableSize() {
val spec = AllocationSpec(Format(FP32, TensorEncoding.Opaque("IQ2_XXS", 0)), 64)
// Opaque carries its raw byte count; zero is "unknown" → physicalBytes may be null or 0 depending on the encoding
val b = spec.bytesOrNull
assertTrue(b == null || b == 0L)
}

@Test
fun defaultsAreAmbientHeapMutableAligned64() {
val s = AllocationSpec(Format.dense(FP32), 8)
assertEquals(MemoryDomain.HOST_HEAP, s.domain)
assertEquals(ScopeKind.AMBIENT, s.scope)
assertTrue(s.mutable)
assertEquals(64, s.alignment)
assertFalse(s.format.isDense.not())
}

@Test
fun validation() {
assertFailsWith<IllegalArgumentException> { AllocationSpec(Format.dense(FP32), -1) }
assertFailsWith<IllegalArgumentException> { AllocationSpec(Format.dense(FP32), 1, alignment = 48) }
assertFailsWith<IllegalArgumentException> { AllocationSpec(Format.dense(FP32), 1, alignment = 0) }
}

@Test
fun storageSpecConvertsToAllocationSpec() {
val weights = StorageSpec.q4k(Placement.MMAP_WEIGHTS).toAllocationSpec(1024)
assertEquals(Format(FP32, TensorEncoding.Q4_K), weights.format)
assertEquals(1024L, weights.elementCount)
assertEquals(MemoryDomain.MMAP_FILE, weights.domain)
assertEquals(ScopeKind.MODEL, weights.scope) // persistent placement → model lifetime
assertFalse(weights.mutable) // borrowed packed bytes

val owned = StorageSpec.fromDType(BF16).toAllocationSpec(10)
assertEquals(Format.dense(BF16), owned.format)
assertEquals(ScopeKind.AMBIENT, owned.scope)
assertTrue(owned.mutable)
assertEquals(20L, owned.bytes)
assertEquals(Ownership.OWNED, StorageSpec.fromDType(BF16).ownership)
}

@Test
fun scopeKindHasTheThreeLifetimes() {
assertEquals(listOf(ScopeKind.MODEL, ScopeKind.FORWARD, ScopeKind.AMBIENT), ScopeKind.entries)
assertNull(ScopeKind.entries.firstOrNull { it.name == "DEVICE" })
}
}
Loading
Loading