Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,55 @@
package sk.ainet.sk.ainet.exec.tensor.ops

import sk.ainet.context.DirectCpuExecutionContext
import sk.ainet.context.forwardScope
import sk.ainet.lang.memory.ExperimentalMemoryApi
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.data.StorageFloatTensorData
import sk.ainet.lang.types.FP32
import kotlin.test.Test
import kotlin.test.assertContentEquals
import kotlin.test.assertEquals
import kotlin.test.assertTrue

/**
* #1145 guard for the offset-0 trap: a slab-backed tensor has a nonzero `arrayOffset`, and a
* fast path that grabbed the raw buffer would read the whole slab. `StorageFloatTensorData`
* deliberately stays off `FloatArrayTensorData`, so real CPU ops must produce numbers identical
* to the Ambient path — for tensors sliced from anywhere in the slab, across resets.
*/
@OptIn(ExperimentalMemoryApi::class)
class ScopedCreationOpsParityTest {

@Test
fun opsOnSlabBackedTensorsMatchAmbient() {
val ctx = DirectCpuExecutionContext()
val aVals = FloatArray(6) { (it + 1).toFloat() } // 2×3
val bVals = FloatArray(12) { (it % 5 - 2).toFloat() } // 3×4

val ambientMatmul: FloatArray
val ambientAdd: FloatArray
run {
val a = ctx.fromFloatArray<FP32, Float>(Shape(2, 3), FP32::class, aVals)
val b = ctx.fromFloatArray<FP32, Float>(Shape(3, 4), FP32::class, bVals)
ambientMatmul = ctx.ops.matmul(a, b).data.copyToFloatArray()
ambientAdd = ctx.ops.add(a, a).data.copyToFloatArray()
}

ctx.forwardScope(slabFloats = 64) { scoped, scope ->
repeat(3) { step ->
// A leading allocation pushes the later tensors deeper into the slab, so the
// offsets under test are nonzero and different from the previous step's layout.
scoped.zeros<FP32, Float>(Shape(1 + step), FP32::class)
val a = scoped.fromFloatArray<FP32, Float>(Shape(2, 3), FP32::class, aVals)
val b = scoped.fromFloatArray<FP32, Float>(Shape(3, 4), FP32::class, bVals)
assertTrue(a.data is StorageFloatTensorData<*>, "step $step: creation must draw from the slab")
assertTrue((a.data as StorageFloatTensorData<*>).storage.arrayOffset > 0, "offset under test must be nonzero")

assertContentEquals(ambientMatmul, ctx.ops.matmul(a, b).data.copyToFloatArray(), "step $step: matmul")
assertContentEquals(ambientAdd, ctx.ops.add(a, a).data.copyToFloatArray(), "step $step: add")
assertEquals(Shape(2, 4), ctx.ops.matmul(a, b).shape)
scope.reset()
}
}
}
}
47 changes: 47 additions & 0 deletions skainet-lang/skainet-lang-core/api/jvm/skainet-lang-core.api
Original file line number Diff line number Diff line change
Expand Up @@ -396,6 +396,40 @@ public final class sk/ainet/context/ResettableExecutionObserver$DefaultImpls {
public static fun onTensorMaterialized (Lsk/ainet/context/ResettableExecutionObserver;Lsk/ainet/context/ExecutionContext;Lsk/ainet/lang/tensor/Tensor;)V
}

public final class sk/ainet/context/ScopedExecutionContext : sk/ainet/context/ExecutionContext {
public fun <init> (Lsk/ainet/context/ExecutionContext;Lsk/ainet/lang/memory/Scope;)V
public fun fromByteArray (Lsk/ainet/lang/tensor/Shape;Lkotlin/reflect/KClass;[B)Lsk/ainet/lang/tensor/Tensor;
public fun fromData (Lsk/ainet/lang/tensor/data/TensorData;Lkotlin/reflect/KClass;)Lsk/ainet/lang/tensor/Tensor;
public fun fromFloatArray (Lsk/ainet/lang/tensor/Shape;Lkotlin/reflect/KClass;[F)Lsk/ainet/lang/tensor/Tensor;
public fun fromIntArray (Lsk/ainet/lang/tensor/Shape;Lkotlin/reflect/KClass;[I)Lsk/ainet/lang/tensor/Tensor;
public fun full (Lsk/ainet/lang/tensor/Shape;Lkotlin/reflect/KClass;Ljava/lang/Number;)Lsk/ainet/lang/tensor/Tensor;
public fun getExecutionStats ()Lsk/ainet/context/ExecutionStats;
public fun getHooks ()Lsk/ainet/lang/nn/hooks/ForwardHooks;
public fun getInTraining ()Z
public fun getMemoryInfo ()Lsk/ainet/context/MemoryInfo;
public fun getMemoryScope ()Lsk/ainet/lang/memory/Scope;
public fun getMemoryTracker ()Lsk/ainet/lang/tensor/storage/MemoryTracker;
public fun getObservers ()Lsk/ainet/context/ExecutionObserverRegistry;
public fun getOps ()Lsk/ainet/lang/tensor/ops/TensorOps;
public fun getPhase ()Lsk/ainet/context/Phase;
public fun getScratch ()Lsk/ainet/lang/tensor/scratch/ScratchPool;
public fun getTensorDataFactory ()Lsk/ainet/lang/tensor/data/TensorDataFactory;
public fun getTraceSink ()Lsk/ainet/lang/memory/trace/TraceSink;
public fun isRecording ()Z
public fun ones (Lsk/ainet/lang/tensor/Shape;Lkotlin/reflect/KClass;)Lsk/ainet/lang/tensor/Tensor;
public fun placeholder (Lsk/ainet/lang/tensor/Shape;Lkotlin/reflect/KClass;)Lsk/ainet/lang/tensor/Tensor;
public fun registerObserver (Lsk/ainet/context/ExecutionObserver;)V
public fun unregisterObserver (Lsk/ainet/context/ExecutionObserver;)V
public fun wrapByteArray (Lsk/ainet/lang/tensor/Shape;Lkotlin/reflect/KClass;[B)Lsk/ainet/lang/tensor/Tensor;
public fun wrapFloatArray (Lsk/ainet/lang/tensor/Shape;Lkotlin/reflect/KClass;[F)Lsk/ainet/lang/tensor/Tensor;
public fun wrapIntArray (Lsk/ainet/lang/tensor/Shape;Lkotlin/reflect/KClass;[I)Lsk/ainet/lang/tensor/Tensor;
public fun zeros (Lsk/ainet/lang/tensor/Shape;Lkotlin/reflect/KClass;)Lsk/ainet/lang/tensor/Tensor;
}

public final class sk/ainet/context/ScopedExecutionContextKt {
public static final fun forwardScope (Lsk/ainet/context/ExecutionContext;ILkotlin/jvm/functions/Function2;)Ljava/lang/Object;
}

public abstract interface class sk/ainet/context/TrainingExecutionContext : sk/ainet/context/ExecutionContext {
public abstract fun backward (Ljava/util/List;Ljava/util/List;)V
public abstract fun startRecording ()V
Expand Down Expand Up @@ -5361,6 +5395,19 @@ public abstract interface class sk/ainet/lang/tensor/data/RowDequantSource {
public abstract fun dequantRow (I)[F
}

public final class sk/ainet/lang/tensor/data/StorageFloatTensorData : sk/ainet/lang/tensor/data/TensorData {
public fun <init> (Lsk/ainet/lang/tensor/Shape;Lsk/ainet/lang/memory/Storage$Heap;)V
public fun copyToFloatArray ()[F
public fun get ([I)Ljava/lang/Float;
public synthetic fun get ([I)Ljava/lang/Object;
public fun getEncoding ()Lsk/ainet/lang/tensor/storage/TensorEncoding;
public fun getShape ()Lsk/ainet/lang/tensor/Shape;
public final fun getStorage ()Lsk/ainet/lang/memory/Storage$Heap;
public fun getView ()Lsk/ainet/lang/memory/TensorView;
public fun set ([IF)V
public synthetic fun set ([ILjava/lang/Object;)V
}

public abstract interface class sk/ainet/lang/tensor/data/TensorData : sk/ainet/lang/tensor/data/ItemsAccessor {
public fun copyToFloatArray ()[F
public fun getEncoding ()Lsk/ainet/lang/tensor/storage/TensorEncoding;
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,12 @@ public interface ExecutionContext {
* Callers MUST acquire inside an active [ScratchPool.scope] block;
* acquires outside a scope succeed but the buffer is not returned to the
* pool when dropped.
*
* Boundary with [memoryScope], on purpose (#1135): `scratch` is untyped *intra-kernel*
* workspace — raw arrays inside one op invocation, returned when its block exits.
* [memoryScope] governs *inter-op* activation lifetime — typed tensors that live across ops
* within a step and are recycled by `ForwardScope.reset()`. They are different layers and
* stay separate; neither replaces the other.
*/
public val scratch: ScratchPool get() = NoopScratchPool

Expand All @@ -74,16 +80,46 @@ public interface ExecutionContext {
observers.unregister(observer)
}

/**
* Dense FP32 data drawn from [memoryScope] when a scope other than `Ambient` is active — the
* creation-path reader of the Scope split (#1145). `null` on the Ambient default, so the
* factory path is untouched for every context that never opts in. The region is *not* cleared:
* a slab slice after `reset()` holds old bytes, so callers fill it themselves.
*/
@sk.ainet.lang.memory.ExperimentalMemoryApi
private fun <T : DType> scopedDenseFloats(
shape: Shape,
dtype: KClass<T>,
): sk.ainet.lang.tensor.data.StorageFloatTensorData<T>? {
val scope = memoryScope
if (scope === sk.ainet.lang.memory.Scope.Ambient || dtype != sk.ainet.lang.types.FP32::class) return null
return sk.ainet.lang.tensor.data.StorageFloatTensorData(shape, scope.allocateFloats(shape.volume))
}

@OptIn(sk.ainet.lang.memory.ExperimentalMemoryApi::class)
public fun <T : DType, V> full(shape: Shape, dtype: KClass<T>, value: Number): Tensor<T, V> {
scopedDenseFloats(shape, dtype)?.let { scoped ->
val s = scoped.storage
s.floats!!.fill(value.toFloat(), s.arrayOffset, s.arrayOffset + shape.volume)
@Suppress("UNCHECKED_CAST")
return fromData(scoped as TensorData<T, V>, dtype)
}
val data = tensorDataFactory.full<T, V>(shape, dtype, value)
return fromData(data, dtype)
}


@OptIn(sk.ainet.lang.memory.ExperimentalMemoryApi::class)
public fun <T : DType, V> zeros(
shape: Shape,
dtype: KClass<T>
): Tensor<T, V> {
scopedDenseFloats(shape, dtype)?.let { scoped ->
val s = scoped.storage
s.floats!!.fill(0f, s.arrayOffset, s.arrayOffset + shape.volume)
@Suppress("UNCHECKED_CAST")
return fromData(scoped as TensorData<T, V>, dtype)
}
val data = tensorDataFactory.zeros<T, V>(shape, dtype)
return fromData(data, dtype)
}
Expand All @@ -102,10 +138,12 @@ public interface ExecutionContext {
return fromData(data, dtype)
}

@OptIn(sk.ainet.lang.memory.ExperimentalMemoryApi::class)
public fun <T : DType, V> ones(
shape: Shape,
dtype: KClass<T>
): Tensor<T, V> {
if (memoryScope !== sk.ainet.lang.memory.Scope.Ambient) return full(shape, dtype, 1)
val data = tensorDataFactory.ones<T, V>(shape, dtype)
return fromData(data, dtype)
}
Expand All @@ -116,11 +154,18 @@ public interface ExecutionContext {
ops
)

@OptIn(sk.ainet.lang.memory.ExperimentalMemoryApi::class)
public fun <T : DType, V> fromFloatArray(
shape: Shape,
dtype: KClass<T>,
data: FloatArray
): Tensor<T, V> {
scopedDenseFloats(shape, dtype)?.let { scoped ->
val s = scoped.storage
data.copyInto(s.floats!!, s.arrayOffset, 0, shape.volume)
@Suppress("UNCHECKED_CAST")
return fromData(scoped as TensorData<T, V>, dtype)
}
val data = tensorDataFactory.fromFloatArray<T, V>(shape, dtype, data)
return fromData(data, dtype)
}
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,64 @@
package sk.ainet.context

import sk.ainet.lang.memory.ExperimentalMemoryApi
import sk.ainet.lang.memory.ForwardScope
import sk.ainet.lang.memory.Scope
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.Tensor
import sk.ainet.lang.types.DType
import kotlin.reflect.KClass

/**
* [base] with a [memoryScope] — the wiring #1135 asked about (#1145).
*
* The mechanism existed before this class did: [ForwardScope] bump-allocates a pre-sized slab and
* `reset()` recycles it per step; what was missing was any reader of `ExecutionContext.memoryScope`.
* This decorator is that reader's other half: creation methods on [ExecutionContext] consult
* `memoryScope` when it is not [Scope.Ambient], so `zeros`/`full`/`ones`/`fromFloatArray` draw
* from the scope's slab and their tensors die at `reset()`.
*
* `Ambient` remains the default everywhere — nothing changes for a context that never opts in.
* The boundary with [ExecutionContext.scratch] is deliberate: `scratch` is untyped *intra-kernel*
* workspace inside one op invocation; `memoryScope` governs *inter-op* activation lifetime across
* a step. They stay separate.
*/
@ExperimentalMemoryApi
public class ScopedExecutionContext(
private val base: ExecutionContext,
override val memoryScope: Scope,
) : ExecutionContext by base {

override fun <T : DType, V> zeros(shape: Shape, dtype: KClass<T>): Tensor<T, V> =
super.zeros(shape, dtype)

override fun <T : DType, V> ones(shape: Shape, dtype: KClass<T>): Tensor<T, V> =
super.ones(shape, dtype)

override fun <T : DType, V> full(shape: Shape, dtype: KClass<T>, value: Number): Tensor<T, V> =
super.full(shape, dtype, value)

override fun <T : DType, V> fromFloatArray(shape: Shape, dtype: KClass<T>, data: FloatArray): Tensor<T, V> =
super.fromFloatArray(shape, dtype, data)
}

/**
* Run [block] with a [ForwardScope] of [slabFloats] floats active on this context, closing the
* scope (and everything it handed out) afterwards. Call `reset()` on the scope between steps:
*
* ```kotlin
* ctx.forwardScope(slabFloats = 1 shl 20) { scoped, scope ->
* while (decoding) {
* step(scoped)
* scope.reset() // steady state: zero new slab bytes per step
* }
* }
* ```
*/
@ExperimentalMemoryApi
public inline fun <R> ExecutionContext.forwardScope(
slabFloats: Int,
block: (ctx: ScopedExecutionContext, scope: ForwardScope) -> R,
): R {
val scope = ForwardScope(slabFloats, traceSink)
return scope.use { block(ScopedExecutionContext(this, scope), scope) }
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,78 @@
package sk.ainet.lang.tensor.data

import sk.ainet.lang.memory.ExperimentalMemoryApi
import sk.ainet.lang.memory.Storage
import sk.ainet.lang.memory.TensorView
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.types.DType
import sk.ainet.lang.types.FP32

/**
* Dense FP32 tensor data over a [Storage.Heap] region — the creation-path end of the Scope split
* (#1145): what `ExecutionContext.zeros/full/fromFloatArray` hand out when a [sk.ainet.lang.memory.Scope]
* other than `Ambient` is active, so an activation's bytes come from the forward slab and die at
* `reset()` instead of waiting for the GC.
*
* Every access goes through [Storage.checkAlive], so a use-after-reset is a
* [sk.ainet.lang.memory.StorageClosedException] naming the storage — not silent corruption.
*
* **Deliberately not a [FloatArrayTensorData].** A slab slice has a nonzero
* [Storage.Heap.arrayOffset], and the ops fast paths that unwrap `buffer` assume offset 0; exposing
* this data through that interface would hand them the whole slab. Element access and
* [copyToFloatArray] are offset-correct; kernels that want zero-copy take [view], which carries the
* offset properly.
*/
@ExperimentalMemoryApi
public class StorageFloatTensorData<T : DType>(
initialShape: Shape,
public val storage: Storage.Heap,
) : TensorData<T, Float> {

override val shape: Shape = Shape(initialShape.dimensions.copyOf())
private val strides: IntArray = shape.computeStrides()

init {
requireNotNull(storage.floats) { "StorageFloatTensorData needs float-backed storage" }
require(storage.elementCount >= shape.volume) {
"storage holds ${storage.elementCount} floats, shape $shape needs ${shape.volume}"
}
}

private fun flatIndex(indices: IntArray): Int {
require(indices.size == shape.dimensions.size) {
"Number of indices (${indices.size}) must match tensor dimensions (${shape.dimensions.size})"
}
var flat = 0
for (i in indices.indices) {
val idx = indices[i]
require(idx >= 0 && idx < shape.dimensions[i]) {
"Index $idx out of bounds for dimension $i with size ${shape.dimensions[i]}"
}
flat += idx * strides[i]
}
return flat
}

override fun get(vararg indices: Int): Float {
storage.checkAlive()
return storage.floats!![storage.arrayOffset + flatIndex(indices)]
}

override fun set(vararg indices: Int, value: Float) {
storage.checkAlive()
storage.floats!![storage.arrayOffset + flatIndex(indices)] = value
}

override fun copyToFloatArray(): FloatArray {
storage.checkAlive()
val off = storage.arrayOffset
return storage.floats!!.copyOfRange(off, off + shape.volume)
}

/** A dense view over the same region — the storage carries the offset, nothing is copied. */
override val view: TensorView
get() {
storage.checkAlive()
return TensorView.dense(storage, shape, FP32)
}
}
Loading
Loading