Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
52 changes: 51 additions & 1 deletion scripts/check_engine_json.sh
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,43 @@ expected_schema = sys.argv[1]
required_top = {"schema_version","suite","scenario","published_at","runtime","system","config","metrics","samples"}
required_metrics = {"primary_metric","unit","value_mean","value_stddev","value_min","value_max","cov_percent"}
required_runtime = {"name","version","commit","backend","kernel_provider","available_providers"}
# #1035: generation-loop metrics. Optional — a matmul microbenchmark has no decode loop — but a
# record that claims to have run one must carry the numbers a dashboard plots, with plausible
# values. A silently empty or negative field is the failure mode this catches.
required_generation = {"prefill_tokens","decode_steps","ms_per_decode_step","bytes_read","adapter_count","adapter_bytes"}
non_negative_generation = required_generation | {"decode_tokens_per_second","prefill_tokens_per_second",
"ttft_ms","effective_bandwidth_bytes_per_second",
"bandwidth_utilization_percent","kernel_share_of_decode_percent",
"page_faults","page_faults_per_second"}
fail = 0

def check_generation(path, gen):
"""Returns a list of problems with a record's `generation` block."""
problems = []
missing = required_generation - gen.keys()
if missing:
problems.append(f"generation missing {sorted(missing)}")
return problems
for key in sorted(non_negative_generation & gen.keys()):
value = gen[key]
if value is None:
continue
if not isinstance(value, (int, float)) or value < 0:
problems.append(f"generation.{key}={value!r} must be a non-negative number or null")
for key in ("bandwidth_utilization_percent", "kernel_share_of_decode_percent"):
value = gen.get(key)
if isinstance(value, (int, float)) and value > 100.0:
problems.append(f"generation.{key}={value} exceeds 100%")
if gen["decode_steps"] > 0 and gen.get("decode_tokens_per_second") is None and gen["ms_per_decode_step"] == 0:
problems.append("generation: decode steps were recorded but never timed")
breakdown = gen.get("module_breakdown_ms", {})
if not isinstance(breakdown, dict):
problems.append("generation.module_breakdown_ms must be an object of module -> milliseconds")
else:
for module, ms in breakdown.items():
if not isinstance(ms, (int, float)) or ms < 0:
problems.append(f"generation.module_breakdown_ms[{module!r}]={ms!r} must be non-negative")
return problems
for path in sys.argv[2:]:
try:
with open(path) as f:
Expand All @@ -48,6 +84,20 @@ for path in sys.argv[2:]:
rm = required_runtime - rec["runtime"].keys()
if rm:
print(f"FAIL {path}: runtime missing {sorted(rm)}", file=sys.stderr); fail += 1; continue
print(f"OK {path} {rec['scenario']} mean={rec['metrics']['value_mean']:.4f} {rec['metrics']['unit']}")
gen = rec.get("generation")
if gen is not None:
problems = check_generation(path, gen)
if problems:
for p in problems:
print(f"FAIL {path}: {p}", file=sys.stderr)
fail += 1
continue
suffix = ""
if gen is not None:
tok = gen.get("decode_tokens_per_second")
bw = gen.get("effective_bandwidth_bytes_per_second")
suffix = " decode=" + (f"{tok:.2f} tok/s" if isinstance(tok, (int, float)) else "n/a")
suffix += " bw=" + (f"{bw / 1e9:.2f} GB/s" if isinstance(bw, (int, float)) else "n/a")
print(f"OK {path} {rec['scenario']} mean={rec['metrics']['value_mean']:.4f} {rec['metrics']['unit']}{suffix}")
sys.exit(1 if fail else 0)
PY
Original file line number Diff line number Diff line change
Expand Up @@ -17,4 +17,10 @@ public data class BenchmarkRecord(
val metrics: MetricSet,
val samples: List<Double>,
val unstable: Boolean = false,
/**
* Generation-loop metrics (#1035), present only for scenarios that run one — TTFT, tok/s,
* effective bandwidth, page faults, per-module breakdown. `scripts/check_engine_json.sh`
* validates the block whenever it appears.
*/
val generation: GenerationMetricsRecord? = null,
)
Original file line number Diff line number Diff line change
@@ -0,0 +1,69 @@
package sk.ainet.bench.publish.schema

import kotlinx.serialization.SerialName
import kotlinx.serialization.Serializable
import sk.ainet.lang.memory.ExperimentalMemoryApi
import sk.ainet.lang.memory.trace.GenerationMetrics

/**
* The generation-loop half of a benchmark record (SKEEP-003 §4.9, #1035): what a decode run
* reported about itself, in the same JSON the dashboards and the Phoronix upload already read.
*
* Optional on [BenchmarkRecord] — a matmul microbenchmark has no generation loop and omits it —
* and validated by `scripts/check_engine_json.sh` whenever it is present. Rates are nullable for
* the same reason they are nullable on [GenerationMetrics]: a span too short for the platform
* clock produces no rate rather than an infinite one.
*/
@Serializable
public data class GenerationMetricsRecord(
@SerialName("prefill_tokens")
val prefillTokens: Int,
@SerialName("prefill_tokens_per_second")
val prefillTokensPerSecond: Double? = null,
@SerialName("decode_steps")
val decodeSteps: Int,
@SerialName("decode_tokens_per_second")
val decodeTokensPerSecond: Double? = null,
@SerialName("ttft_ms")
val ttftMs: Double? = null,
@SerialName("ms_per_decode_step")
val msPerDecodeStep: Double,
@SerialName("bytes_read")
val bytesRead: Long,
@SerialName("effective_bandwidth_bytes_per_second")
val effectiveBandwidthBytesPerSecond: Double? = null,
@SerialName("bandwidth_utilization_percent")
val bandwidthUtilizationPercent: Double? = null,
@SerialName("kernel_share_of_decode_percent")
val kernelShareOfDecodePercent: Double? = null,
@SerialName("adapter_count")
val adapterCount: Int,
@SerialName("adapter_bytes")
val adapterBytes: Long,
@SerialName("page_faults")
val pageFaults: Long? = null,
@SerialName("page_faults_per_second")
val pageFaultsPerSecond: Double? = null,
@SerialName("module_breakdown_ms")
val moduleBreakdownMs: Map<String, Double> = emptyMap(),
)

/** This run's metrics as the record the benchmark JSON carries. */
@OptIn(ExperimentalMemoryApi::class)
public fun GenerationMetrics.toRecord(): GenerationMetricsRecord = GenerationMetricsRecord(
prefillTokens = prefillTokens,
prefillTokensPerSecond = prefillTokensPerSecond,
decodeSteps = decodeSteps,
decodeTokensPerSecond = decodeTokensPerSecond,
ttftMs = timeToFirstTokenNanos?.let { it / 1_000_000.0 },
msPerDecodeStep = nanosPerDecodeStep / 1_000_000.0,
bytesRead = bytesReadDuringDecode,
effectiveBandwidthBytesPerSecond = effectiveBandwidthBytesPerSecond,
bandwidthUtilizationPercent = bandwidthUtilization?.let { it * 100.0 },
kernelShareOfDecodePercent = kernelShareOfDecode?.let { it * 100.0 },
adapterCount = adapterCount,
adapterBytes = adapterBytes,
pageFaults = pageFaultsDuringDecode,
pageFaultsPerSecond = pageFaultsPerSecond,
moduleBreakdownMs = modules.associate { it.path to it.nanos / 1_000_000.0 },
)
Original file line number Diff line number Diff line change
@@ -0,0 +1,103 @@
package sk.ainet.bench.publish.schema

import kotlinx.serialization.json.Json
import sk.ainet.lang.memory.ExperimentalMemoryApi
import sk.ainet.lang.memory.trace.GenerationMetrics
import sk.ainet.lang.memory.trace.ModuleCost
import java.io.File
import kotlin.test.Test
import kotlin.test.assertEquals
import kotlin.test.assertFalse
import kotlin.test.assertTrue

/**
* #1035: the generation metrics survive the trip into the benchmark JSON that dashboards and the
* Phoronix upload read, under the names `scripts/check_engine_json.sh` validates.
*
* The written fixture is the script's own test input: `./scripts/check_engine_json.sh
* skainet-backends/benchmarks/jvm-cpu-publish/build/engine-json-check` must pass on it.
*/
@OptIn(ExperimentalMemoryApi::class)
class GenerationMetricsRecordTest {

private val json = Json { prettyPrint = true; encodeDefaults = true; explicitNulls = false }

private val metrics = GenerationMetrics(
prefillTokens = 128,
prefillNanos = 320_000_000L,
decodeSteps = 64,
decodeNanos = 1_280_000_000L,
sampleNanos = 6_400_000L,
timeToFirstTokenNanos = 340_000_000L,
bytesReadDuringDecode = 40L * 1024 * 1024 * 1024,
bytesWrittenDuringDecode = 4L * 1024 * 1024,
kernelNanosDuringDecode = 1_024_000_000L,
kernelRunsDuringDecode = 4_096,
adapterCount = 2,
adapterBytes = 1_048_576L,
modules = listOf(ModuleCost("model.layers[0].attn", 400_000_000L, 64), ModuleCost("model.layers[0].mlp", 600_000_000L, 64)),
pageFaultsDuringDecode = 3L,
peakBytesPerSecond = 50L * 1024 * 1024 * 1024,
)

private fun record(generation: GenerationMetricsRecord?) = BenchmarkRecord(
schemaVersion = "1.0.0",
suite = "skainet-engine",
scenario = "decode-synthetic",
publishedAt = "2026-08-24T00:00:00Z",
runtime = RuntimeInfo(
version = "0.40.1", commit = "abcdef0", backend = "cpu",
kernelProvider = "scalar", availableProviders = listOf("scalar"),
),
system = SystemInfo(
os = "linux", arch = "x86_64", cpu = "test", cpuLogicalCores = 8,
memoryGib = 32L, jdk = "25", jdkVendor = "test",
),
config = RunConfig(
warmupRuns = 1, measuredRuns = 3, seed = 1L,
parameters = mapOf("ctx" to "512"), jvmArgs = emptyList(), smokeMode = true,
),
metrics = MetricSet("decode_tokens_per_second", "tok/s", 50.0, 0.5, 49.0, 51.0, 1.0),
samples = listOf(49.0, 50.0, 51.0),
generation = generation,
)

@Test
fun `the metrics map onto the published field names`() {
val rec = metrics.toRecord()
assertEquals(128, rec.prefillTokens)
assertEquals(64, rec.decodeSteps)
assertEquals(50.0, rec.decodeTokensPerSecond!!, 1e-9, "64 steps in 1.28 s")
assertEquals(340.0, rec.ttftMs!!, 1e-9)
assertEquals(20.0, rec.msPerDecodeStep, 1e-9)
assertEquals(62.5, rec.bandwidthUtilizationPercent!!, 1e-6, "31.25 GB/s of a 50 GB/s device")
assertEquals(80.0, rec.kernelShareOfDecodePercent!!, 1e-9)
assertEquals(2, rec.adapterCount)
assertEquals(3L, rec.pageFaults)
assertEquals(setOf("model.layers[0].attn", "model.layers[0].mlp"), rec.moduleBreakdownMs.keys)
assertEquals(600.0, rec.moduleBreakdownMs.getValue("model.layers[0].mlp"), 1e-9)
}

@Test
fun `a record with generation metrics serializes under the names the checker requires`() {
val text = json.encodeToString(BenchmarkRecord.serializer(), record(metrics.toRecord()))
for (key in listOf(
"\"generation\"", "\"prefill_tokens\"", "\"decode_steps\"", "\"ms_per_decode_step\"",
"\"bytes_read\"", "\"adapter_count\"", "\"adapter_bytes\"",
"\"decode_tokens_per_second\"", "\"effective_bandwidth_bytes_per_second\"",
"\"bandwidth_utilization_percent\"", "\"module_breakdown_ms\"", "\"ttft_ms\"",
)) {
assertTrue(text.contains(key), "missing $key in:\n$text")
}
val dir = File("build/engine-json-check").apply { mkdirs() }
File(dir, "decode-synthetic.json").writeText(text)
}

@Test
fun `a scenario without a generation loop omits the block entirely`() {
val text = json.encodeToString(BenchmarkRecord.serializer(), record(null))
assertFalse(text.contains("\"generation\""), "a matmul scenario must not carry an empty generation block")
File("build/engine-json-check").apply { mkdirs() }
File("build/engine-json-check/matmul-only.json").writeText(text)
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,11 @@ import sk.ainet.lang.memory.plan.PlanInput
import sk.ainet.lang.memory.plan.PlanTensor
import sk.ainet.lang.memory.trace.RecordingTraceSink
import sk.ainet.lang.memory.trace.TraceEvent
import sk.ainet.lang.memory.trace.decodeStep
import sk.ainet.lang.memory.trace.module
import sk.ainet.lang.memory.trace.phase
import sk.ainet.lang.memory.trace.prefill
import sk.ainet.lang.memory.trace.sample
import sk.ainet.lang.tensor.Shape
import sk.ainet.lang.tensor.TensorId
import sk.ainet.lang.tensor.data.Q8_0BlockTensorData
Expand Down Expand Up @@ -105,28 +109,51 @@ public class DecodeHarness(
return MemoryPlans.plan(PlanInput("harness", "llama", tensors, geometry, ctx, prefillChunk = 1, kvMode = KvCacheMode.FP32))
}

/**
* The prompt pass: [tokens] positions through the same stack, inside a `prefill` span so
* [metrics] can price it (#1035). Deliberately the same work as a decode step — the harness is
* about memory behaviour, not about being a fast prefill.
*/
public fun prefill(tokens: Int) {
sink.prefill(tokens) {
repeat(tokens) { runStack(step = 0) }
}
}

/** Run [steps] decode steps; each allocates activations, runs the stack and resets the scope. */
public fun decode(steps: Int) {
val x = FloatArray(hidden) { (it % 7) * 0.125f }
for (step in 1..steps) {
sink.phase("decode", step) {
val act = forward.allocateFloats(hidden, TensorId(listOf("model"), "hidden", "step=$step"))
x.copyInto(act.floats!!, act.arrayOffset)
val actView = TensorView.dense(act, Shape(1, hidden), FP32, TensorId(listOf("model"), "hidden", "step=$step"))
for (w in weights) {
if (w.shape[1] != hidden) continue
val out = forward.allocateFloats(w.shape[0], TensorId(listOf("model"), "proj", "step=$step"))
val outView = TensorView.dense(out, Shape(1, w.shape[0]), FP32)
KernelDispatch.matmul(actView, w, outView, forward, sink)
}
sink.decodeStep(step) {
runStack(step)
// one token into the KV ring (all layers), as a decode step does
val k = FloatArray(kvHeads * (hidden / heads)) { 0.5f }
if (kv.currentSeqLen < ctx) for (l in 0 until layers) kv.appendToken(l, k, k)
forward.reset()
}
sink.sample(step) { /* argmax over a synthetic logit row: nothing to allocate */ }
}
}

/** One pass over the weight stack, each weight timed as its own module span. */
private fun runStack(step: Int) {
val x = FloatArray(hidden) { (it % 7) * 0.125f }
val act = forward.allocateFloats(hidden, TensorId(listOf("model"), "hidden", "step=$step"))
x.copyInto(act.floats!!, act.arrayOffset)
val actView = TensorView.dense(act, Shape(1, hidden), FP32, TensorId(listOf("model"), "hidden", "step=$step"))
for (w in weights) {
if (w.shape[1] != hidden) continue
sink.module(w.id!!, step) {
val out = forward.allocateFloats(w.shape[0], TensorId(listOf("model"), "proj", "step=$step"))
val outView = TensorView.dense(out, Shape(1, w.shape[0]), FP32)
KernelDispatch.matmul(actView, w, outView, forward, sink)
}
}
}

/** The generation metrics this run produced (#1035); [peakBytesPerSecond] enables utilization. */
public fun metrics(peakBytesPerSecond: Long? = null): sk.ainet.lang.memory.trace.GenerationMetrics =
sk.ainet.lang.memory.trace.GenerationMetrics.from(sink, peakBytesPerSecond)

/** Live bytes per scope as the event stream saw them after the last step. */
public fun liveBytes(): Map<ScopeKind, Long> {
val live = HashMap<ScopeKind, Long>()
Expand Down
Loading
Loading