Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@ set(SKAINET_KERNEL_SOURCES
${SKAINET_KERNELS_ROOT}/src/q5_0_matmul.c
${SKAINET_KERNELS_ROOT}/src/q5_1_matmul.c
${SKAINET_KERNELS_ROOT}/src/q8_0_matmul.c
${SKAINET_KERNELS_ROOT}/src/skainet_row_threads.c
${SKAINET_KERNELS_ROOT}/src/q4k_matmul.c
${SKAINET_KERNELS_ROOT}/src/q5k_matmul.c
${SKAINET_KERNELS_ROOT}/src/q6k_matmul.c
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -105,6 +105,10 @@ class M2A5DeviceMeasurement {
val prepack = args.getString("prepack")?.toBoolean() ?: false
val steps = arg("steps", 16)
val warmup = arg("warmup", 4)
// `residency=heap` (default mapped) restores heap staging — the #1193 A/B lever for
// models that fit the cap. The plan below prices the same form the load uses (#1190).
val residency = if (args.getString("residency") == "heap") WeightResidency.HEAP else WeightResidency.MAPPED
val loadForm = WeightForm(shape = WeightShapeOrientation.OUT_IN, residency = residency)

val report = StringBuilder()
fun line(s: String = "") { report.append(s).append('\n') }
Expand All @@ -116,13 +120,10 @@ class M2A5DeviceMeasurement {
line("- device: ${Build.MANUFACTURER} ${Build.MODEL}, Android ${Build.VERSION.RELEASE} (SDK ${Build.VERSION.SDK_INT}), ABI ${Build.SUPPORTED_ABIS.firstOrNull()}")
line("- ART heap cap (Runtime.maxMemory): ${mb(heapCap)}")
line("- model: ${modelFile.name}, ${mb(modelFile.length())} on disk")
line("- ctx=$ctxLen, decode steps=$steps (warm-up $warmup), prepack=$prepack")
line("- ctx=$ctxLen, decode steps=$steps (warm-up $warmup), prepack=$prepack, residency=${residency.name.lowercase()}")
line()

// ---- plan, from the header only --------------------------------------------------
// The plan gets the same WeightForm the load below uses, so mapped-servable weights are
// budgeted against the page cache instead of the heap cap (#1189).
val loadForm = WeightForm(shape = WeightShapeOrientation.OUT_IN, residency = WeightResidency.MAPPED)
val (plan, geometry) = MappedRandomAccessSource.open(modelPath).let { src ->
StreamingGGUFReader.open(src).use { reader ->
val input = reader.planInput(ctx = ctxLen, formFor = { loadForm })
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@ endif()
set(SKAINET_KERNEL_SOURCES
src/skainet_smoke.c
src/skainet_cpu_features.c
src/skainet_row_threads.c
src/q4k_matmul.c
src/q5k_matmul.c
src/q6k_matmul.c
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -48,6 +48,11 @@ SKAINET_API void skainet_smoke_double(const float* input, float* output, int32_t
*
* Caller owns input/weight/output memory; the kernel does not retain
* pointers past return. input_dim must be a multiple of 256.
*
* Threads over output rows (disjoint out[] slices, pthreads, up to 4)
* when output_dim >= 512 (#1195); single-threaded below that and on
* MSVC. Results are bit-identical either way — per output row the
* accumulation order over blocks never changes.
*/
SKAINET_API void skainet_q4k_matmul(
const float* input,
Expand All @@ -68,6 +73,9 @@ SKAINET_API void skainet_q4k_matmul(
* i.e. the bytes exactly as they sit in a .gguf file — which is what lets
* mmap'd weights be fed to this kernel with no relayout copy (each row's
* blocks are read strictly sequentially).
*
* Threads over output rows when output_dim >= 512 (#1195) — see
* skainet_q4k_matmul; bit-identical to the single-threaded result.
*/
SKAINET_API void skainet_q4k_matmul_rm(
const float* input,
Expand Down Expand Up @@ -120,6 +128,9 @@ SKAINET_API void skainet_q5k_matmul(
* weight + weight_byte_offset + (block_idx * output_dim + o) * 210
*
* input_dim must be a multiple of 256.
*
* Threads over output rows when output_dim >= 512 (#1195) — see
* skainet_q4k_matmul; bit-identical to the single-threaded result.
*/
SKAINET_API void skainet_q6k_matmul(
const float* input,
Expand All @@ -139,6 +150,9 @@ SKAINET_API void skainet_q6k_matmul(
* weight + weight_byte_offset + (o * blocks_per_row + block_idx) * 210
* — the bytes exactly as they sit in a .gguf file, so an mmap'd weight
* needs no relayout copy.
*
* Threads over output rows when output_dim >= 512 (#1195) — see
* skainet_q4k_matmul; bit-identical to the single-threaded result.
*/
SKAINET_API void skainet_q6k_matmul_rm(
const float* input,
Expand Down
Loading
Loading