Kev-0.8B Core AI: Metal-kernel graph, 128 tokens per call (K128) replaces the unrolled S=16 bundle; card v2
Browse files- .gitattributes +2 -0
- README.md +213 -153
- SHA256SUMS +9 -9
- gpu-pipelined/{kev_0_8b_decode_fp16_pf16 → kev_0_8b_decode_fp16_metal_pf128}/head/head.safetensors +0 -0
- gpu-pipelined/{kev_0_8b_decode_fp16_pf16 → kev_0_8b_decode_fp16_metal_pf128}/head/kev_head.json +0 -0
- gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128/kev_0_8b_decode_fp16_metal_pf128.aimodel/main.hash +1 -0
- gpu-pipelined/{kev_0_8b_decode_fp16_pf16/kev_0_8b_decode_fp16_pf16.aimodel → kev_0_8b_decode_fp16_metal_pf128/kev_0_8b_decode_fp16_metal_pf128.aimodel}/main.mlirb +2 -2
- gpu-pipelined/{kev_0_8b_decode_fp16_pf16/kev_0_8b_decode_fp16_pf16.aimodel → kev_0_8b_decode_fp16_metal_pf128/kev_0_8b_decode_fp16_metal_pf128.aimodel}/metadata.json +1 -1
- gpu-pipelined/{kev_0_8b_decode_fp16_pf16 → kev_0_8b_decode_fp16_metal_pf128}/metadata.json +94 -93
- gpu-pipelined/{kev_0_8b_decode_fp16_pf16 → kev_0_8b_decode_fp16_metal_pf128}/tokenizer/tokenizer.json +0 -0
- gpu-pipelined/{kev_0_8b_decode_fp16_pf16 → kev_0_8b_decode_fp16_metal_pf128}/tokenizer/tokenizer_config.json +0 -0
- gpu-pipelined/kev_0_8b_decode_fp16_pf16/kev_0_8b_decode_fp16_pf16.aimodel/main.hash +0 -1
.gitattributes
CHANGED
|
@@ -35,3 +35,5 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
gpu-pipelined/kev_0_8b_decode_fp16_pf16/kev_0_8b_decode_fp16_pf16.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
|
| 37 |
gpu-pipelined/kev_0_8b_decode_fp16_pf16/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
gpu-pipelined/kev_0_8b_decode_fp16_pf16/kev_0_8b_decode_fp16_pf16.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
|
| 37 |
gpu-pipelined/kev_0_8b_decode_fp16_pf16/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128/kev_0_8b_decode_fp16_metal_pf128.aimodel/main.mlirb filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128/tokenizer/tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
README.md
CHANGED
|
@@ -24,16 +24,16 @@ Apple Core AI (`.aimodel`) conversion of [jaredpalmer/kev-0.8b](https://huggingf
|
|
| 24 |
|
| 25 |
| Variant | Path | Size | Requires | Tested on |
|
| 26 |
|---|---|---:|---|---|
|
| 27 |
-
| Decoder, fp16, with the host's pointer head | `gpu-pipelined/
|
| 28 |
|
| 29 |
-
One question of 94 tokens takes
|
| 30 |
|
| 31 |
```swift
|
| 32 |
import Kev
|
| 33 |
|
| 34 |
-
let bundle = URL(filePath: "Kev-0.8B-CoreAI/gpu-pipelined/
|
| 35 |
let kev = try await KevDecider(bundle: bundle) // asset: nil = the .aimodel, specialized here (GPU, frequent reshapes)
|
| 36 |
-
let response = try await kev.decide(requestJSON: requestData, shared: true) // shared: the state's
|
| 37 |
print(PythonFormat.dumps(response, asciiOnly: false))
|
| 38 |
```
|
| 39 |
|
|
@@ -56,14 +56,14 @@ option's closing token against the question's last token. The author's card repo
|
|
| 56 |
results; none of them are re-measured here.
|
| 57 |
|
| 58 |
This port merges the adapter into the base with the author's own merge script and exports the text
|
| 59 |
-
backbone as one Core AI graph. The graph takes token ids
|
| 60 |
-
|
| 61 |
-
|
| 62 |
-
|
| 63 |
-
130.
|
| 64 |
|
| 65 |
-
It runs on the Mac GPU and on an iPhone 18 Pro. On
|
| 66 |
-
|
| 67 |
|
| 68 |
## Readout contract
|
| 69 |
|
|
@@ -96,19 +96,19 @@ row form (`rows_of`), with the base model's tokenizer:
|
|
| 96 |
`(p_max − 1/K) / (1 − 1/K)` and every p; `score` → Σ level · p, the legend, every p and a confidence
|
| 97 |
`max(0, 1 − E|level − mode| / D)`; values rounded to 4 decimals. `usage` = `{input_tokens,
|
| 98 |
output_tokens}` with the author's counting (the packed request; the tokens of `json.dumps(answers)`).
|
| 99 |
-
- Length: a row holds at most **
|
| 100 |
-
position bound
|
| 101 |
-
before any call.
|
| 102 |
- Shared prefix (optional, exact): every row of a request starts with the same state tokens, so the
|
| 103 |
-
|
| 104 |
-
from there. Every row's hidden state equals its direct run bit for bit (below).
|
| 105 |
|
| 106 |
`conversion/kev/oracle_kev.py` runs the author's package unchanged (`Checkpoint.load("cpu", fp32)`,
|
| 107 |
`kev.model.admit`, the row-form `forward`, one CPU thread) and records the reference:
|
| 108 |
[`fixtures-kev-0.8b.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/fixtures-kev-0.8b.json). The fixture is 384 records and 434 questions:
|
| 109 |
-
|
| 110 |
-
tag `kev-1.0`; MMLU, 4 options),
|
| 111 |
-
|
| 112 |
written for the port (70 questions: `noul`, `choice` with 2–10 options, `score` with 2–7 levels,
|
| 113 |
JSON states, three states of 1,431–1,746 tokens, one record of 8 questions). The longest row is 1,802
|
| 114 |
tokens. Fourteen questions have an oracle top-2 margin of 0.02 or less (near-ties) and are listed apart.
|
|
@@ -119,27 +119,31 @@ The held-out set is the next 130 transfer-v4 records (mmlu 61–100, records 21
|
|
| 119 |
|
| 120 |
`conversion/kev/qwen3_5_kev_decoder.py`: the overlay's stateful Qwen3.5 text decoder
|
| 121 |
(`Qwen3_5StatefulForCausalLM`) on the merged weights, the vocabulary head replaced by the identity, the
|
| 122 |
-
output the final-norm hidden state at every position. One function, `main`, at a static
|
| 123 |
-
Gated DeltaNet
|
|
|
|
|
|
|
| 124 |
|
| 125 |
| | name | shape, type |
|
| 126 |
|---|---|---|
|
| 127 |
-
| inputs | `input_ids` | [1,
|
| 128 |
| | `position_ids` | [1, seq] int32, the ramp 0 .. seq − 1 |
|
| 129 |
| states | `keyCache`, `valueCache` | [6, 1, 2, ctx, 256] fp16, ctx up to 4,096 |
|
| 130 |
| | `convState`, `recState` | [18, 1, 6144, 3], [18, 1, 16, 128, 128] fp16 |
|
| 131 |
-
| output | `hidden` | [1,
|
| 132 |
|
| 133 |
-
Per row of T ids, from zeroed states: ⌈T /
|
| 134 |
-
0..
|
| 135 |
rows of every call, cut to T, are the backbone's final-norm `last_hidden_state [T, 1024]`. The bundle's
|
| 136 |
-
`metadata.json` says `kind: decision-backbone` and
|
| 137 |
-
the render rules, the head formula and the response shape.
|
| 138 |
|
| 139 |
**Host.** Builds the rows, runs the calls, converts the hidden rows at the readout positions to float64,
|
| 140 |
applies the head (`head/head.safetensors`, `head/kev_head.json`) and the per-question softmax in float64,
|
| 141 |
rounds p to fp32 once, and writes the response. `conversion/kev/decide.py` is the Python reference;
|
| 142 |
-
[`apps/Kev`](https://github.com/john-rocky/coreai-model-zoo/tree/main/apps/Kev) is the Swift one.
|
|
|
|
|
|
|
| 143 |
|
| 144 |
No engine is involved: the hosts drive the low-level runtime (`AIModel` + `loadFunction`, four zeroed
|
| 145 |
states per row). No runtime patch.
|
|
@@ -148,10 +152,10 @@ states per row). No runtime patch.
|
|
| 148 |
|
| 149 |
The bar, fixed before any graph ran: the argmax equal to the oracle's on every question whose oracle
|
| 150 |
top-2 margin is above 0.02 (near-ties listed apart), max |Δp| ≤ 0.02 over every option of every
|
| 151 |
-
question, the mean over rows of each row's mean |Δp| ≤ 0.002, and every process re-running its
|
| 152 |
bit for bit. The Python gates load the AOT `.aimodelc` (`coreai-build compile … --platform macOS
|
| 153 |
--preferred-compute gpu --architecture h16c --expect-frequent-reshapes`) with
|
| 154 |
-
`SpecializationOptions.default()`
|
| 155 |
|
| 156 |
### Before any graph: the merge and fp32 torch
|
| 157 |
|
|
@@ -161,11 +165,12 @@ bit for bit. The Python gates load the AOT `.aimodelc` (`coreai-build compile
|
|
| 161 |
merged tensors bit for bit (the adapter moves them by 2.8–3.6 %). The base alone moves p by up to
|
| 162 |
0.789 (argmax 25/53), the base's Gated DeltaNet projections put back by up to 0.282 (46/53): the
|
| 163 |
check can fail ([`gate-kev-0.8b-merge.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-merge.json)).
|
| 164 |
-
- The decoder module in fp32 on the CPU, driven like the graph
|
| 165 |
-
argmax (the 14 near-ties included), max |Δp| 3.5e-6, lowest per-position
|
| 166 |
-
on the 21 rows the oracle keeps. It equals the overlay's plain text decoder
|
| 167 |
-
|
| 168 |
-
|
|
|
|
| 169 |
|
| 170 |
### The graph alone on the Mac GPU
|
| 171 |
|
|
@@ -173,39 +178,60 @@ The oracle's ids in, the fp16 hidden rows read through the author's fp32 head:
|
|
| 173 |
|
| 174 |
| set | questions | argmax (margin > 0.02) | near-ties agreeing | max \|Δp\| | mean of row means | bar |
|
| 175 |
|---|---:|---:|---:|---:|---:|---|
|
| 176 |
-
| **fixture** | 434 | **420/420** | **
|
| 177 |
-
| held out | 130 | 125/125 | 5/5 | 0.
|
| 178 |
|
| 179 |
-
The
|
| 180 |
-
|
| 181 |
-
|
| 182 |
-
|
| 183 |
-
|
| 184 |
-
[`gate-kev-0.8b-readout-fp16_pf16.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-readout-fp16_pf16.json),
|
| 185 |
-
[`gate-kev-0.8b-heldout.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-heldout.json).
|
| 186 |
|
| 187 |
### From the request: Python and Swift
|
| 188 |
|
| 189 |
- `conversion/kev/host.py`, without the author's package, rebuilds every oracle row from the raw
|
| 190 |
-
request (ids, `<|fim_suffix|>` / `<|box_end|>` indices, keys):
|
| 191 |
-
implementations
|
| 192 |
-
|
| 193 |
-
|
| 194 |
-
|
| 195 |
-
|
| 196 |
-
|
| 197 |
-
|
| 198 |
-
|
| 199 |
-
|
| 200 |
-
word changed in one question turns its record red.
|
| 201 |
- JIT: the `.aimodel` specialized by the Swift runtime (GPU preferred, `expectFrequentReshapes`) equals
|
| 202 |
-
the AOT asset bit for bit on 25 rows of 10 records
|
| 203 |
-
|
| 204 |
-
- Rows near the limit (a diagnosis, not a gate set): rows of 4,072, 4,030, 3,039 and 2,997 tokens built
|
| 205 |
-
from the fixture's long states keep every argmax, max |Δp| 0.0023, and the hidden state's cosine to
|
| 206 |
-
the oracle's is at least 0.99994 at every position
|
| 207 |
([`gate-kev-0.8b-swift.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-swift.json)).
|
| 208 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 209 |
### iPhone 18 Pro (iOS 27.0 24A437, Core AI arch h19p, 2026-10-04)
|
| 210 |
|
| 211 |
Through [`apps/KevGate`](https://github.com/john-rocky/coreai-model-zoo/tree/main/apps/KevGate), a headless gate app on the Swift host, Release, without the
|
|
@@ -213,73 +239,95 @@ increased-memory-limit entitlement (3,529 MB available at launch). The phone was
|
|
| 213 |
battery at 80 % and charging; every bench item started at thermal state nominal. Every p the app wrote
|
| 214 |
was re-scored on the Mac from its bit patterns.
|
| 215 |
|
| 216 |
-
|
| 217 |
-
|
| 218 |
-
|
| 219 |
-
|
| 220 |
-
| h19p AOT asset (2,505,674,625 B), `.default` | cold | 4.64 | 3.07 | 1.22 | 2,505,681,950 B | 223 MB |
|
| 221 |
-
|
| 222 |
-
| set | asset | questions | argmax (margin > 0.02) | near-ties agreeing | max \|Δp\| | mean | bar |
|
| 223 |
-
|---|---|---:|---:|---:|---:|---:|---|
|
| 224 |
-
| fixture | specialized on the phone | 434 | 420/420 | 13/14 | 0.0102 | 0.00098 | PASS |
|
| 225 |
-
| held out | specialized on the phone | 130 | 125/125 | 5/5 | 0.0049 | 0.00083 | PASS |
|
| 226 |
-
| fixture 60 + held out 30 | h19p AOT | 90 | 80/80 | 10/10 | 0.0071 | 0.00078 | PASS |
|
| 227 |
-
|
| 228 |
-
On the phone the AOT asset and the device specialization agree bit for bit (90/90 rows). No hidden row
|
| 229 |
-
is bit-equal to the Mac's (M4 Max, AOT h16c): their p differ by at most 0.0027, and one near-tie
|
| 230 |
-
(`tv4_023`, oracle margin 0.0020) lands on the oracle's side on the phone and not on the Mac. The shared prefix equals the direct run on
|
| 231 |
-
all 20 multi-question records. The first decision after the cold specialization took 3.58 s; in the bench
|
| 232 |
-
below a 16-token call's median was 25.9–26.6 ms on every item, whatever the row length.
|
| 233 |
-
|
| 234 |
-
Bench (60 s of rest before each item, one warm-up, then the decisions back to back; `latency_ms` = state
|
| 235 |
-
resets, graph calls and the head):
|
| 236 |
-
|
| 237 |
-
| request | tokens | calls | median ms (specialized on the phone) | runs | h19p AOT median ms |
|
| 238 |
-
|---|---:|---:|---:|---|---:|
|
| 239 |
-
| one question (`tv4_000`) | 94 | 6 | 156.6 | 5: 156.3–157.3 | 157.2 |
|
| 240 |
-
| one question (`own_j03` q0) | 380 | 24 | 629.2 | 5: 628.9–630.1 | 627.8 |
|
| 241 |
-
| one question (`own_L02` q0) | 1,518 | 95 | 2,500.0 | 5: 2,499.4–2,502.5 | — |
|
| 242 |
-
| one question (`own_L01` q2) | 1,802 | 113 | 2,975.9 | 5: 2,974.1–2,981.4 | — |
|
| 243 |
-
| 5 questions, 137-token state (`own_m01`), direct | 240 | 51 | 1,330.5 | 5: 1,329.1–1,333.5 | — |
|
| 244 |
-
| the same, shared prefix | 240 | 19 | 507.8 | 5: 507.1–508.8 | 508.2 |
|
| 245 |
-
| 8 questions (`own_m01`), direct | 320 | 83 | 2,167.5 | 3: 2,163.8–2,182.6 | — |
|
| 246 |
-
| the same, shared prefix | 320 | 27 | 722.4 | 3: 721.2–732.9 | — |
|
| 247 |
-
| 4 questions, 1,477-token state (`own_L02`), direct | 1,622 | 380 | 10,026.3 | 2: 10,018.3–10,034.3 | — |
|
| 248 |
-
| the same, shared prefix | 1,622 | 104 | 2,761.9 | 2: 2,752.4–2,771.5 | — |
|
| 249 |
-
|
| 250 |
-
The last item ran 38 s back to back and ended at thermal state fair. Kev-4B on the phone: see the
|
| 251 |
-
[Kev-4B card](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/README.md). Transcript: [`gate-kev-0.8b-iphone.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-iphone.json).
|
| 252 |
-
|
| 253 |
-
### Time per decision on the Mac
|
| 254 |
-
|
| 255 |
-
Swift, Release CLI, AOT asset, in one machine-wide GPU lock window (2026-10-04 00:11–00:33 JST) with no
|
| 256 |
-
other GPU job and a CPU load of 4–10. A B A B: two processes, one warm-up, the median over 20 decisions
|
| 257 |
-
(p10–p90):
|
| 258 |
|
| 259 |
-
|
|
| 260 |
-
|---|---:|---:|---:|---|
|
| 261 |
-
|
|
| 262 |
-
|
|
| 263 |
-
|
| 264 |
-
|
| 265 |
-
|
| 266 |
-
|
| 267 |
-
|
| 268 |
-
|
| 269 |
-
|
| 270 |
-
|
| 271 |
-
|
| 272 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 273 |
|
| 274 |
## Precision
|
| 275 |
|
| 276 |
-
**int8 was measured and is not shipped.**
|
|
|
|
| 277 |
(`symmetric_with_clipping`, weights only; the embedding table, the Gated DeltaNet conv1d and every norm
|
| 278 |
fp16) fails the bar:
|
| 279 |
|
| 280 |
-
| decoder | `main.mlirb` bytes | fixture max \|Δp\| / mean | held out max \|Δp\| / mean | bar |
|
| 281 |
|---|---:|---:|---:|---|
|
| 282 |
-
| fp16
|
| 283 |
| int8lin: every linear int8 | 1,040,063,350 | 0.0431 / 0.00326 | 0.0273 / 0.00294 | FAIL |
|
| 284 |
| int8mix: int8 except layers 0–11 | 1,273,276,920 | 0.0089 / 0.00103 | 0.0043 / 0.00082 | PASS |
|
| 285 |
|
|
@@ -291,21 +339,15 @@ fp16, 45.5 % of the weights, 0.0141). Layers 0–11 kept fp16, exactly half and
|
|
| 291 |
give 0.0046 on the bisect rows; that is int8mix. It keeps 85 % of fp16's bytes, and it was chosen on the
|
| 292 |
fixture, so it does not ship ([`gate-kev-0.8b-int8.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-int8.json)).
|
| 293 |
|
| 294 |
-
**What the compiled asset holds.** `--expect-frequent-reshapes`
|
| 295 |
-
|
| 296 |
-
|
| 297 |
-
linears) compiled with a fixed input length, the int8
|
| 298 |
-
macOS and iOS, with and without the flag, the bytes of
|
| 299 |
-
`main.mlirb` of 10,623,080 bytes. An int8 `.aimodel`
|
| 300 |
-
smaller by the same amount
|
| 301 |
([`../kev-4b/gate-kev-4b-int8.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/gate-kev-4b-int8.json)).
|
| 302 |
|
| 303 |
-
**Chunk width.** S = 16, 32, 64 and 128 all pass the bar (max |Δp| 0.0117, 0.0118, 0.0100, and 0.0060 on
|
| 304 |
-
130 rows). A call's time grows with S: the Gated DeltaNet layers run their S steps one after another
|
| 305 |
-
inside the call. In one interleaved run over the fixture's 423 warm rows, S = 16 took 72.8 s of graph
|
| 306 |
-
calls and S = 32 74.4 s, and S = 16 pads 5.1 % of the fixture against 10.4 %. S = 16 ships
|
| 307 |
-
([`gate-kev-0.8b-readout-s.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-readout-s.json)).
|
| 308 |
-
|
| 309 |
## ⬇️ Bundle
|
| 310 |
|
| 311 |
[mlboydaisuke/Kev-0.8B-CoreAI](https://huggingface.co/mlboydaisuke/Kev-0.8B-CoreAI), one folder under
|
|
@@ -313,29 +355,32 @@ calls and S = 32 74.4 s, and S = 16 pads 5.1 % of the fixture against 10.4 %. S
|
|
| 313 |
|
| 314 |
| file | what | bytes | sha256 |
|
| 315 |
|---|---|---:|---|
|
| 316 |
-
| `
|
| 317 |
-
| `metadata.json` | `kind: decision-backbone`, the readout contract, the gate |
|
| 318 |
| `tokenizer/tokenizer.json` | the base model's, verbatim | 12,807,196 | `fe000e3e…d50d2927` |
|
| 319 |
| `tokenizer/tokenizer_config.json` | the base model's, verbatim | 16,712 | `e611fbcc…c47885de` |
|
| 320 |
-
| `head/head.safetensors` | the pointer head: q / k weight `[256, 1024]` and bias, fp32 | 2,099,576 | `12038b02…
|
| 321 |
-
| `head/kev_head.json` | head size, scale, temperature, delimiter ids, provenance | 2,046 | `73d51070…
|
| 322 |
|
| 323 |
The repository root carries the base model's `config.json`, the adapter's `adapter_config.json` and the
|
| 324 |
-
Apache-2.0 `LICENSE`, verbatim. `SHA256SUMS` lists every file.
|
| 325 |
-
|
| 326 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 327 |
|
| 328 |
No AOT asset ships: the Swift runtime specializes the `.aimodel` correctly on the Mac and on the phone (the
|
| 329 |
-
JIT rows above). To compile one anyway:
|
| 330 |
|
| 331 |
```bash
|
| 332 |
-
xcrun coreai-build compile
|
| 333 |
-
--platform macOS --architecture h16c --expect-frequent-reshapes
|
| 334 |
```
|
| 335 |
|
| 336 |
-
`--expect-frequent-reshapes` is required (without it the runtime specializes again for every new row
|
| 337 |
-
length) and makes the asset 2.5 GB. An iOS asset must not be loaded on a Mac.
|
| 338 |
-
|
| 339 |
## Use it
|
| 340 |
|
| 341 |
Swift, with the [`Kev`](https://github.com/john-rocky/coreai-model-zoo/tree/main/apps/Kev) package (macOS 27 / iOS 27; the system CoreAI framework,
|
|
@@ -344,18 +389,22 @@ Accelerate and swift-transformers' tokenizer), on a download of the repository:
|
|
| 344 |
```swift
|
| 345 |
import Kev
|
| 346 |
|
| 347 |
-
let bundle = URL(filePath: "Kev-0.8B-CoreAI/gpu-pipelined/
|
| 348 |
let kev = try await KevDecider(bundle: bundle) // asset: nil = the .aimodel, specialized here (GPU, frequent reshapes)
|
| 349 |
-
let response = try await kev.decide(requestJSON: requestData, shared: true) // shared: the state's
|
| 350 |
print(PythonFormat.dumps(response, asciiOnly: false))
|
| 351 |
-
// {"model": "
|
| 352 |
// "probabilities": {…}}, "urgent": {"type": "noul", "noul": …}}, "usage": {"input_tokens": …, "output_tokens": …}, "latency_ms": …}
|
|
|
|
|
|
|
|
|
|
|
|
|
| 353 |
```
|
| 354 |
|
| 355 |
The same from the Mac CLI (`swift build -c release --package-path apps/Kev` builds `kev`):
|
| 356 |
|
| 357 |
```bash
|
| 358 |
-
kev run --bundle Kev-0.8B-CoreAI/gpu-pipelined/
|
| 359 |
--request req.json --shared --out resp.json
|
| 360 |
```
|
| 361 |
|
|
@@ -363,7 +412,7 @@ A request, in the SystemOne-compatible request shape:
|
|
| 363 |
|
| 364 |
```json
|
| 365 |
{"model": "kev-0.8b",
|
| 366 |
-
"state": "I was charged twice for
|
| 367 |
"questions": {
|
| 368 |
"team": {"type": "choice", "instructions": "Which team should handle this?",
|
| 369 |
"criteria": {"billing": "Charges and refunds", "shipping": "Deliveries", "returns": "Exchanges"}},
|
|
@@ -386,7 +435,11 @@ with every flag, are in [`conversion/kev/README.md`](https://github.com/john-roc
|
|
| 386 |
(cd $ZOO_WORK_ROOT/_kev/kev-src && python scripts/merge_lora_checkpoint.py \
|
| 387 |
--lora jaredpalmer/kev-0.8b@788ddbdd65715bb03a56788c822f6c632c9a551d --out $ZOO_WORK_ROOT/_kev/merged/kev-0.8b-v1.0)
|
| 388 |
# the decoder (--aot adds the h16c .aimodelc the Python gates load)
|
| 389 |
-
python conversion/kev/export_decoder.py fp16 --prefill-chunk
|
|
|
|
|
|
|
|
|
|
|
|
|
| 390 |
```
|
| 391 |
|
| 392 |
Recipe: [`recipe.toml`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/recipe.toml). Port notes: [`knowledge/kev-port.md`](https://github.com/john-rocky/coreai-model-zoo/blob/main/knowledge/kev-port.md).
|
|
@@ -425,5 +478,12 @@ and numbers.
|
|
| 425 |
From the [author's card](https://huggingface.co/jaredpalmer/kev-0.8b): English only; text generation,
|
| 426 |
chat, tool-call routing and fully automated consequential decisions about people are out of scope; the
|
| 427 |
temperature was fitted on the author's development rows (the card explains how to measure and refit it on
|
| 428 |
-
one's own data). This port adds
|
| 429 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 24 |
|
| 25 |
| Variant | Path | Size | Requires | Tested on |
|
| 26 |
|---|---|---:|---|---|
|
| 27 |
+
| Decoder, fp16, with the host's pointer head | `gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128/` | 1,520 MB | macOS 27 / iOS 27 | M4 Max, macOS 27.0 (26A428), 2026-10-04; iPhone 18 Pro, iOS 27.0 (24A437), 2026-10-04 |
|
| 28 |
|
| 29 |
+
One question of 94 tokens takes 29.9 ms on the M4 Max and 37.8 ms on the iPhone 18 Pro, in Swift ([measured below](#time-per-decision-on-the-mac)).
|
| 30 |
|
| 31 |
```swift
|
| 32 |
import Kev
|
| 33 |
|
| 34 |
+
let bundle = URL(filePath: "Kev-0.8B-CoreAI/gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128")
|
| 35 |
let kev = try await KevDecider(bundle: bundle) // asset: nil = the .aimodel, specialized here (GPU, frequent reshapes)
|
| 36 |
+
let response = try await kev.decide(requestJSON: requestData, shared: true) // shared: the state's whole calls run once
|
| 37 |
print(PythonFormat.dumps(response, asciiOnly: false))
|
| 38 |
```
|
| 39 |
|
|
|
|
| 56 |
results; none of them are re-measured here.
|
| 57 |
|
| 58 |
This port merges the adapter into the base with the author's own merge script and exports the text
|
| 59 |
+
backbone as one Core AI graph. The graph takes 128 token ids per call and returns the final-norm hidden
|
| 60 |
+
state at every position; it has no vocabulary head. Each Gated DeltaNet layer runs its recurrence in an
|
| 61 |
+
fp32 Metal kernel, one GPU dispatch per layer per call. Weights are fp16 (1.51 GB). The host runs the
|
| 62 |
+
pointer head from the author's head weights. The gate is probability parity with the author's own fp32
|
| 63 |
+
code, on every option of every question, on a fixture of 434 questions and on a held-out set of 130.
|
| 64 |
|
| 65 |
+
It runs on the Mac GPU and on an iPhone 18 Pro. On both, the shipped `.aimodel` is specialized where it
|
| 66 |
+
runs.
|
| 67 |
|
| 68 |
## Readout contract
|
| 69 |
|
|
|
|
| 96 |
`(p_max − 1/K) / (1 − 1/K)` and every p; `score` → Σ level · p, the legend, every p and a confidence
|
| 97 |
`max(0, 1 − E|level − mode| / D)`; values rounded to 4 decimals. `usage` = `{input_tokens,
|
| 98 |
output_tokens}` with the author's counting (the packed request; the tokens of `json.dumps(answers)`).
|
| 99 |
+
- Length: a row holds at most **3,968 tokens**. The last call is padded to 128 tokens and must end at or
|
| 100 |
+
below the graph's position bound of 4,095. The author's server accepts states of 65,536 tokens; here a
|
| 101 |
+
longer row is refused before any call.
|
| 102 |
- Shared prefix (optional, exact): every row of a request starts with the same state tokens, so the
|
| 103 |
+
state's whole calls (⌊Ls / 128⌋ of them) can run once, the four states be copied per question, and each row
|
| 104 |
+
continue from there. Every row's hidden state equals its direct run bit for bit (below).
|
| 105 |
|
| 106 |
`conversion/kev/oracle_kev.py` runs the author's package unchanged (`Checkpoint.load("cpu", fp32)`,
|
| 107 |
`kev.model.admit`, the row-form `forward`, one CPU thread) and records the reference:
|
| 108 |
[`fixtures-kev-0.8b.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/fixtures-kev-0.8b.json). The fixture is 384 records and 434 questions:
|
| 109 |
+
lines 1–60 of the author's transfer-v4 development file (`evals/v4/transfer-v4/development.jsonl`,
|
| 110 |
+
tag `kev-1.0`; MMLU, 4 options), records 1–20 of each of its seven other sources (140, `choice` and
|
| 111 |
+
`noul`), its `score` records 1–20, the 144 `choice` items of SemIf authored144, and 20 records
|
| 112 |
written for the port (70 questions: `noul`, `choice` with 2–10 options, `score` with 2–7 levels,
|
| 113 |
JSON states, three states of 1,431–1,746 tokens, one record of 8 questions). The longest row is 1,802
|
| 114 |
tokens. Fourteen questions have an oracle top-2 margin of 0.02 or less (near-ties) and are listed apart.
|
|
|
|
| 119 |
|
| 120 |
`conversion/kev/qwen3_5_kev_decoder.py`: the overlay's stateful Qwen3.5 text decoder
|
| 121 |
(`Qwen3_5StatefulForCausalLM`) on the merged weights, the vocabulary head replaced by the identity, the
|
| 122 |
+
output the final-norm hidden state at every position. One function, `main`, at a static 128 tokens. Each
|
| 123 |
+
Gated DeltaNet layer runs its recurrence in the overlay's fp32 Metal chunk kernel
|
| 124 |
+
(`qwen3_5_gdn_metal`): one GPU dispatch per layer per call, the recurrent state kept in fp32 through the
|
| 125 |
+
call.
|
| 126 |
|
| 127 |
| | name | shape, type |
|
| 128 |
|---|---|---|
|
| 129 |
+
| inputs | `input_ids` | [1, 128] int32 |
|
| 130 |
| | `position_ids` | [1, seq] int32, the ramp 0 .. seq − 1 |
|
| 131 |
| states | `keyCache`, `valueCache` | [6, 1, 2, ctx, 256] fp16, ctx up to 4,096 |
|
| 132 |
| | `convState`, `recState` | [18, 1, 6144, 3], [18, 1, 16, 128, 128] fp16 |
|
| 133 |
+
| output | `hidden` | [1, 128, 1024] fp16, every position |
|
| 134 |
|
| 135 |
+
Per row of T ids, from zeroed states: ⌈T / 128⌉ calls, call k with ids[128k : 128k + 128] and position_ids
|
| 136 |
+
0..128k + 127. The last call is padded with `<|endoftext|>` (248044) and its padded rows are dropped. The
|
| 137 |
rows of every call, cut to T, are the backbone's final-norm `last_hidden_state [T, 1024]`. The bundle's
|
| 138 |
+
`metadata.json` says `kind: decision-backbone` and `language.prefill_chunk: 128`, and carries this order,
|
| 139 |
+
the row layout, the delimiter ids, the render rules, the head formula and the response shape.
|
| 140 |
|
| 141 |
**Host.** Builds the rows, runs the calls, converts the hidden rows at the readout positions to float64,
|
| 142 |
applies the head (`head/head.safetensors`, `head/kev_head.json`) and the per-question softmax in float64,
|
| 143 |
rounds p to fp32 once, and writes the response. `conversion/kev/decide.py` is the Python reference;
|
| 144 |
+
[`apps/Kev`](https://github.com/john-rocky/coreai-model-zoo/tree/main/apps/Kev) is the Swift one. With `shared: true` the state's whole calls run once. A
|
| 145 |
+
prepared state (`prepare(state:)`, then `decide(prepared:questionsJSON:)`) runs the same calls in two steps,
|
| 146 |
+
so questions that arrive later skip the state.
|
| 147 |
|
| 148 |
No engine is involved: the hosts drive the low-level runtime (`AIModel` + `loadFunction`, four zeroed
|
| 149 |
states per row). No runtime patch.
|
|
|
|
| 152 |
|
| 153 |
The bar, fixed before any graph ran: the argmax equal to the oracle's on every question whose oracle
|
| 154 |
top-2 margin is above 0.02 (near-ties listed apart), max |Δp| ≤ 0.02 over every option of every
|
| 155 |
+
question, the mean over rows of each row's mean |Δp| ≤ 0.002, and every process re-running its opening row
|
| 156 |
bit for bit. The Python gates load the AOT `.aimodelc` (`coreai-build compile … --platform macOS
|
| 157 |
--preferred-compute gpu --architecture h16c --expect-frequent-reshapes`) with
|
| 158 |
+
`SpecializationOptions.default()`.
|
| 159 |
|
| 160 |
### Before any graph: the merge and fp32 torch
|
| 161 |
|
|
|
|
| 165 |
merged tensors bit for bit (the adapter moves them by 2.8–3.6 %). The base alone moves p by up to
|
| 166 |
0.789 (argmax 25/53), the base's Gated DeltaNet projections put back by up to 0.282 (46/53): the
|
| 167 |
check can fail ([`gate-kev-0.8b-merge.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-merge.json)).
|
| 168 |
+
- The decoder module in fp32 on the CPU, driven like the graph with the recurrence unrolled, read through
|
| 169 |
+
the author's head: 434/434 argmax (the 14 near-ties included), max |Δp| 3.5e-6, lowest per-position
|
| 170 |
+
hidden cosine 0.99999999999 on the 21 rows the oracle keeps. It equals the overlay's plain text decoder
|
| 171 |
+
bit for bit (3 rows) ([`gate-kev-0.8b-torch-parity.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-torch-parity.json)).
|
| 172 |
+
- The Metal kernel has no torch values: its `torch_defn` returns zeros of the right shapes. The GPU gate
|
| 173 |
+
below is its parity check.
|
| 174 |
|
| 175 |
### The graph alone on the Mac GPU
|
| 176 |
|
|
|
|
| 178 |
|
| 179 |
| set | questions | argmax (margin > 0.02) | near-ties agreeing | max \|Δp\| | mean of row means | bar |
|
| 180 |
|---|---:|---:|---:|---:|---:|---|
|
| 181 |
+
| **fixture** | 434 | **420/420** | **13/14** | **0.0124** | **0.00096** | **PASS** |
|
| 182 |
+
| held out | 130 | 125/125 | 5/5 | 0.0058 | 0.00083 | PASS |
|
| 183 |
|
| 184 |
+
The near-tie that flips is a SemIf item with an oracle margin of 0.0032; it moves by 0.0026. The worst
|
| 185 |
+
row is a SemIf item (0.0124). Every value is finite and every process's re-run is bit-equal. Two red arms
|
| 186 |
+
move p on the same graph: a state swapped for another record's moves 4 of 5 argmaxes (max |Δp| 0.905); a
|
| 187 |
+
grammatical "not" in five yes / no questions moves 2 of 5 (0.759). Transcript:
|
| 188 |
+
[`gate-kev-0.8b-readout.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-readout.json).
|
|
|
|
|
|
|
| 189 |
|
| 190 |
### From the request: Python and Swift
|
| 191 |
|
| 192 |
- `conversion/kev/host.py`, without the author's package, rebuilds every oracle row from the raw
|
| 193 |
+
request (ids, `<|fim_suffix|>` / `<|box_end|>` indices, keys): 434/434 and 130/130 rows with two
|
| 194 |
+
tokenizer implementations. `decide.py` on the AOT graph: hidden rows bit-equal to the gate's on 434/434
|
| 195 |
+
and 130/130 rows ([`gate-kev-0.8b-host.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-host.json)).
|
| 196 |
+
- The Swift host ([`apps/Kev`](https://github.com/john-rocky/coreai-model-zoo/tree/main/apps/Kev), Release): ids from the raw request 434/434 and 130/130;
|
| 197 |
+
on the same AOT asset every row's hidden state and every p equal the Python reference's bit for bit
|
| 198 |
+
(434/434), every answer set's `json.dumps` and usage byte for byte (384/384), and the bar is the gate's
|
| 199 |
+
(fixture 0.0124 / 0.00096). With the shared prefix every row's hidden state and p equal the direct run's
|
| 200 |
+
(434/434). Its text half equals CPython on 2,787 rendered values, 599 requests (21 refused with the
|
| 201 |
+
same messages), 203,635 doubles (`str`, `repr`, `round`) and 1,093 `json.dumps` texts. One word changed
|
| 202 |
+
in one question turns its record red.
|
|
|
|
| 203 |
- JIT: the `.aimodel` specialized by the Swift runtime (GPU preferred, `expectFrequentReshapes`) equals
|
| 204 |
+
the AOT asset bit for bit on 25 rows of 10 records. The specialization took 3.34 s (4.05 s with the
|
| 205 |
+
tokenizer and the head); with the runtime's cache warm a load took 0.69 s
|
|
|
|
|
|
|
|
|
|
| 206 |
([`gate-kev-0.8b-swift.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-swift.json)).
|
| 207 |
|
| 208 |
+
### Time per decision on the Mac
|
| 209 |
+
|
| 210 |
+
Swift, Release CLI, the AOT asset, in one machine-wide GPU lock window (2026-10-04 12:13–12:27 JST). Two
|
| 211 |
+
processes per graph counted only when no other GPU job ran and the one-minute load average was at most 12;
|
| 212 |
+
each made 10 decisions per item after one warm-up, and the table gives the median (p10–p90) over the 20.
|
| 213 |
+
The earlier upload's graph (below, under Bundle) ran in the same window:
|
| 214 |
+
|
| 215 |
+
| request | tokens | calls | ms | p10–p90 | earlier upload's graph, ms |
|
| 216 |
+
|---|---:|---:|---:|---|---:|
|
| 217 |
+
| one question | 94 | 1 | 29.9 | 29.6–30.1 | 100.1 |
|
| 218 |
+
| one question | 380 | 3 | 89.2 | 88.7–90.4 | 404.4 |
|
| 219 |
+
| one question | 1,518 | 12 | 357.0 | 355.4–359.3 | 1,600.1 |
|
| 220 |
+
| one question | 1,802 | 15 | 447.9 | 446.6–449.8 | 1,897.6 |
|
| 221 |
+
| five questions on one 137-token state, each row from zero | 240 | 10 | 296.6 | 294.6–298.5 | 845.7 |
|
| 222 |
+
| the same five questions, the state run once (shared) | 240 | 6 | 186.4 | 185.4–187.7 | 327.7 |
|
| 223 |
+
| eight questions on that state, each row from zero | 320 | 16 | 474.2 | 472.2–475.9 | 1,377.0 |
|
| 224 |
+
| the same eight questions, shared | 320 | 9 | 278.8 | 277.2–281.7 | 461.0 |
|
| 225 |
+
| four questions on one 1,477-token state, each row from zero | 1,622 | 49 | 1,455.9 | 1,453.7–1,460.4 | 6,366.1 |
|
| 226 |
+
| the same four questions, shared | 1,622 | 16 | 485.4 | 483.8–489.4 | 1,749.0 |
|
| 227 |
+
| a follow-up question on a prepared 137-token state | — | — | 30.3 | — | 34.1 |
|
| 228 |
+
| a follow-up question on a prepared 1,477-token state | — | — | 31.5 | — | 51.5 |
|
| 229 |
+
|
| 230 |
+
Preparing the two states took 34.1–34.7 ms and 335.9–339.6 ms (the two processes' medians). In round 11's
|
| 231 |
+
lock window (09:43–10:45 JST) the `.aimodel` specialized by Swift, the shipped path, took 29.7 ms for the
|
| 232 |
+
94-token question and 186.5 ms for the five questions shared, against 30.0 and 189.1 ms for the AOT asset;
|
| 233 |
+
the Python reference took 33.4 and 207.1 ms. Transcript: [`gate-kev-0.8b-timing-mac.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-timing-mac.json).
|
| 234 |
+
|
| 235 |
### iPhone 18 Pro (iOS 27.0 24A437, Core AI arch h19p, 2026-10-04)
|
| 236 |
|
| 237 |
Through [`apps/KevGate`](https://github.com/john-rocky/coreai-model-zoo/tree/main/apps/KevGate), a headless gate app on the Swift host, Release, without the
|
|
|
|
| 239 |
battery at 80 % and charging; every bench item started at thermal state nominal. Every p the app wrote
|
| 240 |
was re-scored on the Mac from its bit patterns.
|
| 241 |
|
| 242 |
+
- **Load.** The `.aimodel` specialized on the phone (GPU preferred, `expectFrequentReshapes`) took 4.26 s cold,
|
| 243 |
+
3.60 s of it in `AIModel(contentsOf:)`, with +2,502,056,463 B of runtime cache and a peak footprint of 233 MB
|
| 244 |
+
(round 13, an earlier KevGate build). With the cache warm, the gate run below loaded it in 1.50 s.
|
| 245 |
+
- **Gate** (run `20261004-140048`, KevGate built on round 15's host):
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 246 |
|
| 247 |
+
| set | questions | argmax (margin > 0.02) | near-ties agreeing | max \|Δp\| | mean of row means | bar |
|
| 248 |
+
|---|---:|---:|---:|---:|---:|---|
|
| 249 |
+
| fixture | 434 | 420/420 | 13/14 | 0.0110 | 0.00097 | PASS |
|
| 250 |
+
| held out | 130 | 125/125 | 5/5 | 0.0059 | 0.00083 | PASS |
|
| 251 |
+
|
| 252 |
+
The shared prefix equals the direct run on all 20 multi-question records, and the re-run of the opening
|
| 253 |
+
record is bit-equal. No hidden row equals the Mac's (a different GPU): their p differ by at most 0.0026,
|
| 254 |
+
with every argmax equal (434/434). The fixture pass peaked at a 389 MB footprint.
|
| 255 |
+
- **Bench** (60 s of rest before each item, one warm-up, then 5 decisions; `latency_ms` = state resets,
|
| 256 |
+
graph calls and the head). The earlier upload's graph ran in the same session:
|
| 257 |
+
|
| 258 |
+
| request | tokens | ms | earlier upload's graph, ms |
|
| 259 |
+
|---|---:|---:|---:|
|
| 260 |
+
| one question | 94 | 37.8 | 156.7 |
|
| 261 |
+
| one question | 380 | 110.6 | 627.9 |
|
| 262 |
+
| one question | 1,518 | 444.1 | 2,500.9 |
|
| 263 |
+
| one question | 1,802 | 559.1 | 2,977.8 |
|
| 264 |
+
| five questions on one 137-token state, each row from zero / shared | 240 | 361.9 / 223.2 | 1,331.4 / 507.4 |
|
| 265 |
+
| eight questions on that state, each row from zero / shared | 320 | 583.2 / 333.4 | 2,167.4 / 723.3 |
|
| 266 |
+
| four questions on one 1,477-token state, each row from zero / shared | 1,622 | 1,822.2 / 613.9 | 10,061.6 / 2,769.8 |
|
| 267 |
+
| the five questions on a prepared state (the prepare: 38.8 / 211.6 ms) | 240 | 183.4 | 295.5 |
|
| 268 |
+
| the four questions on a prepared state (the prepare: 415.3 / 2,451.4 ms) | 1,622 | 196.5 | 342.4 |
|
| 269 |
+
|
| 270 |
+
An earlier session on the same phone, with an earlier KevGate build, gave 36.5 ms for the 94-token question
|
| 271 |
+
against 157.1 ms for the earlier upload's graph. Transcript: [`gate-kev-0.8b-iphone.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-iphone.json).
|
| 272 |
+
|
| 273 |
+
Kev-4B on the phone: see the [Kev-4B card](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/README.md).
|
| 274 |
+
|
| 275 |
+
## Forms measured
|
| 276 |
+
|
| 277 |
+
Every form below was exported, compiled and gated the same way. The times come from different windows, so
|
| 278 |
+
compare forms only within one window. Full records: [`gate-kev-0.8b-forms.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-forms.json).
|
| 279 |
+
|
| 280 |
+
Mac (Swift Release CLI, the 94-token question, median of the counted processes):
|
| 281 |
+
|
| 282 |
+
| form | 94-token decision, ms | window | fixture max \|Δp\| | note |
|
| 283 |
+
|---|---:|---|---:|---|
|
| 284 |
+
| Gated DeltaNet unrolled, S = 16 (the earlier upload) | 100.1 | round 15 | 0.0117 | |
|
| 285 |
+
| unrolled, S = 32 | 139.0 | round 11 | 0.0118 | |
|
| 286 |
+
| unrolled, S = 64 | — | — | 0.0100 | not timed in a lock window (68.06 ms per call on a shared GPU) |
|
| 287 |
+
| in-graph chunk scan, S = 16 | — | — | 0.0135 | not timed in a lock window (10.90 ms per call on a shared GPU) |
|
| 288 |
+
| in-graph chunk scan, S = 32 | — | — | — | fails: non-near-tie argmax 112/127, max \|Δp\| 0.125, non-finite rows; breaks in fp32 torch too |
|
| 289 |
+
| Metal kernel, S = 16 / 32 / 64 | 63.6 / 42.7 / 41.5 | round 11 | 0.0115 / 0.0134 / 0.0115 | |
|
| 290 |
+
| **Metal kernel, S = 128 (this release)** | **29.9** | round 15 | **0.0124** | |
|
| 291 |
+
| Metal kernel, query length 2..512, host calls of at most 512 ids | 25.2 | round 15 | 0.0112 | not shipped: memory grows (below) |
|
| 292 |
+
| the same graph, host calls of at most 256 / 128 ids | 25.1 / 25.0 | round 15 | 0.0060 / 0.0060 (130 rows) | not shipped |
|
| 293 |
+
| int8, every linear (unrolled S = 16) | — | — | 0.0431 | fails the bar (Precision) |
|
| 294 |
+
| any form on the Neural Engine | — | — | — | not run: the recurrence's fp32 state is not an ANE element type ([`qwen3.5-static-ane.md`](https://github.com/john-rocky/coreai-model-zoo/blob/main/knowledge/qwen3.5-static-ane.md)) |
|
| 295 |
+
|
| 296 |
+
iPhone 18 Pro (KevGate bench, the 94-token question):
|
| 297 |
+
|
| 298 |
+
| form | ms | session | fixture max \|Δp\| |
|
| 299 |
+
|---|---:|---|---:|
|
| 300 |
+
| unrolled, S = 16 (the earlier upload) | 156.7 | round 13, stage C | 0.0102 |
|
| 301 |
+
| Metal kernel, S = 16 / 32 / 64 | 92.4 / 58.5 / 49.3 | round 13, stage A | 0.0110 / 0.0133 / 0.0098 |
|
| 302 |
+
| **Metal kernel, S = 128 (this release)** | **37.8** | round 13, stage C | **0.0110** |
|
| 303 |
+
| Metal kernel, query length 2..512, host calls of at most 512 ids | 32.6 | round 13, stage C | gate stopped by memory (below) |
|
| 304 |
+
|
| 305 |
+
**Several short questions on one state.** For these, the S = 64 graph takes less time than S = 128. In
|
| 306 |
+
round 11's Mac window, five questions on one 137-token state, shared, took 145.0 ms with S = 64 and
|
| 307 |
+
189.1 ms with S = 128; one 94-token question took 41.5 and 30.0 ms. On the iPhone (round 13, stage A)
|
| 308 |
+
the same pairs took 180.3 against 224.4 ms and 49.3 against 36.5 ms. The S = 64 bundle is not in this
|
| 309 |
+
release: it would need a new export and a new gate.
|
| 310 |
+
|
| 311 |
+
**Why the dynamic-length graph does not ship.** The graph whose query length is dynamic (2..512) passed the
|
| 312 |
+
gate on the Mac and decided the 94-token question in 25.2 ms. A process that keeps changing its call length
|
| 313 |
+
keeps growing, though. Cycling four call lengths (128, 256, 384, 512) three times added 467.1, 166.9 and
|
| 314 |
+
168.5 MB per lap in Swift and 460.7, 169.5 and 169.6 MB in Python. Alternating between two lengths did not
|
| 315 |
+
grow after the opening calls; bringing in a third added 19.1 MB, then 1.3 MB. On the iPhone the gate run
|
| 316 |
+
reached 3,497 MB with 43 MB available after 439 calls and stopped writing. A 2-second sleep gave back part of
|
| 317 |
+
the memory and loading the function again gave back none; creating the `AIModel` again gave it back. Records:
|
| 318 |
+
[`gate-kev-0.8b-forms.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-forms.json); notes:
|
| 319 |
+
[`knowledge/kev-port.md`](https://github.com/john-rocky/coreai-model-zoo/blob/main/knowledge/kev-port.md).
|
| 320 |
|
| 321 |
## Precision
|
| 322 |
|
| 323 |
+
**int8 was measured and is not shipped.** It was measured on the unrolled S = 16 graph, the earlier upload;
|
| 324 |
+
the Metal-kernel graphs were not measured in int8. int8 per block of 32 over every decoder linear
|
| 325 |
(`symmetric_with_clipping`, weights only; the embedding table, the Gated DeltaNet conv1d and every norm
|
| 326 |
fp16) fails the bar:
|
| 327 |
|
| 328 |
+
| decoder (unrolled S = 16) | `main.mlirb` bytes | fixture max \|Δp\| / mean | held out max \|Δp\| / mean | bar |
|
| 329 |
|---|---:|---:|---:|---|
|
| 330 |
+
| fp16 | 1,506,481,909 | 0.0117 / 0.00097 | 0.0057 / 0.00081 | PASS |
|
| 331 |
| int8lin: every linear int8 | 1,040,063,350 | 0.0431 / 0.00326 | 0.0273 / 0.00294 | FAIL |
|
| 332 |
| int8mix: int8 except layers 0–11 | 1,273,276,920 | 0.0089 / 0.00103 | 0.0043 / 0.00082 | PASS |
|
| 333 |
|
|
|
|
| 339 |
give 0.0046 on the bisect rows; that is int8mix. It keeps 85 % of fp16's bytes, and it was chosen on the
|
| 340 |
fixture, so it does not ship ([`gate-kev-0.8b-int8.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/gate-kev-0.8b-int8.json)).
|
| 341 |
|
| 342 |
+
**What the compiled asset holds.** `--expect-frequent-reshapes` adds an fp16 copy of every linear weight
|
| 343 |
+
to the compiled asset: for the unrolled graph the iPhone h19p asset is 2,505,674,625 bytes with it and
|
| 344 |
+
1,506,400,620 without, and this release's Mac h16c asset is 2,502,047,194 bytes against a 1,505,385,733-byte
|
| 345 |
+
`main.mlirb`. On a toy (an embedding and two linears) compiled with a fixed input length, the int8
|
| 346 |
+
version's `resources.bin` is 12,582,936 bytes on macOS and iOS, with and without the flag, the bytes of
|
| 347 |
+
the fp16 toy compiled without it, against an int8 `main.mlirb` of 10,623,080 bytes. An int8 `.aimodel`
|
| 348 |
+
downloads smaller; the compiled asset is not smaller by the same amount
|
| 349 |
([`../kev-4b/gate-kev-4b-int8.json`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-4b/gate-kev-4b-int8.json)).
|
| 350 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 351 |
## ⬇️ Bundle
|
| 352 |
|
| 353 |
[mlboydaisuke/Kev-0.8B-CoreAI](https://huggingface.co/mlboydaisuke/Kev-0.8B-CoreAI), one folder under
|
|
|
|
| 355 |
|
| 356 |
| file | what | bytes | sha256 |
|
| 357 |
|---|---|---:|---|
|
| 358 |
+
| `kev_0_8b_decode_fp16_metal_pf128.aimodel/main.mlirb` | the decoder, fp16 | 1,505,385,733 | `19d5a480…2c3f684e` |
|
| 359 |
+
| `metadata.json` | `kind: decision-backbone`, the readout contract, the gate | 12,590 | `6e50e41a…47282227` |
|
| 360 |
| `tokenizer/tokenizer.json` | the base model's, verbatim | 12,807,196 | `fe000e3e…d50d2927` |
|
| 361 |
| `tokenizer/tokenizer_config.json` | the base model's, verbatim | 16,712 | `e611fbcc…c47885de` |
|
| 362 |
+
| `head/head.safetensors` | the pointer head: q / k weight `[256, 1024]` and bias, fp32 | 2,099,576 | `12038b02…00893a42` |
|
| 363 |
+
| `head/kev_head.json` | head size, scale, temperature, delimiter ids, provenance | 2,046 | `73d51070…9d5c9d7c` |
|
| 364 |
|
| 365 |
The repository root carries the base model's `config.json`, the adapter's `adapter_config.json` and the
|
| 366 |
+
Apache-2.0 `LICENSE`, verbatim. `SHA256SUMS` lists every file. The Hub revision of this bundle is recorded
|
| 367 |
+
here after the upload.
|
| 368 |
+
|
| 369 |
+
**The earlier upload.** Hub revision
|
| 370 |
+
[`7793533`](https://huggingface.co/mlboydaisuke/Kev-0.8B-CoreAI/tree/7793533783de7c8f916c1cba1144a561fb70b321)
|
| 371 |
+
held another graph of this port, `kev_0_8b_decode_fp16_pf16`: the Gated DeltaNet recurrence unrolled step by
|
| 372 |
+
step, 16 tokens per call. It passed the same gate (fixture max |Δp| 0.0117, held out 0.0057). In the Mac
|
| 373 |
+
window above it took 100.1 ms for the 94-token question against 29.9 ms for this release, and on the iPhone
|
| 374 |
+
156.7 ms against 37.8 ms.
|
| 375 |
|
| 376 |
No AOT asset ships: the Swift runtime specializes the `.aimodel` correctly on the Mac and on the phone (the
|
| 377 |
+
JIT rows above). To compile one for the Mac anyway:
|
| 378 |
|
| 379 |
```bash
|
| 380 |
+
xcrun coreai-build compile kev_0_8b_decode_fp16_metal_pf128.aimodel --output aot --preferred-compute gpu \
|
| 381 |
+
--platform macOS --architecture h16c --expect-frequent-reshapes
|
| 382 |
```
|
| 383 |
|
|
|
|
|
|
|
|
|
|
| 384 |
## Use it
|
| 385 |
|
| 386 |
Swift, with the [`Kev`](https://github.com/john-rocky/coreai-model-zoo/tree/main/apps/Kev) package (macOS 27 / iOS 27; the system CoreAI framework,
|
|
|
|
| 389 |
```swift
|
| 390 |
import Kev
|
| 391 |
|
| 392 |
+
let bundle = URL(filePath: "Kev-0.8B-CoreAI/gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128")
|
| 393 |
let kev = try await KevDecider(bundle: bundle) // asset: nil = the .aimodel, specialized here (GPU, frequent reshapes)
|
| 394 |
+
let response = try await kev.decide(requestJSON: requestData, shared: true) // shared: the state's whole calls run once
|
| 395 |
print(PythonFormat.dumps(response, asciiOnly: false))
|
| 396 |
+
// {"model": "kev_0_8b_decode_fp16_metal_pf128", "answers": {"team": {"type": "choice", "choice": "billing", "confidence": …,
|
| 397 |
// "probabilities": {…}}, "urgent": {"type": "noul", "noul": …}}, "usage": {"input_tokens": …, "output_tokens": …}, "latency_ms": …}
|
| 398 |
+
|
| 399 |
+
// questions that arrive later, on the same state
|
| 400 |
+
let prepared = try await kev.prepare(state: try JSONParser.parse(stateData)) // the state's whole calls, once
|
| 401 |
+
let later = try await kev.decide(prepared: prepared, questionsJSON: questionsData) // only the questions' rows run
|
| 402 |
```
|
| 403 |
|
| 404 |
The same from the Mac CLI (`swift build -c release --package-path apps/Kev` builds `kev`):
|
| 405 |
|
| 406 |
```bash
|
| 407 |
+
kev run --bundle Kev-0.8B-CoreAI/gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128 --asset jit \
|
| 408 |
--request req.json --shared --out resp.json
|
| 409 |
```
|
| 410 |
|
|
|
|
| 412 |
|
| 413 |
```json
|
| 414 |
{"model": "kev-0.8b",
|
| 415 |
+
"state": "I was charged twice for my last order. Please refund one of the charges.",
|
| 416 |
"questions": {
|
| 417 |
"team": {"type": "choice", "instructions": "Which team should handle this?",
|
| 418 |
"criteria": {"billing": "Charges and refunds", "shipping": "Deliveries", "returns": "Exchanges"}},
|
|
|
|
| 435 |
(cd $ZOO_WORK_ROOT/_kev/kev-src && python scripts/merge_lora_checkpoint.py \
|
| 436 |
--lora jaredpalmer/kev-0.8b@788ddbdd65715bb03a56788c822f6c632c9a551d --out $ZOO_WORK_ROOT/_kev/merged/kev-0.8b-v1.0)
|
| 437 |
# the decoder (--aot adds the h16c .aimodelc the Python gates load)
|
| 438 |
+
python conversion/kev/export_decoder.py fp16 --gdn-scan metal --prefill-chunk 128 --aot
|
| 439 |
+
# the published bundle: the gated .aimodel bytes and their metadata, without exporting again
|
| 440 |
+
python conversion/kev/export_decoder.py fp16 --gdn-scan metal --prefill-chunk 128 --metadata-only \
|
| 441 |
+
--from-aimodel <bundles>/kev_0_8b_decode_fp16_metal_pf128/kev_0_8b_decode_fp16_metal_pf128.aimodel \
|
| 442 |
+
--bundle-dir <ship>/kev_0_8b_decode_fp16_metal_pf128
|
| 443 |
```
|
| 444 |
|
| 445 |
Recipe: [`recipe.toml`](https://github.com/john-rocky/coreai-model-zoo/blob/main/models/kev-0.8b/recipe.toml). Port notes: [`knowledge/kev-port.md`](https://github.com/john-rocky/coreai-model-zoo/blob/main/knowledge/kev-port.md).
|
|
|
|
| 478 |
From the [author's card](https://huggingface.co/jaredpalmer/kev-0.8b): English only; text generation,
|
| 479 |
chat, tool-call routing and fully automated consequential decisions about people are out of scope; the
|
| 480 |
temperature was fitted on the author's development rows (the card explains how to measure and refit it on
|
| 481 |
+
one's own data). This port adds:
|
| 482 |
+
|
| 483 |
+
- A row holds at most 3,968 tokens, while the author's server accepts states up to 65,536.
|
| 484 |
+
- A process's opening call is slower. On the Mac the opening 128-token call of each Python gate process
|
| 485 |
+
took 211–1,188 ms, against a median of 29.9 ms for its later calls. On the phone the opening decision
|
| 486 |
+
took 66.9 ms, against 37.8 ms in the bench of the same build.
|
| 487 |
+
- If you export this decoder with a dynamic query length yourself, keep the call lengths to one or two
|
| 488 |
+
values. A process that cycles more lengths grows its memory until the `AIModel` is created again (measured
|
| 489 |
+
on the Mac and the iPhone; which runtime layer keeps the memory is not isolated).
|
SHA256SUMS
CHANGED
|
@@ -1,12 +1,12 @@
|
|
| 1 |
50cbab8a892c5f2993b8c7351a99182507472def3b1374558308605d99b86b32 LICENSE
|
| 2 |
-
|
| 3 |
748acb2cda88454cb1ba69d745ba336f3fcb5486eac349e90960c8b8d8d3e854 adapter_config.json
|
| 4 |
b90b86f35c8e6925ef74ee04d0e758f0a845c83a42089ad82bbaa948de9b4204 config.json
|
| 5 |
-
12038b02ecf7ece3ffb008b22ae97988293e521a86e7aadc8036c9c700893a42 gpu-pipelined/
|
| 6 |
-
73d510708790422d3c84bfc915e311e536d527bc032f0a07394e76999d5c9d7c gpu-pipelined/
|
| 7 |
-
|
| 8 |
-
|
| 9 |
-
|
| 10 |
-
|
| 11 |
-
fe000e3ed39ed12b8d2481d527d44f93c65d37e87645d2dcc80d1bf9d50d2927 gpu-pipelined/
|
| 12 |
-
e611fbccc7c29ef3b1cafb1cb7ea548d189968632901d678fd62be68c47885de gpu-pipelined/
|
|
|
|
| 1 |
50cbab8a892c5f2993b8c7351a99182507472def3b1374558308605d99b86b32 LICENSE
|
| 2 |
+
3d86d2c9190033fda5f58eaa257b93069a0ee5a955c4b9bfbf56e7ac96e34df7 README.md
|
| 3 |
748acb2cda88454cb1ba69d745ba336f3fcb5486eac349e90960c8b8d8d3e854 adapter_config.json
|
| 4 |
b90b86f35c8e6925ef74ee04d0e758f0a845c83a42089ad82bbaa948de9b4204 config.json
|
| 5 |
+
12038b02ecf7ece3ffb008b22ae97988293e521a86e7aadc8036c9c700893a42 gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128/head/head.safetensors
|
| 6 |
+
73d510708790422d3c84bfc915e311e536d527bc032f0a07394e76999d5c9d7c gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128/head/kev_head.json
|
| 7 |
+
7ad0c96ad0ce3cef5315f9504b15b5dacdd8b958b2848f1600e8edbd8b762bfc gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128/kev_0_8b_decode_fp16_metal_pf128.aimodel/main.hash
|
| 8 |
+
19d5a4809e348c7cdd314c17cf18646170e2af2ef62f5a32beb8948b2c3f684e gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128/kev_0_8b_decode_fp16_metal_pf128.aimodel/main.mlirb
|
| 9 |
+
33d232378eb37701453e8974f12141c95bedcf856951feffa29854b24b127a8f gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128/kev_0_8b_decode_fp16_metal_pf128.aimodel/metadata.json
|
| 10 |
+
6e50e41a92eebe440029974a403441cd960d2f6ef4ff00828bb2aa8f47282227 gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128/metadata.json
|
| 11 |
+
fe000e3ed39ed12b8d2481d527d44f93c65d37e87645d2dcc80d1bf9d50d2927 gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128/tokenizer/tokenizer.json
|
| 12 |
+
e611fbccc7c29ef3b1cafb1cb7ea548d189968632901d678fd62be68c47885de gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128/tokenizer/tokenizer_config.json
|
gpu-pipelined/{kev_0_8b_decode_fp16_pf16 → kev_0_8b_decode_fp16_metal_pf128}/head/head.safetensors
RENAMED
|
File without changes
|
gpu-pipelined/{kev_0_8b_decode_fp16_pf16 → kev_0_8b_decode_fp16_metal_pf128}/head/kev_head.json
RENAMED
|
File without changes
|
gpu-pipelined/kev_0_8b_decode_fp16_metal_pf128/kev_0_8b_decode_fp16_metal_pf128.aimodel/main.hash
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
դ��4�|�1L�dap�.�/Z2����,?hN
|
gpu-pipelined/{kev_0_8b_decode_fp16_pf16/kev_0_8b_decode_fp16_pf16.aimodel → kev_0_8b_decode_fp16_metal_pf128/kev_0_8b_decode_fp16_metal_pf128.aimodel}/main.mlirb
RENAMED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:19d5a4809e348c7cdd314c17cf18646170e2af2ef62f5a32beb8948b2c3f684e
|
| 3 |
+
size 1505385733
|
gpu-pipelined/{kev_0_8b_decode_fp16_pf16/kev_0_8b_decode_fp16_pf16.aimodel → kev_0_8b_decode_fp16_metal_pf128/kev_0_8b_decode_fp16_metal_pf128.aimodel}/metadata.json
RENAMED
|
@@ -1,5 +1,5 @@
|
|
| 1 |
{
|
| 2 |
"assetVersion" : "2.0",
|
| 3 |
"producer" : "coreai-core 1.0.0b2",
|
| 4 |
-
"creationDate" : "
|
| 5 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"assetVersion" : "2.0",
|
| 3 |
"producer" : "coreai-core 1.0.0b2",
|
| 4 |
+
"creationDate" : "20261003T234731Z"
|
| 5 |
}
|
gpu-pipelined/{kev_0_8b_decode_fp16_pf16 → kev_0_8b_decode_fp16_metal_pf128}/metadata.json
RENAMED
|
@@ -1,18 +1,18 @@
|
|
| 1 |
{
|
| 2 |
"metadata_version": "0.2",
|
| 3 |
"kind": "decision-backbone",
|
| 4 |
-
"name": "
|
| 5 |
"assets": {
|
| 6 |
-
"main": "
|
| 7 |
},
|
| 8 |
"language": {
|
| 9 |
"tokenizer": "Qwen/Qwen3.5-0.8B-Base",
|
| 10 |
"vocab_size": 248320,
|
| 11 |
"max_context_length": 4096,
|
| 12 |
"embedded_tokenizer": true,
|
| 13 |
-
"prefill_chunk":
|
| 14 |
"static_inputs": [],
|
| 15 |
-
"output": "hidden [1,
|
| 16 |
"function_map": {
|
| 17 |
"main": [
|
| 18 |
"main"
|
|
@@ -43,15 +43,14 @@
|
|
| 43 |
"formula": "W + (B @ A) * alpha / r (alpha / r = 2.0) on every adapted weight, in fp32, written fp32 (12 target kinds, 186 adapted tensors); every other weight unchanged",
|
| 44 |
"weights_sha256": "ab6bd41853ef2a45d56923202e15c8c9c3dcedaec152b1fb8e529f1ec208415f",
|
| 45 |
"model_safetensors_sha256": "b5ebf92a9994a96d5c0c21b52f5eae23049c9e8b5a85fc625b4cea64076408db",
|
| 46 |
-
"tensors": 320
|
| 47 |
-
"script": "scripts/merge_lora_checkpoint.py (github.com/jaredpalmer/kev tag kev-1.0)"
|
| 48 |
},
|
| 49 |
"hf_model_id": "jaredpalmer/kev-0.8b",
|
| 50 |
"hf_revision": "bf75a6a8848ea6960ff2ed108d9ed44c2941174f",
|
| 51 |
"tokenizer": {
|
| 52 |
"repo": "Qwen/Qwen3.5-0.8B-Base",
|
| 53 |
"revision": "dc7cdfe2ee4154fa7e30f5b51ca41bfa40174e68",
|
| 54 |
-
"
|
| 55 |
"tokenizer.json": "fe000e3ed39ed12b8d2481d527d44f93c65d37e87645d2dcc80d1bf9d50d2927",
|
| 56 |
"tokenizer_config.json": "e611fbccc7c29ef3b1cafb1cb7ea548d189968632901d678fd62be68c47885de"
|
| 57 |
},
|
|
@@ -62,7 +61,7 @@
|
|
| 62 |
"compression": null,
|
| 63 |
"decision": {
|
| 64 |
"output": "hidden [1, S, 1024] per call: the final-norm hidden state at every position (no vocabulary head in the graph)",
|
| 65 |
-
"readout": "one row per question, fresh zero states per row; a row of T ids runs as ceil(T /
|
| 66 |
"row": {
|
| 67 |
"source": "the author's kev.model.encode + rows_of (row form, https://github.com/jaredpalmer/kev tag kev-1.0); conversion/kev/oracle_kev.py records the oracle's rows",
|
| 68 |
"layout": "[<state>] + user_tokens(render(state)) + [<q>] + user_tokens(render(instructions)) + for each option ([<opt>] + user_tokens(option_text) + [</opt>]) + [<decide>]",
|
|
@@ -124,18 +123,7 @@
|
|
| 124 |
"choice": "the criteria names, in order",
|
| 125 |
"score": "'0'..'n-1'"
|
| 126 |
},
|
| 127 |
-
"max_options": 255
|
| 128 |
-
"validation": {
|
| 129 |
-
"source": "kev.api.SystemOneRequest (pydantic v2)",
|
| 130 |
-
"model": "string, default 'kev-latest'",
|
| 131 |
-
"state": "required, any JSON value (null allowed)",
|
| 132 |
-
"questions": "object of question id -> question, at least one",
|
| 133 |
-
"question": "type noul | choice | score; instructions: any JSON (optional); unknown fields ignored",
|
| 134 |
-
"noul": "criteria: object or null (optional); only 'true' and 'false' are read",
|
| 135 |
-
"choice": "criteria: object of 1..255 name -> description (any JSON; null or '' = the name alone)",
|
| 136 |
-
"score": "criteria: array of 1..255 levels (any JSON)",
|
| 137 |
-
"reject": "anything else (the author's server answers 422)"
|
| 138 |
-
}
|
| 139 |
},
|
| 140 |
"head": {
|
| 141 |
"files": [
|
|
@@ -146,15 +134,7 @@
|
|
| 146 |
"scale": 0.0625,
|
| 147 |
"temperature": "kev_head.json temperature (2.3510958125672174, the author's calibration in head.pt)",
|
| 148 |
"code": "kev.model.PointerHead.forward (q, k = nn.Linear(1024, 256)); weights from head.pt, fp32",
|
| 149 |
-
"dtype_of_record": "fp32 (the gate reads the graph's fp16 hidden through the fp32 head)"
|
| 150 |
-
"files_sha256": {
|
| 151 |
-
"head/head.safetensors": "12038b02ecf7ece3ffb008b22ae97988293e521a86e7aadc8036c9c700893a42",
|
| 152 |
-
"head/kev_head.json": "73d510708790422d3c84bfc915e311e536d527bc032f0a07394e76999d5c9d7c"
|
| 153 |
-
},
|
| 154 |
-
"temperature_value": 2.3510958125672174,
|
| 155 |
-
"hidden_size": 1024,
|
| 156 |
-
"head_dim": 256,
|
| 157 |
-
"host_arithmetic": "the round-5 reference host (conversion/kev/host.py) computes the head and the softmax in float64 from the fp16 hidden rows and the fp32 weights and rounds p to fp32 once; the author's PyTorch head is fp32 (on the same hidden rows they differ by fp32 rounding: max |dp| 3.0e-07 over the fixture and held-out rows)"
|
| 158 |
},
|
| 159 |
"softmax": "per question, fp32, over that question's options, after the temperature",
|
| 160 |
"response": {
|
|
@@ -164,17 +144,28 @@
|
|
| 164 |
"score": "{type: 'score', score: sum(i * p_i), legend: {key: render(level)}, probabilities: {key: p}, confidence: max(0, 1 - E|level - mode| / D), D = mean |i - (L-1)/2| over the L levels, mode = the first most likely level (1 when L = 1)}",
|
| 165 |
"normalize": "the confidences normalize p to sum 1 first (all zeros -> uniform)",
|
| 166 |
"rounding": "4 decimals (kev.api.round_prob)",
|
| 167 |
-
"usage": "input_tokens = the encoded request's tokens (the author's packed form: state once + every branch); output_tokens = tokens of json.dumps(answers) (kev.api.output_tokens)"
|
| 168 |
-
|
| 169 |
-
|
| 170 |
-
|
| 171 |
-
|
| 172 |
-
|
| 173 |
-
"
|
| 174 |
-
|
| 175 |
-
|
| 176 |
-
|
| 177 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 178 |
},
|
| 179 |
"gate": {
|
| 180 |
"bar": {
|
|
@@ -185,95 +176,105 @@
|
|
| 185 |
"finite": true
|
| 186 |
},
|
| 187 |
"oracle": "the author's fp32 checkpoint on the CPU (kev.model.admit + the row-form forward, threads 1)",
|
|
|
|
| 188 |
"fixture_434": {
|
| 189 |
"result": "PASS",
|
| 190 |
"rows": 434,
|
| 191 |
"argmax_non_near_tie": "420/420",
|
| 192 |
-
"argmax_near_tie": "
|
| 193 |
-
"max_abs_dp": 0.
|
| 194 |
-
"mean_of_run_mean_abs_dp": 0.
|
| 195 |
"worst_row": {
|
| 196 |
"id": "semif_46b7029b9a704138b77a",
|
| 197 |
"q": 0,
|
| 198 |
-
"max_abs_dp": 0.
|
| 199 |
},
|
| 200 |
"reset_bit_equal_all_processes": true,
|
| 201 |
"finite": true,
|
| 202 |
"oracle_sha256": "1a72a45831eecc2af7ddfc68a86e9250f572d6bf9274362f96cc62bcb803cc6d",
|
| 203 |
-
"
|
| 204 |
-
"
|
| 205 |
-
"
|
| 206 |
-
"transcript": "coreai-model-zoo models/kev-0.8b/gate-kev-0.8b-readout-fp16_pf16.json"
|
| 207 |
},
|
| 208 |
"heldout_130": {
|
| 209 |
"result": "PASS",
|
| 210 |
"rows": 130,
|
| 211 |
"argmax_non_near_tie": "125/125",
|
| 212 |
"argmax_near_tie": "5/5",
|
| 213 |
-
"max_abs_dp": 0.
|
| 214 |
-
"mean_of_run_mean_abs_dp": 0.
|
| 215 |
"worst_row": {
|
| 216 |
"id": "tv4h_11",
|
| 217 |
"q": 0,
|
| 218 |
-
"max_abs_dp": 0.
|
| 219 |
},
|
| 220 |
"reset_bit_equal_all_processes": true,
|
| 221 |
"finite": true,
|
| 222 |
"oracle_sha256": "f715653331a81bd0ff48d48d1f9d294b7b9b3343713a855fd0eed16dd1e505e6",
|
| 223 |
-
"
|
| 224 |
-
"
|
| 225 |
-
"
|
| 226 |
-
"transcript": "coreai-model-zoo models/kev-0.8b/gate-kev-0.8b-heldout.json"
|
| 227 |
},
|
| 228 |
-
"
|
| 229 |
-
"
|
| 230 |
-
|
| 231 |
-
|
| 232 |
-
|
| 233 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 234 |
},
|
| 235 |
-
"
|
| 236 |
-
|
| 237 |
-
|
| 238 |
-
|
| 239 |
-
|
| 240 |
-
|
| 241 |
-
"
|
| 242 |
-
"
|
| 243 |
-
"
|
| 244 |
-
"
|
|
|
|
|
|
|
| 245 |
},
|
| 246 |
-
"
|
| 247 |
-
"
|
| 248 |
-
"
|
| 249 |
-
"
|
| 250 |
-
"
|
| 251 |
-
"
|
| 252 |
-
"
|
| 253 |
-
"host_bar_max_abs_dp": 0.005739,
|
| 254 |
-
"host_bar_mean": 0.0008054,
|
| 255 |
-
"lane_file": "results/e2e_kev-0.8b_heldout.json"
|
| 256 |
},
|
| 257 |
-
"transcript": "coreai-model-zoo models/kev-0.8b/gate-kev-0.8b-
|
| 258 |
},
|
| 259 |
-
"
|
| 260 |
-
"
|
| 261 |
-
"tree_sha256": "de1b6a1a12a0a7a3726bd043e9e5859b3ddea295e5c5dcec7fa87b33d25cef02"
|
| 262 |
},
|
| 263 |
"fixture": "coreai-model-zoo models/kev-0.8b/fixtures-kev-0.8b.json"
|
| 264 |
},
|
| 265 |
-
"compilation": {
|
| 266 |
-
"date": "2026-10-03T07:38:20.417082+00:00",
|
| 267 |
-
"targets": []
|
| 268 |
-
},
|
| 269 |
"metadata_of_record": {
|
| 270 |
-
"exported_metadata_json_sha256": "
|
| 271 |
"edits": [
|
| 272 |
"language.tokenizer = the base repo id (was the local id of the merged checkpoint) + tokenizer_revision",
|
| 273 |
"source.hf_model_id / hf_revision = the Kev checkpoint (was the local HF-cache id of the merged weights)",
|
| 274 |
-
"
|
| 275 |
-
"gate
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 276 |
],
|
| 277 |
-
"written": "2026-10-04 (round
|
| 278 |
}
|
| 279 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"metadata_version": "0.2",
|
| 3 |
"kind": "decision-backbone",
|
| 4 |
+
"name": "kev_0_8b_decode_fp16_metal_pf128",
|
| 5 |
"assets": {
|
| 6 |
+
"main": "kev_0_8b_decode_fp16_metal_pf128.aimodel"
|
| 7 |
},
|
| 8 |
"language": {
|
| 9 |
"tokenizer": "Qwen/Qwen3.5-0.8B-Base",
|
| 10 |
"vocab_size": 248320,
|
| 11 |
"max_context_length": 4096,
|
| 12 |
"embedded_tokenizer": true,
|
| 13 |
+
"prefill_chunk": 128,
|
| 14 |
"static_inputs": [],
|
| 15 |
+
"output": "hidden [1, 128, 1024] fp16, every position",
|
| 16 |
"function_map": {
|
| 17 |
"main": [
|
| 18 |
"main"
|
|
|
|
| 43 |
"formula": "W + (B @ A) * alpha / r (alpha / r = 2.0) on every adapted weight, in fp32, written fp32 (12 target kinds, 186 adapted tensors); every other weight unchanged",
|
| 44 |
"weights_sha256": "ab6bd41853ef2a45d56923202e15c8c9c3dcedaec152b1fb8e529f1ec208415f",
|
| 45 |
"model_safetensors_sha256": "b5ebf92a9994a96d5c0c21b52f5eae23049c9e8b5a85fc625b4cea64076408db",
|
| 46 |
+
"tensors": 320
|
|
|
|
| 47 |
},
|
| 48 |
"hf_model_id": "jaredpalmer/kev-0.8b",
|
| 49 |
"hf_revision": "bf75a6a8848ea6960ff2ed108d9ed44c2941174f",
|
| 50 |
"tokenizer": {
|
| 51 |
"repo": "Qwen/Qwen3.5-0.8B-Base",
|
| 52 |
"revision": "dc7cdfe2ee4154fa7e30f5b51ca41bfa40174e68",
|
| 53 |
+
"files_sha256": {
|
| 54 |
"tokenizer.json": "fe000e3ed39ed12b8d2481d527d44f93c65d37e87645d2dcc80d1bf9d50d2927",
|
| 55 |
"tokenizer_config.json": "e611fbccc7c29ef3b1cafb1cb7ea548d189968632901d678fd62be68c47885de"
|
| 56 |
},
|
|
|
|
| 61 |
"compression": null,
|
| 62 |
"decision": {
|
| 63 |
"output": "hidden [1, S, 1024] per call: the final-norm hidden state at every position (no vocabulary head in the graph)",
|
| 64 |
+
"readout": "one row per question, fresh zero states per row; a row of T ids runs as ceil(T / 128) calls of 'main' (static S = 128): call k gets ids[128k : 128k + 128] with position_ids 0..128k+127; the last call is padded with <|endoftext|> (248044) and the hidden rows of the padded positions are discarded (causal: they cannot reach a real position). The hidden rows [1, 128, 1024] of every call, concatenated and cut to T, are the backbone's final-norm last_hidden_state [T, 1024] the head reads. A row fits when ceil(T / 128) * 128 <= 4095 (the position dim's upper bound; the KV sequence dim is allocated at 4096).",
|
| 65 |
"row": {
|
| 66 |
"source": "the author's kev.model.encode + rows_of (row form, https://github.com/jaredpalmer/kev tag kev-1.0); conversion/kev/oracle_kev.py records the oracle's rows",
|
| 67 |
"layout": "[<state>] + user_tokens(render(state)) + [<q>] + user_tokens(render(instructions)) + for each option ([<opt>] + user_tokens(option_text) + [</opt>]) + [<decide>]",
|
|
|
|
| 123 |
"choice": "the criteria names, in order",
|
| 124 |
"score": "'0'..'n-1'"
|
| 125 |
},
|
| 126 |
+
"max_options": 255
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 127 |
},
|
| 128 |
"head": {
|
| 129 |
"files": [
|
|
|
|
| 134 |
"scale": 0.0625,
|
| 135 |
"temperature": "kev_head.json temperature (2.3510958125672174, the author's calibration in head.pt)",
|
| 136 |
"code": "kev.model.PointerHead.forward (q, k = nn.Linear(1024, 256)); weights from head.pt, fp32",
|
| 137 |
+
"dtype_of_record": "fp32 (the gate reads the graph's fp16 hidden through the fp32 head)"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 138 |
},
|
| 139 |
"softmax": "per question, fp32, over that question's options, after the temperature",
|
| 140 |
"response": {
|
|
|
|
| 144 |
"score": "{type: 'score', score: sum(i * p_i), legend: {key: render(level)}, probabilities: {key: p}, confidence: max(0, 1 - E|level - mode| / D), D = mean |i - (L-1)/2| over the L levels, mode = the first most likely level (1 when L = 1)}",
|
| 145 |
"normalize": "the confidences normalize p to sum 1 first (all zeros -> uniform)",
|
| 146 |
"rounding": "4 decimals (kev.api.round_prob)",
|
| 147 |
+
"usage": "input_tokens = the encoded request's tokens (the author's packed form: state once + every branch); output_tokens = tokens of json.dumps(answers) (kev.api.output_tokens)"
|
| 148 |
+
}
|
| 149 |
+
},
|
| 150 |
+
"gdn_scan": {
|
| 151 |
+
"form": "metal",
|
| 152 |
+
"kernel": "qwen3_5_gdn_chunk_s128",
|
| 153 |
+
"chunk_max": 128,
|
| 154 |
+
"kernel_source": "coreai_models.models.macos.qwen3_5_gdn_metal.build_gdn_chunk_kernel (overlay, unchanged)",
|
| 155 |
+
"threads_per_grid": [
|
| 156 |
+
128,
|
| 157 |
+
16,
|
| 158 |
+
1
|
| 159 |
+
],
|
| 160 |
+
"threads_per_thread_group": [
|
| 161 |
+
128,
|
| 162 |
+
1,
|
| 163 |
+
1
|
| 164 |
+
]
|
| 165 |
+
},
|
| 166 |
+
"compilation": {
|
| 167 |
+
"date": "2026-10-04T04:58:42.572786+00:00",
|
| 168 |
+
"targets": []
|
| 169 |
},
|
| 170 |
"gate": {
|
| 171 |
"bar": {
|
|
|
|
| 176 |
"finite": true
|
| 177 |
},
|
| 178 |
"oracle": "the author's fp32 checkpoint on the CPU (kev.model.admit + the row-form forward, threads 1)",
|
| 179 |
+
"graph": "the AOT h16c .aimodelc of this bundle's .aimodel, Python runtime, SpecializationOptions.default(), 128 ids per call",
|
| 180 |
"fixture_434": {
|
| 181 |
"result": "PASS",
|
| 182 |
"rows": 434,
|
| 183 |
"argmax_non_near_tie": "420/420",
|
| 184 |
+
"argmax_near_tie": "13/14",
|
| 185 |
+
"max_abs_dp": 0.012426,
|
| 186 |
+
"mean_of_run_mean_abs_dp": 0.0009621,
|
| 187 |
"worst_row": {
|
| 188 |
"id": "semif_46b7029b9a704138b77a",
|
| 189 |
"q": 0,
|
| 190 |
+
"max_abs_dp": 0.012425720691680908
|
| 191 |
},
|
| 192 |
"reset_bit_equal_all_processes": true,
|
| 193 |
"finite": true,
|
| 194 |
"oracle_sha256": "1a72a45831eecc2af7ddfc68a86e9250f572d6bf9274362f96cc62bcb803cc6d",
|
| 195 |
+
"when": "2026-10-04T00:00:04+00:00",
|
| 196 |
+
"lane_file_sha256": "e15434cd82af6c6d2adcde98c5db3f4f9e36a94b1e83730b5a9f6996148490ab",
|
| 197 |
+
"transcript": "coreai-model-zoo models/kev-0.8b/gate-kev-0.8b-readout.json"
|
|
|
|
| 198 |
},
|
| 199 |
"heldout_130": {
|
| 200 |
"result": "PASS",
|
| 201 |
"rows": 130,
|
| 202 |
"argmax_non_near_tie": "125/125",
|
| 203 |
"argmax_near_tie": "5/5",
|
| 204 |
+
"max_abs_dp": 0.005783,
|
| 205 |
+
"mean_of_run_mean_abs_dp": 0.0008296,
|
| 206 |
"worst_row": {
|
| 207 |
"id": "tv4h_11",
|
| 208 |
"q": 0,
|
| 209 |
+
"max_abs_dp": 0.0057830810546875
|
| 210 |
},
|
| 211 |
"reset_bit_equal_all_processes": true,
|
| 212 |
"finite": true,
|
| 213 |
"oracle_sha256": "f715653331a81bd0ff48d48d1f9d294b7b9b3343713a855fd0eed16dd1e505e6",
|
| 214 |
+
"when": "2026-10-04T00:00:20+00:00",
|
| 215 |
+
"lane_file_sha256": "801f8420c6a0f001e62e6f848d0fef1c332373b1a0dd43cd7ccf4fe844c0276f",
|
| 216 |
+
"transcript": "coreai-model-zoo models/kev-0.8b/gate-kev-0.8b-readout.json"
|
|
|
|
| 217 |
},
|
| 218 |
+
"red_arms_v2": {
|
| 219 |
+
"result": "RED (every arm)",
|
| 220 |
+
"transcript": "coreai-model-zoo models/kev-0.8b/gate-kev-0.8b-readout.json"
|
| 221 |
+
},
|
| 222 |
+
"swift": {
|
| 223 |
+
"result": "PASS",
|
| 224 |
+
"jit_vs_aot": {
|
| 225 |
+
"records": 10,
|
| 226 |
+
"rows": 25,
|
| 227 |
+
"hidden_sha256_equal": 25,
|
| 228 |
+
"p_bit_equal": 25
|
| 229 |
},
|
| 230 |
+
"transcript": "coreai-model-zoo models/kev-0.8b/gate-kev-0.8b-swift.json"
|
| 231 |
+
},
|
| 232 |
+
"iphone_18_pro": {
|
| 233 |
+
"run": "20261004-140048",
|
| 234 |
+
"pass": true,
|
| 235 |
+
"fixture_434": {
|
| 236 |
+
"argmax_equal_non_near_tie": 420,
|
| 237 |
+
"questions_non_near_tie": 420,
|
| 238 |
+
"argmax_equal_near_tie": 13,
|
| 239 |
+
"near_tie_questions": 14,
|
| 240 |
+
"max_abs_dp": 0.01100119948387146,
|
| 241 |
+
"mean_of_run_mean_abs_dp": 0.0009723608388956548
|
| 242 |
},
|
| 243 |
+
"heldout_130": {
|
| 244 |
+
"argmax_equal_non_near_tie": 125,
|
| 245 |
+
"questions_non_near_tie": 125,
|
| 246 |
+
"argmax_equal_near_tie": 5,
|
| 247 |
+
"near_tie_questions": 5,
|
| 248 |
+
"max_abs_dp": 0.005920350551605225,
|
| 249 |
+
"mean_of_run_mean_abs_dp": 0.0008314862770240348
|
|
|
|
|
|
|
|
|
|
| 250 |
},
|
| 251 |
+
"transcript": "coreai-model-zoo models/kev-0.8b/gate-kev-0.8b-iphone.json"
|
| 252 |
},
|
| 253 |
+
"host": {
|
| 254 |
+
"transcript": "coreai-model-zoo models/kev-0.8b/gate-kev-0.8b-host.json"
|
|
|
|
| 255 |
},
|
| 256 |
"fixture": "coreai-model-zoo models/kev-0.8b/fixtures-kev-0.8b.json"
|
| 257 |
},
|
|
|
|
|
|
|
|
|
|
|
|
|
| 258 |
"metadata_of_record": {
|
| 259 |
+
"exported_metadata_json_sha256": "a1b6e5a35ab4d744b8a469caa8c22ad574437a80b5864dd93f6bbc87979d873c",
|
| 260 |
"edits": [
|
| 261 |
"language.tokenizer = the base repo id (was the local id of the merged checkpoint) + tokenizer_revision",
|
| 262 |
"source.hf_model_id / hf_revision = the Kev checkpoint (was the local HF-cache id of the merged weights)",
|
| 263 |
+
"source.tokenizer.shipped_in added",
|
| 264 |
+
"gate: the published form's gate numbers and the zoo transcripts that hold them"
|
| 265 |
+
],
|
| 266 |
+
"host_read_keys_unchanged": [
|
| 267 |
+
"name",
|
| 268 |
+
"kind",
|
| 269 |
+
"assets",
|
| 270 |
+
"language.vocab_size",
|
| 271 |
+
"language.max_context_length",
|
| 272 |
+
"language.prefill_chunk",
|
| 273 |
+
"language.query_len_range",
|
| 274 |
+
"language.query_len_call_max",
|
| 275 |
+
"language.query_len_multiple",
|
| 276 |
+
"decision"
|
| 277 |
],
|
| 278 |
+
"written": "2026-10-04 (round 17)"
|
| 279 |
}
|
| 280 |
}
|
gpu-pipelined/{kev_0_8b_decode_fp16_pf16 → kev_0_8b_decode_fp16_metal_pf128}/tokenizer/tokenizer.json
RENAMED
|
File without changes
|
gpu-pipelined/{kev_0_8b_decode_fp16_pf16 → kev_0_8b_decode_fp16_metal_pf128}/tokenizer/tokenizer_config.json
RENAMED
|
File without changes
|
gpu-pipelined/kev_0_8b_decode_fp16_pf16/kev_0_8b_decode_fp16_pf16.aimodel/main.hash
DELETED
|
@@ -1 +0,0 @@
|
|
| 1 |
-
A1�(H�x{�l�l�*ph�G�N%��˗[��rh-
|
|
|
|
|
|