Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -266,3 +266,10 @@ examples/demo.py # minimal API walkthrough
## License

Apache-2.0, including vendored third-party code (see [NOTICE](NOTICE)).

### Persistent conversation caching (opt-in)

Reuse processed prompt prefixes across requests and process restarts with
`edge0 serve /path/to/model --cache-dir /path/to/conversation-cache`.
Defaults: 20 GiB of checkpoint payloads and checkpoints every 2,048 tokens.
See [configuration, guarantees, and benchmarks](docs/conversation-cache.md).
265 changes: 265 additions & 0 deletions docs/benchmarks/conversation-cache-m5.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,265 @@
[
{
"case": "disabled",
"output_tokens": [
1092,
13321,
300,
268,
25825,
198,
198,
678
],
"ttft_s": 6.5136767080002755,
"wall_s": 7.46466375,
"post_first_token_effective_tok_s": 7.360773271190409,
"tokenization_s": 0.018990666999343375,
"prefill_s": 6.318684333999954,
"decode_s": 1.1266685829996277,
"decode_tok_s": 7.100579638691002,
"peak_mlx_bytes": 2731136504,
"peak_rss_bytes": 4904845312,
"io_read_blocks": 0,
"io_write_blocks": 0,
"system_disk_read_bytes": 573726720,
"system_disk_write_bytes": 179904512,
"usage": {
"prompt_tokens": 2799,
"completion_tokens": 8,
"total_tokens": 2807
}
},
{
"case": "first_write",
"output_tokens": [
1092,
13321,
300,
268,
25825,
198,
198,
678
],
"ttft_s": 6.774661375000505,
"wall_s": 7.7653235410007255,
"post_first_token_effective_tok_s": 7.0659809572241645,
"tokenization_s": 0.006916833000104816,
"prefill_s": 5.296447042001091,
"decode_s": 0.5170845000002373,
"decode_tok_s": 15.471359129883663,
"peak_mlx_bytes": 3387898722,
"peak_rss_bytes": 5075402752,
"io_read_blocks": 0,
"io_write_blocks": 0,
"system_disk_read_bytes": 260599808,
"system_disk_write_bytes": 1423396864,
"usage": {
"prompt_tokens": 2799,
"completion_tokens": 8,
"total_tokens": 2807,
"prompt_tokens_details": {
"cached_tokens": 0
}
},
"cache": {
"lookup_s": 0.0003477079999356647,
"writes": 3,
"write_s": 1.9365312069994616,
"stored_bytes": 761533603,
"prefill_s": 5.296447042001091,
"remaining_prefill_tokens": 2799,
"tokenization_s": 0.006814792000113812
}
},
{
"case": "repeat",
"output_tokens": [
1092,
13321,
300,
268,
25825,
198,
198,
678
],
"ttft_s": 0.5033716249999998,
"wall_s": 1.0222226670002783,
"post_first_token_effective_tok_s": 13.491348062082608,
"tokenization_s": 0.0013498750004146132,
"prefill_s": 1.583999619469978e-06,
"decode_s": 0.6283079169998018,
"decode_tok_s": 12.732610529882155,
"peak_mlx_bytes": 1777306570,
"peak_rss_bytes": 4981604352,
"io_read_blocks": 0,
"io_write_blocks": 0,
"system_disk_read_bytes": 27803648,
"system_disk_write_bytes": 25305088,
"usage": {
"prompt_tokens": 2799,
"completion_tokens": 8,
"total_tokens": 2807,
"prompt_tokens_details": {
"cached_tokens": 2799
}
},
"cache": {
"lookup_s": 0.00028974999986530747,
"restore_s": 0.38771641599942086,
"reused_tokens": 2799,
"prefill_s": 1.583999619469978e-06,
"remaining_prefill_tokens": 0,
"tokenization_s": 0.0012944999998580897
}
},
{
"case": "appended_turn",
"output_tokens": [
198,
198,
6952,
698,
268,
9379,
3189,
391
],
"ttft_s": 1.6190897500000574,
"wall_s": 2.5677901669996572,
"post_first_token_effective_tok_s": 7.378514728747033,
"tokenization_s": 0.005887457999961043,
"prefill_s": 0.5801039170000877,
"decode_s": 0.4516456660003314,
"decode_tok_s": 17.71300070439319,
"peak_mlx_bytes": 1819663332,
"peak_rss_bytes": 4993744896,
"io_read_blocks": 0,
"io_write_blocks": 0,
"system_disk_read_bytes": 66650112,
"system_disk_write_bytes": 782823424,
"usage": {
"prompt_tokens": 2832,
"completion_tokens": 8,
"total_tokens": 2840,
"prompt_tokens_details": {
"cached_tokens": 2807
}
},
"cache": {
"lookup_s": 0.0002899589999287855,
"restore_s": 0.3203155000001061,
"reused_tokens": 2807,
"writes": 2,
"write_s": 1.206859083999916,
"stored_bytes": 992792613,
"prefill_s": 0.5801039170000877,
"remaining_prefill_tokens": 25,
"tokenization_s": 0.005829499999890686
}
},
{
"case": "branch",
"output_tokens": [
198,
198,
6952,
698,
1099,
297,
6474,
3832
],
"ttft_s": 1.5322595829993588,
"wall_s": 2.667079666999598,
"post_first_token_effective_tok_s": 6.168378669616958,
"tokenization_s": 0.005418582999482169,
"prefill_s": 0.46757583300041006,
"decode_s": 0.6374825420007255,
"decode_tok_s": 12.549363273372427,
"peak_mlx_bytes": 1840880720,
"peak_rss_bytes": 5205737472,
"io_read_blocks": 0,
"io_write_blocks": 0,
"system_disk_read_bytes": 78704640,
"system_disk_write_bytes": 906326016,
"usage": {
"prompt_tokens": 2831,
"completion_tokens": 8,
"total_tokens": 2839,
"prompt_tokens_details": {
"cached_tokens": 2807
}
},
"cache": {
"lookup_s": 0.0002689579996513203,
"restore_s": 0.3277814999992188,
"reused_tokens": 2807,
"writes": 2,
"write_s": 1.2258794170002147,
"stored_bytes": 1223715239,
"prefill_s": 0.46757583300041006,
"remaining_prefill_tokens": 24,
"tokenization_s": 0.005361083000025246
}
},
{
"case": "process_restart",
"output_tokens": [
1092,
13321,
300,
268,
25825,
198,
198,
678
],
"ttft_s": 0.6099199159998534,
"wall_s": 1.224041457999192,
"post_first_token_effective_tok_s": 11.398395140497348,
"tokenization_s": 0.005186958000194863,
"prefill_s": 6.250011210795492e-07,
"decode_s": 0.8147992079993855,
"decode_tok_s": 9.818369877460697,
"peak_mlx_bytes": 1190999148,
"peak_rss_bytes": 2118565888,
"io_read_blocks": 0,
"io_write_blocks": 0,
"system_disk_read_bytes": 38510592,
"system_disk_write_bytes": 36405248,
"usage": {
"prompt_tokens": 2799,
"completion_tokens": 8,
"total_tokens": 2807,
"prompt_tokens_details": {
"cached_tokens": 2799
}
},
"cache": {
"lookup_s": 0.003613083999880473,
"restore_s": 0.3949573749996489,
"reused_tokens": 2799,
"prefill_s": 6.250011210795492e-07,
"remaining_prefill_tokens": 0,
"tokenization_s": 0.005138458000146784
}
},
{
"case": "radix_lookup",
"checkpoints": 1000,
"lookup_us": 1.5507749999414955
},
{
"case": "radix_lookup",
"checkpoints": 10000,
"lookup_us": 1.5693625000494649
},
{
"case": "radix_lookup",
"checkpoints": 100000,
"lookup_us": 1.5628167000613757
}
]
101 changes: 101 additions & 0 deletions docs/conversation-cache-results.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,101 @@
# Local cache measurements

Measured September 12, 2026 on an Apple M5 MacBook Air with 24 GB unified memory,
local `edge0-8b`, its shipped LoRA and prerouter, MLX 0.30.6, and mlx-lm 0.31.0.
The fixed coding fixture contains 2,799 prompt tokens. Each request generates eight
greedy tokens. The appended turn and branch contain 2,832 and 2,831 prompt tokens.
The interval is 2,048; the checkpoint budget is 20 GiB.

These are sequential single samples on an active laptop, not confidence intervals.
The OS page cache was warm. The baseline runs first and therefore includes more
kernel/expert warmup. Initial model loading and artifact hashing are excluded from
request latency. A fresh process is measured separately; no claim of cold-SSD
performance is made. [Raw measurements](benchmarks/conversation-cache-m5.json)
include output token IDs, memory, decode, and disk counters.

| Request | Reused tokens | TTFT (s) | Wall (s) | Restore (s) | Remaining prefill (s) | Writes (s) |
|---|---:|---:|---:|---:|---:|---:|
| Disabled | 0 | 6.514 | 7.465 | — | 6.319 | — |
| First write | 0 | 6.775 | 7.765 | — | 5.296 | 1.937 |
| Repeat | 2,799 | 0.503 | 1.022 | 0.388 | <0.001 | 0 |
| Appended turn | 2,807 | 1.619 | 2.568 | 0.320 | 0.580 | 1.207 |
| Branch | 2,807 | 1.532 | 2.667 | 0.328 | 0.468 | 1.226 |
| Process restart | 2,799 | 0.610 | 1.224 | 0.395 | <0.001 | 0 |

Write time includes prompt and completed-generation snapshots, so not all of it
falls before the first token. Prompt and completion counts remain unchanged by
reuse. Disabled, first-write, repeat, and restarted requests produced identical
eight-token outputs. Appended/branched requests processed only 25/24 unmatched
tokens through prefill.

At this measured prefix length, the first cached request cost 0.261 s extra TTFT
and 0.301 s extra total time. One repeat saved 6.010 s TTFT and 6.442 s total time
against the disabled sample, so the first repeat amortized the observed initial
cost. Even charging the full 1.937 s publication time as overhead, one repeat
covered it. This is a measured **reuse-count** break-even at 2,799 tokens, not a
measured minimum token-length threshold. Short-prompt break-even, randomized
request-order trials, sustained workloads, and cold filesystem trials remain
unmeasured. Earlier development samples ranged from 5.45–6.36 s disabled TTFT
and 0.49 s repeat TTFT; laptop load and warmup affect the absolute values.

| Request | Peak MLX (GiB) | Sampled peak RSS (GiB) | Decode (tokens/s) |
|---|---:|---:|---:|
| Disabled | 2.54 | 4.57 | 7.10 |
| First write | 3.16 | 4.73 | 15.47 |
| Repeat | 1.66 | 4.64 | 12.73 |
| Appended turn | 1.69 | 4.65 | 17.71 |
| Branch | 1.71 | 4.85 | 12.55 |
| Process restart | 1.11 | 1.97 | 9.82 |

Publication increased peak MLX allocation by about 24% in this sample. The active
attention cache still lives in RAM; caching is not active-context offloading.
RSS includes expert caches and other allocations, and is sampled every 20 ms.
Decode values cover only eight tokens and have substantial warmup/noise; they do
not establish a decode-speed improvement. Including final synchronous publication,
the effective rate after the first token was 7.07 tokens/s for first write versus
13.49 for repeat.

Unique payloads occupied 761,533,603 bytes after the first conversation, then
1,223,715,239 bytes after both branches. System-wide disk counters reported about
1.42 GB written during first write, 0.78 GB for the appended turn, and 0.91 GB for
the branch. These include other processes and filesystem behavior. Per-process
block counters remained zero on this macOS run. Content deduplication saves
retained storage, but the current synchronous codec reserializes shared blocks;
it does not eliminate their write/checksum work.

A separate radix-only benchmark performed 10,000 lookups on 130-token keys:

| Checkpoints | Mean lookup (µs) |
|---:|---:|
| 1,000 | 1.55 |
| 10,000 | 1.57 |
| 100,000 | 1.56 |

This isolates index traversal. It excludes SQLite startup/rebuild, lock contention,
and payload restore. End-to-end warm lookup was about 0.27–0.35 ms in the verified
run; first lookup after restart, including index construction, was 3.61 ms.

## Correctness coverage and remaining limits

The tests cover radix branches and exact hits, incremental local index updates,
namespace/artifact invalidation, reference cleanup and LRU eviction, physical
storage limits, missing/corrupt payloads and metadata, repair of shared corrupt
blocks, interrupted publication including abrupt process exit, and concurrent
threads/processes. Continuation coverage includes attention plus recurrent state,
family prerouter fields and expert staging, exact hits, one-token suffixes,
intermediate prefill boundaries, full sampling history, EOS, generation limits,
and callback cancellation. HTTP session tests verify released idle context and
unchanged total token accounting.

Real 8B continuation logits matched within `rtol=1e-4, atol=1e-4`; restored prompt
logits used `1e-5`. The actual small Qwen gated-delta/attention backbone also passed
with random weights at `1e-5`. Real-weight **35B validation remains outstanding**
because that checkpoint is unavailable locally.

A checkpoint resumes the saved execution boundary. Hybrid prerouter behavior can
depend on prefill versus decode execution and chunk boundaries; the continuation
contract is equality with the same uninterrupted saved trajectory, not arbitrary
re-chunking of a previously decoded conversation. Corrupt entries become misses;
whole-database destruction or cache filesystem failure is not a recovery mechanism
for the inference service itself. No background write queue or quantized KV format
is included in this phase.
Loading