-
Notifications
You must be signed in to change notification settings - Fork 3
Expand file tree
/
Copy pathrecipes.json
More file actions
186 lines (186 loc) · 12.2 KB
/
Copy pathrecipes.json
File metadata and controls
186 lines (186 loc) · 12.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
{
"schema": 1,
"updated": "2026-09-27",
"about": "Which GPU runs which model, and how. scripts/recipe.py turns a recipe id into the environment scripts/up.sh launches with; the frontend lists what is here. Every number carries {v, measured, src}: measured means read on that card, anything else is an estimate and says so.",
"gpus": [
{
"id": "a100-40g",
"card": "A100-40G",
"accelerator": "A100",
"shape": "st",
"vram_gb": 40,
"ram_gb": 83,
"min_vram_gib": 36,
"cu_per_hour": 5.37,
"facts": {
"cu_per_hour": {"v": 5.37, "measured": false, "src": "Colab UI figure, never read from ccu-info"},
"vram": {"v": 39, "unit": "GiB", "measured": true, "src": "a 40 GB draw probes at ~39 GiB (scripts/restore.py)"},
"ram": {"v": 83, "unit": "GiB", "measured": false, "src": "Colab's standard A100 shape; not probed here"}
}
},
{
"id": "a100-80g-hm",
"card": "A100-80G High-RAM",
"accelerator": "A100",
"shape": "hm",
"vram_gb": 80,
"ram_gb": 167,
"min_vram_gib": 70,
"cu_per_hour": 6.77,
"facts": {
"cu_per_hour": {"v": 6.77, "measured": true, "src": "Colab /tun/m/ccu-info consumptionRateHourly with this box as the account's only assignment, 2026-09-25 (older docs quote 7.52)"},
"vram": {"v": 79.3, "unit": "GiB", "measured": true, "src": "restore.py probe, docs/MEASURED.md"},
"ram": {"v": 167.1, "unit": "GiB", "measured": true, "src": "restore.py probe, docs/MEASURED.md"}
}
}
],
"models": [
{
"id": "qwen38-fn",
"name": "Qwen3.8-Flash-Next",
"short": "EXL3 4-bit",
"params": "125B-A6B MoE",
"quant": "EXL3 4.05 bpw",
"repo": "turboderp/Qwen3.8-Flash-Next-exl3",
"branch": "4.05bpw_h6_ng6",
"revision": "55a732e0c4c3d4614bc42b68493bb930d9b02c0a",
"dir": "/content/exl3",
"served_id": "qwen3.8-flash-next-exl3",
"bytes": 107463600896,
"native_ctx": 262144,
"vision": true,
"mtp": true,
"ngram": true,
"recurrent": true
},
{
"id": "qwen38-27b",
"name": "Qwen3.8-27B",
"short": "EXL3 4-bit",
"params": "27B dense, hybrid DeltaNet",
"quant": "EXL3 4.00 bpw, self-calibrated (head 5, vision 6 bits)",
"repo": "turboderp/Qwen3.8-27B-exl3",
"branch": "SC_4.00bpw_H5_V6",
"revision": "516bf129059031c6da9416768ea6b7a1be00a8fc",
"dir": "/content/models/qwen38-27b",
"served_id": "qwen3.8-27b-exl3",
"bytes": 16384243706,
"native_ctx": 262144,
"vision": true,
"mtp": true,
"ngram": false,
"recurrent": true
},
{
"id": "qwen38-fn-strata",
"name": "Qwen3.8-Flash-Next",
"short": "Strata 3.5-bit",
"params": "125B-A6B MoE",
"quant": "GGUF IQ3_S, 3.50 bpw (ISTA-DASLab GSQ-RCO), on Strata",
"engine": "strata",
"repo": "ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF",
"branch": "main",
"revision": "2a55d75962e22f7a4a1d9963ab6eae3678537831",
"dir": "/content/strata-data",
"served_id": "qwen3.8-flash-next-iq3_s",
"bytes": 83600000000,
"native_ctx": 262144,
"vision": true,
"mtp": true,
"ngram": true,
"recurrent": true,
"strata": {
"repo": "https://github.com/architectds/Strata",
"commit": "e416ba57f589f7c2561a1ed14ef72d4358f14027",
"release": "https://github.com/architectds/Strata/releases/download/v0.1.24-linux-cuda12.8/",
"sha256": "f9cdd1952fec36db0c577417c48c09e14dbe37174dd0ff4bf977c0b12fc40371",
"cuda": 12,
"quant": "IQ3_S"
}
}
],
"recipes": [
{
"id": "a100-40g/qwen38-27b",
"gpu": "a100-40g",
"model": "qwen38-27b",
"status": "unmeasured",
"eta_min": 6,
"env": {
"CACHE_SIZE": 819200,
"CACHE_QUANT": 4,
"CPU_CACHE_GB": 16,
"RECURRENT_CACHE_GB": 16,
"NDT": 4,
"GCS": 8192,
"VISION": 1,
"YARN_FACTOR": 1.5625,
"CONCURRENCY": 2
},
"facts": {
"weights": {"v": 15.26, "unit": "GiB", "measured": true, "src": "16,384,243,706 B at branch SC_4.00bpw_H5_V6 (Hugging Face tree), embeddings, 6-bit vision tower and 4-bit MTP layer included; the pack's own plot puts this quant at KL 0.0062 vs ~0.015 for the plain 3.50 bpw it replaces"},
"vram_mib": {"v": 33500, "measured": false, "src": "derived: 17.1 GiB measured outside the KV on 2026-09-27 (26,746 MiB in all with Q8 KV at 262K) + Q4 KV for 819,200 tokens at 19,584 B each (16 attention layers + the MTP layer) = 14.9 GiB + the second recurrent state slot for two-way decode, 728 MiB (48 GDN layers x 48 heads x 128 x 128 fp32 x 5 for ndt 4, plus conv state); not yet loaded this way"},
"prefill_tps": {"v": 2544, "measured": true, "src": "32,551-token cold prompt, second of a pair, on the A100-40G 2026-09-27 with Q8 KV (4K: 2,935; 119K: 1,435)"},
"decode_tps": {"v": 50.4, "measured": true, "src": "512 tokens at 32K context, MTP ndt 4, on the A100-40G 2026-09-27 with Q8 KV (short context: 62.9; ~128K: ~44); Q4 reads half the KV per token, so long contexts should come out a little faster"},
"eta_min": {"v": 6, "measured": true, "src": "billing to READY in 5 min 51 s on 2026-09-27, 43.6 s of it a warm-up since cut to ~8K tokens"},
"context": {"v": 409600, "measured": false, "src": "per conversation: YaRN 1.5625 over the native 262,144 (the model card documents YaRN up to 1M); the 819,200-token cache holds two such conversations, decoding together; quality past 262K not measured"},
"vision": {"v": true, "measured": false, "src": "the tower is quantized to 6 bits inside the shards (needs ExLlamaV3 >= 1.4.4; 1.5.1 here); never loaded"},
"concurrency": {"v": 2, "measured": false, "src": "two jobs per engine step: -ambs 2 gives the second recurrent state slot, and api_server's one conductor thread drives the Generator for both; 2 x 409,600 positions fit the 819,200-token cache; a third request waits its turn; not yet run on this card"},
"host_ram": {"v": "ccs 16 + rcs 16 GiB", "measured": false, "src": "36.7 of 83.5 GiB in use on 2026-09-27; rcs is 16, not 8: a GDN checkpoint here is ~150 MB (48 layers x 48 heads x 128 x 128 fp32), one every ~2048 tokens, so an 86K-token Codex prompt leaves ~42 of them (~6 GB)"}
}
},
{
"id": "a100-80g/qwen38-fn",
"gpu": "a100-80g-hm",
"model": "qwen38-fn",
"status": "verified",
"eta_min": 11,
"env": {
"CACHE_SIZE": 500224,
"CACHE_QUANT": 4,
"CPU_CACHE_GB": 32,
"RECURRENT_CACHE_GB": 24,
"NDT": 4,
"GCS": 8192,
"VISION": 1,
"EXL3_VISION_PINNED": 1,
"YARN_FACTOR": 2,
"CONCURRENCY": 1
},
"facts": {
"weights": {"v": 100.1, "unit": "GiB", "measured": true, "src": "107,463,600,896 B at branch 4.05bpw_h6_ng6 (Hugging Face tree): the 4.05 bpw weights with the 36.36 GiB n-gram table (host RAM, -ngr) and the 6-bit vision tower (561 MB)"},
"prefill_tps": {"v": 3882, "measured": true, "src": "gcs sweep at 8192, docs/MEASURED.md (2,806 at gcs 2048)"},
"decode_tps": {"v": 97.4, "measured": true, "src": "MTP ndt 4 at a 30K context, docs/MEASURED.md"},
"load_s": {"v": 259.5, "measured": true, "src": "serve.sh at cache 500224, docs/MEASURED.md"},
"eta_min": {"v": 11, "measured": true, "src": "download 280 s + load 259.5 s + install, docs/MEASURED.md"},
"vram_mib": {"v": 76437, "measured": true, "src": "after load at cache 500224, gcs 4096, vision off; with gcs 8192 and the tower together: not measured"},
"context": {"v": 500224, "measured": false, "src": "the KV fits (measured); positions past the native 262,144 come from YaRN x2 (rope_parameters + max_position_embeddings 524288, which is how exllamav3 derives the factor); quality there not measured"},
"vision": {"v": true, "measured": false, "src": "the pack carries vision_k6.safetensors (561 MB); its weights sit in pinned host RAM (EXL3_VISION_PINNED) so the tower costs no VRAM for weights; image encoding speed not measured"},
"concurrency": {"v": 1, "measured": true, "src": "one job per engine step (CONCURRENCY 1); how many live jobs would fit is computed in docs/CONCURRENCY.md, never exercised"},
"host_ram": {"v": "ngr 36.4 + ccs 32 + rcs 24 GiB", "measured": false, "src": "the verified run used -ccs 16 -rcs 16"}
}
},
{
"id": "a100-40g/qwen38-fn-strata",
"gpu": "a100-40g",
"model": "qwen38-fn-strata",
"status": "verified",
"eta_min": 20,
"env": {
"STRATA_CONTEXT": 262144,
"STRATA_KV": "int8"
},
"facts": {
"weights": {"v": 77.9, "unit": "GiB", "measured": true, "src": "83.6 GB on Hugging Face: two IQ3_S shards (3.50 bpw on average; GSQ-RCO picks each tensor's type, the experts mixing Q2_0 and IQ4_NL by layer) with the 28.8 GB n-gram table; ISTA-DASLab, Apache-2.0. Their benchmarks put it level with BF16 (AIME25, GPQA-Diamond, LiveCodeBench v6: 93.26 vs 93.12)"},
"prefill_tps": {"v": 2082, "measured": true, "src": "engine 0.1.24 on the A100-40G, 2026-09-29, real text (Python's standard library, nothing cached), the n-gram table in RAM: 1,448 t/s at 16K, 2,082 at 36K, 2,492 at 70K, 2,497 at 144K (2,464 on a 41K prompt; 2,056-2,513 at 55-118K on mixed code and docs). 0.1.20 read 905-1,376 at 4K-59K: 0.1.22 moved the prompt's attention onto tensor cores and 0.1.24 its block selection, the part that grew with length. Short prompts vary most (825-1,702 at ~5K) with the text and a warm row cache"},
"decode_tps": {"v": 58.8, "measured": true, "src": "a 512-token answer at a short context, engine 0.1.24 on the A100-40G, 2026-09-29, expert hit rate 0.96; 66.1 after a 41K prompt, 72.3 on a short next turn. The same prompt gave 58.0-61.0 on 0.1.20: 0.1.24 changes prompts, not decode. 16,251 of the 24,576 experts live in VRAM with images on; Colab's Xeon (6 cores, 2.2 GHz, AVX2) computes the rest"},
"eta_min": {"v": 20, "measured": true, "src": "engine 0.1.24 on 2026-09-29: READY 18 min 30 s after the box was assigned (20 min 41 s on the app's billing clock, which started at the confirm that got it): the model's 84.5 GB at its pinned revision in 225 s (Xet, ~375 MB/s; 413 s that morning), Strata's setup and model pack ~1 min, the MTP layer (fetched from Qwen's checkpoint and converted) ~8.5 min (4 min that morning), the engine's start 4.5 min (46.84 GiB of experts at 0.35 GiB/s); the n-gram table then goes into RAM in the background (77 s). Colab had no A100 free for 15 min before it: 'Service Unavailable', no box, no billing"},
"vram_mib": {"v": 40737, "measured": true, "src": "the whole card at 262K with images, engine 0.1.24, 2026-09-29: what the dense layers, MTP, a 32K-position KV page pool and the image encoder's 700 MiB leave becomes the expert cache (16,251 of 24,576 experts)"},
"context": {"v": 262144, "measured": true, "src": "the native window: Strata's engine reports max_context 262,144 on the A100-40G, 2026-09-29, with the 8-bit KV streamed (a 32K-position page pool in VRAM, the rest in pinned RAM, ~13.7 KB/token). Strata's own setup capped it at 128K below 90 GB of RAM; the fork's setup counts the RAM instead (Niko1221/Strata#134). YaRN beyond 262K: Strata's RoPE has no scaling yet"},
"vision": {"v": true, "measured": true, "src": "the image encoder on the card (llama.cpp mtmd, 700 MiB reserved from the expert cache): a 640x400 test image read right (shapes, colors, places, its text) in 3.1 s with engine 0.1.24, which thinks before it answers (208 tokens; 0.1.20 answered in 123)"},
"concurrency": {"v": 1, "measured": true, "src": "Strata serves one request at a time, by design; and no /v1/responses (chat completions and Anthropic messages only), so Codex cannot use it yet; ModelDock can, with its chat transport"},
"host_ram": {"v": "56 of 83.5 GiB", "measured": true, "src": "engine 0.1.24 at 262K with images, 2026-09-29: 56.3 GiB in use; the engine holds the 46.84 GiB pinned expert arena, 3.9 GiB of pinned KV and buffers and its heap, the rest of the box ~1.5 GiB; the ~27 GiB left is page cache, where strata_warm keeps ~90% of the 26.8 GiB n-gram table"}
}
}
]
}