-
Notifications
You must be signed in to change notification settings - Fork 1k
Expand file tree
/
Copy pathsetup.py
More file actions
4422 lines (4001 loc) · 260 KB
/
Copy pathsetup.py
File metadata and controls
4422 lines (4001 loc) · 260 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
#!/usr/bin/env python3
"""Strata one-click setup and start (Windows and Linux, NVIDIA or AMD GPUs).
START-HERE.bat (Windows) / ./setup.sh (Linux) - they install Python if needed and run this file
The first time it asks four questions - which model (the original Qwen3.8-Flash-Next or the Swift 1.5 fine-tune),
which size, how much context, and whether the model should also read images - then installs everything and starts the model on http://127.0.0.1:8080 (OpenAI- and Anthropic-compatible
API; a small page there shows that it runs). Every later start skips straight to running the model: nothing that
is already downloaded, installed or prepared is done again.
What the first run does (each step is skipped when it is already done):
1. checks your PC: NVIDIA or AMD GPU and driver, RAM, CPU, free disk space
2. asks the questions
3. installs the Python packages it needs into .venv (numpy, jinja2, ..., and NVIDIA's CUDA libraries)
4. gets the Strata engine: a ready-made build for RTX 20/30/40/50 cards (no compiler needed); if none fits your PC,
it installs the build tools (asks first) and compiles the engine for your GPU. AMD (--backend hip, chosen by
itself on a PC with no usable NVIDIA card): the ready-made HIP engine on Windows, compiled here on Linux
5. downloads the model from Hugging Face (resumable), and the vision encoder if you want images
6. prepares the model for Strata and fetches the MTP draft layer (~5 GB, from the original Qwen checkpoint)
7. writes run-<model>.bat / run-<model>.sh and starts the model
Options: --family qwen|swift, --model Q2_0|IQ2_XS|IQ3_XXS|IQ3_S, --context 32768, --rope-scaling none|linear|yarn
(--rope-scale F; past the trained 262144 the setup adds yarn and the factor is the final context over 262144,
at least 1 - an explicit --rope-scaling none is refused for such a context), --vision yes|no|gpu|cpu, --port
8080, --yes (recommended
answers, no questions), --setup (install another model / change settings instead of starting), --no-start,
--host 0.0.0.0 --api-key KEY (reach it from other devices on your network), --experimental-speed-projection on|off
(EXPERIMENTAL, off by default),
--models-dir DIR, --gguf-dir DIR (use GGUF files you already have), --build (compile instead of the ready-made
engine), --check (only check this PC), --resident-budget-gib N (UD-Q4_K_XL's or UD-IQ4_XS's experts in RAM),
--kv-streaming on|off|auto.
Setup recommends, it never forces: the recommended answers are the defaults (--yes, or Enter), and a bigger choice
than it recommends - a longer context, more GPUs, a bigger RAM budget, a size it thinks will not fit - is kept, with
what it risks. With --yes, an explicit flag (--model, --gpus, ...) is the consent to a risk setup would otherwise
stop at; --yes alone is not.
"""
from __future__ import annotations
import argparse
import ctypes
import hashlib
import json
import math
import os
import platform
import re
import shutil
import struct
import subprocess
import sys
import textwrap
import time
import urllib.error
import urllib.request
import zipfile
from pathlib import Path
ROOT = Path(__file__).resolve().parent
WIN = os.name == "nt"
# #214: every Hugging Face file comes from a fixed commit of its repository (the `sha` of
# https://huggingface.co/api/models/<repo> when this was pinned), so a checkout installs the same files on any
# day. A revision the repository no longer has falls back to its current files, with a message (download()).
HF_REVISIONS = {
"ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF": "ed59f92082b1e93c0e96d60a8b11aab089b52f09", # 2026-09-29
"ukisai/Swift-1.5-Qwen3.8-Flash-Next-GSQ-RCO-GGUF": "b22d729eae29b5796f76fb70f91aef549b9fc52c", # 2026-09-24
"ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-Coder-GGUF": "5348543e0147355ac9cbcb031184a3546350988e", # 2026-09-29
"unsloth/Qwen3.8-Flash-Next-GGUF": "38bb39ee97821de2c9009abb7e93950eec396e66", # 2026-09-30
}
HF_DEFAULT = "https://huggingface.co"
def hf_endpoint() -> str:
"""#495: the Hugging Face host - HF_ENDPOINT as huggingface_hub reads it (a mirror, e.g. https://hf-mirror.com),
else huggingface.co. The pinned revisions and the SHA-256 checks are the same whichever host serves the files."""
return (os.environ.get("HF_ENDPOINT") or "").strip().rstrip("/") or HF_DEFAULT
def hf(repo: str) -> str:
"""The download folder of a Hugging Face repository at its pinned revision."""
return f"{hf_endpoint()}/{repo}/resolve/{HF_REVISIONS[repo]}/"
def hf_unpinned(url: str) -> str:
"""The same file at the repository's current revision (main)."""
return re.sub(r"^(https?://[^/]+/.+?/resolve/)[0-9a-f]{40}/", r"\1main/", url, count=1)
HF = hf("ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF")
LLAMA_CPP_COMMIT = "3cf03257f219afbe7334045ff7c6a06ac68c627d"
LLAMA_CPP_ZIP = f"https://github.com/ggml-org/llama.cpp/archive/{LLAMA_CPP_COMMIT}.zip"
# The ready-made engine: <PREBUILT_URL><asset>, a zip with strata(.exe), strata-vision(.exe) and BUILD.json, built
# by tools/make_release.py. Set this to the GitHub release download folder when publishing, e.g.
# "https://github.com/<you>/Strata/releases/latest/download/" (or pass --prebuilt / set STRATA_PREBUILT_URL).
# With the default, the release of this checkout's own version (PREBUILT_TAG_URL, CMakeLists.txt's version) is
# tried first and the latest release is the fallback (#214): an older checkout keeps the engine it shipped with.
PREBUILT_URL = "https://github.com/Niko1221/Strata/releases/latest/download/"
PREBUILT_TAG_URL = "https://github.com/Niko1221/Strata/releases/download/v{version}/"
PREBUILT_ASSET = "strata-windows-x64.zip" if WIN else "strata-linux-x64.zip"
# the CUDA libraries the ready-made engine loads (the same CUDA 13.0 it is built with), from NVIDIA's pip packages
CUDA_WHEELS = ["nvidia-cublas==13.0.2.14", "nvidia-cuda-runtime==13.0.96"]
MIN_DRIVER = 580 # CUDA 13.0
# Older NVIDIA GPUs (experimental): CUDA 13 dropped Pascal (sm_60/61) and Volta (sm_70), so a model whose GPUs include
# one runs a second engine, built with CUDA 12.9 (-DSTRATA_EXPERIMENTAL_SM60=ON) and kept in its own folder: the
# ready-made one is CUDA12_ASSET (Windows; on Linux it is compiled here with a CUDA 12.x toolkit). One engine runs per
# model, so the choice is per model config, by its oldest GPU; --cuda 12|13 overrides it (docs/OLDER_GPUS.md).
CUDA13_MIN_ARCH = 75 # the oldest compute capability CUDA 13 compiles for (sm_75, RTX 20)
CUDA12_ASSET = "strata-windows-x64-cuda12.zip" if WIN else "strata-linux-x64-cuda12.zip"
CUDA12_WHEELS = ["nvidia-cublas-cu12==12.9.1.4", "nvidia-cuda-runtime-cu12==12.9.79"]
# CUDA 12.x minor-version compatibility (NVIDIA's table: Linux 525.60.13, Windows 527.41); the wheels match the 12.9.1
# toolkit the CUDA 12 zip is built with (cuBLAS 12.9.1.4, runtime 12.9.79). Not tested on such an old driver here.
CUDA12_MIN_DRIVER = 528 if WIN else 525
ENGINE12_DIR = "engine-cuda12"
MIN_ENGINE = (0, 1, 39) # v0.1.39: the #577 file-tier regression fixed, the OpenAI Responses API (#451, Codex), a reply stuck on one token ended (#606), the head before the arena (#620), effort_position (#458), --vram-reserve hot resize opt-in (#533), PR batch; v0.1.38: prompts faster (one gather per expert group #372, the first chunk's PLE rows beside layer 0 #374, DeltaNet three heads per thread #413), --kv q4_0 prompts on tensor cores (#452), Q5_0 experts on the GPU (#473), IQ4_XS on AVX-2 (#415), unbuffered expert loading on Windows (#357 #362), --peer-device (#531), a 6 GB card starts (#496), PR batch; v0.1.37: a silent engine is restarted (#481), Windows AMD counts the desktop's VRAM (#380 #377 #497), a steadier PCIe probe (#485), fixes #496 #495 #498 #505 #493; v0.1.36: a cancelled prompt logged as read so far (#471), the draft-head hint (#474), UPDATE.bat (#475), --expert-profile-save (#477); v0.1.35: Windows AMD uses its bundled HIP runtime (#468 #461), the low-RAM resident mode on Windows 32 GB (#467), fixes #460 #459 #446 #447 #457 #448 #444; v0.1.34: AMD on Windows (a ready-made HIP engine), an MCP server for AI assistants (tools/strata_mcp.py), a shorter README; v0.1.33: a portable image encoder again (#411 #412), setup recommends instead of forcing (#406 #403 #364 #384), fixes #352 #365 #369 #371 #375 #393 #408 #414; v0.1.32: split prompts faster (#340), AMD router +12%, Unsloth Q4 in setup, faster Q4 prompts, #326/#327/#342/#344 fixes, PR batch; v0.1.31: Unsloth UD-Q4_K_XL (experimental), GGUF-in-place low-RAM mode, Windows GGUF load 2x, server race + tokenizer fixes, AMD intrinsics; v0.1.30: short prompts faster (streaming from 1024 tokens), resident low-RAM variant, multi-GPU session carve, RDNA4; v0.1.29: sampled answers faster (split top-k), #154 correctness fixes; v0.1.28: the expert cache reserves the draft head, a cancelled request no longer fails the next; v0.1.27: RTX 20 (sm_75) in the ready-made engine, the HIP build without CUDA headers; v0.1.26: the draft layer's prompt pass in batches; v0.1.25: faster prompts (grouping off the copy engine, fused hyper-connection kernels), AMD HIP backend, --kv k8v4; v0.1.24: long prompts faster (QSA select on tensor cores); v0.1.23: image requests honor sampling, 8 GB cards start, batched verify window; v0.1.22: faster prompts (tensor-core attention), multi-GPU across images/steering/KV streaming; v0.1.21: multi-GPU layer split (--gpus); v0.1.20: system-prompt checkpoint, PCIe probe, hit rate; v0.1.19: penalties
PY_PACKAGES = ["numpy", "jinja2", "regex", "pyyaml", "tqdm", "requests", "cmake", "ninja", "pillow", "psutil"]
REQUIREMENTS = ROOT / "requirements.txt" # the same packages and their dependencies, pinned (#214)
MODELS = {
# the original model only for now: Swift 1.5's Q2_0 files split one layer's experts across the two shards, which
# the pack tool (tools/iq_pack.py) cannot prepare yet (#171)
"Q2_0": {"about": "2-bit, the fastest", "download_gb": 66.4, "ram_gb": 48, "arena_gb": 34.0, "families": ("qwen",)},
"IQ2_XS": {"about": "2-bit i-quant, a little better quality, close in speed", "download_gb": 68.0, "ram_gb": 48,
"arena_gb": 35.5},
"IQ3_XXS": {"about": "3-bit i-quant, better quality, slower (more CPU work per token)", "download_gb": 75.8,
"ram_gb": 60, "arena_gb": 42.9},
# the original model only (Swift 1.5 has no IQ3_S): matches the full BF16 model on the published benchmarks
"IQ3_S": {"about": "3.5-bit i-quant, the best quality (matches the full model), the slowest; needs a 64 GB PC "
"with little else running", "download_gb": 83.6, "ram_gb": 62, "arena_gb": 50.3,
"families": ("qwen",)},
# the Coder release: 256 of the 512 experts kept (the ones code, tools and vision use), IQ2_S-IQ4_XS like IQ3_S
"IQ1_M": {"about": "the Coder's only size: half the experts, stored like IQ3_S (3.5 bits)", "download_gb": 58.4,
"ram_gb": 32, "arena_gb": 23.4, "families": ("coder",)},
# EXPERIMENTAL (docs/UNSLOTH_Q4.md): Unsloth's 4-bit file; its 77 GB of experts do not fit a 64 GB PC, so the engine
# keeps a RAM budget of them (--resident-budget-gib, chosen below) and reads the rest from the GGUF on the SSD
"UD-Q4_K_XL": {"about": "4-bit (Unsloth Dynamic), EXPERIMENTAL: the best quality, but most experts come from the "
"SSD on a 64 GB PC (7-8.5 tokens/s measured)", "download_gb": 111.3, "ram_gb": 48,
"arena_gb": 77.0, "families": ("unsloth",), "budget": True, "nvidia_only": True,
"experimental": True},
# #621: Unsloth's UD-IQ4_XS - IQ3_S gate/up experts with IQ4_NL (43 layers) or Q8_0 (5) downs, the dense side as
# UD-Q4_K_XL's; three shards. A regular choice from 0.1.39 (no longer experimental). Its 59.5 GB of experts: a
# RAM budget of them, like UD-Q4_K_XL, but far fewer read from the SSD on a 64 GB PC and none from ~80 GB of RAM.
# Images: the vision path has no restriction for this pack (the same base model and image encoder), so setup asks
"UD-IQ4_XS": {"about": "~4-bit i-quant (Unsloth Dynamic), between IQ3_S and UD-Q4_K_XL in quality; on a PC with "
"less than ~80 GB of RAM part of its experts are read from the SSD",
"download_gb": 93.7, "ram_gb": 48, "arena_gb": 59.5, "families": ("unsloth",), "budget": True,
"shards": 3, "file": "Qwen3.8-Flash-Next-{q}-0000{i}-of-00003.gguf", "engine": (0, 1, 38),
"vision": True},
}
# The experimental Unsloth file's four shards at the pinned revision: name -> (bytes, sha256), checked after the
# download (setup trusts no other model file by name and size alone either: check_shards reads their directories).
UNSLOTH_SHARDS = {
"Qwen3.8-Flash-Next-UD-Q4_K_XL-00001-of-00004.gguf":
(10946624, "4448186216b3af4cc558bbce2c3213f01608f8f8b2e5267a9767971dd3ec8082"),
"Qwen3.8-Flash-Next-UD-Q4_K_XL-00002-of-00004.gguf":
(49859583136, "3f342f1c1580473f1ee94ddd5b28206e8c07a70fa1a366f59d1d6c922919a6c9"),
"Qwen3.8-Flash-Next-UD-Q4_K_XL-00003-of-00004.gguf":
(49376141504, "56758f40269cad5cd9b0d3d6fbae0f40f6d5be6de49e4ab392dbe83157d9cbd3"),
"Qwen3.8-Flash-Next-UD-Q4_K_XL-00004-of-00004.gguf":
(12087983520, "753bda48b98ba4f1636134a90a967de1b2d3908a236c026e464777342e53510a"),
}
# #621: UD-IQ4_XS's three shards at the same revision (sizes and SHA-256: the Hub's LFS pointers)
UNSLOTH_IQ4_XS_SHARDS = {
"Qwen3.8-Flash-Next-UD-IQ4_XS-00001-of-00003.gguf":
(10946624, "5ce89370720f8bf90890f439361282104c1aa1482d4013bb9a50923e758e71a4"),
"Qwen3.8-Flash-Next-UD-IQ4_XS-00002-of-00003.gguf":
(49835229856, "577a38a2392b40ca2193cea502e1d92f60b8cd370675d308e0ec21885d9daaa7"),
"Qwen3.8-Flash-Next-UD-IQ4_XS-00003-of-00003.gguf":
(43836407744, "d4634e6d84f0ebb0940be15c90d3790bf6464e3dea3a1cddc567dc0e83ad8833"),
}
UNSLOTH_ENGINE = (0, 1, 32) # the first engine setup configures for UD-Q4_K_XL (0.1.31 ran it by hand)
UNSLOTH_RAM_LEFT_GB = 24 # RAM beside the budget: the OS, the engine, and the file cache the rest is read through
# Contexts past 262144 (the model's trained length) extend it by rope scaling: for the context it will
# serve the setup resolves the method (yarn, or one question when interactive) and derives the factor
# from the final context (final / 262144, at least 1) itself (below), keeps an explicit
# --rope-scaling/--rope-scale, and refuses an explicit --rope-scaling none there - the stock angles past
# the trained range are out of spec.
CONTEXTS = [8192, 32768, 65536, 131072, 262144, 393216, 524288]
# The model families: the same architecture, weights in the same three GSQ-RCO sizes, different files.
FAMILIES = {
"qwen": {"title": "Qwen3.8-Flash-Next", "by": "Qwen; GSQ-RCO quants by ISTA-DASLab",
"about": "the original model",
"hf": hf("ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF") + "{q}/",
"file": "Qwen3.8-Flash-Next-GSQ-RCO-{q}-0000{i}-of-00002.gguf", "tag": "",
"mmproj_hf": hf("ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF"),
"mmproj": "mmproj-Qwen3.8-Flash-Next-BF16.gguf", "name": "qwen3.8-flash-next"},
"swift": {"title": "Swift 1.5", "by": "UkisAI's fine-tune of Qwen3.8-Flash-Next",
"about": "thinks much shorter (-63% thinking tokens, 1.8x sooner answers by its authors' numbers)",
"hf": hf("ukisai/Swift-1.5-Qwen3.8-Flash-Next-GSQ-RCO-GGUF"),
"file": "Swift-Qwen3.8-Flash-Next-GSQ-RCO-{q}-0000{i}-of-00002.gguf", "tag": "swift-",
"mmproj_hf": hf("ukisai/Swift-1.5-Qwen3.8-Flash-Next-GSQ-RCO-GGUF"),
"mmproj": "mmproj-Swift-Qwen3.8-Flash-Next-BF16.gguf", "name": "swift-1.5",
"license": "Swift Open License 1.0: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-Flash-Next-GSQ-RCO-GGUF"},
# ISTA-DASLab's expert-pruned release: half of each layer's experts removed, chosen for code, agentic tool use and
# vision; its shard 2 (the n-gram table) and vision encoder are the original's files, shared with it
"coder": {"title": "Qwen3.8-Flash-Next Coder", "by": "ISTA-DASLab's coding version",
"about": "half the experts (code, tools, images kept): needs ~32 GB of RAM, faster; weaker outside coding",
"hf": hf("ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-Coder-GGUF") + "{q}/",
"file": "Qwen3.8-Flash-Next-GSQ-RCO-{q}-0000{i}-of-00002.gguf", "tag": "coder-",
"mmproj_hf": hf("ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-Coder-GGUF"),
"mmproj": "mmproj-Qwen3.8-Flash-Next-BF16.gguf", "name": "qwen3.8-flash-next-coder",
"profile": "expert-profile-coder.bin"},
# Unsloth's UD-IQ4_XS (three shards, #621; a regular choice from 0.1.39) and the EXPERIMENTAL UD-Q4_K_XL (four)
# of the original model (docs/UNSLOTH_Q4.md); "experimental" and "vision" are per model (MODELS)
"unsloth": {"title": "Qwen3.8-Flash-Next (Unsloth)", "by": "Unsloth's ~4-bit quantizations",
"about": "UD-IQ4_XS: a 94 GB download; with less than ~80 GB of RAM part of its experts are read from "
"the SSD (UD-Q4_K_XL, 111 GB: experimental)",
"hf": hf("unsloth/Qwen3.8-Flash-Next-GGUF") + "{q}/",
"file": "Qwen3.8-Flash-Next-{q}-0000{i}-of-00004.gguf", "shards": 4, "tag": "unsloth-",
"mmproj_hf": hf("ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF"),
"mmproj": "mmproj-Qwen3.8-Flash-Next-BF16.gguf", "name": "qwen3.8-flash-next-unsloth",
"vision": False, "pack_args": ["--compat-bf16"],
"sha256": {**UNSLOTH_SHARDS, **UNSLOTH_IQ4_XS_SHARDS}},
}
MMPROJ = "mmproj-Qwen3.8-Flash-Next-BF16.gguf"
# EXPERIMENTAL, off by default (setup asks): a control vector shipped with the repository, see its README
ESP_VECTOR = ROOT / "data" / "experimental-speed-projection" / "Qwen3.8-Flash-Next-experimental-speed-projection.gguf"
# the image encoder on the GPU (~1.2 GB at 1024 image tokens) warms up before the engine starts, so the engine
# sizes its expert slots around it and the default reserve (700 MiB) is enough; engines before 0.1.2 need more
VISION_GPU_SMALL_RESERVE_MIB = 1000 # the tip for images on a <= 12 GB card (the engine's LOW line asked ~1003)
VISION = {"gpu": {"max_tokens": 1024, "reserve_mib": 700},
"cpu": {"max_tokens": 300, "reserve_mib": 700}}
EXE = "strata.exe" if WIN else "strata"
VEXE = "strata-vision.exe" if WIN else "strata-vision"
# ------------------------------------------------------------------------------------------------ output
def say(msg=""):
print(msg, flush=True)
def step(n, title):
say()
say(f"=== Step {n}: {title} ===")
def ok(msg):
say(f" [ok] {msg}")
def warn(msg):
say(f" [!] {msg}")
def fail(msg, hint=None):
say(f"\n [X] {msg}")
if hint:
say(f" {hint}")
say("\nSetup stopped. Fix the item above and run it again - everything already done is kept and skipped.")
sys.exit(1)
def ask(question, choices, default, yes):
if yes:
return default
while True:
try:
a = input(f"{question} [{default}]: ").strip()
except EOFError:
fail("input ended before a setup answer was received",
"run setup in a terminal, or pass --yes to accept the recommended answers")
if not a:
return default
if a.lower() in [c.lower() for c in choices]:
return next(c for c in choices if c.lower() == a.lower())
say(f" please answer one of: {', '.join(choices)}")
def run(cmd, cwd=None, env=None, check=True, quiet=False):
say(" > " + " ".join(str(c) for c in cmd))
r = subprocess.run([str(c) for c in cmd], cwd=cwd, env=env,
stdout=subprocess.PIPE if quiet else None, stderr=subprocess.STDOUT if quiet else None,
text=True)
if check and r.returncode != 0:
if quiet and r.stdout:
say(r.stdout[-4000:])
fail(f"command failed (exit {r.returncode}): {Path(str(cmd[0])).name}")
return r
def out(cmd):
try:
return subprocess.run(cmd, capture_output=True, text=True, timeout=60).stdout
except (OSError, subprocess.TimeoutExpired):
return ""
def done(path: Path) -> bool:
"""A step's finish mark: <path>.done exists (written only after the step completed)."""
return path.with_name(path.name + ".done").exists()
def mark(path: Path, text=""):
path.with_name(path.name + ".done").write_text(text or time.strftime("%Y-%m-%d %H:%M"), encoding="utf-8")
# ------------------------------------------------------------------------------------------------ the PC
def _memory_status():
"""Windows' GlobalMemoryStatusEx: RAM, and the commit limit (ullTotalPageFile = RAM + page file)."""
class MS(ctypes.Structure):
_fields_ = [("dwLength", ctypes.c_ulong), ("dwMemoryLoad", ctypes.c_ulong),
("ullTotalPhys", ctypes.c_ulonglong), ("ullAvailPhys", ctypes.c_ulonglong),
("ullTotalPageFile", ctypes.c_ulonglong), ("ullAvailPageFile", ctypes.c_ulonglong),
("ullTotalVirtual", ctypes.c_ulonglong), ("ullAvailVirtual", ctypes.c_ulonglong),
("ullAvailExtendedVirtual", ctypes.c_ulonglong)]
m = MS()
m.dwLength = ctypes.sizeof(MS)
ctypes.windll.kernel32.GlobalMemoryStatusEx(ctypes.byref(m))
return m
def ram_gb():
if WIN:
return _memory_status().ullTotalPhys / 2**30
for line in open("/proc/meminfo"):
if line.startswith("MemTotal"):
return int(line.split()[1]) * 1024 / 2**30
return 0.0
def page_file_gb():
"""The page file's current size (GB) on Windows, None elsewhere. The graphics card's memory needs room there
too: under Windows' driver model every allocation on the card is also charged to the commit (RAM + page file),
so with the page file off or tiny the engine cannot use the free VRAM (issue #60)."""
if not WIN:
return None
m = _memory_status()
return max(0.0, (m.ullTotalPageFile - m.ullTotalPhys) / 2**30)
def cpu_cores():
"""#642: (performance cores, efficiency cores) of a hybrid CPU (Intel 12th gen+, AMD Zen 5 + Zen 5c), counted
as the engine's pool counts them (detect_cpu_topology: physical cores, by Windows' EfficiencyClass or Linux's
cpu_capacity); None on a CPU whose cores are all alike, or when the OS does not say."""
classes = [] # one entry per physical core: its efficiency/capacity class
try:
if WIN:
k32 = ctypes.windll.kernel32
n = ctypes.c_ulong(0)
k32.GetLogicalProcessorInformationEx(0, None, ctypes.byref(n)) # RelationProcessorCore: the size
if not n.value:
return None
buf = ctypes.create_string_buffer(n.value)
if not k32.GetLogicalProcessorInformationEx(0, buf, ctypes.byref(n)):
return None
raw, at = buf.raw[:n.value], 0
while at + 10 <= len(raw): # SYSTEM_LOGICAL_PROCESSOR_INFORMATION_EX: Relationship, Size, then
rel, size = struct.unpack_from("<II", raw, at) # PROCESSOR_RELATIONSHIP (Flags,
if size <= 0: # EfficiencyClass, ...)
break
if rel == 0:
classes.append(raw[at + 9])
at += size
else:
seen = {}
for cpu in sorted(Path("/sys/devices/system/cpu").glob("cpu[0-9]*"), key=lambda p: int(p.name[3:])):
cap = cpu / "cpu_capacity"
pkg, core = cpu / "topology" / "physical_package_id", cpu / "topology" / "core_id"
if not cap.exists():
return None
key = (pkg.read_text().strip(), core.read_text().strip()) if pkg.exists() and core.exists() else cpu.name
seen.setdefault(key, int(cap.read_text().strip()))
classes = list(seen.values())
except (OSError, ValueError, AttributeError):
return None
if not classes or max(classes) == min(classes):
return None
p = sum(1 for c in classes if c == max(classes))
return p, len(classes) - p
def hybrid_pool_workers(cores) -> int | None:
"""#642 (Hardin22's measurement): on a hybrid CPU the expert pool runs best on the P-cores but the host loop's one
plus HALF of the E-cores - an E-core runs the expert kernels ~2.2x slower and each layer waits for its slowest
part (i9-14900KF, 8P + 16E: 15 workers decoded 165 / 116 tok/s against 106 / 84 with all 23). Only on a CPU with
more E-cores than P-cores: on an i7-13700KF (8P + 8E, docs/AMD_HIP.md's gfx1030 report) all 15 workers decoded
38-42 tok/s against 36 with 8, so there the engine's own count stays. None: the engine's own default (one worker
per physical core but the host's) stays."""
if not cores:
return None
p, e = cores
if e <= p:
return None
return max(1, p - 1 + e // 2)
def recommend_pool_workers(args: list) -> list:
"""`args` with setup's recommended `--pool-workers` for a hybrid CPU, unless they set one already (a calibration's
measured count, or the user's own). A recommendation: the config line can be edited or removed."""
n = hybrid_pool_workers(cpu_cores())
if n is None or "--pool-workers" in args:
return args
p, e = cpu_cores()
ok(f"hybrid CPU ({p} performance + {e} efficiency cores): {n} CPU expert workers - the performance cores and half "
"of the efficiency cores (--pool-workers in the config; START-HERE --calibrate measures it on this PC)")
return [*args, "--pool-workers", str(n)]
def cpu_info():
"""(name, avx2, avx512): avx512 means everything Strata's fast AVX-512 kernels use (F, BW, VL, VNNI, VBMI),
the same test the engine makes (cpu_avx512_ok), not just AVX-512F."""
name, avx2, avx512 = platform.processor() or "unknown CPU", False, False
if WIN:
pf = ctypes.windll.kernel32.IsProcessorFeaturePresent
avx2 = bool(pf(40)) or _cpuid_avx2() # PF_AVX2_INSTRUCTIONS_AVAILABLE, else the CPU itself (#159)
n = out(["powershell", "-NoProfile", "-Command", "(Get-CimInstance Win32_Processor).Name"]).strip()
name = n or name
avx512 = bool(pf(41)) and _cpuid_avx512_full()
else:
try:
txt = open("/proc/cpuinfo").read()
flags = set(re.search(r"^flags\s*:\s*(.*)$", txt, re.M).group(1).split())
avx2 = "avx2" in flags
avx512 = {"avx512f", "avx512bw", "avx512vl", "avx512_vnni", "avx512vbmi"} <= flags
m = re.search(r"^model name\s*:\s*(.*)$", txt, re.M)
name = m.group(1) if m else name
except OSError:
pass
return name, avx2, avx512
def _cpuid_floor() -> str:
"""Below AVX2 (Windows): "avx" when the CPU has AVX and the OS saves the YMM registers, "sse4.2" with SSE4.2 and
POPCNT, else ""."""
try:
regs = (ctypes.c_uint32 * 4)()
_run_stub(bytes([0x53, 0x49, 0x89, 0xC8, 0x89, 0xD0, 0x31, 0xC9, 0x0F, 0xA2, # push rbx; r8=rcx; eax=edx; ecx=0; cpuid
0x41, 0x89, 0x00, 0x41, 0x89, 0x58, 0x04, 0x41, 0x89, 0x48, 0x08, # [r8]=eax, [r8+4]=ebx, [r8+8]=ecx
0x41, 0x89, 0x50, 0x0C, 0x5B, 0xC3]), # [r8+12]=edx; pop rbx
ctypes.addressof(regs), 1)
ecx1 = regs[2]
if (ecx1 >> 27) & 1 and (ecx1 >> 28) & 1: # OSXSAVE, AVX
xcr0 = (ctypes.c_uint32 * 2)()
_run_stub(bytes([0x49, 0x89, 0xC8, 0x31, 0xC9, 0x0F, 0x01, 0xD0, # r8=rcx; ecx=0; xgetbv
0x41, 0x89, 0x00, 0x41, 0x89, 0x50, 0x04, 0xC3]), ctypes.addressof(xcr0))
if xcr0[0] & 6 == 6:
return "avx"
return "sse4.2" if (ecx1 >> 20) & 1 and (ecx1 >> 23) & 1 else ""
except Exception:
return ""
def cpu_floor(avx2: bool) -> str:
"""The experimental older-CPU build this PC needs (#394 #595 #623): "" with AVX2 (the normal engine), "avx" (Sandy /
Ivy Bridge, AMD Bulldozer), "none" (SSE4.2 + POPCNT: Nehalem, Westmere), or "unsupported". STRATA_ISA_FLOOR=avx|
none asks for that build on any PC (testing it on a newer one)."""
forced = os.environ.get("STRATA_ISA_FLOOR", "").strip().lower()
if forced in ("avx", "none"):
return forced
if avx2:
return ""
if WIN:
f = _cpuid_floor()
else:
try:
txt = open("/proc/cpuinfo").read()
flags = set(re.search(r"^flags\s*:\s*(.*)$", txt, re.M).group(1).split())
except (OSError, AttributeError):
flags = set()
f = "avx" if "avx" in flags else "sse4.2" if {"sse4_2", "popcnt"} <= flags else ""
return {"avx": "avx", "sse4.2": "none"}.get(f, "unsupported")
def _cpuid_avx512_full() -> bool:
"""Windows has no feature bit for VNNI / VBMI: ask the CPU (CPUID leaf 7) through a tiny machine-code stub."""
try:
code = bytes([0x53, 0x49, 0x89, 0xC8, 0xB8, 0x07, 0x00, 0x00, 0x00, 0x31, 0xC9, 0x0F, 0xA2, # push rbx; r8=rcx; cpuid(7,0)
0x41, 0x89, 0x18, 0x41, 0x89, 0x48, 0x04, 0x5B, 0xC3]) # [r8]=ebx,[r8+4]=ecx; pop rbx
k32 = ctypes.windll.kernel32
k32.VirtualAlloc.restype = ctypes.c_void_p
buf = k32.VirtualAlloc(None, len(code), 0x3000, 0x40)
if not buf:
return False
ctypes.memmove(buf, code, len(code))
regs = (ctypes.c_uint32 * 2)()
ctypes.CFUNCTYPE(None, ctypes.c_void_p)(buf)(ctypes.addressof(regs))
ebx, ecx = regs[0], regs[1]
need_ebx = (1 << 16) | (1 << 30) | (1 << 31) # F, BW, VL
need_ecx = (1 << 1) | (1 << 11) # VBMI, VNNI
return (ebx & need_ebx) == need_ebx and (ecx & need_ecx) == need_ecx
except Exception:
return False
def _run_stub(code: bytes, *args) -> None:
"""Runs a few bytes of x64 machine code (Windows calling convention: the arguments in rcx, rdx)."""
k32 = ctypes.windll.kernel32
k32.VirtualAlloc.restype = ctypes.c_void_p
k32.VirtualFree.argtypes = (ctypes.c_void_p, ctypes.c_size_t, ctypes.c_uint32)
buf = k32.VirtualAlloc(None, len(code), 0x3000, 0x40)
if not buf:
raise OSError("VirtualAlloc failed")
try:
ctypes.memmove(buf, code, len(code))
ctypes.CFUNCTYPE(None, *[ctypes.c_void_p] * len(args))(buf)(*args)
finally:
k32.VirtualFree(buf, 0, 0x8000)
def _cpuid_avx2() -> bool:
"""AVX2 asked from the CPU (CPUID leaf 7 EBX bit 5), with the OS saving the YMM registers (OSXSAVE + XCR0):
Windows' IsProcessorFeaturePresent(PF_AVX2) says no on some PCs whose CPU has it (a Ryzen 9 3950X, #159)."""
try:
def cpuid(leaf):
regs = (ctypes.c_uint32 * 4)()
_run_stub(bytes([0x53, 0x49, 0x89, 0xC8, 0x89, 0xD0, 0x31, 0xC9, 0x0F, 0xA2, # push rbx; r8=rcx; eax=edx; ecx=0; cpuid
0x41, 0x89, 0x00, 0x41, 0x89, 0x58, 0x04, 0x41, 0x89, 0x48, 0x08, # [r8]=eax, [r8+4]=ebx, [r8+8]=ecx
0x41, 0x89, 0x50, 0x0C, 0x5B, 0xC3]), # [r8+12]=edx; pop rbx
ctypes.addressof(regs), leaf)
return list(regs)
if cpuid(0)[0] < 7:
return False
ecx1 = cpuid(1)[2]
if not (ecx1 >> 27) & 1 or not (ecx1 >> 28) & 1: # OSXSAVE, AVX
return False
xcr0 = (ctypes.c_uint32 * 2)()
_run_stub(bytes([0x49, 0x89, 0xC8, 0x31, 0xC9, 0x0F, 0x01, 0xD0, # r8=rcx; ecx=0; xgetbv
0x41, 0x89, 0x00, 0x41, 0x89, 0x50, 0x04, 0xC3]), ctypes.addressof(xcr0))
if xcr0[0] & 6 != 6: # the OS saves XMM and YMM
return False
return bool((cpuid(7)[1] >> 5) & 1)
except Exception:
return False
def gpus():
"""Every NVIDIA GPU, numbered as nvidia-smi numbers them (by PCI bus, the order the engine is told to use)."""
s = out(["nvidia-smi", "--query-gpu=index,name,memory.total,compute_cap,driver_version",
"--format=csv,noheader,nounits"])
found = []
for line in s.strip().splitlines():
try:
idx, name, mem, cc, drv = [x.strip() for x in line.split(",")]
found.append({"index": int(idx), "name": name, "vram_gb": float(mem) / 1024.0, "arch": cc.replace(".", ""),
"driver": drv})
except ValueError:
continue
return found
GPU_PICK = None # --gpu N (issue #51); None: the card with the most VRAM
SPLIT_MIN_VRAM_GB = 8 # a card sharing a model holds the dense weights and its own
# prompt buffers too (docs/MULTI_GPU.md)
SPLIT_PROMPT_VRAM_GB = 12 # #448: a split stage lends a prompt chunk's buffers from its
# own cache; below this one cannot fund a 4096-token chunk
# (a 10 GB RTX 3080 beside a 32 GB card: 512 tokens, prompts
# 6.2x slower), while the big card alone could
def cc(g) -> str:
return f"{g['arch'][:-1]}.{g['arch'][-1]}"
OLD_GPUS = None # why Pascal / Volta cards are admitted in this run (old_gpus_opt_in), None: they are not
def experimental_sm60() -> bool:
"""#295: STRATA_EXPERIMENTAL_SM60=1 admits Pascal (6.x) and Volta (7.0) cards, run by the experimental CUDA 12
engine (-DSTRATA_EXPERIMENTAL_SM60=ON). So does naming such a card (--gpu N / --gpus), --cuda 12, or a PC that
has no newer card (old_gpus_opt_in)."""
return os.environ.get("STRATA_EXPERIMENTAL_SM60", "").strip() == "1" or OLD_GPUS is not None
def sm60_card(arch) -> bool:
return 60 <= int(arch) <= 70
def old_gpus_opt_in(found, named=(), cuda=None, other=False):
"""Why this run may use Pascal / Volta cards (the experimental CUDA 12 engine), or None. The cards are an opt-in:
the user named one (`named`: --gpu / --gpus), asked for --cuda 12, set STRATA_EXPERIMENTAL_SM60=1, or the PC has
no card the ready-made engine runs on and no supported AMD card (`other`; it used to stop there). A PC with a
newer card keeps recommending it."""
old = [g for g in found if sm60_card(g["arch"])]
if not old:
return None
if os.environ.get("STRATA_EXPERIMENTAL_SM60", "").strip() == "1":
return "STRATA_EXPERIMENTAL_SM60=1"
if str(cuda) == "12":
return "--cuda 12"
picked = [g for g in old if g["index"] in set(named)]
if picked:
return "you chose " + ", ".join(f"GPU {g['index']} ({g['name']})" for g in picked)
if not other and not any(int(g["arch"]) >= CUDA13_MIN_ARCH for g in found):
return "it is the only kind of NVIDIA GPU in this PC"
return None
def named_gpus(gpu, gpus) -> list:
"""The card numbers --gpu / --gpus name (an unreadable value: none; parse_gpus says what is wrong later)."""
try:
if gpus and str(gpus).strip().lower() != "all":
return [int(x) for x in str(gpus).split(",") if x.strip()]
return [int(gpu)] if gpu is not None else []
except ValueError:
return []
def cuda_choice(archs, cuda=None):
"""The CUDA toolkit of one model's engine: (12 or 13, why). 13 (the ready-made engine) unless a card is older
than CUDA 13 supports (Pascal / Volta: CUDA 13 cannot compile for them) - one engine runs per model, so its oldest
card decides. `cuda` (--cuda 12|13) overrides it; setup recommends, it does not refuse (the caller warns)."""
archs = sorted({int(x) for x in archs})
old = [a for a in archs if a < CUDA13_MIN_ARCH]
if str(cuda) == "13":
return 13, ("--cuda 13 (as you chose)" + (f"; CUDA 13 has no code for sm_{old[0]}: the engine will not run "
"on that card" if old else ""))
if str(cuda) == "12":
return 12, "--cuda 12 (as you chose" + ("; RTX 50 (sm_120) engines built with CUDA 12.8 crashed on long "
"prompts, #220" if archs and archs[-1] >= 120 else "") + ")"
if old:
return 12, (f"sm_{old[0]} is older than CUDA 13 supports (it dropped Pascal and Volta): this model runs the "
"experimental CUDA 12 engine")
return 13, None
def engine_dir(toolkit=13) -> Path:
"""The folder of the engine a model runs: engine/ (CUDA 13, or HIP), engine-cuda12/ (the experimental one)."""
return ROOT / (ENGINE12_DIR if int(toolkit) == 12 else "engine")
def config_toolkit(cfg: dict) -> int:
"""12 when a model config runs the experimental CUDA 12 engine (its exe is in engine-cuda12/), else 13."""
return 12 if cfg.get("cuda") == 12 or Path(str(cfg.get("exe", ""))).parent.name == ENGINE12_DIR else 13
def gpu_problem(g, together=False):
"""Why Strata cannot use this card, in plain words (None: it can)."""
if int(g["arch"]) < 75 and not (sm60_card(g["arch"]) and experimental_sm60()):
return (f"not supported - older than the RTX 20 series (compute capability {cc(g)}; Strata needs 7.5 or "
"newer" + ("; experimental: choose it with --gpu " + str(g["index"]) + " (the CUDA 12 engine, "
"docs/OLDER_GPUS.md)" if sm60_card(g["arch"]) else "") + ")")
if together and g["vram_gb"] < SPLIT_MIN_VRAM_GB - 0.5:
return (f"not supported together with other GPUs - {g['vram_gb']:.0f} GB of VRAM (a card sharing the model "
f"needs {SPLIT_MIN_VRAM_GB} GB or more)")
return None
def gpu_rank(g):
"""The order cards share a model in: the newest generation first (it gets the first layers and most of the
work), then the most VRAM."""
return (-int(g["arch"]), -round(g["vram_gb"]), g["index"])
def gpu_name(g) -> str:
return f"GPU {g['index']} ({g['name']}, {g['vram_gb']:.0f} GB)"
def gpu_table(found) -> None:
say(" Your NVIDIA GPUs:")
for g in found:
p = gpu_problem(g)
say(f" GPU {g['index']}: {g['name']}, {g['vram_gb']:.0f} GB VRAM - " + ("can be used" if p is None else p))
def together_ok(found) -> list:
"""The cards that can share one model, in the order they would (empty if fewer than two)."""
ok_ = sorted([g for g in found if gpu_problem(g, together=True) is None], key=gpu_rank)
return ok_ if len(ok_) >= 2 else []
def split_short(cards) -> list:
"""#448: the later cards of a split that would cap its prompt chunk below what the first card alone reads (a card
under SPLIT_PROMPT_VRAM_GB beside one that has it). Empty: the split is recommended as before."""
if len(cards) < 2 or cards[0]["vram_gb"] < SPLIT_PROMPT_VRAM_GB - 0.5:
return []
return [g for g in cards[1:] if g["vram_gb"] < SPLIT_PROMPT_VRAM_GB - 0.5]
def split_short_note(g) -> str:
return (f"GPU {g['index']} ({g['name']}, {g['vram_gb']:.0f} GB) is too small to lend a split its prompt buffers: "
"it would cap prompt reading at 512-2048-token chunks, several times slower than the first card alone "
"(#448). It can serve as a helper expert cache instead (docs/SECOND_GPU.md)")
def parse_gpus(text, found) -> list:
"""--gpus / --gpu with several: "0,2" or "all" (every card that can share the model)."""
if str(text).strip().lower() == "all":
sel = [g["index"] for g in together_ok(found)]
if not sel:
gpu_table(found)
fail("--gpus all: this PC does not have two GPUs Strata can use together")
return sel
try:
sel = [int(x) for x in str(text).split(",") if x.strip()]
except ValueError:
fail(f"--gpus takes GPU numbers as nvidia-smi numbers them, e.g. --gpus 0,2 (or --gpus all), not {text!r}")
if len(sel) < 2 or len(set(sel)) != len(sel):
fail("--gpus takes two or more different GPUs, e.g. --gpus 0,2 (one GPU: --gpu 0)")
return sel
def check_gpus(sel, found, what="", yes=False, named=False) -> None:
"""Stops with a plain message when a chosen card is missing or cannot be used, and says what can. named: the user
named these cards (--gpus 0,1, or a config that has them): a card that is only short of VRAM for sharing the model
is then a risk to confirm, not a stop (the owner's rule; --yes with the named cards is the consent)."""
together = len(sel) > 1
for i in sel:
g = next((x for x in found if x["index"] == i), None)
p = "not found on this PC" if g is None else gpu_problem(g, together)
if p is None:
continue
if named and g is not None and gpu_problem(g) is None: # it runs Strata; only its VRAM is small
confirm_risk(f"GPU {i} ({g['name']}) has {g['vram_gb']:.0f} GB of VRAM: a card sharing the model needs "
f"{SPLIT_MIN_VRAM_GB} GB or more (it holds the dense weights of its layers and its own prompt "
"buffers), so the model may not start, or run slower than without it", True, yes,
f"GPU {i} ({g['name']}) {what}is not used together with other GPUs: {p}",
"leave it out of --gpus, or answer y to use it anyway", " Use it anyway?")
warn(f"GPU {i} ({g['name']}) is used together with the others, as you chose")
continue
say()
gpu_table(found)
can = together_ok(found)
single = [x for x in found if gpu_problem(x) is None]
ones = " or ".join(f"--gpu {x['index']}" for x in single)
both = "--gpus " + ",".join(str(x["index"]) for x in can) if can else ""
hint = ((f"use these together: {both}" + (f" (or one card: {ones})" if not together else "")) if can else
f"use one card: {ones}" if single else "Strata needs an NVIDIA RTX 20 series or newer card")
fail(f"GPU {i}{'' if g is None else ' (' + g['name'] + ')'} {what}cannot be used: {p}", hint)
def engine_archs(toolkit=13):
"""The GPU generations the installed engine has code for: (archs, ptx), or None when there is none."""
info = engine_dir(toolkit) / "BUILD.json"
try:
meta = json.loads(info.read_text())
except (OSError, ValueError):
return None
return [int(x) for x in meta.get("archs", [])], bool(meta.get("ptx"))
def engine_archs_hip():
"""The AMD architectures the installed HIP engine was compiled for ("gfx1201", ...), or None."""
try:
meta = json.loads((ROOT / "engine" / "BUILD.json").read_text())
except (OSError, ValueError):
return None
return [str(x) for x in meta.get("archs", [])] if meta.get("backend") == "hip" else None
def engine_runs_on(g, toolkit=13) -> bool:
ea = engine_archs(toolkit)
if ea is None or not ea[0]:
return True
archs, ptx = ea
return int(g["arch"]) in archs or (ptx and int(g["arch"]) > max(archs))
def start_gpus(text):
"""--gpus when starting an installed model: NVIDIA cards as nvidia-smi numbers them, or on a PC whose AMD cards
are the ones Strata can use, AMD cards as setup lists them ("all": every supported AMD card)."""
if not text:
return None
if str(text).strip().lower() == "all" and not WIN and not together_ok(gpus()):
amd = amd_gpus()
if len([g for g in amd if amd_problem(g) is None]) >= 2:
return [g["index"] for g in amd_parse_gpus("all", amd)]
return parse_gpus(text, gpus())
def choose_gpus(a, found) -> list:
"""Which cards this install uses: --gpus / --gpu, or asked when two or more can share the model (the two best
together recommended), else the supported card with the most VRAM. Returns their numbers, the main one first."""
if a.gpus:
sel = parse_gpus(a.gpus, found)
check_gpus(sel, found, yes=a.yes, named=str(a.gpus).strip().lower() != "all")
return sel
if a.gpu is not None:
check_gpus([a.gpu], found)
return [a.gpu]
single = sorted([g for g in found if gpu_problem(g) is None], key=lambda x: (-round(x["vram_gb"]), x["index"]))
if not single:
gpu_table(found)
fail("none of your GPUs can run Strata", "it needs an NVIDIA RTX 20 series or newer (compute capability 7.5+)")
can = together_ok(found)
if not can:
return [single[0]["index"]]
say()
say(f" Strata can run the model on one GPU, or share it across {'these' if len(can) > 2 else 'both'}: then each"
" card holds the")
say(" experts of its own layers, so together they hold about twice as many, and prompts are read about 20%")
say(" faster (details: docs/MULTI_GPU.md). A much slower extra card can also make it slower.")
opts = [can[:2]] + ([can] if len(can) > 2 else []) + [[g] for g in single]
# #448: a pair whose second card cannot lend a 4096-token chunk recommends the first card alone (still offered)
short = split_short(can[:2])
rec = next(i for i, o in enumerate(opts, 1) if o == [can[0]]) if short else 1
for i, o in enumerate(opts, 1):
label = (" + ".join(gpu_name(g) for g in o) + " together") if len(o) > 1 else gpu_name(o[0]) + " only"
say(f" {i}) {label}" + (" (recommended)" if i == rec else ""))
for g in found:
if gpu_problem(g, together=True) is not None:
say(f" (GPU {g['index']}, {g['name']}: {gpu_problem(g, together=True)})")
for g in short:
say(f" ({split_short_note(g)})")
pick = opts[int(ask("Which GPUs?", [str(i) for i in range(1, len(opts) + 1)], str(rec), a.yes or a.check)) - 1]
return [g["index"] for g in pick]
def split_mmap(cfg: dict) -> bool:
"""#364 #384: the low-RAM mode's resident variant (--resident-experts) has no layer split yet. A config with it
that runs on several GPUs reads the experts the GPUs do not hold through the OS file cache instead
(--mmap-experts: the placement those reports measured 1.3-1.6x faster than one GPU), said plainly - the engine
used to refuse the pair. True when the config changed."""
a = cfg.get("args", [])
if "--resident-experts" not in a:
return False
a[a.index("--resident-experts")] = "--mmap-experts"
warn("the low-RAM mode's resident variant (--resident-experts) has no layer split yet: on several GPUs the experts "
"the GPUs do not hold are read through the OS file cache (--mmap-experts) instead, and RAM can fill up to 0 "
"free during long prompts. One GPU keeps them in RAM (steady RAM use): START-HERE --setup, or --gpu N for a "
"start")
return True
def model_file(fam: dict, model: str, i: int) -> str:
"""Shard i's file name: the family's pattern, or the model's own (#621: UD-IQ4_XS has three shards, not four)."""
return MODELS.get(model, {}).get("file", fam["file"]).format(q=model, i=i)
def model_shards(fam: dict, model: str) -> int:
return MODELS.get(model, {}).get("shards", fam.get("shards", 2))
def budget_model(cfg: dict) -> str:
"""The Unsloth model a config with a RAM budget runs, from its --native shard's name (UD-Q4_K_XL by default)."""
a = cfg.get("args", [])
native = Path(a[a.index("--native") + 1]).name.upper() if "--native" in a and a.index("--native") + 1 < len(a) \
else ""
return next((m for m, d in MODELS.items() if d.get("budget") and f"-{m}-" in native), "UD-Q4_K_XL")
def unsloth_split_need_gb(model="UD-Q4_K_XL") -> float:
"""#498: the RAM UD-Q4_K_XL needs on several GPUs, where it has no RAM budget (the engine refuses
--resident-budget-gib with a layer split): its GGUF files and UNSLOTH_RAM_LEFT_GB more (~135 GB). Measured safe
at 165 GiB (2x RTX 3090: MemAvailable never under 68 GiB); the 0-free case of #384 was 47 GB with a 70 GB model."""
return MODELS[model]["download_gb"] + UNSLOTH_RAM_LEFT_GB
def split_budget(cfg: dict) -> bool:
"""#498: a UD-Q4_K_XL config (its RAM budget, --resident-budget-gib) started on several GPUs. The engine refuses
the budget with a layer split (it exited with code 2), so the split runs without it - all the experts loaded into
RAM at start - where the RAM holds the GGUFs and 24 GB more; else setup stops, before the config is saved. True
when the config changed."""
a = cfg.get("args", [])
if "--resident-budget-gib" not in a:
return False
model = budget_model(cfg)
need, ram = unsloth_split_need_gb(model), ram_gb()
if ram < need:
fail(f"{model} cannot share its RAM budget across GPUs (the engine has no layer split with it), and without "
f"the budget it needs ~{need:.0f} GB of RAM (its GGUF files and {UNSLOTH_RAM_LEFT_GB} GB more); this PC "
f"has {ram:.0f} GB", "start it on one GPU: START-HERE.bat --gpu N (Linux: ./setup.sh --gpu N)")
i = a.index("--resident-budget-gib")
del a[i:i + 2]
ok(f"{model} on several GPUs: no RAM budget (the engine has none with a layer split) - all its experts are "
"loaded into RAM from the model files at start, and the files pass through the OS file cache (#498)")
return True
REMOTE_EXPERT_OPT = "--remote-expert-opt"
def recommend_remote_expert_opt(cfg: dict, off: bool = False) -> None:
"""0.1.39b (#578): a config on two or more GPUs gets --remote-expert-opt - the helper expert caches
(--expert-cache-device1..3) then stay complementary to the main GPU's, return their rows already weighted and skip
the CPU's activation quantization where no expert is left to it (dual RTX 4090: +63% mixed, +132% code over the
plain helper path). The engine uses it only with a helper cache; a layer split runs as before. A recommendation:
`off` (setup's --no-remote-expert-opt) or "remote_expert_opt": false in the config keeps it out, and a single-GPU
config is not touched."""
if not isinstance(cfg.get("gpu"), list) or len(cfg["gpu"]) < 2:
return
args = cfg.setdefault("args", [])
if off or cfg.get("remote_expert_opt") is False:
if REMOTE_EXPERT_OPT in args:
args.remove(REMOTE_EXPERT_OPT)
return
if REMOTE_EXPERT_OPT not in args:
args.append(REMOTE_EXPERT_OPT)
ok("multi-GPU: --remote-expert-opt (helper expert caches complementary to the main GPU's, #578; "
"--no-remote-expert-opt leaves it out)")
def offer_together(cfg_path: Path, cfg: dict, yes: bool) -> dict:
"""Starting a model set up for one card on a PC with two or more that can share it: asked once (the answer is
saved in its config)."""
if isinstance(cfg.get("gpu"), list) or cfg.get("gpus_asked"):
return cfg
found = gpus()
can = together_ok(found)
if not can:
return cfg
# #498: UD-Q4_K_XL's RAM budget has no layer split; without it the RAM must hold the GGUFs and 24 GB more
budget = "--resident-budget-gib" in cfg.get("args", [])
if budget and ram_gb() < unsloth_split_need_gb(budget_model(cfg)):
return cfg
pair = can[:2]
cfg["gpus_asked"] = True
# #364 #384: the resident low-RAM variant stays on one card unless the user says otherwise (its RAM use is steady)
resident = "--resident-experts" in cfg.get("args", [])
say()
say(" This PC has " + " and ".join(gpu_name(g) for g in pair) + ": Strata can share the model across both.")
say(" Together they hold about twice the model's experts and read prompts about 20% faster (docs/MULTI_GPU.md).")
if resident:
say(" This model runs in the low-RAM mode with its experts kept in RAM, on one GPU (recommended: steady RAM")
say(" use). On both, the experts the GPUs do not hold are read through the OS file cache instead: faster in")
say(" two reports (#364, #384), but RAM can fill up to 0 free during long prompts.")
if budget:
say(f" This model ({budget_model(cfg)}) runs on one GPU with a RAM budget of its experts (recommended: the tested")
say(" setup). On both it has no budget: all its experts are loaded into RAM at start, which this PC's RAM")
say(" holds - about twice as fast in #498 (2x RTX 3090: 31 -> 64-78 tokens/s).")
short = split_short(pair) # #448: one card recommended (asked "n" by default), as for --resident
for g in short:
say(f" {split_short_note(g)}.")
tk = config_toolkit(cfg)
missing = [g for g in pair if not (engine_runs_on(g) if tk == 13 else engine_runs_on(g, tk))]
if missing:
say(" The installed engine has no code for " + ", ".join(g["name"] for g in missing) + ": to use them "
"together, run START-HERE.bat --setup --gpus " + ",".join(str(g["index"]) for g in pair))
elif ask(" Use both from now on? (you can change it later: START-HERE.bat --gpu N for one card)",
["y", "n"], "n" if resident or short or budget else "y", yes) == "y":
cfg["gpu"] = [g["index"] for g in pair]
cfg["layer_split"] = cfg.get("layer_split") or "auto"
split_mmap(cfg)
split_budget(cfg)
recommend_remote_expert_opt(cfg)
ok("from now on this model runs on " + " + ".join(gpu_name(g) for g in pair))
else:
ok("staying on one GPU (START-HERE.bat --gpus " + ",".join(str(g["index"]) for g in pair) + " switches)")
write_config(cfg_path, cfg)
return cfg
def gpu_info(pick=None):
"""The GPU Strata runs on: `pick` (nvidia-smi's number) if given, else the one with the most VRAM (ties: the
lower number). None when there is no NVIDIA GPU. The dict also says how many there are ("count")."""
found = gpus()
if not found:
return None
pick = GPU_PICK if pick is None else pick
if pick is not None:
g = next((x for x in found if x["index"] == pick), None)
if g is None:
fail(f"there is no GPU {pick}: " + ", ".join(f"{x['index']} = {x['name']}" for x in found))
else:
g = max(found, key=lambda x: (round(x["vram_gb"]), -x["index"]))
return {**g, "count": len(found)}
def find_nvcc(below=None):
"""The newest CUDA toolkit's nvcc and its (major, minor); with `below`, the newest older than that version.
#601: STRATA_NVCC=<path to nvcc> is the only one considered (a newer toolkit beside it that cannot build on this
PC - CUDA 12.9 with glibc 2.43 - is not taken instead)."""
pick = os.environ.get("STRATA_NVCC")
if pick:
if not Path(pick).exists():
warn(f"STRATA_NVCC={pick}: no such file; looking for a CUDA toolkit as usual")
else:
v = re.search(r"release (\d+)\.(\d+)", out([pick, "--version"]))
ver = (int(v.group(1)), int(v.group(2))) if v else None
if ver and below is not None and ver >= below:
warn(f"STRATA_NVCC={pick} is CUDA {ver[0]}.{ver[1]}; this build needs one older than "
f"{below[0]}.{below[1]}")
return (None, None)
return (pick, ver) if ver else (None, None)
cands = [shutil.which("nvcc")]
for var in ("CUDA_PATH", "CUDA_HOME"): # CUDA_HOME: Linux's usual name (#601)
if os.environ.get(var):
cands.append(str(Path(os.environ[var]) / "bin" / ("nvcc.exe" if WIN else "nvcc")))
if WIN:
base = Path(r"C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA")
if base.exists():
cands += [str(p / "bin" / "nvcc.exe") for p in sorted(base.iterdir(), reverse=True)]
else:
cands += [str(p / "bin" / "nvcc") for p in sorted(Path("/usr/local").glob("cuda*"), reverse=True)]
cands += [str(p / "bin" / "nvcc") for p in sorted(Path("/opt").glob("cuda*"), reverse=True)] # Arch (#46)
best = (None, None)
for c in dict.fromkeys(cands): # every toolkit found; the newest wins
if c and Path(c).exists():
v = re.search(r"release (\d+)\.(\d+)", out([c, "--version"]))
ver = (int(v.group(1)), int(v.group(2))) if v else None
if ver and (below is None or ver < below) and (best[1] is None or ver > best[1]):
best = (c, ver)
return best
def find_vcvars():
vswhere = Path(os.environ.get("ProgramFiles(x86)", r"C:\Program Files (x86)")) / "Microsoft Visual Studio/Installer/vswhere.exe"
if not vswhere.exists():
return None
# CUDA 13 accepts Visual Studio 2019 and 2022 only: a newer one (2026 = version 18) installed next to them
# must not be picked ("unsupported Microsoft Visual Studio version"); with only a newer one there is none
p = out([str(vswhere), "-latest", "-products", "*", "-version", "[16.0,18.0)", "-requires",
"Microsoft.VisualStudio.Component.VC.Tools.x86.x64", "-property", "installationPath"]).strip()
v = Path(p) / "VC/Auxiliary/Build/vcvars64.bat" if p else None
return v if v and v.exists() else None
def find_tool(name):