Text Generation
Transformers
Safetensors
GGUF
English
qwen3_5
image-text-to-text
decision-model
typed-decisions
calibration
calibrated-probabilities
classification
tool-selection
agent-routing
decision-index
jevbench
jev-compatible
systemone
wald
wald-q4b
qwen3.5
4b
vllm
reasoning
llama.cpp
conversational
Eval Results (legacy)
Instructions to use org2ai/Wald-4B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use org2ai/Wald-4B with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="org2ai/Wald-4B") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# pip install -U transformers accelerate # Load model directly from transformers import AutoProcessor, AutoModelForMultimodalLM processor = AutoProcessor.from_pretrained("org2ai/Wald-4B") model = AutoModelForMultimodalLM.from_pretrained("org2ai/Wald-4B", device_map="auto") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=256) print(processor.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use org2ai/Wald-4B with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "org2ai/Wald-4B" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "org2ai/Wald-4B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/org2ai/Wald-4B
- SGLang
How to use org2ai/Wald-4B with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "org2ai/Wald-4B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "org2ai/Wald-4B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "org2ai/Wald-4B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "org2ai/Wald-4B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use org2ai/Wald-4B with Docker Model Runner:
docker model run hf.co/org2ai/Wald-4B
Commit ·
78ebf43
0
Parent(s):
Wald-4B v1.0
Browse files- .gitattributes +36 -0
- CONTAMINATION.md +253 -0
- Dockerfile +20 -0
- LICENSE +202 -0
- MANIFEST.json +34 -0
- NOTICE +19 -0
- README.md +201 -0
- RUNBOOK.md +110 -0
- chat_template.jinja +154 -0
- config.json +83 -0
- contamination/trained-on-di-ids.json +0 -0
- docs/lora-cli.md +96 -0
- docs/observations/one-lora-all-verticals.md +112 -0
- docs/readmes/README.zh.md +173 -0
- figures/data.json +1584 -0
- figures/di-areas.svg +84 -0
- figures/di-vs-size.svg +111 -0
- figures/latency-cost.svg +52 -0
- figures/lora-cli-architecture.svg +154 -0
- figures/verticals.svg +162 -0
- generation_config.json +6 -0
- model-files/serving.json +6 -0
- model.safetensors +3 -0
- run.sh +13 -0
- server/README.md +14 -0
- server/pyproject.toml +22 -0
- server/src/wald_serve/__init__.py +2 -0
- server/src/wald_serve/__main__.py +3 -0
- server/src/wald_serve/engine.py +285 -0
- server/src/wald_serve/prompt.py +97 -0
- server/src/wald_serve/server.py +152 -0
- server/src/wald_serve/wire.py +96 -0
- server/tests/fakevllm.py +66 -0
- server/tests/test_parity.py +102 -0
- server/tests/test_server.py +251 -0
- serving.json +6 -0
- temperature.json +154 -0
- tokenizer.json +3 -0
- tokenizer_config.json +32 -0
- tools/export_contamination.py +90 -0
- tools/export_figure_data.py +166 -0
- tools/make_figures.py +299 -0
- tools/prepare_model_dir.py +31 -0
- tools/secret_scan.py +28 -0
.gitattributes
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
*.7z filter=lfs diff=lfs merge=lfs -text
|
| 2 |
+
*.arrow filter=lfs diff=lfs merge=lfs -text
|
| 3 |
+
*.bin filter=lfs diff=lfs merge=lfs -text
|
| 4 |
+
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
| 5 |
+
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
| 6 |
+
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
+
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
+
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
+
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
+
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
+
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
+
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
+
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
+
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
+
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
+
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
+
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
+
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
+
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
+
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
+
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
+
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
+
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
+
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
+
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
+
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
| 27 |
+
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
| 28 |
+
*.tar filter=lfs diff=lfs merge=lfs -text
|
| 29 |
+
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
+
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
+
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
+
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
+
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
+
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
+
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
CONTAMINATION.md
ADDED
|
@@ -0,0 +1,253 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Contamination statement
|
| 2 |
+
|
| 3 |
+
This covers the checkpoint submitted as Wald-4B and **every stage of its lineage**, back to the base model. The Decision
|
| 4 |
+
Index rule is that any test item an entry was trained on counts as wrong. The list below is for the maintainers to apply
|
| 5 |
+
that rule. We also count these items wrong in our own blocklist-adjusted scores.
|
| 6 |
+
|
| 7 |
+
## 1. Decision Index items in our training data
|
| 8 |
+
|
| 9 |
+
**[contamination/trained-on-di-ids.json](https://huggingface.co/Harry19081/Wald-4B/blob/main/contamination/trained-on-di-ids.json)** lists 363 item ids. For each id it
|
| 10 |
+
gives the benchmark, the lineage stage whose data contains it, and why it is there. Please count all 363 as trained on.
|
| 11 |
+
|
| 12 |
+
| benchmark | ids | scored in 0.2.1 | why they are in our data |
|
| 13 |
+
|---|---:|---|---|
|
| 14 |
+
| ToolRet (#2) | 67 | yes | **shared upstream source.** ToolRet aggregates the user queries of public tool-calling datasets. Our data contains the train splits of ToolACE and Glaive function-calling. |
|
| 15 |
+
| BANKING77 (#4) | 49 | yes | **public train split duplicates a test sentence.** The BANKING77 train split is in our data, and these test sentences also appear verbatim in train. |
|
| 16 |
+
| CLINC150+OOS (#5) | 44 | yes | **public train split duplicates a test sentence** (as above, for the CLINC-OOS train split). |
|
| 17 |
+
| MMLU-Pro (#57) | 6 | yes | **shared upstream source.** MMLU-Pro includes MATH / TheoremQA problems, and the same problems are in our math data (MATH train). |
|
| 18 |
+
| SATA-Bench (#33) | 4 | yes | **shared upstream source.** Reading passages from RACE, which makes up most of MMLU auxiliary_train. |
|
| 19 |
+
| ANLI (#12) | 3 | yes | **shared upstream text.** The premise text also appears in other public datasets in our data. |
|
| 20 |
+
| WinoGrande (#28) | 2 | yes | **public train split duplicates a test sentence.** |
|
| 21 |
+
| BRIGHT (#36) | 1 | yes | **shared upstream source.** A public programming problem statement. |
|
| 22 |
+
| RouterBench (#6) | 170 | no (removed in 0.2.1) | **shared upstream source.** RouterBench prompts embed GSM8K / MMLU questions, and those questions are in our math and QA training data (GSM8K train and others). |
|
| 23 |
+
| MMLU (#24), ARC-Easy (#26), ARC-Challenge (#27) | 6 / 7 / 4 | display only | **public train split duplicates** (ARC train, MMLU auxiliary_train). These are the same kind of duplicates the board already records for other entries. |
|
| 24 |
+
|
| 25 |
+
That is 176 ids on scored benchmarks. On our 6,948-request sample, counting them wrong moves the index by 0.3 points:
|
| 26 |
+
53.91 → 53.62 for v1.0 (effort medium, `repeat_state_plain`), and 49.62 → 49.34 for its parent at high-k4.
|
| 27 |
+
|
| 28 |
+
**How the items got in.** We never trained on a Decision Index test split. Every hit above reached us through a public
|
| 29 |
+
train split or an upstream dataset that the benchmark itself draws from. Stages 1–3 of the lineage were built before we
|
| 30 |
+
had the full blocklist (below), so those stages contain these items. Stage 4 (the task-coverage LoRA) was filtered with the full blocklist before training: 0 hits.
|
| 31 |
+
|
| 32 |
+
## 2. Training data, by source
|
| 33 |
+
|
| 34 |
+
This section names every public source in the lineage and the split we read. It gives no proportions: no ratios, token
|
| 35 |
+
or row counts per source, or mixing weights. The training data itself is not released.
|
| 36 |
+
|
| 37 |
+
### 2.1 The lineage
|
| 38 |
+
|
| 39 |
+
| stage | what it is | data |
|
| 40 |
+
|---|---|---|
|
| 41 |
+
| 1 | full-parameter decision training of Qwen/Qwen3.5-4B-Base | converted public datasets (§2.2), code-generated decision families, and synthetic tool, document-QA and style items |
|
| 42 |
+
| 2 | LoRA refinement with KL replay (merged) | a curated mix of human-labelled public sets and code-generated items, synthetic hard items, and a KL-only replay of stage-1 rows |
|
| 43 |
+
| 3 | short-thought distillation (LoRA, merged) | synthetic short thoughts on questions from the stage-2 pool, one-pass rows from the same pool, and replay rows whose target is the stage-2 model's own distribution |
|
| 44 |
+
| 4 | task-coverage LoRA (merged; the submitted checkpoint) | the train splits of iSarcasmEval, API-Bank, ContractNLI, VAST, NLI4CT, ACOS and RAGTruth, new Home-appliance households from the Decision Index kit's generator, and a KL-only replay of stage-2 rows |
|
| 45 |
+
|
| 46 |
+
Stage 1 drew on a public decision-training mixture. Its constituent datasets are listed one by one below, and its custom
|
| 47 |
+
and routing parts were left out. Stage 4 teaches the formats of eight Decision Index benchmarks from their source train
|
| 48 |
+
splits and the kit's own generator. No test item is included. **Home appliance scores 1.00 (skill) for this checkpoint
|
| 49 |
+
on our sample, so please weigh that benchmark with this training in mind.** The maintainers can regenerate the rows
|
| 50 |
+
with the kit's generator and any new seed, and compare them with their test rows.
|
| 51 |
+
|
| 52 |
+
### 2.2 Public sources
|
| 53 |
+
|
| 54 |
+
Every public dataset was read through its **train split only**, unless the split column says otherwise. No test split
|
| 55 |
+
of any source was used.
|
| 56 |
+
|
| 57 |
+
In the stage column, **(replay)** means that the stage re-used rows from that source only as KL replay. For those rows
|
| 58 |
+
the training target is the model's own earlier distribution, not the source label.
|
| 59 |
+
|
| 60 |
+
The licence column gives the licence **as the source states it**. "card" means the Hugging Face dataset card, and
|
| 61 |
+
"upstream" means the original release, used where the card is empty or says unknown. We have not reviewed these terms
|
| 62 |
+
here.
|
| 63 |
+
|
| 64 |
+
#### Constituents of the stage-1 public mixture
|
| 65 |
+
|
| 66 |
+
| dataset | split(s) used | stages | licence (as stated) | note |
|
| 67 |
+
|---|---|---|---|---|
|
| 68 |
+
| [cais/mmlu](https://huggingface.co/datasets/cais/mmlu) (all) | auxiliary_train | 1, 2, 3, 4 (replay) | MIT (card) | dev / validation / test never used. auxiliary_train is mostly RACE passages (see §1, SATA-Bench) |
|
| 69 |
+
| [allenai/ai2_arc](https://huggingface.co/datasets/allenai/ai2_arc) (ARC-Challenge, ARC-Easy) | train | 1, 2, 3, 4 (replay) | CC-BY-SA-4.0 (card) | also in the wide replay (ARC-Easy). Some test questions duplicate train questions (§1) |
|
| 70 |
+
| [tau/commonsense_qa](https://huggingface.co/datasets/tau/commonsense_qa) | train | 1, 2, 3, 4 (replay) | MIT (card) | also read directly as a human-labelled set |
|
| 71 |
+
| [google/boolq](https://huggingface.co/datasets/google/boolq) | train | 1, 2, 3, 4 (replay) | CC-BY-SA-3.0 (card) | also read directly as a human-labelled set |
|
| 72 |
+
| [nyu-mll/multi_nli](https://huggingface.co/datasets/nyu-mll/multi_nli) | train | 1, 2, 3, 4 (replay) | mixed: CC-BY-3.0 / CC-BY-SA-3.0 / MIT / OANC (card) | validation_matched / mismatched never used. Also read directly |
|
| 73 |
+
| [stanfordnlp/snli](https://huggingface.co/datasets/stanfordnlp/snli) | train | 1, 2, 3, 4 (replay) | CC-BY-SA-4.0 (card) | |
|
| 74 |
+
| [mteb/banking77](https://huggingface.co/datasets/mteb/banking77) | train | 1, 2, 3, 4 (replay) | MIT (card); upstream PolyAI release CC-BY-4.0 | some test sentences also appear verbatim in train (§1) |
|
| 75 |
+
| [allenai/openbookqa](https://huggingface.co/datasets/allenai/openbookqa) (main) | train | 1, 2, 3, 4 (replay) | Apache-2.0 (upstream allenai/OpenBookQA) | also read directly as a human-labelled set |
|
| 76 |
+
| [GBaker/MedQA-USMLE-4-options](https://huggingface.co/datasets/GBaker/MedQA-USMLE-4-options) | train | 1, 2, 3, 4 (replay) | CC-BY-4.0 (card); upstream MIT | |
|
| 77 |
+
| [allenai/winogrande](https://huggingface.co/datasets/allenai/winogrande) (winogrande_xl) | train | 1, 2, 3, 4 (replay) | CC-BY-4.0 (upstream allenai/winogrande) | also in the wide replay. Some test sentences duplicate train (§1) |
|
| 78 |
+
| [clinc/clinc_oos](https://huggingface.co/datasets/clinc/clinc_oos) (plus) | train | 1, 2, 3, 4 (replay) | CC-BY-3.0 (card) | also in the wide replay. Some test sentences duplicate train (§1) |
|
| 79 |
+
| [SetFit/amazon_massive_intent_en-US](https://huggingface.co/datasets/SetFit/amazon_massive_intent_en-US) | train | 1, 2, 3, 4 (replay) | CC-BY-4.0 (upstream Amazon MASSIVE) | |
|
| 80 |
+
| [bitext/Bitext-customer-support-llm-chatbot-training-dataset](https://huggingface.co/datasets/bitext/Bitext-customer-support-llm-chatbot-training-dataset) | train (the only split) | 1, 2, 3, 4 (replay) | CDLA-Sharing-1.0 (card) | |
|
| 81 |
+
| [fancyzhx/dbpedia_14](https://huggingface.co/datasets/fancyzhx/dbpedia_14) | train | 1, 2, 3, 4 (replay) | CC-BY-SA-3.0 (card) | also read directly as a human-labelled set |
|
| 82 |
+
| [google-research-datasets/go_emotions](https://huggingface.co/datasets/google-research-datasets/go_emotions) (simplified) | train | 1, 2, 3, 4 (replay) | Apache-2.0 (card) | |
|
| 83 |
+
| [SetFit/hate_speech_offensive](https://huggingface.co/datasets/SetFit/hate_speech_offensive) | train | 1, 2, 3, 4 (replay) | MIT (upstream t-davidson/hate-speech-and-offensive-language) | |
|
| 84 |
+
| [SetFit/amazon_counterfactual_en](https://huggingface.co/datasets/SetFit/amazon_counterfactual_en) | train | 1, 2, 3, 4 (replay) | CC-BY-SA-4.0 (upstream amazon-research/amazon-multilingual-counterfactual-dataset) | |
|
| 85 |
+
| [google/civil_comments](https://huggingface.co/datasets/google/civil_comments) | train (a leading slice) | 1, 2, 3, 4 (replay) | CC0-1.0 (card) | also in the wide replay |
|
| 86 |
+
| [ucirvine/sms_spam](https://huggingface.co/datasets/ucirvine/sms_spam) | train (the only split) | 1, 2, 3, 4 (replay) | CC-BY-4.0 (upstream UCI SMS Spam Collection) | |
|
| 87 |
+
| [allenai/qasc](https://huggingface.co/datasets/allenai/qasc) | train | 1, 2, 3, 4 (replay) | CC-BY-4.0 (card) | also in the wide replay |
|
| 88 |
+
| [Rowan/hellaswag](https://huggingface.co/datasets/Rowan/hellaswag) | train | 1, 2, 3, 4 (replay) | MIT (upstream rowanz/hellaswag) | validation / test never used |
|
| 89 |
+
| [ybisk/piqa](https://huggingface.co/datasets/ybisk/piqa) | train | 1, 2, 3, 4 (replay) | unknown | also in the wide replay |
|
| 90 |
+
| [LabHC/bias_in_bios](https://huggingface.co/datasets/LabHC/bias_in_bios) | train | 1, 2, 3, 4 (replay) | MIT (card) | |
|
| 91 |
+
| [nvidia/HelpSteer2](https://huggingface.co/datasets/nvidia/HelpSteer2) | train | 2, 3 | CC-BY-4.0 (card) | |
|
| 92 |
+
| [nvidia/HelpSteer3](https://huggingface.co/datasets/nvidia/HelpSteer3) (preference) | train | 1, 2, 3, 4 (replay) | CC-BY-4.0 (card) | also in the wide replay |
|
| 93 |
+
| [ucberkeley-dlab/measuring-hate-speech](https://huggingface.co/datasets/ucberkeley-dlab/measuring-hate-speech) | train (the only split) | 1, 2, 3, 4 (replay) | CC-BY-4.0 (card) | |
|
| 94 |
+
| [chengxuphd/liar2](https://huggingface.co/datasets/chengxuphd/liar2) | train | 1, 2, 3, 4 (replay) | Apache-2.0 (card) | |
|
| 95 |
+
| [allenai/prosocial-dialog](https://huggingface.co/datasets/allenai/prosocial-dialog) | train | 1, 2, 3, 4 (replay) | CC-BY-4.0 (card) | also in the wide replay |
|
| 96 |
+
| [HuggingFaceH4/ultrafeedback_binarized](https://huggingface.co/datasets/HuggingFaceH4/ultrafeedback_binarized) | train_prefs | 1, 2, 3, 4 (replay) | MIT (card) | test_prefs never used |
|
| 97 |
+
| [Anthropic/hh-rlhf](https://huggingface.co/datasets/Anthropic/hh-rlhf) | train | 1, 2, 3, 4 (replay) | MIT (card) | |
|
| 98 |
+
| [glaiveai/glaive-function-calling-v2](https://huggingface.co/datasets/glaiveai/glaive-function-calling-v2) | train (the only split) | 1, 2, 3, 4 (replay) | Apache-2.0 (card) | also a seed source for the synthetic tool items. Shares user queries with ToolRet (§1) |
|
| 99 |
+
| [Team-ACE/ToolACE](https://huggingface.co/datasets/Team-ACE/ToolACE) | train (the only split) | 1, 2, 3, 4 (replay) | Apache-2.0 (card) | also in the wide replay and the tool items. Shares user queries with ToolRet (§1) |
|
| 100 |
+
| [copenlu/fever_gold_evidence](https://huggingface.co/datasets/copenlu/fever_gold_evidence) | train | 1, 2, 3, 4 (replay) | unknown | |
|
| 101 |
+
| [openlifescienceai/medmcqa](https://huggingface.co/datasets/openlifescienceai/medmcqa) | train | 1, 2, 3, 4 (replay) | Apache-2.0 (card); upstream MIT | also in the wide replay |
|
| 102 |
+
| [osunlp/Mind2Web](https://huggingface.co/datasets/osunlp/Mind2Web) | train | 1, 2, 3, 4 (replay) | CC-BY-4.0 (card) | also a seed source for the synthetic tool items |
|
| 103 |
+
|
| 104 |
+
#### Wide replay of public train splits (sources not listed above)
|
| 105 |
+
|
| 106 |
+
| dataset | split(s) used | stages | licence (as stated) | note |
|
| 107 |
+
|---|---|---|---|---|
|
| 108 |
+
| [allenai/cosmos_qa](https://huggingface.co/datasets/allenai/cosmos_qa) | train | 1, 2 (replay), 3, 4 (replay) | CC-BY-4.0 (card) | |
|
| 109 |
+
| [ucinlp/drop](https://huggingface.co/datasets/ucinlp/drop) | train | 1, 2 (replay), 3, 4 (replay) | CC-BY-SA-4.0 (card) | |
|
| 110 |
+
| [allenai/quoref](https://huggingface.co/datasets/allenai/quoref) | train | 1, 2 (replay), 3, 4 (replay) | CC-BY-4.0 (card) | |
|
| 111 |
+
| [allenai/quartz](https://huggingface.co/datasets/allenai/quartz) | train | 1, 2 (replay), 3, 4 (replay) | CC-BY-4.0 (card) | |
|
| 112 |
+
| [allenai/ropes](https://huggingface.co/datasets/allenai/ropes) (plain_text) | train | 1, 2 (replay), 3, 4 (replay) | CC-BY-4.0 (card) | |
|
| 113 |
+
| [alisawuffles/WANLI](https://huggingface.co/datasets/alisawuffles/WANLI) | train | 1, 2 (replay), 3, 4 (replay) | CC-BY-4.0 (card) | crowd labels that revise model-written candidates |
|
| 114 |
+
| [allenai/social_i_qa](https://huggingface.co/datasets/allenai/social_i_qa) | train | 1, 2 (replay), 3, 4 (replay) | CC-BY-4.0 (card text) | |
|
| 115 |
+
| [deepmind/aqua_rat](https://huggingface.co/datasets/deepmind/aqua_rat) (raw) | train | 1, 2 (replay), 3, 4 (replay) | Apache-2.0 (card) | |
|
| 116 |
+
| [ChilleD/SVAMP](https://huggingface.co/datasets/ChilleD/SVAMP) | train | 1, 2 (replay), 3, 4 (replay) | MIT (card) | |
|
| 117 |
+
| [math-eval/TAL-SCQ5K](https://huggingface.co/datasets/math-eval/TAL-SCQ5K) (EN) | train | 1, 2 (replay), 3, 4 (replay) | MIT (card) | |
|
| 118 |
+
| [tasksource/ruletaker](https://huggingface.co/datasets/tasksource/ruletaker) | train | 1, 2 (replay), 3, 4 (replay) | Apache-2.0 (card) | |
|
| 119 |
+
| [ImperialCollegeLondon/health_fact](https://huggingface.co/datasets/ImperialCollegeLondon/health_fact) (PUBHEALTH) | train | 1, 2 (replay), 3, 4 (replay) | MIT (card) | |
|
| 120 |
+
| [wenhu/tab_fact](https://huggingface.co/datasets/wenhu/tab_fact) | train | 1, 2 (replay), 3, 4 (replay) | CC-BY-4.0 (card) | |
|
| 121 |
+
| [lighteval/wikitablequestions](https://huggingface.co/datasets/lighteval/wikitablequestions) | train | 1, 2 (replay), 3, 4 (replay) | CC-BY-SA-4.0 (upstream ppasupat/WikiTableQuestions) | |
|
| 122 |
+
| [AmazonScience/massive](https://huggingface.co/datasets/AmazonScience/massive) (en-US) | train | 1, 2 (replay), 3, 4 (replay) | CC-BY-4.0 (card) | |
|
| 123 |
+
| [google-research-datasets/schema_guided_dstc8](https://huggingface.co/datasets/google-research-datasets/schema_guided_dstc8) (SGD) | train | 1, 2 (replay), 3, 4 (replay) | CC-BY-SA-4.0 (card) | |
|
| 124 |
+
| [tasksource/esci](https://huggingface.co/datasets/tasksource/esci) (us) | train | 1, 2 (replay), 3, 4 (replay) | Apache-2.0 (card) | |
|
| 125 |
+
| [google-research-datasets/poem_sentiment](https://huggingface.co/datasets/google-research-datasets/poem_sentiment) | train | 1, 2 (replay), 3, 4 (replay) | CC-BY-4.0 (card) | |
|
| 126 |
+
| [zeroshot/twitter-financial-news-sentiment](https://huggingface.co/datasets/zeroshot/twitter-financial-news-sentiment) | train | 1, 2 (replay), 3, 4 (replay) | MIT (card) | |
|
| 127 |
+
| [NousResearch/hermes-function-calling-v1](https://huggingface.co/datasets/NousResearch/hermes-function-calling-v1) (glaive subset) | train | 1, 2 (replay), 3, 4 (replay) | Apache-2.0 (card) | |
|
| 128 |
+
| [openbmb/UltraFeedback](https://huggingface.co/datasets/openbmb/UltraFeedback) (evol_instruct, ultrachat) | train | 1, 2 (replay), 3, 4 (replay) | MIT (card) | ratings in the source are model-written |
|
| 129 |
+
| [jackhhao/jailbreak-classification](https://huggingface.co/datasets/jackhhao/jailbreak-classification) | train | 1, 2 (replay), 3, 4 (replay) | Apache-2.0 (card) | |
|
| 130 |
+
| [deepset/prompt-injections](https://huggingface.co/datasets/deepset/prompt-injections) | train | 1, 2 (replay), 3, 4 (replay) | Apache-2.0 (card) | |
|
| 131 |
+
| [tonytan48/TempReason](https://huggingface.co/datasets/tonytan48/TempReason) | train_l1 | 1, 2 (replay), 3, 4 (replay) | CC-BY-SA-3.0 (card) | answers recomputed by code |
|
| 132 |
+
|
| 133 |
+
#### Other public decision sets
|
| 134 |
+
|
| 135 |
+
| dataset | split(s) used | stages | licence (as stated) | note |
|
| 136 |
+
|---|---|---|---|---|
|
| 137 |
+
| [jaredpalmer/kev-suites](https://huggingface.co/datasets/jaredpalmer/kev-suites) | v7/decision-v7/train.jsonl | 2, 3 | Apache-2.0 (card); rows derive from public datasets under their own terms | Kev's public training partition. Its rows come from the BANKING77, BoolQ, DBpedia-14 and MNLI train splits |
|
| 138 |
+
| [thu-coai/cold](https://huggingface.co/datasets/thu-coai/cold) (COLD, Chinese) | train | 2, 3 | Apache-2.0 (repo) | test never used for training |
|
| 139 |
+
| [ZefanCai/Open-Jev](https://huggingface.co/datasets/ZefanCai/Open-Jev) | train | 1, 2, 3, 4 (replay) | CC0-1.0 (card); code MIT | labels by rules, solvers and generator latents. Its calibration / validation / test / OOD rows were never used. `customer-control-v1` excluded |
|
| 140 |
+
| [ZefanCai/Open-Jev-v1.1](https://huggingface.co/datasets/ZefanCai/Open-Jev-v1.1) | train (`wanli-decisions-v1`, `community-diversity-v2`) | 2, 3 | CC0-1.0 (community-diversity-v2); CC-BY-4.0 (wanli-decisions-v1, from WANLI) | |
|
| 141 |
+
|
| 142 |
+
#### Prompt sources for routing and verification items
|
| 143 |
+
|
| 144 |
+
The items themselves are ours. A routing item labels a public prompt by the kind of source it came from, and a
|
| 145 |
+
verification item plants an error in a human-written solution by code.
|
| 146 |
+
|
| 147 |
+
| dataset | split(s) used | stages | licence (as stated) | note |
|
| 148 |
+
|---|---|---|---|---|
|
| 149 |
+
| [openai/gsm8k](https://huggingface.co/datasets/openai/gsm8k) (main) | train | 1, 2, 3, 4 (replay) | MIT (card) | routing prompts and the worked solutions of the verification items. Test never used. RouterBench embeds GSM8K questions (§1) |
|
| 150 |
+
| [EleutherAI/hendrycks_math](https://huggingface.co/datasets/EleutherAI/hendrycks_math) | train | 1, 2, 3, 4 (replay) | MIT (card) | test never used. MMLU-Pro includes MATH problems (§1) |
|
| 151 |
+
| [deepmind/code_contests](https://huggingface.co/datasets/deepmind/code_contests) | train | 1, 2, 3, 4 (replay) | CC-BY-4.0 (card) | valid / test never used |
|
| 152 |
+
| [google-research-datasets/mbpp](https://huggingface.co/datasets/google-research-datasets/mbpp) (full) | train, validation, prompt | 2, 3 | CC-BY-4.0 (card) | test never used |
|
| 153 |
+
| [princeton-nlp/SWE-bench](https://huggingface.co/datasets/princeton-nlp/SWE-bench) | train | 1, 2, 3, 4 (replay) | MIT (card) | test (the benchmark) never used |
|
| 154 |
+
| [mandarjoshi/trivia_qa](https://huggingface.co/datasets/mandarjoshi/trivia_qa) (rc.nocontext) | train | 1, 2, 3, 4 (replay) | Apache-2.0 (card) | |
|
| 155 |
+
| [abisee/cnn_dailymail](https://huggingface.co/datasets/abisee/cnn_dailymail) (3.0.0) | train | 1, 2, 3, 4 (replay) | Apache-2.0 (card) | |
|
| 156 |
+
| [OpenAssistant/oasst1](https://huggingface.co/datasets/OpenAssistant/oasst1) | train | 1, 2, 3, 4 (replay) | Apache-2.0 (card) | |
|
| 157 |
+
| [HuggingFaceM4/WebSight](https://huggingface.co/datasets/HuggingFaceM4/WebSight) (v0.2) | train (the only split) | 1, 2, 3 | CC-BY-4.0 (card) | |
|
| 158 |
+
|
| 159 |
+
#### Text embedded in long states
|
| 160 |
+
|
| 161 |
+
| dataset | split(s) used | stages | licence (as stated) | note |
|
| 162 |
+
|---|---|---|---|---|
|
| 163 |
+
| [HuggingFaceFW/fineweb-edu](https://huggingface.co/datasets/HuggingFaceFW/fineweb-edu) (sample-10BT) | train (the only split) | 1, 2, 3, 4 (replay) | ODC-BY-1.0 (card) | documents placed inside code-generated long-state items |
|
| 164 |
+
| [wikimedia/wikipedia](https://huggingface.co/datasets/wikimedia/wikipedia) (20231101.en) | train (the only split) | 1, 2, 3, 4 (replay) | CC-BY-SA-3.0 and GFDL (card) | as above |
|
| 165 |
+
|
| 166 |
+
#### Seed sources of the synthetic tool items
|
| 167 |
+
|
| 168 |
+
| dataset | split(s) used | stages | licence (as stated) | note |
|
| 169 |
+
|---|---|---|---|---|
|
| 170 |
+
| [nvidia/When2Call](https://huggingface.co/datasets/nvidia/When2Call) | train_sft | 1 | CC-BY-4.0 (card) | test never used |
|
| 171 |
+
| [liminghao1630/API-Bank](https://huggingface.co/datasets/liminghao1630/API-Bank) | training-data: lv1 (stage 1); lv1 + lv2 (stage 4) | 1, 4 | MIT (card); the GitHub repo AlibabaResearch/DAMO-ConvAI (api-bank) states Apache-2.0 | test-data never used. In stage 4, 53-option tool catalogs are built the kit's way from the train tool pool |
|
| 172 |
+
| [AgentGym/AgentTraj-L](https://huggingface.co/datasets/AgentGym/AgentTraj-L) | train | 1 | unknown | |
|
| 173 |
+
|
| 174 |
+
#### Stage 4: Decision Index benchmark sources
|
| 175 |
+
|
| 176 |
+
| dataset | split(s) used | stages | licence (as stated) | note |
|
| 177 |
+
|---|---|---|---|---|
|
| 178 |
+
| [iSarcasmEval](https://github.com/iabufarha/iSarcasmEval) (SemEval-2022 Task 6) | train.En.csv (tasks A-En, B-En), train.Ar.csv (task A-Ar) | 4 | MIT (repo LICENSE); the README asks for the task citation | test never used |
|
| 179 |
+
| [ContractNLI](https://github.com/stanfordnlp/contract-nli) | train | 4 | CC BY 4.0 (README / project site) | train rows only. Rows that share clauses with test contracts were dropped |
|
| 180 |
+
| [VAST](https://github.com/emilyallaway/zero-shot-stance) @ e7c4775 | data/VAST/vast_train.csv | 4 | unknown | test never used |
|
| 181 |
+
| [NLI4CT](https://github.com/ai-systems/Task-2-SemEval-2024) (SemEval-2024 Task 2) @ 7f32fa6 | train.json, dev.json | 4 | unknown | test never used |
|
| 182 |
+
| [ACOS](https://github.com/NUSTM/ACOS) @ 45d179a | data/*/*_quad_train.tsv | 4 | unknown | test never used |
|
| 183 |
+
| [RAGTruth](https://github.com/ParticleMedia/RAGTruth) @ c103204 | response.jsonl split = train, source_info.jsonl | 4 | MIT (repo LICENSE); contexts: unknown | test never used |
|
| 184 |
+
| [Decision Index kit](https://github.com/apolinario/decision-index) @ 19ad28e, Home-appliance generator (`decision_index/suite/build/home_appliance.py`) | new households from our own seed | 4 | MIT (repo) | code, not a dataset. 0 states are shared with the benchmark's dev + test rows |
|
| 185 |
+
|
| 186 |
+
That is **84 public sources**: 83 datasets and one generator.
|
| 187 |
+
|
| 188 |
+
### 2.3 Synthetic data
|
| 189 |
+
|
| 190 |
+
Everything that is not a public source above falls into four categories.
|
| 191 |
+
|
| 192 |
+
- **Code-generated and code-labelled items.** These are decision families whose answer a program computes: rule
|
| 193 |
+
application, dates and business days, numeric extraction, lookup, ordering, long policies with amendments and decoys,
|
| 194 |
+
temporal traces, multi-hop and buried-evidence long states, and tool-use worlds. The judge items plant an error in a
|
| 195 |
+
human-written GSM8K solution by code, so the label holds by construction. The routing items take their label from the
|
| 196 |
+
prompt's source dataset. This category also includes the rule-based renderings (JSON, described, wide and
|
| 197 |
+
single-question forms) and the rule-twin items that ship with the stage-1 mixture's public code, and the stage-4
|
| 198 |
+
Home-appliance households.
|
| 199 |
+
- **Synthetic items and short thoughts.** Part of the data is synthetic: items and short thoughts written or labelled
|
| 200 |
+
by much larger frontier LLMs — far above 120B parameters where the size is published. An item is kept only when
|
| 201 |
+
checks by code or independent answers agree and it passes our schema, shortcut and overlap gates; a thought is kept
|
| 202 |
+
only if its final answer matches the gold label.
|
| 203 |
+
- **Our own model's errors, used for judge data.** None of the four stages' training files contain this category. The
|
| 204 |
+
judge items of v1.0 are the code-planted errors described above, plus synthetic judge items.
|
| 205 |
+
- **KL replay.** Stages 2, 3 and 4 re-show earlier rows, and the target is the model's own distribution before that
|
| 206 |
+
stage (§2.2, "(replay)"). This keeps earlier abilities while the new rows are learned. No label is added.
|
| 207 |
+
|
| 208 |
+
The training data is not released.
|
| 209 |
+
|
| 210 |
+
### 2.4 How Decision Index items were screened
|
| 211 |
+
|
| 212 |
+
- **Stages 1–3** were built before the full blocklist existed. For stage 3, we removed the 5 rows that held the strictly
|
| 213 |
+
verified hits known at the time. Every remaining hit in the stage-1, stage-2 and stage-3 training files is listed in
|
| 214 |
+
§1: 363 ids in [contamination/trained-on-di-ids.json](https://huggingface.co/Harry19081/Wald-4B/blob/main/contamination/trained-on-di-ids.json). Please count them as
|
| 215 |
+
trained on.
|
| 216 |
+
- **The blocklist.** Since 2026-09-27, every Decision Index 0.2 request we rebuilt (all but HLE) is on a permanent
|
| 217 |
+
blocklist in our training repository (`trainer/data/blocklists`). Items are stored as fingerprints, not text. The data acceptance gate fails any
|
| 218 |
+
training row that matches it (§3).
|
| 219 |
+
- **Stage 4** was checked row by row before training against the full blocklist, our sample rows, the calibration
|
| 220 |
+
holdouts and our vertical test sets. It passed the gate with 0 rows that contain a Decision Index item.
|
| 221 |
+
|
| 222 |
+
## 3. How we scanned
|
| 223 |
+
|
| 224 |
+
- **Suite.** The Decision Index 0.2 suite was rebuilt with the maintainers' public kit (github.com/apolinario/decision-index
|
| 225 |
+
@ 19ad28e) from its pinned public sources. It has 154,877 requests; HLE (501 rows, gated) is not included.
|
| 226 |
+
- **Match rule** (applied to every string of every training record):
|
| 227 |
+
- Words are lower-case `[a-z0-9]+`.
|
| 228 |
+
- Each item's state is taken after stripping the benchmark's fixed instruction line. A shared prompt prefix is
|
| 229 |
+
stripped too, so training text without the DI prompt is still recognized.
|
| 230 |
+
- A state of 4–11 words must appear verbatim.
|
| 231 |
+
- A state of 12 words or more needs ≥ 80 % of its 8-word shingles in one record. Shingles that occur in more than 20
|
| 232 |
+
states are dropped as boilerplate.
|
| 233 |
+
- Benchmarks whose options are the item's content (MMLU / ARC / MMLU-Pro / GPQA / BBH / HellaSwag / WinoGrande …)
|
| 234 |
+
also need ≥ 50 % of the option text to match.
|
| 235 |
+
- States under 8 words are recorded as "weak" and not counted.
|
| 236 |
+
- **Coverage.** The scan covered every training file of every stage of this lineage, plus sibling and ancestor corpora.
|
| 237 |
+
The per-file scan outputs (line counts, file sha256, hit counts) are summarized in the `scans` field of
|
| 238 |
+
trained-on-di-ids.json.
|
| 239 |
+
- **Blocklist.** Since 2026-09-27 every Decision Index item is on a permanent blocklist, stored as fingerprints, not
|
| 240 |
+
text. The data acceptance gate fails any training row that hits it.
|
| 241 |
+
- **Items the scan cannot see.** Items without free text cannot be checked this way: CLINC150 short intents, POP909 and
|
| 242 |
+
cfcolor. Nothing in the pretraining data of the base model was scanned.
|
| 243 |
+
|
| 244 |
+
## 4. What we did not do
|
| 245 |
+
|
| 246 |
+
- We never used a Decision Index test split, the kit's suite rows or the sample rows for training, selection or
|
| 247 |
+
calibration.
|
| 248 |
+
- We never tuned the prompt, the readout or the effort policy per benchmark. The knockout for more than 26 options is one
|
| 249 |
+
rule for every question.
|
| 250 |
+
- We selected checkpoints on our own holdouts only. The Decision Index was read once per candidate checkpoint.
|
| 251 |
+
- One disclosure: the benchmarks that stage 4 covers were chosen from our Decision Index sample reads, as the
|
| 252 |
+
benchmarks where we trailed Jev most. That choice used scores only. No items were used, and the stage has no
|
| 253 |
+
checkpoint selection.
|
Dockerfile
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Wald-4B /v1/systemone server: vLLM 0.30.0 (bf16) on the weights as a loopback-only sidecar + wald-serve in front.
|
| 2 |
+
#
|
| 3 |
+
# docker build -t wald-serve .
|
| 4 |
+
# docker run --gpus all -v /path/to/Wald-4B:/model:ro -p 8000:8000 wald-serve
|
| 5 |
+
# # ready when GET http://127.0.0.1:8000/health returns {"ok": true, ...}
|
| 6 |
+
#
|
| 7 |
+
# The weights directory holds config, tokenizer, bf16 safetensors, temperature.json and serving.json (declared policy,
|
| 8 |
+
# prompt format, context limit). Extra vLLM arguments: -e WALD_VLLM_ARGS="...". Another policy: -e WALD_EFFORT=high.
|
| 9 |
+
FROM python:3.12-slim
|
| 10 |
+
|
| 11 |
+
RUN pip install --no-cache-dir uv==0.9.5 \
|
| 12 |
+
&& uv pip install --system --no-cache "vllm==0.30.0" \
|
| 13 |
+
&& python -c "import importlib.metadata as m; print('vllm', m.version('vllm'), 'torch', m.version('torch'))"
|
| 14 |
+
COPY server /app/server
|
| 15 |
+
RUN uv pip install --system --no-cache /app/server
|
| 16 |
+
ENV HF_HUB_OFFLINE=1 TRANSFORMERS_OFFLINE=1 TOKENIZERS_PARALLELISM=false PYTHONUNBUFFERED=1 \
|
| 17 |
+
VLLM_USE_FLASHINFER_SAMPLER=0 VLLM_NO_USAGE_STATS=1 DO_NOT_TRACK=1 \
|
| 18 |
+
WALD_EFFORT="" WALD_VLLM_ARGS=""
|
| 19 |
+
EXPOSE 8000
|
| 20 |
+
CMD ["sh", "-c", "exec wald-serve --model /model --port 8000 ${WALD_EFFORT:+--effort $WALD_EFFORT} --vllm-args \"$WALD_VLLM_ARGS\""]
|
LICENSE
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Apache License
|
| 2 |
+
Version 2.0, January 2004
|
| 3 |
+
http://www.apache.org/licenses/
|
| 4 |
+
|
| 5 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 6 |
+
|
| 7 |
+
1. Definitions.
|
| 8 |
+
|
| 9 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 10 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 11 |
+
|
| 12 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 13 |
+
the copyright owner that is granting the License.
|
| 14 |
+
|
| 15 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 16 |
+
other entities that control, are controlled by, or are under common
|
| 17 |
+
control with that entity. For the purposes of this definition,
|
| 18 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 19 |
+
direction or management of such entity, whether by contract or
|
| 20 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 21 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 22 |
+
|
| 23 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 24 |
+
exercising permissions granted by this License.
|
| 25 |
+
|
| 26 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 27 |
+
including but not limited to software source code, documentation
|
| 28 |
+
source, and configuration files.
|
| 29 |
+
|
| 30 |
+
"Object" form shall mean any form resulting from mechanical
|
| 31 |
+
transformation or translation of a Source form, including but
|
| 32 |
+
not limited to compiled object code, generated documentation,
|
| 33 |
+
and conversions to other media types.
|
| 34 |
+
|
| 35 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 36 |
+
Object form, made available under the License, as indicated by a
|
| 37 |
+
copyright notice that is included in or attached to the work
|
| 38 |
+
(an example is provided in the Appendix below).
|
| 39 |
+
|
| 40 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 41 |
+
form, that is based on (or derived from) the Work and for which the
|
| 42 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 43 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 44 |
+
of this License, Derivative Works shall not include works that remain
|
| 45 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 46 |
+
the Work and Derivative Works thereof.
|
| 47 |
+
|
| 48 |
+
"Contribution" shall mean any work of authorship, including
|
| 49 |
+
the original version of the Work and any modifications or additions
|
| 50 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 51 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 52 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 53 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 54 |
+
means any form of electronic, verbal, or written communication sent
|
| 55 |
+
to the Licensor or its representatives, including but not limited to
|
| 56 |
+
communication on electronic mailing lists, source code control systems,
|
| 57 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 58 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 59 |
+
excluding communication that is conspicuously marked or otherwise
|
| 60 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 61 |
+
|
| 62 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 63 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 64 |
+
subsequently incorporated within the Work.
|
| 65 |
+
|
| 66 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 67 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 68 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 69 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 70 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 71 |
+
Work and such Derivative Works in Source or Object form.
|
| 72 |
+
|
| 73 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 74 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 75 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 76 |
+
(except as stated in this section) patent license to make, have made,
|
| 77 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 78 |
+
where such license applies only to those patent claims licensable
|
| 79 |
+
by such Contributor that are necessarily infringed by their
|
| 80 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 81 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 82 |
+
institute patent litigation against any entity (including a
|
| 83 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 84 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 85 |
+
or contributory patent infringement, then any patent licenses
|
| 86 |
+
granted to You under this License for that Work shall terminate
|
| 87 |
+
as of the date such litigation is filed.
|
| 88 |
+
|
| 89 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 90 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 91 |
+
modifications, and in Source or Object form, provided that You
|
| 92 |
+
meet the following conditions:
|
| 93 |
+
|
| 94 |
+
(a) You must give any other recipients of the Work or
|
| 95 |
+
Derivative Works a copy of this License; and
|
| 96 |
+
|
| 97 |
+
(b) You must cause any modified files to carry prominent notices
|
| 98 |
+
stating that You changed the files; and
|
| 99 |
+
|
| 100 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 101 |
+
that You distribute, all copyright, patent, trademark, and
|
| 102 |
+
attribution notices from the Source form of the Work,
|
| 103 |
+
excluding those notices that do not pertain to any part of
|
| 104 |
+
the Derivative Works; and
|
| 105 |
+
|
| 106 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 107 |
+
distribution, then any Derivative Works that You distribute must
|
| 108 |
+
include a readable copy of the attribution notices contained
|
| 109 |
+
within such NOTICE file, excluding those notices that do not
|
| 110 |
+
pertain to any part of the Derivative Works, in at least one
|
| 111 |
+
of the following places: within a NOTICE text file distributed
|
| 112 |
+
as part of the Derivative Works; within the Source form or
|
| 113 |
+
documentation, if provided along with the Derivative Works; or,
|
| 114 |
+
within a display generated by the Derivative Works, if and
|
| 115 |
+
wherever such third-party notices normally appear. The contents
|
| 116 |
+
of the NOTICE file are for informational purposes only and
|
| 117 |
+
do not modify the License. You may add Your own attribution
|
| 118 |
+
notices within Derivative Works that You distribute, alongside
|
| 119 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 120 |
+
that such additional attribution notices cannot be construed
|
| 121 |
+
as modifying the License.
|
| 122 |
+
|
| 123 |
+
You may add Your own copyright statement to Your modifications and
|
| 124 |
+
may provide additional or different license terms and conditions
|
| 125 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 126 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 127 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 128 |
+
the conditions stated in this License.
|
| 129 |
+
|
| 130 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 131 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 132 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 133 |
+
this License, without any additional terms or conditions.
|
| 134 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 135 |
+
the terms of any separate license agreement you may have executed
|
| 136 |
+
with Licensor regarding such Contributions.
|
| 137 |
+
|
| 138 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 139 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 140 |
+
except as required for reasonable and customary use in describing the
|
| 141 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 142 |
+
|
| 143 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 144 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 145 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 146 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 147 |
+
implied, including, without limitation, any warranties or conditions
|
| 148 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 149 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 150 |
+
appropriateness of using or redistributing the Work and assume any
|
| 151 |
+
risks associated with Your exercise of permissions under this License.
|
| 152 |
+
|
| 153 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 154 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 155 |
+
unless required by applicable law (such as deliberate and grossly
|
| 156 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 157 |
+
liable to You for damages, including any direct, indirect, special,
|
| 158 |
+
incidental, or consequential damages of any character arising as a
|
| 159 |
+
result of this License or out of the use or inability to use the
|
| 160 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 161 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 162 |
+
other commercial damages or losses), even if such Contributor
|
| 163 |
+
has been advised of the possibility of such damages.
|
| 164 |
+
|
| 165 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 166 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 167 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 168 |
+
or other liability obligations and/or rights consistent with this
|
| 169 |
+
License. However, in accepting such obligations, You may act only
|
| 170 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 171 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 172 |
+
defend, and hold each Contributor harmless for any liability
|
| 173 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 174 |
+
of your accepting any such warranty or additional liability.
|
| 175 |
+
|
| 176 |
+
END OF TERMS AND CONDITIONS
|
| 177 |
+
|
| 178 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 179 |
+
|
| 180 |
+
To apply the Apache License to your work, attach the following
|
| 181 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 182 |
+
replaced with your own identifying information. (Don't include
|
| 183 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 184 |
+
comment syntax for the file format. We also recommend that a
|
| 185 |
+
file or class name and description of purpose be included on the
|
| 186 |
+
same "printed page" as the copyright notice for easier
|
| 187 |
+
identification within third-party archives.
|
| 188 |
+
|
| 189 |
+
Copyright [yyyy] [name of copyright owner]
|
| 190 |
+
|
| 191 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 192 |
+
you may not use this file except in compliance with the License.
|
| 193 |
+
You may obtain a copy of the License at
|
| 194 |
+
|
| 195 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 196 |
+
|
| 197 |
+
Unless required by applicable law or agreed to in writing, software
|
| 198 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 199 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 200 |
+
See the License for the specific language governing permissions and
|
| 201 |
+
limitations under the License.
|
| 202 |
+
|
MANIFEST.json
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"chat_template.jinja": {
|
| 3 |
+
"bytes": 7756,
|
| 4 |
+
"sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715"
|
| 5 |
+
},
|
| 6 |
+
"config.json": {
|
| 7 |
+
"bytes": 1979,
|
| 8 |
+
"sha256": "4ccab9e001f1b53e8213882a6e36c9aa3db09413aa2cf1e11caf18085077aabe"
|
| 9 |
+
},
|
| 10 |
+
"generation_config.json": {
|
| 11 |
+
"bytes": 116,
|
| 12 |
+
"sha256": "62153eb6c69f2e1f426beaa8002b7186437e949c7588167085df14e10e9c0a73"
|
| 13 |
+
},
|
| 14 |
+
"model.safetensors": {
|
| 15 |
+
"bytes": 8411558400,
|
| 16 |
+
"sha256": "dabeb7bc3f43bf6ea7d20ecfdac8758b3945517a991809a5772988829598e893"
|
| 17 |
+
},
|
| 18 |
+
"serving.json": {
|
| 19 |
+
"bytes": 126,
|
| 20 |
+
"sha256": "34ae0348497abc99bedee0e8f65b4f6de5bc7888f7347ca30cb926b75acf8ca6"
|
| 21 |
+
},
|
| 22 |
+
"temperature.json": {
|
| 23 |
+
"bytes": 2717,
|
| 24 |
+
"sha256": "a0f72cd2d0a653e81051e5a0c77fc1a69131552a8102b580a93a6dbe7908b2da"
|
| 25 |
+
},
|
| 26 |
+
"tokenizer.json": {
|
| 27 |
+
"bytes": 19989509,
|
| 28 |
+
"sha256": "bd53432f0de26d67a83b634040d4f043053da4ecd0c759e7f4b24ad4f8bb9a81"
|
| 29 |
+
},
|
| 30 |
+
"tokenizer_config.json": {
|
| 31 |
+
"bytes": 1127,
|
| 32 |
+
"sha256": "171ecbe7ddae98d11840698f7df2b8d5b4722139db0f0620d3bbf429bd656250"
|
| 33 |
+
}
|
| 34 |
+
}
|
NOTICE
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Wald-4B
|
| 2 |
+
Copyright 2026 the Wald-4B authors
|
| 3 |
+
|
| 4 |
+
The weights and the serving code in this repository are licensed under the Apache License, Version 2.0 (see LICENSE).
|
| 5 |
+
|
| 6 |
+
Base model
|
| 7 |
+
The weights are derived from Qwen/Qwen3.5-4B-Base (https://huggingface.co/Qwen/Qwen3.5-4B-Base), licensed under the
|
| 8 |
+
Apache License, Version 2.0, copyright the Qwen team, Alibaba Cloud. Its licence and notices continue to apply.
|
| 9 |
+
|
| 10 |
+
Third-party code and text in server/
|
| 11 |
+
server/src/wald_serve/wire.py adapts the request models and the state / option rendering (render, option_text,
|
| 12 |
+
question_keys, to_record) of Kev (https://github.com/jaredpalmer/kev), Copyright Jared Palmer, licensed under the
|
| 13 |
+
Apache License, Version 2.0. Modified: the record carries each question's type and keys; unused fields removed.
|
| 14 |
+
|
| 15 |
+
server/src/wald_serve/prompt.py contains the sentence STATE_REPEAT, copied verbatim from simple-jev
|
| 16 |
+
(https://github.com/featherless-ai/simple-jev, commit dae340e, hf-server/hf_prompt_policies.py), licensed under the
|
| 17 |
+
Apache License, Version 2.0. It is used only by the `repeat_state_plain` prompt format.
|
| 18 |
+
|
| 19 |
+
No benchmark items or training data are included in this repository.
|
README.md
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: apache-2.0
|
| 3 |
+
base_model: Qwen/Qwen3.5-4B-Base
|
| 4 |
+
base_model_relation: finetune
|
| 5 |
+
language:
|
| 6 |
+
- en
|
| 7 |
+
tags:
|
| 8 |
+
- decision-model
|
| 9 |
+
- calibration
|
| 10 |
+
- typesafe
|
| 11 |
+
- decision-index
|
| 12 |
+
pipeline_tag: text-generation
|
| 13 |
+
---
|
| 14 |
+
|
| 15 |
+
<div align="center">
|
| 16 |
+
<h1>Wald-4B v1.0</h1>
|
| 17 |
+
<p><strong>A calibrated decision model: one forward pass turns a state and a question into a probability for every option.</strong></p>
|
| 18 |
+
</div>
|
| 19 |
+
|
| 20 |
+
**W**ait **A** bit, **L**ook, then **D**ecide. Sure? Decide now. Not sure? Look once more. Also named after Abraham Wald (1902–1950), who founded sequential analysis: stop as soon as the evidence is enough.
|
| 21 |
+
|
| 22 |
+
---
|
| 23 |
+
|
| 24 |
+
<p align="center"><a href="https://huggingface.co/Harry19081/Wald-4B">English</a> · <a href="https://huggingface.co/Harry19081/Wald-4B/blob/main/docs/readmes/README.zh.md">简体中文</a></p>
|
| 25 |
+
|
| 26 |
+
---
|
| 27 |
+
|
| 28 |
+
**Wald-4B turns a decision into one forward pass.** Give it a state and a question, and it returns a calibrated
|
| 29 |
+
probability for every option: which tool to call, whether to ask the user, which intent, whether an output is safe. Act
|
| 30 |
+
above 0.9, ask a person below 0.6, and send the cases in between to a larger model. Under effort `medium` it thinks
|
| 31 |
+
(≤ 512 tokens) only when its top probability is below 0.7, about one question in nine; the answer is always a
|
| 32 |
+
distribution, never parsed text.
|
| 33 |
+
|
| 34 |
+
## At a glance
|
| 35 |
+
|
| 36 |
+
- **Decision Index 0.2.1: 53.91** (sample estimate), #6 of 67 on the board and the best 4B entry ([Decision Index](#decision-index-021)).
|
| 37 |
+
- **XL hidden split: 63.2**, #2 behind Jev and the best 4B ([XL](#xl-hidden-split)).
|
| 38 |
+
- **Your task for under $2:** one LoRA beats Jev on 9 of 12 public tasks ([Verticals](#verticals-a-quick-lora-per-task)).
|
| 39 |
+
- **~50 ms per decision** at p50, about 1/5 of Jev's price per 1k decisions ([Latency and cost](#latency-and-cost)).
|
| 40 |
+
|
| 41 |
+

|
| 42 |
+
|
| 43 |
+
## How it works
|
| 44 |
+
|
| 45 |
+
A 4B model built on Qwen/Qwen3.5-4B-Base @ `1001bb4d826a52d1f399e183466143f4da7b741b`: full-parameter decision training,
|
| 46 |
+
then three merged LoRAs (refinement with KL replay, short-thought distillation, task coverage). Each question becomes one
|
| 47 |
+
plain prompt with no chat template and no new tokens. The state is written twice (format `repeat_state_plain`; the
|
| 48 |
+
second copy is introduced by one fixed sentence, see NOTICE), then:
|
| 49 |
+
|
| 50 |
+
```
|
| 51 |
+
Question: {instructions}
|
| 52 |
+
(A) {option 1}
|
| 53 |
+
(B) {option 2}
|
| 54 |
+
Answer: (
|
| 55 |
+
```
|
| 56 |
+
|
| 57 |
+
The option-letter logits at the last position give the distribution. Above 26 options, options are read in ⌈n / 26⌉
|
| 58 |
+
chunks and the chunk winners are read again (knockout). A temperature per (question type, option count), fitted on our own
|
| 59 |
+
held-out rows, changes confidence only. The prompt format, the policy and the context limit are fixed in `serving.json`
|
| 60 |
+
and are the same for every benchmark.
|
| 61 |
+
|
| 62 |
+
**Training data** (not released): four stages over 84 public sources (83 datasets read through their train splits only, NLI4CT also its dev split,
|
| 63 |
+
plus the Decision Index kit's MIT Home-appliance generator), code-generated and code-labelled decision items, and KL
|
| 64 |
+
replay of earlier rows. Part of the data is synthetic: items and short thoughts written or labelled by much larger
|
| 65 |
+
frontier LLMs — far above 120B parameters where the size is published. Every source, and how Decision Index items
|
| 66 |
+
were screened: [CONTAMINATION.md](https://huggingface.co/Harry19081/Wald-4B/blob/main/CONTAMINATION.md).
|
| 67 |
+
|
| 68 |
+
## Benchmarks
|
| 69 |
+
|
| 70 |
+
What we measured: the Decision Index 0.2.1 (the board), XL (a decision benchmark with a hidden split), task LoRAs on
|
| 71 |
+
public verticals, and serving latency. All numbers are our own reads, zero-shot unless marked, with the same requests
|
| 72 |
+
for every system; rows marked *board* are the official leaderboard's.
|
| 73 |
+
|
| 74 |
+
### Decision Index 0.2.1
|
| 75 |
+
|
| 76 |
+
Our read is a **stratified sample of 6,948 requests** (37,469 questions; the kit's own sampler and seed; HLE not
|
| 77 |
+
rebuilt), scored with the 0.2.1 rules. Jev reads 57.19 on this sample against its board 57.89, so the sample reads about
|
| 78 |
+
0.7 points low. Board rows are official full-suite numbers ([board](https://huggingface.co/spaces/multimodalart/jev-decision-index)).
|
| 79 |
+
|
| 80 |
+
| system | size | Decision Index 0.2.1 | source |
|
| 81 |
+
|---|---|---|---|
|
| 82 |
+
| Jev (hosted API, jev-1.13.0) | undisclosed | 57.19 [55.29, 58.60] · board 57.89 | ours, sample · board |
|
| 83 |
+
| simple-jev · Qwen3.8-27B (board #4) · Jebadiah 27B (#5) | 27B | 55.74 · 54.67 | board |
|
| 84 |
+
| **Wald-4B v1.0 · effort medium** | 4B | **53.91** [51.87, 55.27] | ours, sample |
|
| 85 |
+
| reflex 27B (board #6) · Decider chat · Qwen3.6-27B (#7) | 27B | 52.16 · 51.35 | board |
|
| 86 |
+
| Qwen3.8-27B, untrained, our letter readout (one pass) | 27B | 47.78 [45.95, 49.17] | ours, sample |
|
| 87 |
+
| Decider 35B-A3B (board #11) | 35B-A3B | 47.11 | board |
|
| 88 |
+
| Decider 4B (board #15) | 4B | 40.70 | board |
|
| 89 |
+
| Kev 9B (board #23) · Kev 4B (#28) | 9B · 4B | 38.48 · 34.64 | board |
|
| 90 |
+
|
| 91 |
+
53.91 places #6 on the 67-entry board (between Jebadiah 27B and reflex 27B); against Jev on the same sample it is
|
| 92 |
+
−3.3 [−4.9, −1.8]. On the same 4B base it is +19.3 over Kev 4B.
|
| 93 |
+
|
| 94 |
+

|
| 95 |
+
|
| 96 |
+
**Disclosures.** 363 Decision Index item ids reached our training data through public train splits or shared upstream
|
| 97 |
+
sources (never a test split); counting them wrong gives **53.62**. Home appliance scores 1.00 (skill): the last stage
|
| 98 |
+
trained on new households from the benchmark's own MIT generator (new seed, 0 shared states); with Home appliance held
|
| 99 |
+
at the parent build's answers the index is 51.57 [49.56, 52.95]. Details: [CONTAMINATION.md](https://huggingface.co/Harry19081/Wald-4B/blob/main/CONTAMINATION.md).
|
| 100 |
+
|
| 101 |
+
**Same-request suite.** JevBench public 231 (accuracy %): Wald-4B v1.0 **88.3** (packaged server), Jev 86.6, raw
|
| 102 |
+
Qwen3.5-4B-Base 67.5, Laya 421M 58.0, CLM-8B 39.0. CLM-8B and Laya have no Decision Index read.
|
| 103 |
+
|
| 104 |
+
### XL (hidden split)
|
| 105 |
+
|
| 106 |
+
A decision benchmark we built (13 families; ranked on a hidden split of 1,895 items, chance-corrected composite XL-Int).
|
| 107 |
+
We are its authors, so no number is independent; our models score far higher on its public split than on the hidden one,
|
| 108 |
+
so only the hidden split is reported.
|
| 109 |
+
|
| 110 |
+
| system | XL-Int (hidden) | rank |
|
| 111 |
+
|---|---:|---:|
|
| 112 |
+
| Jev (API, v1.13) | 73.0 [69.8, 75.9] | #1 |
|
| 113 |
+
| **Wald-4B v1.0 · effort medium** | **63.2** [60.0, 66.4] | **#2**, best 4B |
|
| 114 |
+
| Cygnet (gemma-4-12B-it + shim) | 59.8 [56.7, 62.7] | #4 |
|
| 115 |
+
| Laya 421M · CLM-8B | 13.4 · 12.7 | #24 · #25 |
|
| 116 |
+
|
| 117 |
+
### Verticals: a quick LoRA per task
|
| 118 |
+
|
| 119 |
+

|
| 120 |
+
|
| 121 |
+
One LoRA on the task's labels costs $0.12–$1.81 of GPU time (< 2 GPU-hours); every other system is zero-shot on the
|
| 122 |
+
same fixed test items and byte-identical requests. Differences to Jev are paired bootstrap 95 % CIs; ▲ = the CI excludes
|
| 123 |
+
0. These reads use the pre-release build v0.9 (see Versioning).
|
| 124 |
+
|
| 125 |
+
| task (metric) | Wald-4B v0.9 + LoRA | Jev |
|
| 126 |
+
|---|---:|---:|
|
| 127 |
+
| MetaTool (tool selection, accuracy) | **97.0** | 83.0 |
|
| 128 |
+
| BANKING77 (77 intents, macro-F1) | **92.8** | 78.1 |
|
| 129 |
+
| AndroidControl (phone-agent action, accuracy) | **85.8**¹ | 73.4 |
|
| 130 |
+
| ToxicChat (toxic-class F1) | **84.9** | 80.4 |
|
| 131 |
+
| COLD, Chinese offensive language (macro-F1, full 5,323-item test) | **84.1** [83.2, 85.1] | 75.3 |
|
| 132 |
+
| When2Call (call / ask / refuse, accuracy) | **83.0**² | 72.6 |
|
| 133 |
+
|
| 134 |
+
¹ Trained on all training steps; the figure shows the first all-labels arm (81.6). ² Best of 8 LoRA configurations read
|
| 135 |
+
on test; the dev-selected one scores 82.6. COLD: paired +8.8 [7.7, 10.0]; 83.8 on the 4,823 items not used for
|
| 136 |
+
temperature fitting; ties the best published 83.7.
|
| 137 |
+
|
| 138 |
+
#### With few labels, starting from Wald beats LoRA on the raw base
|
| 139 |
+
|
| 140 |
+
Same LoRA recipe, labels and test items; Wald-4B (v0.9) minus raw Qwen3.5-4B-Base, points, paired 95 % CIs.
|
| 141 |
+
**Bold** = the CI excludes 0.
|
| 142 |
+
|
| 143 |
+
| task (metric) | 0 labels (zero-shot) | 300 labels |
|
| 144 |
+
|---|---:|---:|
|
| 145 |
+
| When2Call (accuracy) | **+21.2** [17.0, 25.2] | +2.0 [−0.2, 4.2] |
|
| 146 |
+
| BANKING77 (macro-F1) | **+10.9** [7.6, 14.7] | **+4.4** [2.0, 7.2] |
|
| 147 |
+
| SGD intent (macro-F1) | +2.8 (n.s.) | −0.1 |
|
| 148 |
+
|
| 149 |
+
### Latency and cost
|
| 150 |
+
|
| 151 |
+

|
| 152 |
+
|
| 153 |
+
One question, one pass, serial: p50 26 ms on one RTX PRO 6000 (51 ms on an H100). Batched: 115 decisions/s (fp8), about
|
| 154 |
+
$0.007 per 1,000 decisions at the GPU-hour price; the Jev API lists $0.04. Under `medium` the median stays at the
|
| 155 |
+
one-pass level and the p95 is about one second. Measured on v0.9, which has the same architecture and serving path.
|
| 156 |
+
|
| 157 |
+
## Train your own task LoRA (CLI, releasing soon)
|
| 158 |
+
|
| 159 |
+
Every vertical above was made with one CLI, which we plan to release soon. It splits a labelled dataset with fixed seeds,
|
| 160 |
+
checks the test items for leaks, trains a LoRA on Wald-4B on a GPU you choose, reads the same test items with your LoRA,
|
| 161 |
+
Jev and zero-shot baselines, and serves the adapter behind the same `/v1/systemone` API.
|
| 162 |
+
|
| 163 |
+

|
| 164 |
+
|
| 165 |
+
Stages, guards, costs and a command preview: [docs/lora-cli.md](https://huggingface.co/Harry19081/Wald-4B/blob/main/docs/lora-cli.md).
|
| 166 |
+
|
| 167 |
+
## Serving
|
| 168 |
+
|
| 169 |
+
```sh
|
| 170 |
+
docker build -t wald-serve .
|
| 171 |
+
docker run --gpus all -v /path/to/Wald-4B:/model:ro -p 8000:8000 wald-serve # ready when GET /health is {"ok": true}
|
| 172 |
+
```
|
| 173 |
+
|
| 174 |
+
Without Docker: `./run.sh /path/to/Wald-4B`. The server takes `--effort` (`none | low | medium | high | high-k<k>`), and a
|
| 175 |
+
request may carry `"effort"` to override it. Declared: effort `medium`, prompt format `repeat_state_plain`, 131,072
|
| 176 |
+
tokens per question prompt (longer → HTTP 422, never truncated), 1–255 options. Commands, limits and runtimes:
|
| 177 |
+
[RUNBOOK.md](https://huggingface.co/Harry19081/Wald-4B/blob/main/RUNBOOK.md). Weights: bf16, `model.safetensors` sha256
|
| 178 |
+
`dabeb7bc3f43bf6ea7d20ecfdac8758b3945517a991809a5772988829598e893`; every file's sha256 is in `MANIFEST.json`.
|
| 179 |
+
|
| 180 |
+
## Limits
|
| 181 |
+
|
| 182 |
+
- Our non-regression gate against the parent build has no failing metric. Two losses exceed the tolerance but are not
|
| 183 |
+
significant after correction: XL long_policy −6.7 [−11.6, −2.0] and JevBench public 231 −2.6 [−5.2, 0.0] (plain-prompt
|
| 184 |
+
read).
|
| 185 |
+
- The temperature table is the parent build's, the one the measured Decision Index read used.
|
| 186 |
+
- Our Decision Index read used a 16,384-token context; prompts longer than that were counted unsupported there and are
|
| 187 |
+
answered by the packaged server.
|
| 188 |
+
|
| 189 |
+
## Versioning
|
| 190 |
+
|
| 191 |
+
| version | status | checkpoint | base | date | policy | prompt format |
|
| 192 |
+
|---|---|---|---|---|---|---|
|
| 193 |
+
| **v1.0** | current release | internal build 021A0-f10 | Qwen/Qwen3.5-4B-Base | 2026-09-27 | effort `medium` | `repeat_state_plain` |
|
| 194 |
+
| v0.9 | pre-release; the vertical LoRAs and latency above | internal build 015D0-f4 | Qwen/Qwen3.5-4B-Base | 2026-09-26 | effort `medium` | plain |
|
| 195 |
+
|
| 196 |
+
v1.x = serving or LoRA refreshes on the same base generation; v2.0 = a new base generation.
|
| 197 |
+
|
| 198 |
+
---
|
| 199 |
+
|
| 200 |
+
Apache-2.0 (weights and code), subject to the licence of the base model Qwen/Qwen3.5-4B-Base (Apache-2.0, © the Qwen
|
| 201 |
+
team, Alibaba Cloud). Third-party notices: [NOTICE](https://huggingface.co/Harry19081/Wald-4B/blob/main/NOTICE).
|
RUNBOOK.md
ADDED
|
@@ -0,0 +1,110 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Runbook for the Decision Index maintainers
|
| 2 |
+
|
| 3 |
+
These steps serve Wald-4B on one NVIDIA RTX PRO 6000 (96 GB) so that the kit's `http` engine can run the full 0.2.1
|
| 4 |
+
suite against it.
|
| 5 |
+
|
| 6 |
+
## 1. Install
|
| 7 |
+
|
| 8 |
+
The weights are in the Hub repo `Harry19081/Wald-4B` (bf16 safetensors, tokenizer, config, `temperature.json`,
|
| 9 |
+
`serving.json`, `MANIFEST.json`). The code is in this repository, pinned by commit.
|
| 10 |
+
|
| 11 |
+
```sh
|
| 12 |
+
huggingface-cli download Harry19081/Wald-4B --local-dir /models/Wald-4B
|
| 13 |
+
cd /models/Wald-4B && python - <<'EOF' # optional: verify every file against MANIFEST.json
|
| 14 |
+
import hashlib, json; m = json.load(open("MANIFEST.json"))
|
| 15 |
+
for f, v in m.items():
|
| 16 |
+
h = hashlib.sha256(open(f, "rb").read()).hexdigest(); assert h == v["sha256"], f
|
| 17 |
+
print("ok", len(m))
|
| 18 |
+
EOF
|
| 19 |
+
```
|
| 20 |
+
|
| 21 |
+
Choose one of the following.
|
| 22 |
+
|
| 23 |
+
- **Docker:**
|
| 24 |
+
```sh
|
| 25 |
+
docker build -t wald-serve .
|
| 26 |
+
docker run --gpus '"device=0"' -v /models/Wald-4B:/model:ro -p 8000:8000 wald-serve
|
| 27 |
+
```
|
| 28 |
+
- **No Docker** (Python 3.12 and `uv`):
|
| 29 |
+
```sh
|
| 30 |
+
./run.sh /models/Wald-4B
|
| 31 |
+
```
|
| 32 |
+
- **By hand:**
|
| 33 |
+
```sh
|
| 34 |
+
uv venv -p 3.12 .venv && VIRTUAL_ENV=.venv uv pip install "vllm==0.30.0" ./server
|
| 35 |
+
VLLM_USE_FLASHINFER_SAMPLER=0 .venv/bin/wald-serve --model /models/Wald-4B --port 8000
|
| 36 |
+
```
|
| 37 |
+
`VLLM_USE_FLASHINFER_SAMPLER=0` is needed because FlashInfer's sampler refuses sm_120 in vLLM 0.30.0.
|
| 38 |
+
|
| 39 |
+
In every case, `wald-serve` starts vLLM 0.30.0 on the weights: bf16, loopback only, `--max-model-len 131072`,
|
| 40 |
+
`--gpu-memory-utilization 0.90`, `--max-num-seqs 256`, `--seed 0`. It then serves `POST /v1/systemone` on port 8000. The
|
| 41 |
+
service is ready when `curl -s localhost:8000/health` returns `{"ok": true, ...}` with the effective configuration. The
|
| 42 |
+
first start takes about 2–4 minutes (weights load and CUDA graph capture).
|
| 43 |
+
|
| 44 |
+
## 2. Run the suite
|
| 45 |
+
|
| 46 |
+
```sh
|
| 47 |
+
python -m decision_index run --engine http --option base_url=http://127.0.0.1:8000 --option model=wald-4b \
|
| 48 |
+
--option timeout=3600 ...
|
| 49 |
+
```
|
| 50 |
+
|
| 51 |
+
- **Send requests concurrently.** We ran 16–32 kit processes over group-balanced shards against one server, and vLLM
|
| 52 |
+
batches them. A single sequential client would be several times slower (section 4).
|
| 53 |
+
- **Use a long timeout (≥ 3,600 s).** Under load, the p95 request latency was several minutes, from the requests with
|
| 54 |
+
many long questions (BRIGHT / ToolRet chunks, ContractNLI, ACOS).
|
| 55 |
+
- **A 422 means a declared capacity limit**, and the body contains `maximum context length`. Count it as unsupported,
|
| 56 |
+
not as an error.
|
| 57 |
+
- **Smoke test:**
|
| 58 |
+
```sh
|
| 59 |
+
curl -s localhost:8000/v1/systemone -H 'content-type: application/json' -d '{"state":"I was charged twice.","questions":{"q":{"type":"noul","instructions":"Is this a billing problem?"}}}'
|
| 60 |
+
```
|
| 61 |
+
|
| 62 |
+
## 3. Declared configuration and limits
|
| 63 |
+
|
| 64 |
+
| item | value |
|
| 65 |
+
|---|---|
|
| 66 |
+
| declared policy | **effort `medium`**: one pass; if the top probability is below 0.7, a thought of at most 512 tokens (temperature 0.6, top-p 0.95, top-k 20, seed = hash of the request and question id), then a second read |
|
| 67 |
+
| prompt format | `repeat_state_plain` (the state written twice), from `serving.json`; the same for every benchmark |
|
| 68 |
+
| readout | option-letter logits at the last position; tempered with `temperature.json` (A table for one-pass answers, B512 after a thought) |
|
| 69 |
+
| > 26 options | knockout: ⌈n / 26⌉ consecutive chunks, then a final over the chunk winners; one pass; up to 676 options (the wire allows 255) |
|
| 70 |
+
| context | 131,072 tokens per question prompt; longer → HTTP 422 `maximum context length`, never truncated |
|
| 71 |
+
| thought that does not fit | if prompt + 512 + 64 tokens exceeds the context, the one-pass answer is returned |
|
| 72 |
+
| questions per request | no fixed limit; read concurrently (8 at a time per request); tested up to 64 |
|
| 73 |
+
| determinism | same request → same thought seed → same answer, up to bf16 batch-order noise (argmax agreement 99.7 % between two reads of 36k questions) |
|
| 74 |
+
|
| 75 |
+
Nothing in the server depends on the benchmark: there is no per-benchmark prompt, option filtering, retry or truncation.
|
| 76 |
+
|
| 77 |
+
## 4. Expected runtime of the full suite on one RTX PRO 6000
|
| 78 |
+
|
| 79 |
+
The full suite has about 150,500 scoreable requests, including the display-only MMLU / ARC and RouterBench (not scored
|
| 80 |
+
in 0.2.1). HLE is not counted here.
|
| 81 |
+
|
| 82 |
+
**How we estimated.** Our basis is the measured wall time of our 6,948-request sample reads:
|
| 83 |
+
- one pass: 875 s;
|
| 84 |
+
- medium: 1,702 s. Both were measured on one card shared with a second engine, with 16 kit processes each.
|
| 85 |
+
- always-think: 5,520 s, on a shared card, including a k = 4 read of the knowledge and language rows.
|
| 86 |
+
|
| 87 |
+
We scaled these by a full-to-sample work ratio of 12–15×. That ratio weights each benchmark's per-request latency by its
|
| 88 |
+
full size; the sample over-represents long-prompt benchmarks. A dedicated card is assumed to be 1.25–1.7× faster than
|
| 89 |
+
our shared runs. `repeat_state_plain` writes the state twice, which adds about 30–60 % to prefill: this is an estimate,
|
| 90 |
+
not measured on the full suite. The in-request question concurrency of this server shortens the tail of requests with
|
| 91 |
+
many questions, and the table does not credit it.
|
| 92 |
+
|
| 93 |
+
| policy | thinks on | plain prompt | repeat_state_plain (declared) |
|
| 94 |
+
|---|---|---|---|
|
| 95 |
+
| none | 0 % | ≈ 2–3 h | ≈ 2.5–4.5 h |
|
| 96 |
+
| low | a few % | ≈ 2.5–4 h | ≈ 3–5.5 h |
|
| 97 |
+
| **medium (declared)** | ≈ 12 % of questions | **≈ 3.5–6 h** | **≈ 4.5–9 h** |
|
| 98 |
+
| high | every 2–26-option question | ≈ 7–12 h | ≈ 9–16 h |
|
| 99 |
+
| high-k4 | every question, 4 thoughts | ≈ 12–20 h | ≈ 15–25 h |
|
| 100 |
+
|
| 101 |
+
These are estimates, and our own full-suite read with the packaged server will replace them (CHECKLIST.md). We declare
|
| 102 |
+
`medium`. `high-k4` would take most of a day on one card: prefill for every question would be about 2.3× that of `high`,
|
| 103 |
+
and decoding 4×. Our measured `high-k4` gain over `high` was not significant (+0.48 [−0.4, +1.5] on the parent
|
| 104 |
+
checkpoint), and that measurement applied k = 4 only to the knowledge and language benchmarks.
|
| 105 |
+
|
| 106 |
+
## 5. Throughput and latency (for reference)
|
| 107 |
+
|
| 108 |
+
- One question, one pass, serial: p50 26 ms on one RTX PRO 6000; 51 ms on an H100.
|
| 109 |
+
- `medium`, serial: p50 ≈ 52 ms, p95 ≈ 1.1 s (the questions that think).
|
| 110 |
+
- Batched one-pass throughput: about 115 decisions/s on one RTX PRO 6000 (fp8). bf16 throughput was not measured separately.
|
chat_template.jinja
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{%- set image_count = namespace(value=0) %}
|
| 2 |
+
{%- set video_count = namespace(value=0) %}
|
| 3 |
+
{%- macro render_content(content, do_vision_count, is_system_content=false) %}
|
| 4 |
+
{%- if content is string %}
|
| 5 |
+
{{- content }}
|
| 6 |
+
{%- elif content is iterable and content is not mapping %}
|
| 7 |
+
{%- for item in content %}
|
| 8 |
+
{%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
|
| 9 |
+
{%- if is_system_content %}
|
| 10 |
+
{{- raise_exception('System message cannot contain images.') }}
|
| 11 |
+
{%- endif %}
|
| 12 |
+
{%- if do_vision_count %}
|
| 13 |
+
{%- set image_count.value = image_count.value + 1 %}
|
| 14 |
+
{%- endif %}
|
| 15 |
+
{%- if add_vision_id %}
|
| 16 |
+
{{- 'Picture ' ~ image_count.value ~ ': ' }}
|
| 17 |
+
{%- endif %}
|
| 18 |
+
{{- '<|vision_start|><|image_pad|><|vision_end|>' }}
|
| 19 |
+
{%- elif 'video' in item or item.type == 'video' %}
|
| 20 |
+
{%- if is_system_content %}
|
| 21 |
+
{{- raise_exception('System message cannot contain videos.') }}
|
| 22 |
+
{%- endif %}
|
| 23 |
+
{%- if do_vision_count %}
|
| 24 |
+
{%- set video_count.value = video_count.value + 1 %}
|
| 25 |
+
{%- endif %}
|
| 26 |
+
{%- if add_vision_id %}
|
| 27 |
+
{{- 'Video ' ~ video_count.value ~ ': ' }}
|
| 28 |
+
{%- endif %}
|
| 29 |
+
{{- '<|vision_start|><|video_pad|><|vision_end|>' }}
|
| 30 |
+
{%- elif 'text' in item %}
|
| 31 |
+
{{- item.text }}
|
| 32 |
+
{%- else %}
|
| 33 |
+
{{- raise_exception('Unexpected item type in content.') }}
|
| 34 |
+
{%- endif %}
|
| 35 |
+
{%- endfor %}
|
| 36 |
+
{%- elif content is none or content is undefined %}
|
| 37 |
+
{{- '' }}
|
| 38 |
+
{%- else %}
|
| 39 |
+
{{- raise_exception('Unexpected content type.') }}
|
| 40 |
+
{%- endif %}
|
| 41 |
+
{%- endmacro %}
|
| 42 |
+
{%- if not messages %}
|
| 43 |
+
{{- raise_exception('No messages provided.') }}
|
| 44 |
+
{%- endif %}
|
| 45 |
+
{%- if tools and tools is iterable and tools is not mapping %}
|
| 46 |
+
{{- '<|im_start|>system\n' }}
|
| 47 |
+
{{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
|
| 48 |
+
{%- for tool in tools %}
|
| 49 |
+
{{- "\n" }}
|
| 50 |
+
{{- tool | tojson }}
|
| 51 |
+
{%- endfor %}
|
| 52 |
+
{{- "\n</tools>" }}
|
| 53 |
+
{{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
|
| 54 |
+
{%- if messages[0].role == 'system' %}
|
| 55 |
+
{%- set content = render_content(messages[0].content, false, true)|trim %}
|
| 56 |
+
{%- if content %}
|
| 57 |
+
{{- '\n\n' + content }}
|
| 58 |
+
{%- endif %}
|
| 59 |
+
{%- endif %}
|
| 60 |
+
{{- '<|im_end|>\n' }}
|
| 61 |
+
{%- else %}
|
| 62 |
+
{%- if messages[0].role == 'system' %}
|
| 63 |
+
{%- set content = render_content(messages[0].content, false, true)|trim %}
|
| 64 |
+
{{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
|
| 65 |
+
{%- endif %}
|
| 66 |
+
{%- endif %}
|
| 67 |
+
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
| 68 |
+
{%- for message in messages[::-1] %}
|
| 69 |
+
{%- set index = (messages|length - 1) - loop.index0 %}
|
| 70 |
+
{%- if ns.multi_step_tool and message.role == "user" %}
|
| 71 |
+
{%- set content = render_content(message.content, false)|trim %}
|
| 72 |
+
{%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
|
| 73 |
+
{%- set ns.multi_step_tool = false %}
|
| 74 |
+
{%- set ns.last_query_index = index %}
|
| 75 |
+
{%- endif %}
|
| 76 |
+
{%- endif %}
|
| 77 |
+
{%- endfor %}
|
| 78 |
+
{%- if ns.multi_step_tool %}
|
| 79 |
+
{{- raise_exception('No user query found in messages.') }}
|
| 80 |
+
{%- endif %}
|
| 81 |
+
{%- for message in messages %}
|
| 82 |
+
{%- set content = render_content(message.content, true)|trim %}
|
| 83 |
+
{%- if message.role == "system" %}
|
| 84 |
+
{%- if not loop.first %}
|
| 85 |
+
{{- raise_exception('System message must be at the beginning.') }}
|
| 86 |
+
{%- endif %}
|
| 87 |
+
{%- elif message.role == "user" %}
|
| 88 |
+
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 89 |
+
{%- elif message.role == "assistant" %}
|
| 90 |
+
{%- set reasoning_content = '' %}
|
| 91 |
+
{%- if message.reasoning_content is string %}
|
| 92 |
+
{%- set reasoning_content = message.reasoning_content %}
|
| 93 |
+
{%- else %}
|
| 94 |
+
{%- if '</think>' in content %}
|
| 95 |
+
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
| 96 |
+
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
| 97 |
+
{%- endif %}
|
| 98 |
+
{%- endif %}
|
| 99 |
+
{%- set reasoning_content = reasoning_content|trim %}
|
| 100 |
+
{%- if loop.index0 > ns.last_query_index %}
|
| 101 |
+
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
|
| 102 |
+
{%- else %}
|
| 103 |
+
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 104 |
+
{%- endif %}
|
| 105 |
+
{%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
|
| 106 |
+
{%- for tool_call in message.tool_calls %}
|
| 107 |
+
{%- if tool_call.function is defined %}
|
| 108 |
+
{%- set tool_call = tool_call.function %}
|
| 109 |
+
{%- endif %}
|
| 110 |
+
{%- if loop.first %}
|
| 111 |
+
{%- if content|trim %}
|
| 112 |
+
{{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 113 |
+
{%- else %}
|
| 114 |
+
{{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 115 |
+
{%- endif %}
|
| 116 |
+
{%- else %}
|
| 117 |
+
{{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 118 |
+
{%- endif %}
|
| 119 |
+
{%- if tool_call.arguments is defined %}
|
| 120 |
+
{%- for args_name, args_value in tool_call.arguments|items %}
|
| 121 |
+
{{- '<parameter=' + args_name + '>\n' }}
|
| 122 |
+
{%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
|
| 123 |
+
{{- args_value }}
|
| 124 |
+
{{- '\n</parameter>\n' }}
|
| 125 |
+
{%- endfor %}
|
| 126 |
+
{%- endif %}
|
| 127 |
+
{{- '</function>\n</tool_call>' }}
|
| 128 |
+
{%- endfor %}
|
| 129 |
+
{%- endif %}
|
| 130 |
+
{{- '<|im_end|>\n' }}
|
| 131 |
+
{%- elif message.role == "tool" %}
|
| 132 |
+
{%- if loop.previtem and loop.previtem.role != "tool" %}
|
| 133 |
+
{{- '<|im_start|>user' }}
|
| 134 |
+
{%- endif %}
|
| 135 |
+
{{- '\n<tool_response>\n' }}
|
| 136 |
+
{{- content }}
|
| 137 |
+
{{- '\n</tool_response>' }}
|
| 138 |
+
{%- if not loop.last and loop.nextitem.role != "tool" %}
|
| 139 |
+
{{- '<|im_end|>\n' }}
|
| 140 |
+
{%- elif loop.last %}
|
| 141 |
+
{{- '<|im_end|>\n' }}
|
| 142 |
+
{%- endif %}
|
| 143 |
+
{%- else %}
|
| 144 |
+
{{- raise_exception('Unexpected message role.') }}
|
| 145 |
+
{%- endif %}
|
| 146 |
+
{%- endfor %}
|
| 147 |
+
{%- if add_generation_prompt %}
|
| 148 |
+
{{- '<|im_start|>assistant\n' }}
|
| 149 |
+
{%- if enable_thinking is defined and enable_thinking is false %}
|
| 150 |
+
{{- '<think>\n\n</think>\n\n' }}
|
| 151 |
+
{%- else %}
|
| 152 |
+
{{- '<think>\n' }}
|
| 153 |
+
{%- endif %}
|
| 154 |
+
{%- endif %}
|
config.json
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"Qwen3_5ForCausalLM"
|
| 4 |
+
],
|
| 5 |
+
"attention_bias": false,
|
| 6 |
+
"attention_dropout": 0.0,
|
| 7 |
+
"attn_output_gate": true,
|
| 8 |
+
"bos_token_id": null,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"eos_token_id": 248044,
|
| 11 |
+
"full_attention_interval": 4,
|
| 12 |
+
"head_dim": 256,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 2560,
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 9216,
|
| 17 |
+
"layer_types": [
|
| 18 |
+
"linear_attention",
|
| 19 |
+
"linear_attention",
|
| 20 |
+
"linear_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"linear_attention",
|
| 23 |
+
"linear_attention",
|
| 24 |
+
"linear_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"linear_attention",
|
| 27 |
+
"linear_attention",
|
| 28 |
+
"linear_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"linear_attention",
|
| 31 |
+
"linear_attention",
|
| 32 |
+
"linear_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"linear_attention",
|
| 35 |
+
"linear_attention",
|
| 36 |
+
"linear_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"linear_attention",
|
| 39 |
+
"linear_attention",
|
| 40 |
+
"linear_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"linear_attention",
|
| 43 |
+
"linear_attention",
|
| 44 |
+
"linear_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"linear_attention",
|
| 47 |
+
"linear_attention",
|
| 48 |
+
"linear_attention",
|
| 49 |
+
"full_attention"
|
| 50 |
+
],
|
| 51 |
+
"linear_conv_kernel_dim": 4,
|
| 52 |
+
"linear_key_head_dim": 128,
|
| 53 |
+
"linear_num_key_heads": 16,
|
| 54 |
+
"linear_num_value_heads": 32,
|
| 55 |
+
"linear_value_head_dim": 128,
|
| 56 |
+
"mamba_ssm_dtype": "float32",
|
| 57 |
+
"max_position_embeddings": 262144,
|
| 58 |
+
"mlp_only_layers": [],
|
| 59 |
+
"model_type": "qwen3_5_text",
|
| 60 |
+
"mtp_num_hidden_layers": 1,
|
| 61 |
+
"mtp_use_dedicated_embeddings": false,
|
| 62 |
+
"num_attention_heads": 16,
|
| 63 |
+
"num_hidden_layers": 32,
|
| 64 |
+
"num_key_value_heads": 4,
|
| 65 |
+
"pad_token_id": null,
|
| 66 |
+
"partial_rotary_factor": 0.25,
|
| 67 |
+
"rms_norm_eps": 1e-06,
|
| 68 |
+
"rope_parameters": {
|
| 69 |
+
"mrope_interleaved": true,
|
| 70 |
+
"mrope_section": [
|
| 71 |
+
11,
|
| 72 |
+
11,
|
| 73 |
+
10
|
| 74 |
+
],
|
| 75 |
+
"partial_rotary_factor": 0.25,
|
| 76 |
+
"rope_theta": 10000000,
|
| 77 |
+
"rope_type": "default"
|
| 78 |
+
},
|
| 79 |
+
"tie_word_embeddings": true,
|
| 80 |
+
"transformers_version": "5.17.0",
|
| 81 |
+
"use_cache": false,
|
| 82 |
+
"vocab_size": 248320
|
| 83 |
+
}
|
contamination/trained-on-di-ids.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
docs/lora-cli.md
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Fine-tune Wald-4B on your task (CLI, releasing soon)
|
| 2 |
+
|
| 3 |
+
> **Preview.** The CLI is not public yet; we plan to release it soon. The commands below are preview syntax and may
|
| 4 |
+
> change before the release.
|
| 5 |
+
|
| 6 |
+
**One dataset in, one task model out, with an honest comparison to Jev.** The `vertical` CLI takes a labelled dataset (a
|
| 7 |
+
Hugging Face repo, a URL, or your own file) and trains a LoRA on Wald-4B for that one task. It then reads the task's
|
| 8 |
+
fixed test items with your LoRA, with Jev and with zero-shot baselines, using byte-identical requests, and writes a report
|
| 9 |
+
with paired confidence intervals and calibration. The verticals in the [README](https://huggingface.co/Harry19081/Wald-4B#verticals-a-quick-lora-per-task)
|
| 10 |
+
were all made this way. One LoRA costs $0.12–$1.81 of GPU time (under 2 GPU-hours).
|
| 11 |
+
|
| 12 |
+

|
| 13 |
+
|
| 14 |
+
## The pipeline
|
| 15 |
+
|
| 16 |
+
Free stages run on your machine. Paid stages run on a GPU backend you choose: any SSH GPU box, RunPod or Modal. Every
|
| 17 |
+
stage is resumable. Reads are cached per item, a finished LoRA is not trained again, and each run keeps a manifest with
|
| 18 |
+
the sha256 of every output.
|
| 19 |
+
|
| 20 |
+
| stage | command | what it does |
|
| 21 |
+
|---|---|---|
|
| 22 |
+
| task spec | `init` · `inspect` · `validate` | Scaffold `task.yaml` from a dataset: the source pinned to a commit, fields, labels, question, metric and split sizes. `inspect` renders example records as the model will see them. |
|
| 23 |
+
| prep | `prep` | Download the data and hash every file. Render each item once, both as a decision record and as the exact request Jev receives. Make a seeded, label-stratified, group-aware split into train · calib · dev · test, and dedupe across splits. |
|
| 24 |
+
| overlap check | `prep` | Compare held-out items with the train pool, with Decision Index items and with Wald-4B's training data. The counts go in the data manifest. |
|
| 25 |
+
| audit | `audit` · `approve` | Free, deterministic rule checks before and after every stage, plus a review of a sample by a cheap model. A FAIL blocks the next paid stage. |
|
| 26 |
+
| augment (optional) | `augment` | Extra training rows: soft labels for unlabelled texts, synthetic items, or paraphrases. The teacher is your own coding agent or any OpenAI-compatible API. Every row is re-checked on ingest (schema, labels, dedupe, overlap with the held-out splits). |
|
| 27 |
+
| baseline | `baseline` | Read calib, dev and test with Jev (cached per request, so re-runs are free) and with zero-shot Wald-4B and the raw base. Optionally add frontier LLMs, plus random and class-prior floors. |
|
| 28 |
+
| train | `train` · `watch` | One LoRA per arm. An arm is a training size, for example 300 labels, 1,000 labels, or all of them. The recipe is rank 32 on every projection, lr 1e-4, 2 epochs, and KL replay toward the base's own answers, so the base keeps its other skills. `watch` tails the log and stops a bad run. |
|
| 29 |
+
| eval | `eval` | One vLLM server holds Wald-4B plus every adapter. Reads use the serving reader: letter readout, knockout above 26 options, and optional thinking effort. |
|
| 30 |
+
| calibrate | `calibrate` | Fit one temperature per system on the calib split, never on test. It changes confidence, never the answer. Jev is scored as served. |
|
| 31 |
+
| report | `report` · `compare` | `report.md` + `report.json`: the test metric with 2,000 paired bootstrap resamples against Jev and the base, ECE, high-confidence errors, and dev numbers for picking a configuration. `compare` gives paired differences between two runs, for example Wald-4B vs the raw base. |
|
| 32 |
+
| deploy | `serve` · `try` · `latency` | Serve the adapter and its temperature on Wald-4B behind the same `/v1/systemone` decision API. The adapter stays a separate file, and Wald-4B itself is never changed. `try` sends one item through Jev and your model side by side, and `latency` measures serial latency on an otherwise idle card. |
|
| 33 |
+
|
| 34 |
+
## Guards
|
| 35 |
+
|
| 36 |
+
A task model is only useful if its numbers are real. The CLI enforces these rules itself:
|
| 37 |
+
|
| 38 |
+
- **Same bytes everywhere.** Each item is rendered once, into one request. Jev receives that request, our reader reads
|
| 39 |
+
it, and your LoRA trains on the same bytes.
|
| 40 |
+
- **Leak gate.** Before training, any test item whose word 5-gram Jaccard similarity with a train or calib item is 0.8 or
|
| 41 |
+
higher stops the run. You can drop those items from this run (`--drop-near-dups`), or a person can approve keeping them.
|
| 42 |
+
- **Audit gate.** A FAIL (exit code 4) blocks every paid stage. Only a person can override it: `vertical approve` asks
|
| 43 |
+
y/N on a real terminal, or you click *Approve override* in the UI. The approval is signed, and it expires when the
|
| 44 |
+
data or the audit result changes. An agent cannot sign it.
|
| 45 |
+
- **Held-out discipline.** Temperatures are fitted on calib. Configurations are chosen on dev. Test is reported once per
|
| 46 |
+
configuration, and a best-of-N read on test is labelled as best-of-N.
|
| 47 |
+
- **Cost gate.** Paid stages print a cost estimate and run only with `--yes`, or when the estimate fits under `--max-usd`.
|
| 48 |
+
`--dry-run` prints the plan and changes nothing.
|
| 49 |
+
- **Training watch.** A NaN or infinite loss, a zero learning rate, the wrong base, or a cost above twice the estimate
|
| 50 |
+
stops the job.
|
| 51 |
+
- **Secrets.** Keys are passed at run time and never reach the GPU box's disk or the logs.
|
| 52 |
+
|
| 53 |
+
## Built for coding agents
|
| 54 |
+
|
| 55 |
+
Every command takes `--json` (one JSON object on stdout, with a `next` hint) and has fixed exit codes: 0 ok, 2 user or
|
| 56 |
+
config error, 3 remote failure, 4 audit FAIL. Spending money and overriding an audit still need a person's yes.
|
| 57 |
+
|
| 58 |
+
## Cost and time
|
| 59 |
+
|
| 60 |
+
- **One LoRA:** $0.12–$1.81 of GPU time on one RTX PRO 6000 at about $1 per hour (under 2 GPU-hours). Training runs at
|
| 61 |
+
about 5,000 tokens/s.
|
| 62 |
+
- **A full run** has cost $0.25–$2.15 on our public tasks. That covers the baseline reads, one to three training sizes,
|
| 63 |
+
eval and the report. Jev reads are billed by Jev and cached, so re-runs do not pay for them again.
|
| 64 |
+
|
| 65 |
+
## Example (preview syntax)
|
| 66 |
+
|
| 67 |
+
A task is one `task.yaml`:
|
| 68 |
+
|
| 69 |
+
```yaml
|
| 70 |
+
name: banking77
|
| 71 |
+
title: Bank customer intent routing
|
| 72 |
+
source: {hf: legacy-datasets/banking77, revision: <commit>, license: CC-BY-4.0, splits: {train: train, test: test}}
|
| 73 |
+
fields: {text: text, label: label}
|
| 74 |
+
labels: {from_features: true}
|
| 75 |
+
state_template: "Customer message: {text}"
|
| 76 |
+
question: Which intent does this bank customer's message express?
|
| 77 |
+
metric: macro_f1
|
| 78 |
+
sizes: {calib: 300, test: 500}
|
| 79 |
+
arms: [n300, n1000, all]
|
| 80 |
+
```
|
| 81 |
+
|
| 82 |
+
```sh
|
| 83 |
+
vertical init banking77 --hf legacy-datasets/banking77 # scaffold task.yaml
|
| 84 |
+
vertical inspect banking77 # rendered records, labels, lengths
|
| 85 |
+
vertical validate banking77
|
| 86 |
+
vertical prep banking77 # download, split, dedupe, overlap checks
|
| 87 |
+
vertical audit banking77 --stage data
|
| 88 |
+
vertical run banking77 --dry-run # cost of every paid stage; nothing runs
|
| 89 |
+
vertical run banking77 --max-usd 3 # baseline → train → eval → calibrate → report
|
| 90 |
+
vertical report banking77 # report.md + report.json
|
| 91 |
+
vertical serve banking77 --arm all --yes # the adapter behind /v1/systemone
|
| 92 |
+
vertical try banking77 --arm all --text "My card still hasn't arrived"
|
| 93 |
+
```
|
| 94 |
+
|
| 95 |
+
Other commands: `list`, `status`, `stop`, `backends --check`, `compare`, `latency`, `note`, `import-read` (adds a read
|
| 96 |
+
made outside the CLI to a run's report).
|
docs/observations/one-lora-all-verticals.md
ADDED
|
@@ -0,0 +1,112 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# One LoRA for all verticals? (observation, 2026-09-27)
|
| 2 |
+
|
| 3 |
+
**中文摘要:** 用一个 LoRA 代替每个任务各自一个 LoRA:在 13 个 vertical 的训练数据(外加几个 Decision Index 弱项的训练数据)上混训一个 LoRA,
|
| 4 |
+
1 epoch,一张 RTX PRO 6000 约 2.5 GPU 小时。在各任务固定测试集上逐题配对:读过 Jev 的 12 个任务里 **7 个显著超过 Jev**;和每个任务自己的
|
| 5 |
+
LoRA 比,When2Call / Mind2Web / RewardBench 打平、SGD intent 只差 1.4、agent 轨迹护栏(hard)反而 +3.5;BANKING77 / AndroidControl /
|
| 6 |
+
MetaTool / 提示注入差 4–7 分;**ToxicChat(−11)和两个路由任务(F1 ≈ 2,"永远不交给强模型")失效**。结论:一个共享适配器能覆盖大多数意图和
|
| 7 |
+
agent 决策任务,代价 1–5 分;正例稀少的护栏任务和按概率排序的路由任务还需要单独的适配器。
|
| 8 |
+
|
| 9 |
+
## 1. Question
|
| 10 |
+
|
| 11 |
+
Can one LoRA serve every vertical, instead of one adapter per task? A customer with a dozen decision points (intent routing,
|
| 12 |
+
tool choice, guardrails, agent next-step) would rather deploy one adapter than twelve.
|
| 13 |
+
|
| 14 |
+
## 2. Setup
|
| 15 |
+
|
| 16 |
+
- **Model:** one LoRA on an intermediate Wald-4B build between v0.9 and v1.0 (the v1.0 base before its final
|
| 17 |
+
task-coverage LoRA was merged; see Versioning in the README). Recipe: LoRA r32 / α64 on every projection, lr 1e-4 cosine,
|
| 18 |
+
5 % warm-up, **1 epoch**, letter readout, and 30 % replay rows trained toward the base's own answer distribution (a KL
|
| 19 |
+
anchor). The LoRA is kept separate, never merged into the released weights.
|
| 20 |
+
- **Data** (commercial licences only): 53,243 rows = 37,270 labelled + 15,973 replay, about 37M prompt tokens.
|
| 21 |
+
- 13 verticals, each capped at 2,500 rows after removing near-duplicates of **every** vertical's test / calibration /
|
| 22 |
+
development items. Rows per task after that: When2Call, BANKING77, SGD intent, ToxicChat, MetaTool, AndroidControl,
|
| 23 |
+
RewardBench, Mind2Web 2,500 each; model routing 1,174 (and 1,174 for a second framing of the same prompts); prompt
|
| 24 |
+
injection 780; agent-trajectory safety 52; **agent-trajectory safety (hard) 0**.
|
| 25 |
+
- Training splits of a few Decision Index benchmarks where the base trails (a home-appliance simulator, iSarcasmEval,
|
| 26 |
+
API-Bank, ContractNLI), about 16.5k rows.
|
| 27 |
+
- Uniform mixing: no per-task sampling weight, no loss re-weighting.
|
| 28 |
+
- **Cost:** about 2.5 GPU-hours of training on one RTX PRO 6000 (≈ $2.4 at the GPU-hour price), about $0.8 for the 13
|
| 29 |
+
evaluation reads.
|
| 30 |
+
- **Evaluation:** each task's **fixed test items**, a temperature fitted on that task's calibration split, paired on the
|
| 31 |
+
same items (1,000 item resamples; SGD intent 2,000) against:
|
| 32 |
+
- **that task's own LoRA** (all labels). **Caveat: every own LoRA sits on Wald-4B v0.9**, a different build, so this
|
| 33 |
+
comparison is paired on items, not on base;
|
| 34 |
+
- **Jev** (same request bytes);
|
| 35 |
+
- **the same base zero-shot**, where a hosted read exists (a different reader than ours).
|
| 36 |
+
|
| 37 |
+
## 3. Results
|
| 38 |
+
|
| 39 |
+
| Task | Metric | one LoRA | own LoRA (v0.9) | Jev | base zero-shot | one − own | one − Jev | one − zero-shot |
|
| 40 |
+
|---|---|---:|---:|---:|---:|---|---|---|
|
| 41 |
+
| When2Call | accuracy | 82.4 | 82.8 | 72.6 | 76.6 | −0.4 [−2.8, +2.2] | **+9.8** [+6.2, +13.2] | +5.8 [+2.2, +9.2] |
|
| 42 |
+
| BANKING77 | macro-F1 | 88.2 | 92.8 | 78.1 | — | −4.6 [−7.2, −2.6] | **+10.1** [+7.3, +14.1] | — |
|
| 43 |
+
| SGD intent | macro-F1 | 96.6 | 98.0 | 92.8 | — | −1.4 [−2.7, −0.2] | **+3.7** [+1.5, +6.6] | — |
|
| 44 |
+
| AndroidControl | accuracy | 81.8 | 85.8 | 73.4 | — | −4.0 [−6.8, −1.0] | **+8.4** [+4.4, +12.4] | — |
|
| 45 |
+
| MetaTool | accuracy | 90.2 | 97.0 | 83.0 | 81.4 | −6.8 [−9.2, −4.4] | **+7.2** [+4.4, +10.4] | +8.8 [+6.2, +11.6] |
|
| 46 |
+
| Agent-trajectory safety (hard) | unsafe F1 | 75.7 | 72.2 | 47.0 | 29.3 | **+3.5** [+1.8, +5.2] | **+28.7** [+21.6, +36.3] | +46.5 [+39.5, +53.7] |
|
| 47 |
+
| Agent-trajectory safety | unsafe F1 | 98.0 | — | 92.2 | 74.8 | — | **+5.8** [+3.2, +8.7] | +23.2 [+18.7, +28.1] |
|
| 48 |
+
| Mind2Web | accuracy | 51.2 | 51.9 | 48.5 | — | −0.6 [−5.0, +3.3] | +2.7 [−1.5, +6.9] | — |
|
| 49 |
+
| RewardBench | accuracy | 82.8 | 83.4 | 90.6 | 83.6 | −0.6 [−2.5, +1.1] | −7.8 [−9.9, −5.7] | −0.8 [−2.4, +0.8] |
|
| 50 |
+
| Prompt injection | injection F1 | 75.0 | 79.4 | 82.8 | 72.7 | −4.4 [−7.9, −1.4] | −7.8 [−14.1, −0.9] | +2.3 [−1.8, +6.4] |
|
| 51 |
+
| ToxicChat | toxic F1 | 73.8 | 84.9 | 80.4 | 71.8 | **−11.1** [−14.5, −7.9] | −6.6 [−10.0, −3.4] | +2.0 [−0.6, +4.6] |
|
| 52 |
+
| Model routing | needs-strong F1 | 1.9 | 26.8 | 0.0 | 10.4 | **−24.9** [−34.0, −14.7] | +1.9 [+0.0, +6.2] | −8.5 [−16.9, −0.6] |
|
| 53 |
+
| Model routing (difficulty framing) | hard F1 | 2.0 | — | — | 16.2 | — | — | −14.2 [−22.7, −6.0] |
|
| 54 |
+
|
| 55 |
+
## 4. Three groups
|
| 56 |
+
|
| 57 |
+
**One adapter is enough.**
|
| 58 |
+
- When2Call (−0.4), Mind2Web (−0.6) and RewardBench (−0.6) are level with their own LoRA; SGD intent is 1.4 behind.
|
| 59 |
+
- Agent-trajectory safety (hard) is **+3.5 ahead** of its own LoRA with **zero** rows of its own task in the mix: the gain is
|
| 60 |
+
transfer from the other agent and guard tasks. Its own LoRA had only 250 training rows and was often confidently wrong,
|
| 61 |
+
so the bar was low.
|
| 62 |
+
|
| 63 |
+
**A small cost (−4 to −7).**
|
| 64 |
+
- BANKING77 −4.6, AndroidControl −4.0, MetaTool −6.8, prompt injection −4.4. All but prompt injection still beat Jev.
|
| 65 |
+
- The first three had 2,500 rows in the mix vs 8,000–30,000 for their own LoRA, so part of the gap is data volume (the own
|
| 66 |
+
LoRAs gain about one point per doubling of labels), not interference.
|
| 67 |
+
|
| 68 |
+
**It breaks.**
|
| 69 |
+
- **ToxicChat −11.1** vs its own LoRA, back to the zero-shot level. 2,500 rows at about 7 % toxic means about 175 positives
|
| 70 |
+
among 37k labelled rows: the rare-positive signal is drowned.
|
| 71 |
+
- **Model routing collapses to "never route"** (F1 1.9 vs 26.8; 2.0 for the second framing). Routing is a probability-
|
| 72 |
+
ranking task: its value is the ordering of P(needs a strong model) across prompts, and the mixed model puts nearly all mass
|
| 73 |
+
on the majority option. It also had only 1,174 rows after near-duplicate removal (own LoRA: 8,000).
|
| 74 |
+
|
| 75 |
+
## 5. Why (hypotheses, not tested)
|
| 76 |
+
|
| 77 |
+
1. **Class imbalance under uniform mixing.** Rare-positive tasks (ToxicChat 7 %, routing 20 %) contribute few positive
|
| 78 |
+
gradients in a 37k-row mix, so the model learns their majority option.
|
| 79 |
+
2. **Ranking / threshold tasks lose their calibrated margin.** The replay anchor and other tasks' confident targets sharpen
|
| 80 |
+
the answer head; routing needs small, well-ordered probability differences that argmax F1 at the default threshold misses.
|
| 81 |
+
3. **Label-space interference.** The letter readout shares (A) / (B) across binary tasks with different meanings and base
|
| 82 |
+
rates, which can pull against each other.
|
| 83 |
+
4. **Data volume and isolation losses.** The per-task cap (2,500) and the cross-task near-duplicate removal (routing
|
| 84 |
+
8,000 → 1,174; hard agent-safety 250 → 0) mean the mix is not the data the dedicated LoRAs saw.
|
| 85 |
+
5. **Base difference.** The one LoRA and the dedicated LoRAs sit on different builds. On one browser task a base swap alone
|
| 86 |
+
changed nothing (−0.4 [−2.9, +2.1]), so this likely explains little, but it is not ruled out per task.
|
| 87 |
+
|
| 88 |
+
## 6. What to try next
|
| 89 |
+
|
| 90 |
+
| Idea | What | Rough cost |
|
| 91 |
+
|---|---|---|
|
| 92 |
+
| Per-task sampling weights | sample tasks ∝ n^α (α ≈ 0.3–0.5), or up-weight rare-positive and ranking tasks | ≈ $2.5 + $0.8 reads |
|
| 93 |
+
| Task-balanced or focal loss | class-balanced weights within each binary task, or focal loss on rare-positive tasks | ≈ $2.5 + $0.8 |
|
| 94 |
+
| Hybrid adapters | the shared adapter plus small separate adapters for ToxicChat, prompt injection and routing | ≈ $0.2–1.5 per extra adapter; multi-LoRA serving |
|
| 95 |
+
| Per-task calibration | per-task temperature and decision threshold on calibration data (routing: rank-based, not argmax) | ≈ $0 (offline) |
|
| 96 |
+
| Same-base comparison | dedicated and shared LoRAs on the same build | ≈ $2.5 |
|
| 97 |
+
| Lighter isolation | remove near-duplicates only against each task's own evaluation items | $0 data + one re-train |
|
| 98 |
+
|
| 99 |
+
## 7. Product takeaway
|
| 100 |
+
|
| 101 |
+
One adapter covers most intent and agent-decision tasks at a 1–5 point cost against dedicated adapters, and still beats Jev
|
| 102 |
+
on 7 of 12 tasks. Imbalanced guard tasks (ToxicChat) and probability-ranking tasks (routing) still need their own adapter,
|
| 103 |
+
or a hybrid of one shared adapter plus small per-task ones.
|
| 104 |
+
|
| 105 |
+
## Caveats
|
| 106 |
+
|
| 107 |
+
- For 7 tasks the Jev numbers are Jev's reads from the dedicated-LoRA evaluations on the same item ids, not a fresh read.
|
| 108 |
+
- Mind2Web was read with plain knockout, because the adapter was trained on knockout-shaped questions; the knockout +
|
| 109 |
+
top-10 re-read gives 49.0.
|
| 110 |
+
- The dedicated LoRAs are on Wald-4B v0.9: paired on items, not on base.
|
| 111 |
+
- The base zero-shot column comes from a hosted read, not our evaluation reader.
|
| 112 |
+
- One seed, one epoch, one mix; no sampling-weight search.
|
docs/readmes/README.zh.md
ADDED
|
@@ -0,0 +1,173 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
<div align="center">
|
| 2 |
+
<h1>Wald-4B v1.0</h1>
|
| 3 |
+
<p><strong>经过校准的决策模型:一次前向,把状态和问题变成每个选项的概率。</strong></p>
|
| 4 |
+
</div>
|
| 5 |
+
|
| 6 |
+
**W**ait **A** bit, **L**ook, then **D**ecide(等一下,看一眼,再决定):有把握,立刻决定;没把握,再看一眼。名字也取自 Abraham Wald(1902–1950),序贯分析的创立者:证据够了就停下来做决定。
|
| 7 |
+
|
| 8 |
+
---
|
| 9 |
+
|
| 10 |
+
<p align="center"><a href="https://huggingface.co/Harry19081/Wald-4B">English</a> · <a href="https://huggingface.co/Harry19081/Wald-4B/blob/main/docs/readmes/README.zh.md">简体中文</a></p>
|
| 11 |
+
|
| 12 |
+
---
|
| 13 |
+
|
| 14 |
+
**Wald-4B 把一次决策变成一次前向。** 给它一段状态和一个问题,它为每个选项返回一个校准过的概率:调哪个工具、要不要问用户、
|
| 15 |
+
属于哪个意图、输出是否安全。概率高于 0.9 直接执行,低于 0.6 交给人,中间的交给更大的模型。在 `medium` 档位下,只有最高概率
|
| 16 |
+
低于 0.7 时它才先想一想(不超过 512 token),大约九题一次;输出始终是概率分布,不解析文本。
|
| 17 |
+
|
| 18 |
+
## 一览
|
| 19 |
+
|
| 20 |
+
- **Decision Index 0.2.1:53.91**(样本估计),67 条榜单中排第 6,4B 中第一(见下文“基准测试”)。
|
| 21 |
+
- **XL 隐藏集:63.2**,第 2,仅次于 Jev,4B 中第一。
|
| 22 |
+
- **你的任务,不到 $2**:每个任务一个 LoRA,12 个公开任务中 9 个超过 Jev。
|
| 23 |
+
- **每次决策约 50 ms**(p50),每千次决策的价格约为 Jev 的 1/5。
|
| 24 |
+
|
| 25 |
+

|
| 26 |
+
|
| 27 |
+
## 工作原理
|
| 28 |
+
|
| 29 |
+
基于 Qwen/Qwen3.5-4B-Base @ `1001bb4d826a52d1f399e183466143f4da7b741b` 的 4B 模型:先做全参数决策训练,再合并三个 LoRA
|
| 30 |
+
(带 KL 回放的精调、短思考蒸馏、任务覆盖)。每个问题变成一条纯文本提示,不用对话模板,不加新 token。状态写两遍(格式
|
| 31 |
+
`repeat_state_plain`;第二遍由一句固定的话引出,见 NOTICE),然后是:
|
| 32 |
+
|
| 33 |
+
```
|
| 34 |
+
Question: {instructions}
|
| 35 |
+
(A) {option 1}
|
| 36 |
+
(B) {option 2}
|
| 37 |
+
Answer: (
|
| 38 |
+
```
|
| 39 |
+
|
| 40 |
+
最后一个位置上选项字母的 logits 给出分布。超过 26 个选项时,按 ⌈n / 26⌉ 块分别读,再把各块的胜者读一次(淘汰赛)。
|
| 41 |
+
每个(题型、选项数)一个温度,在我们自己的留出数据上拟合,只改变置信度。提示格式、策略和上下文上限都固定在 `serving.json`
|
| 42 |
+
里,所有基准一律相同。
|
| 43 |
+
|
| 44 |
+
**训练数据**(不公开):四个阶段,用到 84 个公开来源(83 个数据集,只读训练切分,NLI4CT 另读了 dev 切分;外加 Decision Index 工具包里 MIT 许可的
|
| 45 |
+
Home-appliance 生成器)、由代码生成并由代码标注的决策题,以及对早先数据的 KL 回放。其中一部分是合成数据:题目和短思考
|
| 46 |
+
由大得多的前沿大模型编写或标注——凡公开了参数量的都远超 120B。全部来源及 Decision Index 题目的筛查方法见
|
| 47 |
+
[CONTAMINATION.md](https://huggingface.co/Harry19081/Wald-4B/blob/main/CONTAMINATION.md)。
|
| 48 |
+
|
| 49 |
+
## 基准测试
|
| 50 |
+
|
| 51 |
+
我们测了四项:Decision Index 0.2.1(榜单)、XL(带隐藏集的决策基准)、公开垂直任务上的 LoRA,以及服务延迟。除非标注
|
| 52 |
+
“榜单”,所有数字都是我们自己的读数,默认零样本,所有系统收到相同的请求。
|
| 53 |
+
|
| 54 |
+
### Decision Index 0.2.1
|
| 55 |
+
|
| 56 |
+
我们读的是 **6,948 个请求的分层样本**(37,469 个问题;用工具包自带的采样器和种子;未重建 HLE),按 0.2.1 规则计分。
|
| 57 |
+
Jev 在这个样本上是 57.19,榜单全量是 57.89,所以样本大约偏低 0.7 分。榜单行是官方全量数字
|
| 58 |
+
([榜单](https://huggingface.co/spaces/multimodalart/jev-decision-index))。
|
| 59 |
+
|
| 60 |
+
| 系统 | 规模 | Decision Index 0.2.1 | 来源 |
|
| 61 |
+
|---|---|---|---|
|
| 62 |
+
| Jev(托管 API,jev-1.13.0) | 未公开 | 57.19 [55.29, 58.60] · 榜单 57.89 | 我们的样本 · 榜单 |
|
| 63 |
+
| simple-jev · Qwen3.8-27B(榜单第 4)· Jebadiah 27B(第 5) | 27B | 55.74 · 54.67 | 榜单 |
|
| 64 |
+
| **Wald-4B v1.0 · medium** | 4B | **53.91** [51.87, 55.27] | 我们的样本 |
|
| 65 |
+
| reflex 27B(榜单第 6)· Decider chat · Qwen3.6-27B(第 7) | 27B | 52.16 · 51.35 | 榜单 |
|
| 66 |
+
| Qwen3.8-27B,未训练,用我们的字母读出(一遍) | 27B | 47.78 [45.95, 49.17] | 我们的样本 |
|
| 67 |
+
| Decider 35B-A3B(榜单第 11) | 35B-A3B | 47.11 | 榜单 |
|
| 68 |
+
| Decider 4B(榜单第 15) | 4B | 40.70 | 榜单 |
|
| 69 |
+
| Kev 9B(榜单第 23)· Kev 4B(第 28) | 9B · 4B | 38.48 · 34.64 | 榜单 |
|
| 70 |
+
|
| 71 |
+
53.91 在 67 条榜单中排第 6(在 Jebadiah 27B 与 reflex 27B 之间);在同一样本上比 Jev 低 3.3 [−4.9, −1.8]。
|
| 72 |
+
同样是 4B 底座,比 Kev 4B 高 19.3。
|
| 73 |
+
|
| 74 |
+

|
| 75 |
+
|
| 76 |
+
**披露。** 有 363 个 Decision Index 题目 id 通过公开训练切分或共同的上游来源进入了我们的训练数据(从未用过测试切分);
|
| 77 |
+
��它们全部记错,得分为 **53.62**。Home appliance 得 1.00(skill):最后一个阶段用该基准自己的 MIT 生成器造了新的家庭数据来训练
|
| 78 |
+
(新种子,与测试数据没有共享状态);把 Home appliance 换回父版本的答案,指数为 51.57 [49.56, 52.95]。详见
|
| 79 |
+
[CONTAMINATION.md](https://huggingface.co/Harry19081/Wald-4B/blob/main/CONTAMINATION.md)。
|
| 80 |
+
|
| 81 |
+
**同请求套件。** JevBench public 231(准确率 %):Wald-4B v1.0 **88.3**(打包后的服务),Jev 86.6,未训练的
|
| 82 |
+
Qwen3.5-4B-Base 67.5,Laya 421M 58.0,CLM-8B 39.0。CLM-8B 和 Laya 没有 Decision Index 读数。
|
| 83 |
+
|
| 84 |
+
### XL(隐藏集)
|
| 85 |
+
|
| 86 |
+
我们自己构建的决策基准(13 个题族;按 1,895 题的隐藏集排名,指标为机会校正后的综合分 XL-Int)。我们是它的作者,
|
| 87 |
+
所以没有一个数字是独立的;我们的模型在公开集上的得分远高于隐藏集,因此只报告隐藏集。
|
| 88 |
+
|
| 89 |
+
| 系统 | XL-Int(隐藏集) | 名次 |
|
| 90 |
+
|---|---:|---:|
|
| 91 |
+
| Jev(API,v1.13) | 73.0 [69.8, 75.9] | 第 1 |
|
| 92 |
+
| **Wald-4B v1.0 · medium** | **63.2** [60.0, 66.4] | **第 2**,4B 中第一 |
|
| 93 |
+
| Cygnet(gemma-4-12B-it + shim) | 59.8 [56.7, 62.7] | 第 4 |
|
| 94 |
+
| Laya 421M · CLM-8B | 13.4 · 12.7 | 第 24 · 第 25 |
|
| 95 |
+
|
| 96 |
+
### 垂直任务:每个任务一个快速 LoRA
|
| 97 |
+
|
| 98 |
+

|
| 99 |
+
|
| 100 |
+
用任务标签训练一个 LoRA,花费 $0.12–$1.81 的 GPU 时间(不到 2 GPU 小时);其他系统都是零样本,用同一批固定测试题和逐字节
|
| 101 |
+
相同的请求。与 Jev 的差值为配对 bootstrap 95 % 置信区间;▲ 表示区间不含 0。这些读数来自预发布版本 v0.9(见“版本”)。
|
| 102 |
+
|
| 103 |
+
| 任务(指标) | Wald-4B v0.9 + LoRA | Jev |
|
| 104 |
+
|---|---:|---:|
|
| 105 |
+
| MetaTool(工具选择,准确率) | **97.0** | 83.0 |
|
| 106 |
+
| BANKING77(77 个意图,宏 F1) | **92.8** | 78.1 |
|
| 107 |
+
| AndroidControl(手机智能体动作,准确率) | **85.8**¹ | 73.4 |
|
| 108 |
+
| ToxicChat(有害类 F1) | **84.9** | 80.4 |
|
| 109 |
+
| COLD,中文冒犯语言(宏 F1,完整 5,323 题测试集) | **84.1** [83.2, 85.1] | 75.3 |
|
| 110 |
+
| When2Call(调用 / 追问 / 拒绝,准确率) | **83.0**² | 72.6 |
|
| 111 |
+
|
| 112 |
+
¹ 用全部训练步训练的臂;图中是第一个全标签臂(81.6)。² 8 个 LoRA 配置中在测试集上最好的一个;按 dev 选出的配置为 82.6。
|
| 113 |
+
COLD:配对差 +8.8 [7.7, 10.0];在未用于温度拟合的 4,823 题上为 83.8;与已发表的最好结果 83.7 持平。
|
| 114 |
+
|
| 115 |
+
#### 标签很少时,从 Wald 起步胜过在原始底座上训 LoRA
|
| 116 |
+
|
| 117 |
+
相同的 LoRA 配方、标签和测试题;表中为 Wald-4B(v0.9)减去原始 Qwen3.5-4B-Base 的分差,配对 95 % 置信区间。
|
| 118 |
+
**加粗**表示区间不含 0。
|
| 119 |
+
|
| 120 |
+
| 任务(指标) | 0 个标签(零样本) | 300 个标签 |
|
| 121 |
+
|---|---:|---:|
|
| 122 |
+
| When2Call(准确率) | **+21.2** [17.0, 25.2] | +2.0 [−0.2, 4.2] |
|
| 123 |
+
| BANKING77(宏 F1) | **+10.9** [7.6, 14.7] | **+4.4** [2.0, 7.2] |
|
| 124 |
+
| SGD intent(宏 F1) | +2.8(不显著) | −0.1 |
|
| 125 |
+
|
| 126 |
+
### 延迟与成本
|
| 127 |
+
|
| 128 |
+

|
| 129 |
+
|
| 130 |
+
单个问题、一遍、串行:一张 RTX PRO 6000 上 p50 为 26 ms(H100 上 51 ms)。批处理:115 次决策/秒(fp8),按 GPU 小时价格
|
| 131 |
+
约 $0.007 / 千次决策;Jev API 标价 $0.04。`medium` 下中位延迟与一遍相同,p95 约 1 秒。测于 v0.9,架构与服务路径相同。
|
| 132 |
+
|
| 133 |
+
## 训练你自己的任务 LoRA(CLI,即将发布)
|
| 134 |
+
|
| 135 |
+
上面每个垂直任务都是用同一个 CLI 做的,我们计划很快发布。它用固定种子切分带标签的数据集,检查测试题泄漏,在你选的 GPU 上
|
| 136 |
+
基于 Wald-4B 训练 LoRA,用你的 LoRA、Jev 和零样本基线读同一批测试题,并把适配器放在同样的 `/v1/systemone` API 后面提供服务。
|
| 137 |
+
|
| 138 |
+

|
| 139 |
+
|
| 140 |
+
阶段、防护、成本和命令预览:[docs/lora-cli.md](https://huggingface.co/Harry19081/Wald-4B/blob/main/docs/lora-cli.md)(英文)。
|
| 141 |
+
|
| 142 |
+
## 服务
|
| 143 |
+
|
| 144 |
+
```sh
|
| 145 |
+
docker build -t wald-serve .
|
| 146 |
+
docker run --gpus all -v /path/to/Wald-4B:/model:ro -p 8000:8000 wald-serve # GET /health 返回 {"ok": true} 即就绪
|
| 147 |
+
```
|
| 148 |
+
|
| 149 |
+
不用 Docker:`./run.sh /path/to/Wald-4B`。服务端接受 `--effort`(`none | low | medium | high | high-k<k>`),请求里也可以带
|
| 150 |
+
`"effort"` 覆盖。声明配置:effort `medium`,提示格式 `repeat_state_plain`,每个问题提示最多 131,072 token(更长返回 HTTP 422,
|
| 151 |
+
从不截断),1–255 个选项。命令、上限和运行时间见 [RUNBOOK.md](https://huggingface.co/Harry19081/Wald-4B/blob/main/RUNBOOK.md)(英文)。权重为 bf16,
|
| 152 |
+
`model.safetensors` 的 sha256 为 `dabeb7bc3f43bf6ea7d20ecfdac8758b3945517a991809a5772988829598e893`;每个文件的 sha256 见 `MANIFEST.json`。
|
| 153 |
+
|
| 154 |
+
## 局限
|
| 155 |
+
|
| 156 |
+
- 对父版本的不退步门槛没有失败项。���项下降超出容差,但多重校正后不显著:XL long_policy −6.7 [−11.6, −2.0],
|
| 157 |
+
JevBench public 231 −2.6 [−5.2, 0.0](plain 提示的读数)。
|
| 158 |
+
- 温度表沿用父版本的,也就是 Decision Index 实测读数所用的那张。
|
| 159 |
+
- 我们的 Decision Index 读数用的是 16,384 token 上下文;超出的提示当时记为不支持,打包后的服务会作答。
|
| 160 |
+
|
| 161 |
+
## 版本
|
| 162 |
+
|
| 163 |
+
| 版本 | 状态 | 检查点 | 底座 | 日期 | 策略 | 提示格式 |
|
| 164 |
+
|---|---|---|---|---|---|---|
|
| 165 |
+
| **v1.0** | 当前版本 | 内部版本 021A0-f10 | Qwen/Qwen3.5-4B-Base | 2026-09-27 | effort `medium` | `repeat_state_plain` |
|
| 166 |
+
| v0.9 | 预发布;上文的垂直任务 LoRA 和延迟 | 内部版本 015D0-f4 | Qwen/Qwen3.5-4B-Base | 2026-09-26 | effort `medium` | plain |
|
| 167 |
+
|
| 168 |
+
v1.x = 同一底座代上的服务或 LoRA 更新;v2.0 = 新的底座代。
|
| 169 |
+
|
| 170 |
+
---
|
| 171 |
+
|
| 172 |
+
Apache-2.0(权重和代码),同时受底座模型 Qwen/Qwen3.5-4B-Base 许可约束(Apache-2.0,© 阿里云通义千问团队)。第三方声明:
|
| 173 |
+
[NOTICE](https://huggingface.co/Harry19081/Wald-4B/blob/main/NOTICE)。
|
figures/data.json
ADDED
|
@@ -0,0 +1,1584 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"board": {
|
| 3 |
+
"edition": "release-v2.1",
|
| 4 |
+
"generated_utc": "2026-09-27T03:42:13+00:00",
|
| 5 |
+
"jev": {
|
| 6 |
+
"index": 57.91,
|
| 7 |
+
"areas": {
|
| 8 |
+
"knowledge": 0.514,
|
| 9 |
+
"language": 0.6202,
|
| 10 |
+
"retrieval": 0.5542,
|
| 11 |
+
"tools": 0.7509,
|
| 12 |
+
"arts": 0.3766
|
| 13 |
+
}
|
| 14 |
+
},
|
| 15 |
+
"entries": [
|
| 16 |
+
{
|
| 17 |
+
"rank": 1,
|
| 18 |
+
"name": "Surogate Rune 26B-A4B v3",
|
| 19 |
+
"size_b": 26.0,
|
| 20 |
+
"base_model": "google/gemma-4-26B-A4B-it",
|
| 21 |
+
"kind": "full fine-tune",
|
| 22 |
+
"index": 57.44,
|
| 23 |
+
"areas": {
|
| 24 |
+
"knowledge": 0.4336,
|
| 25 |
+
"language": 0.6306,
|
| 26 |
+
"retrieval": 0.6352,
|
| 27 |
+
"tools": 0.7122,
|
| 28 |
+
"arts": 0.419
|
| 29 |
+
}
|
| 30 |
+
},
|
| 31 |
+
{
|
| 32 |
+
"rank": 2,
|
| 33 |
+
"name": "Decider chat \u00b7 Gemma-4-31B",
|
| 34 |
+
"size_b": 31.0,
|
| 35 |
+
"base_model": "google/gemma-4-31B",
|
| 36 |
+
"kind": "inference technique",
|
| 37 |
+
"index": 57.33,
|
| 38 |
+
"areas": {
|
| 39 |
+
"knowledge": 0.4427,
|
| 40 |
+
"language": 0.6038,
|
| 41 |
+
"retrieval": 0.6308,
|
| 42 |
+
"tools": 0.7557,
|
| 43 |
+
"arts": 0.3834
|
| 44 |
+
}
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"rank": 3,
|
| 48 |
+
"name": "AutoJev-27B",
|
| 49 |
+
"size_b": 27.0,
|
| 50 |
+
"base_model": "Qwen/Qwen3.8-27B",
|
| 51 |
+
"kind": "full fine-tune",
|
| 52 |
+
"index": 56.4,
|
| 53 |
+
"areas": {
|
| 54 |
+
"knowledge": 0.4088,
|
| 55 |
+
"language": 0.6346,
|
| 56 |
+
"retrieval": 0.5486,
|
| 57 |
+
"tools": 0.7935,
|
| 58 |
+
"arts": 0.3939
|
| 59 |
+
}
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"rank": 4,
|
| 63 |
+
"name": "simple-jev \u00b7 Qwen3.8-27B (featherless)",
|
| 64 |
+
"size_b": 27.0,
|
| 65 |
+
"base_model": "Qwen/Qwen3.8-27B",
|
| 66 |
+
"kind": "inference technique",
|
| 67 |
+
"index": 55.74,
|
| 68 |
+
"areas": {
|
| 69 |
+
"knowledge": 0.3656,
|
| 70 |
+
"language": 0.6209,
|
| 71 |
+
"retrieval": 0.6325,
|
| 72 |
+
"tools": 0.7621,
|
| 73 |
+
"arts": 0.3648
|
| 74 |
+
}
|
| 75 |
+
},
|
| 76 |
+
{
|
| 77 |
+
"rank": 5,
|
| 78 |
+
"name": "Jebadiah 27B",
|
| 79 |
+
"size_b": 27.0,
|
| 80 |
+
"base_model": "Qwen/Qwen3.8-27B",
|
| 81 |
+
"kind": "LoRA",
|
| 82 |
+
"index": 54.67,
|
| 83 |
+
"areas": {
|
| 84 |
+
"knowledge": 0.3884,
|
| 85 |
+
"language": 0.6068,
|
| 86 |
+
"retrieval": 0.5389,
|
| 87 |
+
"tools": 0.7813,
|
| 88 |
+
"arts": 0.3873
|
| 89 |
+
}
|
| 90 |
+
},
|
| 91 |
+
{
|
| 92 |
+
"rank": 6,
|
| 93 |
+
"name": "reflex 27B",
|
| 94 |
+
"size_b": 27.0,
|
| 95 |
+
"base_model": "Qwen/Qwen3.8-27B",
|
| 96 |
+
"kind": "inference technique",
|
| 97 |
+
"index": 52.16,
|
| 98 |
+
"areas": {
|
| 99 |
+
"knowledge": 0.3511,
|
| 100 |
+
"language": 0.5416,
|
| 101 |
+
"retrieval": 0.5781,
|
| 102 |
+
"tools": 0.7412,
|
| 103 |
+
"arts": 0.3965
|
| 104 |
+
}
|
| 105 |
+
},
|
| 106 |
+
{
|
| 107 |
+
"rank": 7,
|
| 108 |
+
"name": "Decider chat \u00b7 Qwen3.6-27B",
|
| 109 |
+
"size_b": 27.0,
|
| 110 |
+
"base_model": "Qwen/Qwen3.6-27B",
|
| 111 |
+
"kind": "inference technique",
|
| 112 |
+
"index": 51.35,
|
| 113 |
+
"areas": {
|
| 114 |
+
"knowledge": 0.3699,
|
| 115 |
+
"language": 0.5712,
|
| 116 |
+
"retrieval": 0.5223,
|
| 117 |
+
"tools": 0.7142,
|
| 118 |
+
"arts": 0.3506
|
| 119 |
+
}
|
| 120 |
+
},
|
| 121 |
+
{
|
| 122 |
+
"rank": 8,
|
| 123 |
+
"name": "Winnow-12B",
|
| 124 |
+
"size_b": 12.0,
|
| 125 |
+
"base_model": "google/gemma-4-12B",
|
| 126 |
+
"kind": "LoRA",
|
| 127 |
+
"index": 50.02,
|
| 128 |
+
"areas": {
|
| 129 |
+
"knowledge": 0.3382,
|
| 130 |
+
"language": 0.5599,
|
| 131 |
+
"retrieval": 0.5405,
|
| 132 |
+
"tools": 0.7102,
|
| 133 |
+
"arts": 0.2997
|
| 134 |
+
}
|
| 135 |
+
},
|
| 136 |
+
{
|
| 137 |
+
"rank": 9,
|
| 138 |
+
"name": "JoshuaSP diffusiongemma (open-jev)",
|
| 139 |
+
"size_b": 26.0,
|
| 140 |
+
"base_model": "google/diffusiongemma-26B-A4B-it",
|
| 141 |
+
"kind": "inference technique",
|
| 142 |
+
"index": 49.47,
|
| 143 |
+
"areas": {
|
| 144 |
+
"knowledge": 0.3272,
|
| 145 |
+
"language": 0.5351,
|
| 146 |
+
"retrieval": 0.5837,
|
| 147 |
+
"tools": 0.7016,
|
| 148 |
+
"arts": 0.2665
|
| 149 |
+
}
|
| 150 |
+
},
|
| 151 |
+
{
|
| 152 |
+
"rank": 10,
|
| 153 |
+
"name": "Jevfire",
|
| 154 |
+
"size_b": 27.0,
|
| 155 |
+
"base_model": "Qwen/Qwen3.8-27B",
|
| 156 |
+
"kind": "inference technique",
|
| 157 |
+
"index": 49.37,
|
| 158 |
+
"areas": {
|
| 159 |
+
"knowledge": 0.3048,
|
| 160 |
+
"language": 0.5334,
|
| 161 |
+
"retrieval": 0.5622,
|
| 162 |
+
"tools": 0.7233,
|
| 163 |
+
"arts": 0.3221
|
| 164 |
+
}
|
| 165 |
+
},
|
| 166 |
+
{
|
| 167 |
+
"rank": 11,
|
| 168 |
+
"name": "Decider 35B-A3B",
|
| 169 |
+
"size_b": 35.0,
|
| 170 |
+
"base_model": "Qwen/Qwen3.5-35B-A3B-Base",
|
| 171 |
+
"kind": "full fine-tune",
|
| 172 |
+
"index": 47.11,
|
| 173 |
+
"areas": {
|
| 174 |
+
"knowledge": 0.3178,
|
| 175 |
+
"language": 0.5552,
|
| 176 |
+
"retrieval": 0.5466,
|
| 177 |
+
"tools": 0.5654,
|
| 178 |
+
"arts": 0.326
|
| 179 |
+
}
|
| 180 |
+
},
|
| 181 |
+
{
|
| 182 |
+
"rank": 12,
|
| 183 |
+
"name": "JPT-9B",
|
| 184 |
+
"size_b": 9.0,
|
| 185 |
+
"base_model": "Qwen/Qwen3.5-9B-Base",
|
| 186 |
+
"kind": "LoRA",
|
| 187 |
+
"index": 46.89,
|
| 188 |
+
"areas": {
|
| 189 |
+
"knowledge": 0.317,
|
| 190 |
+
"language": 0.5672,
|
| 191 |
+
"retrieval": 0.4459,
|
| 192 |
+
"tools": 0.6702,
|
| 193 |
+
"arts": 0.2856
|
| 194 |
+
}
|
| 195 |
+
},
|
| 196 |
+
{
|
| 197 |
+
"rank": 13,
|
| 198 |
+
"name": "Decision 1.0 Lux",
|
| 199 |
+
"size_b": 9.0,
|
| 200 |
+
"base_model": "Qwen/Qwen3.5-9B-Base",
|
| 201 |
+
"kind": "head / adapter",
|
| 202 |
+
"index": 43.49,
|
| 203 |
+
"areas": {
|
| 204 |
+
"knowledge": 0.309,
|
| 205 |
+
"language": 0.4797,
|
| 206 |
+
"retrieval": 0.5002,
|
| 207 |
+
"tools": 0.5719,
|
| 208 |
+
"arts": 0.2635
|
| 209 |
+
}
|
| 210 |
+
},
|
| 211 |
+
{
|
| 212 |
+
"rank": 14,
|
| 213 |
+
"name": "Xor",
|
| 214 |
+
"size_b": 35.0,
|
| 215 |
+
"base_model": "Qwen/Qwen3.6-35B-A3B",
|
| 216 |
+
"kind": "full fine-tune",
|
| 217 |
+
"index": 41.48,
|
| 218 |
+
"areas": {
|
| 219 |
+
"knowledge": 0.3433,
|
| 220 |
+
"language": 0.5558,
|
| 221 |
+
"retrieval": 0.2296,
|
| 222 |
+
"tools": 0.5514,
|
| 223 |
+
"arts": 0.3563
|
| 224 |
+
}
|
| 225 |
+
},
|
| 226 |
+
{
|
| 227 |
+
"rank": 15,
|
| 228 |
+
"name": "Decider 4B",
|
| 229 |
+
"size_b": 4.0,
|
| 230 |
+
"base_model": "Qwen/Qwen3.5-4B-Base",
|
| 231 |
+
"kind": "full fine-tune",
|
| 232 |
+
"index": 40.7,
|
| 233 |
+
"areas": {
|
| 234 |
+
"knowledge": 0.2574,
|
| 235 |
+
"language": 0.4597,
|
| 236 |
+
"retrieval": 0.4467,
|
| 237 |
+
"tools": 0.5862,
|
| 238 |
+
"arts": 0.2503
|
| 239 |
+
}
|
| 240 |
+
},
|
| 241 |
+
{
|
| 242 |
+
"rank": 16,
|
| 243 |
+
"name": "Jev-Omni",
|
| 244 |
+
"size_b": 12.0,
|
| 245 |
+
"base_model": "google/gemma-4-12B",
|
| 246 |
+
"kind": "LoRA + head",
|
| 247 |
+
"index": 40.53,
|
| 248 |
+
"areas": {
|
| 249 |
+
"knowledge": 0.2549,
|
| 250 |
+
"language": 0.5052,
|
| 251 |
+
"retrieval": 0.3583,
|
| 252 |
+
"tools": 0.6077,
|
| 253 |
+
"arts": 0.2607
|
| 254 |
+
}
|
| 255 |
+
},
|
| 256 |
+
{
|
| 257 |
+
"rank": 17,
|
| 258 |
+
"name": "djev",
|
| 259 |
+
"size_b": 26.0,
|
| 260 |
+
"base_model": "google/diffusiongemma-26B-A4B-it",
|
| 261 |
+
"kind": "inference technique",
|
| 262 |
+
"index": 40.28,
|
| 263 |
+
"areas": {
|
| 264 |
+
"knowledge": 0.264,
|
| 265 |
+
"language": 0.4876,
|
| 266 |
+
"retrieval": 0.4177,
|
| 267 |
+
"tools": 0.5168,
|
| 268 |
+
"arts": 0.3045
|
| 269 |
+
}
|
| 270 |
+
},
|
| 271 |
+
{
|
| 272 |
+
"rank": 18,
|
| 273 |
+
"name": "Winnow-E4B",
|
| 274 |
+
"size_b": 4.0,
|
| 275 |
+
"base_model": "google/gemma-4-E4B",
|
| 276 |
+
"kind": "LoRA",
|
| 277 |
+
"index": 39.89,
|
| 278 |
+
"areas": {
|
| 279 |
+
"knowledge": 0.2229,
|
| 280 |
+
"language": 0.4511,
|
| 281 |
+
"retrieval": 0.438,
|
| 282 |
+
"tools": 0.6251,
|
| 283 |
+
"arts": 0.2277
|
| 284 |
+
}
|
| 285 |
+
},
|
| 286 |
+
{
|
| 287 |
+
"rank": 19,
|
| 288 |
+
"name": "Hopper",
|
| 289 |
+
"size_b": 4.0,
|
| 290 |
+
"base_model": "Qwen/Qwen3.5-4B-Base",
|
| 291 |
+
"kind": "LoRA",
|
| 292 |
+
"index": 39.67,
|
| 293 |
+
"areas": {
|
| 294 |
+
"knowledge": 0.2327,
|
| 295 |
+
"language": 0.444,
|
| 296 |
+
"retrieval": 0.4446,
|
| 297 |
+
"tools": 0.5905,
|
| 298 |
+
"arts": 0.248
|
| 299 |
+
}
|
| 300 |
+
},
|
| 301 |
+
{
|
| 302 |
+
"rank": 20,
|
| 303 |
+
"name": "Bespoke Nimble 9B v2",
|
| 304 |
+
"size_b": 9.0,
|
| 305 |
+
"base_model": "Qwen/Qwen3.5-9B-Base",
|
| 306 |
+
"kind": "LoRA",
|
| 307 |
+
"index": 39.57,
|
| 308 |
+
"areas": {
|
| 309 |
+
"knowledge": 0.2952,
|
| 310 |
+
"language": 0.4509,
|
| 311 |
+
"retrieval": 0.3493,
|
| 312 |
+
"tools": 0.5921,
|
| 313 |
+
"arts": 0.2471
|
| 314 |
+
}
|
| 315 |
+
},
|
| 316 |
+
{
|
| 317 |
+
"rank": 21,
|
| 318 |
+
"name": "JevK5",
|
| 319 |
+
"size_b": 4.0,
|
| 320 |
+
"base_model": "Qwen/Qwen3.5-4B-Base",
|
| 321 |
+
"kind": "LoRA",
|
| 322 |
+
"index": 38.81,
|
| 323 |
+
"areas": {
|
| 324 |
+
"knowledge": 0.235,
|
| 325 |
+
"language": 0.4329,
|
| 326 |
+
"retrieval": 0.4615,
|
| 327 |
+
"tools": 0.5212,
|
| 328 |
+
"arts": 0.2781
|
| 329 |
+
}
|
| 330 |
+
},
|
| 331 |
+
{
|
| 332 |
+
"rank": 22,
|
| 333 |
+
"name": "lev",
|
| 334 |
+
"size_b": 4.0,
|
| 335 |
+
"base_model": "Qwen/Qwen3.5-4B-Base",
|
| 336 |
+
"kind": "LoRA + head",
|
| 337 |
+
"index": 38.54,
|
| 338 |
+
"areas": {
|
| 339 |
+
"knowledge": 0.2197,
|
| 340 |
+
"language": 0.3852,
|
| 341 |
+
"retrieval": 0.4792,
|
| 342 |
+
"tools": 0.5752,
|
| 343 |
+
"arts": 0.2796
|
| 344 |
+
}
|
| 345 |
+
},
|
| 346 |
+
{
|
| 347 |
+
"rank": 23,
|
| 348 |
+
"name": "Kev 9B",
|
| 349 |
+
"size_b": 9.0,
|
| 350 |
+
"base_model": "Qwen/Qwen3.5-9B-Base",
|
| 351 |
+
"kind": "LoRA + head",
|
| 352 |
+
"index": 38.48,
|
| 353 |
+
"areas": {
|
| 354 |
+
"knowledge": 0.2616,
|
| 355 |
+
"language": 0.4167,
|
| 356 |
+
"retrieval": 0.4367,
|
| 357 |
+
"tools": 0.5448,
|
| 358 |
+
"arts": 0.2242
|
| 359 |
+
}
|
| 360 |
+
},
|
| 361 |
+
{
|
| 362 |
+
"rank": 24,
|
| 363 |
+
"name": "Intern-Decision-4B",
|
| 364 |
+
"size_b": 4.0,
|
| 365 |
+
"base_model": "Qwen/Qwen3.5-4B-Base",
|
| 366 |
+
"kind": "full fine-tune",
|
| 367 |
+
"index": 37.81,
|
| 368 |
+
"areas": {
|
| 369 |
+
"knowledge": 0.2371,
|
| 370 |
+
"language": 0.4188,
|
| 371 |
+
"retrieval": 0.4034,
|
| 372 |
+
"tools": 0.5566,
|
| 373 |
+
"arts": 0.2608
|
| 374 |
+
}
|
| 375 |
+
},
|
| 376 |
+
{
|
| 377 |
+
"rank": 25,
|
| 378 |
+
"name": "razorback16 openjev diffusiongemma (NVFP4, vLLM)",
|
| 379 |
+
"size_b": 26.0,
|
| 380 |
+
"base_model": "google/diffusiongemma-26B-A4B-it",
|
| 381 |
+
"kind": "inference technique",
|
| 382 |
+
"index": 37.25,
|
| 383 |
+
"areas": {
|
| 384 |
+
"knowledge": 0.2618,
|
| 385 |
+
"language": 0.4486,
|
| 386 |
+
"retrieval": 0.3702,
|
| 387 |
+
"tools": 0.4719,
|
| 388 |
+
"arts": 0.285
|
| 389 |
+
}
|
| 390 |
+
},
|
| 391 |
+
{
|
| 392 |
+
"rank": 26,
|
| 393 |
+
"name": "NeoHorse-Jev-4B",
|
| 394 |
+
"size_b": 4.0,
|
| 395 |
+
"base_model": "Qwen/Qwen3.5-4B-Base",
|
| 396 |
+
"kind": "head / adapter",
|
| 397 |
+
"index": 36.75,
|
| 398 |
+
"areas": {
|
| 399 |
+
"knowledge": 0.2259,
|
| 400 |
+
"language": 0.3936,
|
| 401 |
+
"retrieval": 0.4391,
|
| 402 |
+
"tools": 0.5888,
|
| 403 |
+
"arts": 0.1177
|
| 404 |
+
}
|
| 405 |
+
},
|
| 406 |
+
{
|
| 407 |
+
"rank": 27,
|
| 408 |
+
"name": "Solomon v1.1",
|
| 409 |
+
"size_b": 27.0,
|
| 410 |
+
"base_model": "Qwen/Qwen3.8-27B",
|
| 411 |
+
"kind": "head / adapter",
|
| 412 |
+
"index": 36.43,
|
| 413 |
+
"areas": {
|
| 414 |
+
"knowledge": 0.2213,
|
| 415 |
+
"language": 0.5728,
|
| 416 |
+
"retrieval": 0.2597,
|
| 417 |
+
"tools": 0.4697,
|
| 418 |
+
"arts": 0.2116
|
| 419 |
+
}
|
| 420 |
+
},
|
| 421 |
+
{
|
| 422 |
+
"rank": 28,
|
| 423 |
+
"name": "Kev 4B",
|
| 424 |
+
"size_b": 4.0,
|
| 425 |
+
"base_model": "Qwen/Qwen3.5-4B-Base",
|
| 426 |
+
"kind": "LoRA + head",
|
| 427 |
+
"index": 34.64,
|
| 428 |
+
"areas": {
|
| 429 |
+
"knowledge": 0.2286,
|
| 430 |
+
"language": 0.353,
|
| 431 |
+
"retrieval": 0.4098,
|
| 432 |
+
"tools": 0.5256,
|
| 433 |
+
"arts": 0.1792
|
| 434 |
+
}
|
| 435 |
+
},
|
| 436 |
+
{
|
| 437 |
+
"rank": 29,
|
| 438 |
+
"name": "Decision 1.0 Nox",
|
| 439 |
+
"size_b": 4.0,
|
| 440 |
+
"base_model": "Qwen/Qwen3.5-4B-Base",
|
| 441 |
+
"kind": "head / adapter",
|
| 442 |
+
"index": 34.36,
|
| 443 |
+
"areas": {
|
| 444 |
+
"knowledge": 0.2141,
|
| 445 |
+
"language": 0.3659,
|
| 446 |
+
"retrieval": 0.4431,
|
| 447 |
+
"tools": 0.4957,
|
| 448 |
+
"arts": 0.1439
|
| 449 |
+
}
|
| 450 |
+
},
|
| 451 |
+
{
|
| 452 |
+
"rank": 30,
|
| 453 |
+
"name": "Jobe",
|
| 454 |
+
"size_b": 4.0,
|
| 455 |
+
"base_model": "Qwen/Qwen3.5-4B-Base",
|
| 456 |
+
"kind": "inference technique",
|
| 457 |
+
"index": 32.35,
|
| 458 |
+
"areas": {
|
| 459 |
+
"knowledge": 0.1773,
|
| 460 |
+
"language": 0.3454,
|
| 461 |
+
"retrieval": 0.3669,
|
| 462 |
+
"tools": 0.4881,
|
| 463 |
+
"arts": 0.2574
|
| 464 |
+
}
|
| 465 |
+
},
|
| 466 |
+
{
|
| 467 |
+
"rank": 31,
|
| 468 |
+
"name": "mmastrac diffusiongemma (vLLM PR 57250)",
|
| 469 |
+
"size_b": 26.0,
|
| 470 |
+
"base_model": "google/diffusiongemma-26B-A4B-it",
|
| 471 |
+
"kind": "inference technique",
|
| 472 |
+
"index": 32.24,
|
| 473 |
+
"areas": {
|
| 474 |
+
"knowledge": 0.249,
|
| 475 |
+
"language": 0.4593,
|
| 476 |
+
"retrieval": 0.2417,
|
| 477 |
+
"tools": 0.3399,
|
| 478 |
+
"arts": 0.2881
|
| 479 |
+
}
|
| 480 |
+
},
|
| 481 |
+
{
|
| 482 |
+
"rank": 32,
|
| 483 |
+
"name": "open-jev (pngwn)",
|
| 484 |
+
"size_b": 4.0,
|
| 485 |
+
"base_model": "Qwen/Qwen3.5-4B-Base",
|
| 486 |
+
"kind": "LoRA",
|
| 487 |
+
"index": 29.91,
|
| 488 |
+
"areas": {
|
| 489 |
+
"knowledge": 0.1988,
|
| 490 |
+
"language": 0.3546,
|
| 491 |
+
"retrieval": 0.2912,
|
| 492 |
+
"tools": 0.4573,
|
| 493 |
+
"arts": 0.1414
|
| 494 |
+
}
|
| 495 |
+
},
|
| 496 |
+
{
|
| 497 |
+
"rank": 33,
|
| 498 |
+
"name": "Tev1-4B-experimental",
|
| 499 |
+
"size_b": 4.0,
|
| 500 |
+
"base_model": "Qwen/Qwen3.5-4B-Base",
|
| 501 |
+
"kind": "full fine-tune",
|
| 502 |
+
"index": 29.24,
|
| 503 |
+
"areas": {
|
| 504 |
+
"knowledge": 0.2349,
|
| 505 |
+
"language": 0.3919,
|
| 506 |
+
"retrieval": 0.1548,
|
| 507 |
+
"tools": 0.4183,
|
| 508 |
+
"arts": 0.2292
|
| 509 |
+
}
|
| 510 |
+
},
|
| 511 |
+
{
|
| 512 |
+
"rank": 34,
|
| 513 |
+
"name": "Decider 2B",
|
| 514 |
+
"size_b": 2.0,
|
| 515 |
+
"base_model": "Qwen/Qwen3.5-2B-Base",
|
| 516 |
+
"kind": "full fine-tune",
|
| 517 |
+
"index": 28.97,
|
| 518 |
+
"areas": {
|
| 519 |
+
"knowledge": 0.1493,
|
| 520 |
+
"language": 0.3257,
|
| 521 |
+
"retrieval": 0.3729,
|
| 522 |
+
"tools": 0.4245,
|
| 523 |
+
"arts": 0.1464
|
| 524 |
+
}
|
| 525 |
+
},
|
| 526 |
+
{
|
| 527 |
+
"rank": 35,
|
| 528 |
+
"name": "openvons",
|
| 529 |
+
"size_b": 4.0,
|
| 530 |
+
"base_model": "Qwen/Qwen3-4B-Instruct-2507",
|
| 531 |
+
"kind": "inference technique",
|
| 532 |
+
"index": 28.42,
|
| 533 |
+
"areas": {
|
| 534 |
+
"knowledge": 0.2077,
|
| 535 |
+
"language": 0.301,
|
| 536 |
+
"retrieval": 0.2537,
|
| 537 |
+
"tools": 0.4465,
|
| 538 |
+
"arts": 0.2033
|
| 539 |
+
}
|
| 540 |
+
},
|
| 541 |
+
{
|
| 542 |
+
"rank": 36,
|
| 543 |
+
"name": "this-that 1.2",
|
| 544 |
+
"size_b": null,
|
| 545 |
+
"base_model": null,
|
| 546 |
+
"kind": "full fine-tune",
|
| 547 |
+
"index": 28.14,
|
| 548 |
+
"areas": {
|
| 549 |
+
"knowledge": 0.1514,
|
| 550 |
+
"language": 0.3191,
|
| 551 |
+
"retrieval": 0.3283,
|
| 552 |
+
"tools": 0.4497,
|
| 553 |
+
"arts": 0.1183
|
| 554 |
+
}
|
| 555 |
+
},
|
| 556 |
+
{
|
| 557 |
+
"rank": 37,
|
| 558 |
+
"name": "Metask-Jev-4B",
|
| 559 |
+
"size_b": 4.0,
|
| 560 |
+
"base_model": "Qwen/Qwen3.5-4B-Base",
|
| 561 |
+
"kind": "LoRA",
|
| 562 |
+
"index": 26.89,
|
| 563 |
+
"areas": {
|
| 564 |
+
"knowledge": 0.2096,
|
| 565 |
+
"language": 0.3939,
|
| 566 |
+
"retrieval": 0.1291,
|
| 567 |
+
"tools": 0.3759,
|
| 568 |
+
"arts": 0.1828
|
| 569 |
+
}
|
| 570 |
+
},
|
| 571 |
+
{
|
| 572 |
+
"rank": 38,
|
| 573 |
+
"name": "SemIf",
|
| 574 |
+
"size_b": 4.0,
|
| 575 |
+
"base_model": "Qwen/Qwen3.5-4B-Base",
|
| 576 |
+
"kind": "inference technique",
|
| 577 |
+
"index": 25.94,
|
| 578 |
+
"areas": {
|
| 579 |
+
"knowledge": 0.171,
|
| 580 |
+
"language": 0.3379,
|
| 581 |
+
"retrieval": 0.1572,
|
| 582 |
+
"tools": 0.391,
|
| 583 |
+
"arts": 0.2492
|
| 584 |
+
}
|
| 585 |
+
},
|
| 586 |
+
{
|
| 587 |
+
"rank": 39,
|
| 588 |
+
"name": "Decision 1.0 Sol",
|
| 589 |
+
"size_b": 2.0,
|
| 590 |
+
"base_model": "Qwen/Qwen3.5-2B-Base",
|
| 591 |
+
"kind": "head / adapter",
|
| 592 |
+
"index": 25.32,
|
| 593 |
+
"areas": {
|
| 594 |
+
"knowledge": 0.1149,
|
| 595 |
+
"language": 0.2535,
|
| 596 |
+
"retrieval": 0.3271,
|
| 597 |
+
"tools": 0.4611,
|
| 598 |
+
"arts": 0.0818
|
| 599 |
+
}
|
| 600 |
+
},
|
| 601 |
+
{
|
| 602 |
+
"rank": 40,
|
| 603 |
+
"name": "mini-jev",
|
| 604 |
+
"size_b": 4.0,
|
| 605 |
+
"base_model": "Qwen/Qwen3-4B-Instruct-2507",
|
| 606 |
+
"kind": "inference technique",
|
| 607 |
+
"index": 20.98,
|
| 608 |
+
"areas": {
|
| 609 |
+
"knowledge": 0.1763,
|
| 610 |
+
"language": 0.2857,
|
| 611 |
+
"retrieval": 0.0866,
|
| 612 |
+
"tools": 0.2842,
|
| 613 |
+
"arts": 0.2107
|
| 614 |
+
}
|
| 615 |
+
},
|
| 616 |
+
{
|
| 617 |
+
"rank": 41,
|
| 618 |
+
"name": "Bosun v3.1 1.7B",
|
| 619 |
+
"size_b": 1.7,
|
| 620 |
+
"base_model": "Qwen/Qwen3-1.7B-Base",
|
| 621 |
+
"kind": "LoRA + head",
|
| 622 |
+
"index": 20.1,
|
| 623 |
+
"areas": {
|
| 624 |
+
"knowledge": 0.0472,
|
| 625 |
+
"language": 0.124,
|
| 626 |
+
"retrieval": 0.4282,
|
| 627 |
+
"tools": 0.3478,
|
| 628 |
+
"arts": 0.0747
|
| 629 |
+
}
|
| 630 |
+
},
|
| 631 |
+
{
|
| 632 |
+
"rank": 42,
|
| 633 |
+
"name": "Intern-Decision-2B",
|
| 634 |
+
"size_b": 2.0,
|
| 635 |
+
"base_model": "Qwen/Qwen3.5-2B-Base",
|
| 636 |
+
"kind": "full fine-tune",
|
| 637 |
+
"index": 19.38,
|
| 638 |
+
"areas": {
|
| 639 |
+
"knowledge": 0.138,
|
| 640 |
+
"language": 0.2056,
|
| 641 |
+
"retrieval": 0.1566,
|
| 642 |
+
"tools": 0.3334,
|
| 643 |
+
"arts": 0.1272
|
| 644 |
+
}
|
| 645 |
+
},
|
| 646 |
+
{
|
| 647 |
+
"rank": 43,
|
| 648 |
+
"name": "JPT-0.8B",
|
| 649 |
+
"size_b": 0.8,
|
| 650 |
+
"base_model": "Qwen/Qwen3.5-0.8B-Base",
|
| 651 |
+
"kind": "LoRA",
|
| 652 |
+
"index": 19.22,
|
| 653 |
+
"areas": {
|
| 654 |
+
"knowledge": 0.0988,
|
| 655 |
+
"language": 0.2304,
|
| 656 |
+
"retrieval": 0.164,
|
| 657 |
+
"tools": 0.3623,
|
| 658 |
+
"arts": 0.08
|
| 659 |
+
}
|
| 660 |
+
},
|
| 661 |
+
{
|
| 662 |
+
"rank": 44,
|
| 663 |
+
"name": "Decision 1.0 Eos",
|
| 664 |
+
"size_b": 0.8,
|
| 665 |
+
"base_model": "Qwen/Qwen3.5-0.8B-Base",
|
| 666 |
+
"kind": "head / adapter",
|
| 667 |
+
"index": 18.41,
|
| 668 |
+
"areas": {
|
| 669 |
+
"knowledge": 0.0737,
|
| 670 |
+
"language": 0.1459,
|
| 671 |
+
"retrieval": 0.3057,
|
| 672 |
+
"tools": 0.321,
|
| 673 |
+
"arts": 0.0747
|
| 674 |
+
}
|
| 675 |
+
},
|
| 676 |
+
{
|
| 677 |
+
"rank": 45,
|
| 678 |
+
"name": "Kev 0.8B",
|
| 679 |
+
"size_b": 0.8,
|
| 680 |
+
"base_model": "Qwen/Qwen3.5-0.8B-Base",
|
| 681 |
+
"kind": "LoRA + head",
|
| 682 |
+
"index": 14.6,
|
| 683 |
+
"areas": {
|
| 684 |
+
"knowledge": 0.0608,
|
| 685 |
+
"language": 0.1133,
|
| 686 |
+
"retrieval": 0.1914,
|
| 687 |
+
"tools": 0.315,
|
| 688 |
+
"arts": 0.0511
|
| 689 |
+
}
|
| 690 |
+
},
|
| 691 |
+
{
|
| 692 |
+
"rank": 46,
|
| 693 |
+
"name": "Bosun v3.1 0.6B",
|
| 694 |
+
"size_b": 0.6,
|
| 695 |
+
"base_model": "Qwen/Qwen3-0.6B-Base",
|
| 696 |
+
"kind": "LoRA + head",
|
| 697 |
+
"index": 14.32,
|
| 698 |
+
"areas": {
|
| 699 |
+
"knowledge": 0.0283,
|
| 700 |
+
"language": 0.0383,
|
| 701 |
+
"retrieval": 0.3415,
|
| 702 |
+
"tools": 0.2844,
|
| 703 |
+
"arts": 0.0557
|
| 704 |
+
}
|
| 705 |
+
},
|
| 706 |
+
{
|
| 707 |
+
"rank": 47,
|
| 708 |
+
"name": "Tev1-0.8B-experimental",
|
| 709 |
+
"size_b": 0.8,
|
| 710 |
+
"base_model": "Qwen/Qwen3.5-0.8B-Base",
|
| 711 |
+
"kind": "full fine-tune",
|
| 712 |
+
"index": 12.85,
|
| 713 |
+
"areas": {
|
| 714 |
+
"knowledge": 0.0729,
|
| 715 |
+
"language": 0.1405,
|
| 716 |
+
"retrieval": 0.0488,
|
| 717 |
+
"tools": 0.3197,
|
| 718 |
+
"arts": 0.0512
|
| 719 |
+
}
|
| 720 |
+
},
|
| 721 |
+
{
|
| 722 |
+
"rank": 48,
|
| 723 |
+
"name": "Intern-Decision-0.8B",
|
| 724 |
+
"size_b": 0.8,
|
| 725 |
+
"base_model": "Qwen/Qwen3.5-0.8B-Base",
|
| 726 |
+
"kind": "full fine-tune",
|
| 727 |
+
"index": 11.94,
|
| 728 |
+
"areas": {
|
| 729 |
+
"knowledge": 0.0797,
|
| 730 |
+
"language": 0.1035,
|
| 731 |
+
"retrieval": 0.0644,
|
| 732 |
+
"tools": 0.2838,
|
| 733 |
+
"arts": 0.0722
|
| 734 |
+
}
|
| 735 |
+
},
|
| 736 |
+
{
|
| 737 |
+
"rank": 49,
|
| 738 |
+
"name": "MoJev",
|
| 739 |
+
"size_b": 0.8,
|
| 740 |
+
"base_model": "Qwen/Qwen3.5-0.8B-Base",
|
| 741 |
+
"kind": "head / adapter",
|
| 742 |
+
"index": 11.69,
|
| 743 |
+
"areas": {
|
| 744 |
+
"knowledge": 0.0557,
|
| 745 |
+
"language": 0.0908,
|
| 746 |
+
"retrieval": 0.1618,
|
| 747 |
+
"tools": 0.2186,
|
| 748 |
+
"arts": 0.0672
|
| 749 |
+
}
|
| 750 |
+
},
|
| 751 |
+
{
|
| 752 |
+
"rank": 50,
|
| 753 |
+
"name": "GLiNER2.5-Decide",
|
| 754 |
+
"size_b": null,
|
| 755 |
+
"base_model": null,
|
| 756 |
+
"kind": "full fine-tune",
|
| 757 |
+
"index": 11.21,
|
| 758 |
+
"areas": {
|
| 759 |
+
"knowledge": 0.0241,
|
| 760 |
+
"language": 0.0748,
|
| 761 |
+
"retrieval": 0.2532,
|
| 762 |
+
"tools": 0.1478,
|
| 763 |
+
"arts": 0.0887
|
| 764 |
+
}
|
| 765 |
+
},
|
| 766 |
+
{
|
| 767 |
+
"rank": 51,
|
| 768 |
+
"name": "Lavoir",
|
| 769 |
+
"size_b": null,
|
| 770 |
+
"base_model": null,
|
| 771 |
+
"kind": "full fine-tune",
|
| 772 |
+
"index": 8.69,
|
| 773 |
+
"areas": {
|
| 774 |
+
"knowledge": 0.0349,
|
| 775 |
+
"language": 0.1037,
|
| 776 |
+
"retrieval": 0.12,
|
| 777 |
+
"tools": 0.1302,
|
| 778 |
+
"arts": 0.0323
|
| 779 |
+
}
|
| 780 |
+
},
|
| 781 |
+
{
|
| 782 |
+
"rank": 52,
|
| 783 |
+
"name": "jeff",
|
| 784 |
+
"size_b": null,
|
| 785 |
+
"base_model": null,
|
| 786 |
+
"kind": "inference technique",
|
| 787 |
+
"index": 8.04,
|
| 788 |
+
"areas": {
|
| 789 |
+
"knowledge": 0.0305,
|
| 790 |
+
"language": 0.104,
|
| 791 |
+
"retrieval": 0.2131,
|
| 792 |
+
"tools": 0.0119,
|
| 793 |
+
"arts": 0.0077
|
| 794 |
+
}
|
| 795 |
+
},
|
| 796 |
+
{
|
| 797 |
+
"rank": 53,
|
| 798 |
+
"name": "CLM-v0.1-8B",
|
| 799 |
+
"size_b": 8.0,
|
| 800 |
+
"base_model": "Qwen/Qwen3-8B-Base",
|
| 801 |
+
"kind": "full fine-tune",
|
| 802 |
+
"index": 7.4,
|
| 803 |
+
"areas": {
|
| 804 |
+
"knowledge": 0.0311,
|
| 805 |
+
"language": 0.0523,
|
| 806 |
+
"retrieval": 0.1126,
|
| 807 |
+
"tools": 0.1398,
|
| 808 |
+
"arts": 0.043
|
| 809 |
+
}
|
| 810 |
+
},
|
| 811 |
+
{
|
| 812 |
+
"rank": 54,
|
| 813 |
+
"name": "GLiNER 2.5 base",
|
| 814 |
+
"size_b": null,
|
| 815 |
+
"base_model": null,
|
| 816 |
+
"kind": "full fine-tune",
|
| 817 |
+
"index": 6.76,
|
| 818 |
+
"areas": {
|
| 819 |
+
"knowledge": 0.021,
|
| 820 |
+
"language": 0.0748,
|
| 821 |
+
"retrieval": 0.0966,
|
| 822 |
+
"tools": 0.1116,
|
| 823 |
+
"arts": 0.0313
|
| 824 |
+
}
|
| 825 |
+
},
|
| 826 |
+
{
|
| 827 |
+
"rank": 55,
|
| 828 |
+
"name": "LFM2.5-2.6B-RLCD",
|
| 829 |
+
"size_b": 2.6,
|
| 830 |
+
"base_model": "LiquidAI/LFM2.5-2.6B-Base",
|
| 831 |
+
"kind": "full fine-tune",
|
| 832 |
+
"index": 6.76,
|
| 833 |
+
"areas": {
|
| 834 |
+
"knowledge": 0.0742,
|
| 835 |
+
"language": 0.0542,
|
| 836 |
+
"retrieval": 0.1053,
|
| 837 |
+
"tools": 0.0443,
|
| 838 |
+
"arts": 0.0525
|
| 839 |
+
}
|
| 840 |
+
},
|
| 841 |
+
{
|
| 842 |
+
"rank": 56,
|
| 843 |
+
"name": "Decision 1.0 Kai",
|
| 844 |
+
"size_b": null,
|
| 845 |
+
"base_model": null,
|
| 846 |
+
"kind": "head / adapter",
|
| 847 |
+
"index": 6.52,
|
| 848 |
+
"areas": {
|
| 849 |
+
"knowledge": 0.0383,
|
| 850 |
+
"language": 0.0291,
|
| 851 |
+
"retrieval": 0.0932,
|
| 852 |
+
"tools": 0.1423,
|
| 853 |
+
"arts": 0.0308
|
| 854 |
+
}
|
| 855 |
+
},
|
| 856 |
+
{
|
| 857 |
+
"rank": 57,
|
| 858 |
+
"name": "Laya",
|
| 859 |
+
"size_b": null,
|
| 860 |
+
"base_model": null,
|
| 861 |
+
"kind": "full fine-tune",
|
| 862 |
+
"index": 6.04,
|
| 863 |
+
"areas": {
|
| 864 |
+
"knowledge": 0.0356,
|
| 865 |
+
"language": 0.0903,
|
| 866 |
+
"retrieval": 0.0715,
|
| 867 |
+
"tools": 0.0582,
|
| 868 |
+
"arts": 0.0289
|
| 869 |
+
}
|
| 870 |
+
},
|
| 871 |
+
{
|
| 872 |
+
"rank": 58,
|
| 873 |
+
"name": "Julia 1",
|
| 874 |
+
"size_b": null,
|
| 875 |
+
"base_model": null,
|
| 876 |
+
"kind": "full fine-tune",
|
| 877 |
+
"index": 5.54,
|
| 878 |
+
"areas": {
|
| 879 |
+
"knowledge": 0.0661,
|
| 880 |
+
"language": 0.0139,
|
| 881 |
+
"retrieval": 0.0991,
|
| 882 |
+
"tools": 0.072,
|
| 883 |
+
"arts": 0.0174
|
| 884 |
+
}
|
| 885 |
+
},
|
| 886 |
+
{
|
| 887 |
+
"rank": 59,
|
| 888 |
+
"name": "system-one-gemma",
|
| 889 |
+
"size_b": null,
|
| 890 |
+
"base_model": null,
|
| 891 |
+
"kind": "LoRA + head",
|
| 892 |
+
"index": 5.07,
|
| 893 |
+
"areas": {
|
| 894 |
+
"knowledge": 0.0129,
|
| 895 |
+
"language": 0.0108,
|
| 896 |
+
"retrieval": 0.1722,
|
| 897 |
+
"tools": 0.0331,
|
| 898 |
+
"arts": 0.0408
|
| 899 |
+
}
|
| 900 |
+
},
|
| 901 |
+
{
|
| 902 |
+
"rank": 60,
|
| 903 |
+
"name": "Decision 1.0 Lex",
|
| 904 |
+
"size_b": null,
|
| 905 |
+
"base_model": null,
|
| 906 |
+
"kind": "head / adapter",
|
| 907 |
+
"index": 4.54,
|
| 908 |
+
"areas": {
|
| 909 |
+
"knowledge": 0.0188,
|
| 910 |
+
"language": 0.0241,
|
| 911 |
+
"retrieval": 0.0625,
|
| 912 |
+
"tools": 0.1086,
|
| 913 |
+
"arts": 0.0193
|
| 914 |
+
}
|
| 915 |
+
},
|
| 916 |
+
{
|
| 917 |
+
"rank": 61,
|
| 918 |
+
"name": "GLiNER 2.5 multilingual",
|
| 919 |
+
"size_b": null,
|
| 920 |
+
"base_model": null,
|
| 921 |
+
"kind": "full fine-tune",
|
| 922 |
+
"index": 4.26,
|
| 923 |
+
"areas": {
|
| 924 |
+
"knowledge": 0.009,
|
| 925 |
+
"language": 0.0636,
|
| 926 |
+
"retrieval": 0.0531,
|
| 927 |
+
"tools": 0.0587,
|
| 928 |
+
"arts": 0.025
|
| 929 |
+
}
|
| 930 |
+
},
|
| 931 |
+
{
|
| 932 |
+
"rank": 62,
|
| 933 |
+
"name": "GLiNER 2.5 small",
|
| 934 |
+
"size_b": null,
|
| 935 |
+
"base_model": null,
|
| 936 |
+
"kind": "full fine-tune",
|
| 937 |
+
"index": 3.82,
|
| 938 |
+
"areas": {
|
| 939 |
+
"knowledge": 0.0175,
|
| 940 |
+
"language": 0.0375,
|
| 941 |
+
"retrieval": 0.043,
|
| 942 |
+
"tools": 0.0698,
|
| 943 |
+
"arts": 0.0266
|
| 944 |
+
}
|
| 945 |
+
},
|
| 946 |
+
{
|
| 947 |
+
"rank": 63,
|
| 948 |
+
"name": "Qwen-2.5-1B-RLCD",
|
| 949 |
+
"size_b": 1.5,
|
| 950 |
+
"base_model": "Qwen/Qwen2.5-1.5B",
|
| 951 |
+
"kind": "inference technique",
|
| 952 |
+
"index": 3.78,
|
| 953 |
+
"areas": {
|
| 954 |
+
"knowledge": 0.014,
|
| 955 |
+
"language": 0.0471,
|
| 956 |
+
"retrieval": 0.0094,
|
| 957 |
+
"tools": 0.07,
|
| 958 |
+
"arts": 0.0731
|
| 959 |
+
}
|
| 960 |
+
},
|
| 961 |
+
{
|
| 962 |
+
"rank": 64,
|
| 963 |
+
"name": "Lumma-Fev-0.6B",
|
| 964 |
+
"size_b": 0.6,
|
| 965 |
+
"base_model": "FrontiersMind/Lumma-0.6B-Base",
|
| 966 |
+
"kind": "LoRA",
|
| 967 |
+
"index": 2.97,
|
| 968 |
+
"areas": {
|
| 969 |
+
"knowledge": 0.0234,
|
| 970 |
+
"language": 0.0091,
|
| 971 |
+
"retrieval": 0.0888,
|
| 972 |
+
"tools": 0.0112,
|
| 973 |
+
"arts": 0.0144
|
| 974 |
+
}
|
| 975 |
+
},
|
| 976 |
+
{
|
| 977 |
+
"rank": 65,
|
| 978 |
+
"name": "Verdict",
|
| 979 |
+
"size_b": null,
|
| 980 |
+
"base_model": null,
|
| 981 |
+
"kind": "full fine-tune",
|
| 982 |
+
"index": 1.87,
|
| 983 |
+
"areas": {
|
| 984 |
+
"knowledge": 0.0036,
|
| 985 |
+
"language": 0.0145,
|
| 986 |
+
"retrieval": 0.0,
|
| 987 |
+
"tools": 0.0658,
|
| 988 |
+
"arts": 0.0203
|
| 989 |
+
}
|
| 990 |
+
},
|
| 991 |
+
{
|
| 992 |
+
"rank": 66,
|
| 993 |
+
"name": "Lumma-Fev-0.1B",
|
| 994 |
+
"size_b": null,
|
| 995 |
+
"base_model": null,
|
| 996 |
+
"kind": "full fine-tune",
|
| 997 |
+
"index": 1.78,
|
| 998 |
+
"areas": {
|
| 999 |
+
"knowledge": 0.0151,
|
| 1000 |
+
"language": 0.0062,
|
| 1001 |
+
"retrieval": 0.0245,
|
| 1002 |
+
"tools": 0.0265,
|
| 1003 |
+
"arts": 0.0257
|
| 1004 |
+
}
|
| 1005 |
+
},
|
| 1006 |
+
{
|
| 1007 |
+
"rank": 67,
|
| 1008 |
+
"name": "LFM2.5-350M-RLCD",
|
| 1009 |
+
"size_b": null,
|
| 1010 |
+
"base_model": null,
|
| 1011 |
+
"kind": "full fine-tune",
|
| 1012 |
+
"index": 1.38,
|
| 1013 |
+
"areas": {
|
| 1014 |
+
"knowledge": 0.0047,
|
| 1015 |
+
"language": 0.0098,
|
| 1016 |
+
"retrieval": 0.0002,
|
| 1017 |
+
"tools": 0.0384,
|
| 1018 |
+
"arts": 0.0298
|
| 1019 |
+
}
|
| 1020 |
+
}
|
| 1021 |
+
]
|
| 1022 |
+
},
|
| 1023 |
+
"wald_di": [
|
| 1024 |
+
{
|
| 1025 |
+
"id": "v1.0",
|
| 1026 |
+
"label": "Wald-4B v1.0 \u00b7 effort medium",
|
| 1027 |
+
"size_b": 4.0,
|
| 1028 |
+
"index": 53.91,
|
| 1029 |
+
"ci95": [
|
| 1030 |
+
51.87,
|
| 1031 |
+
55.27
|
| 1032 |
+
],
|
| 1033 |
+
"areas": {
|
| 1034 |
+
"knowledge": 0.3945,
|
| 1035 |
+
"language": 0.6176,
|
| 1036 |
+
"retrieval": 0.5193,
|
| 1037 |
+
"tools": 0.782,
|
| 1038 |
+
"arts": 0.3058
|
| 1039 |
+
},
|
| 1040 |
+
"source": "results/decision-index/boot021-0927-021A0-f10.json#arrmed",
|
| 1041 |
+
"status": "committed"
|
| 1042 |
+
}
|
| 1043 |
+
],
|
| 1044 |
+
"effort_di": {
|
| 1045 |
+
"none": {
|
| 1046 |
+
"index": 42.89,
|
| 1047 |
+
"ci95": [
|
| 1048 |
+
41.16,
|
| 1049 |
+
44.43
|
| 1050 |
+
],
|
| 1051 |
+
"areas": {
|
| 1052 |
+
"knowledge": 0.2859,
|
| 1053 |
+
"language": 0.448,
|
| 1054 |
+
"retrieval": 0.463,
|
| 1055 |
+
"tools": 0.6369,
|
| 1056 |
+
"arts": 0.3011
|
| 1057 |
+
},
|
| 1058 |
+
"source": "results/decision-index/boot021-0927-015G0-f4.json#g4one"
|
| 1059 |
+
},
|
| 1060 |
+
"low": {
|
| 1061 |
+
"index": 45.23,
|
| 1062 |
+
"ci95": [
|
| 1063 |
+
43.54,
|
| 1064 |
+
46.69
|
| 1065 |
+
],
|
| 1066 |
+
"areas": {
|
| 1067 |
+
"knowledge": 0.3545,
|
| 1068 |
+
"language": 0.4672,
|
| 1069 |
+
"retrieval": 0.4676,
|
| 1070 |
+
"tools": 0.6355,
|
| 1071 |
+
"arts": 0.3011
|
| 1072 |
+
},
|
| 1073 |
+
"source": "results/decision-index/boot021-0927-015G0-f4.json#g4low"
|
| 1074 |
+
},
|
| 1075 |
+
"medium": {
|
| 1076 |
+
"index": 47.23,
|
| 1077 |
+
"ci95": [
|
| 1078 |
+
45.52,
|
| 1079 |
+
48.86
|
| 1080 |
+
],
|
| 1081 |
+
"areas": {
|
| 1082 |
+
"knowledge": 0.3998,
|
| 1083 |
+
"language": 0.5023,
|
| 1084 |
+
"retrieval": 0.4671,
|
| 1085 |
+
"tools": 0.642,
|
| 1086 |
+
"arts": 0.2819
|
| 1087 |
+
},
|
| 1088 |
+
"source": "results/decision-index/boot021-0927-015G0-f4.json#g4med"
|
| 1089 |
+
},
|
| 1090 |
+
"high": {
|
| 1091 |
+
"index": 49.14,
|
| 1092 |
+
"ci95": [
|
| 1093 |
+
47.21,
|
| 1094 |
+
50.65
|
| 1095 |
+
],
|
| 1096 |
+
"areas": {
|
| 1097 |
+
"knowledge": 0.4415,
|
| 1098 |
+
"language": 0.5279,
|
| 1099 |
+
"retrieval": 0.4698,
|
| 1100 |
+
"tools": 0.6605,
|
| 1101 |
+
"arts": 0.2604
|
| 1102 |
+
},
|
| 1103 |
+
"source": "results/decision-index/boot021-0927-015G0-f4.json#g4high"
|
| 1104 |
+
},
|
| 1105 |
+
"high-k4": {
|
| 1106 |
+
"index": 49.62,
|
| 1107 |
+
"ci95": [
|
| 1108 |
+
47.95,
|
| 1109 |
+
51.16
|
| 1110 |
+
],
|
| 1111 |
+
"areas": {
|
| 1112 |
+
"knowledge": 0.4649,
|
| 1113 |
+
"language": 0.5232,
|
| 1114 |
+
"retrieval": 0.4698,
|
| 1115 |
+
"tools": 0.6605,
|
| 1116 |
+
"arts": 0.2604
|
| 1117 |
+
},
|
| 1118 |
+
"source": "results/decision-index/boot021-0927-015G0-f4.json#g4k4"
|
| 1119 |
+
}
|
| 1120 |
+
},
|
| 1121 |
+
"qwen27_paren": {
|
| 1122 |
+
"index": 47.78,
|
| 1123 |
+
"ci95": [
|
| 1124 |
+
45.95,
|
| 1125 |
+
49.17
|
| 1126 |
+
],
|
| 1127 |
+
"areas": {
|
| 1128 |
+
"knowledge": 0.3485,
|
| 1129 |
+
"language": 0.4867,
|
| 1130 |
+
"retrieval": 0.557,
|
| 1131 |
+
"tools": 0.6954,
|
| 1132 |
+
"arts": 0.2329
|
| 1133 |
+
},
|
| 1134 |
+
"source": "results/decision-index/boot021-bigbase-0927.json#q27paren"
|
| 1135 |
+
},
|
| 1136 |
+
"jev_sample": {
|
| 1137 |
+
"index": 57.19,
|
| 1138 |
+
"ci95": [
|
| 1139 |
+
55.29,
|
| 1140 |
+
58.6
|
| 1141 |
+
],
|
| 1142 |
+
"areas": {
|
| 1143 |
+
"knowledge": 0.5044,
|
| 1144 |
+
"language": 0.5832,
|
| 1145 |
+
"retrieval": 0.566,
|
| 1146 |
+
"tools": 0.763,
|
| 1147 |
+
"arts": 0.3799
|
| 1148 |
+
},
|
| 1149 |
+
"source": "results/decision-index/boot021-0927.json#jev"
|
| 1150 |
+
},
|
| 1151 |
+
"board_named": {
|
| 1152 |
+
"decider_4b": {
|
| 1153 |
+
"rank": 15,
|
| 1154 |
+
"index": 40.7,
|
| 1155 |
+
"base_model": "Qwen/Qwen3.5-4B-Base"
|
| 1156 |
+
},
|
| 1157 |
+
"decider_35b": {
|
| 1158 |
+
"rank": 11,
|
| 1159 |
+
"index": 47.11,
|
| 1160 |
+
"base_model": "Qwen/Qwen3.5-35B-A3B-Base"
|
| 1161 |
+
}
|
| 1162 |
+
},
|
| 1163 |
+
"board_4_5": [
|
| 1164 |
+
{
|
| 1165 |
+
"rank": 4,
|
| 1166 |
+
"name": "simple-jev \u00b7 Qwen3.8-27B (featherless)",
|
| 1167 |
+
"size_b": 27.0,
|
| 1168 |
+
"base_model": "Qwen/Qwen3.8-27B",
|
| 1169 |
+
"kind": "inference technique",
|
| 1170 |
+
"index": 55.74,
|
| 1171 |
+
"areas": {
|
| 1172 |
+
"knowledge": 0.3656,
|
| 1173 |
+
"language": 0.6209,
|
| 1174 |
+
"retrieval": 0.6325,
|
| 1175 |
+
"tools": 0.7621,
|
| 1176 |
+
"arts": 0.3648
|
| 1177 |
+
}
|
| 1178 |
+
},
|
| 1179 |
+
{
|
| 1180 |
+
"rank": 5,
|
| 1181 |
+
"name": "Jebadiah 27B",
|
| 1182 |
+
"size_b": 27.0,
|
| 1183 |
+
"base_model": "Qwen/Qwen3.8-27B",
|
| 1184 |
+
"kind": "LoRA",
|
| 1185 |
+
"index": 54.67,
|
| 1186 |
+
"areas": {
|
| 1187 |
+
"knowledge": 0.3884,
|
| 1188 |
+
"language": 0.6068,
|
| 1189 |
+
"retrieval": 0.5389,
|
| 1190 |
+
"tools": 0.7813,
|
| 1191 |
+
"arts": 0.3873
|
| 1192 |
+
}
|
| 1193 |
+
}
|
| 1194 |
+
],
|
| 1195 |
+
"verticals": [
|
| 1196 |
+
{
|
| 1197 |
+
"laya_0shot": 35.6,
|
| 1198 |
+
"laya_p50_ms": 893.8,
|
| 1199 |
+
"clm_0shot": 34.8,
|
| 1200 |
+
"clm_p50_ms": 1059.5,
|
| 1201 |
+
"task": "When2Call",
|
| 1202 |
+
"metric": "accuracy",
|
| 1203 |
+
"n_test": 500,
|
| 1204 |
+
"ours_lora": 82.8,
|
| 1205 |
+
"ours_ci95": [
|
| 1206 |
+
79.2,
|
| 1207 |
+
86.0
|
| 1208 |
+
],
|
| 1209 |
+
"jev": 72.6,
|
| 1210 |
+
"wald_0shot": 68.6,
|
| 1211 |
+
"raw_qwen_0shot": 47.4,
|
| 1212 |
+
"raw_source": "verticals/when2call/runs/20260926-2141/report.json",
|
| 1213 |
+
"ours_minus_jev_ci95": [
|
| 1214 |
+
6.2,
|
| 1215 |
+
14.2
|
| 1216 |
+
],
|
| 1217 |
+
"train_usd": 1.0,
|
| 1218 |
+
"beats_jev": true
|
| 1219 |
+
},
|
| 1220 |
+
{
|
| 1221 |
+
"laya_0shot": 33.9,
|
| 1222 |
+
"laya_p50_ms": 897.7,
|
| 1223 |
+
"clm_0shot": 0.3,
|
| 1224 |
+
"clm_p50_ms": 1057.1,
|
| 1225 |
+
"task": "BANKING77",
|
| 1226 |
+
"metric": "macro-F1",
|
| 1227 |
+
"n_test": 500,
|
| 1228 |
+
"ours_lora": 92.8,
|
| 1229 |
+
"ours_ci95": [
|
| 1230 |
+
89.5,
|
| 1231 |
+
94.6
|
| 1232 |
+
],
|
| 1233 |
+
"jev": 78.1,
|
| 1234 |
+
"wald_0shot": 77.4,
|
| 1235 |
+
"raw_qwen_0shot": 66.8,
|
| 1236 |
+
"raw_source": "verticals/banking77/runs/20260926-2200/report.json",
|
| 1237 |
+
"ours_minus_jev_ci95": [
|
| 1238 |
+
12.1,
|
| 1239 |
+
18.6
|
| 1240 |
+
],
|
| 1241 |
+
"train_usd": 0.64,
|
| 1242 |
+
"beats_jev": true
|
| 1243 |
+
},
|
| 1244 |
+
{
|
| 1245 |
+
"laya_0shot": 62.6,
|
| 1246 |
+
"laya_p50_ms": 992.6,
|
| 1247 |
+
"clm_0shot": 12.7,
|
| 1248 |
+
"clm_p50_ms": 1064.2,
|
| 1249 |
+
"task": "SGD intent",
|
| 1250 |
+
"metric": "macro-F1",
|
| 1251 |
+
"n_test": 500,
|
| 1252 |
+
"ours_lora": 98.0,
|
| 1253 |
+
"ours_ci95": [
|
| 1254 |
+
92.6,
|
| 1255 |
+
99.2
|
| 1256 |
+
],
|
| 1257 |
+
"jev": 92.8,
|
| 1258 |
+
"wald_0shot": 82.8,
|
| 1259 |
+
"raw_qwen_0shot": 79.7,
|
| 1260 |
+
"raw_source": "verticals/sgd-intent/runs/20260926-2227/report.json",
|
| 1261 |
+
"ours_minus_jev_ci95": [
|
| 1262 |
+
2.9,
|
| 1263 |
+
8.2
|
| 1264 |
+
],
|
| 1265 |
+
"train_usd": 0.83,
|
| 1266 |
+
"beats_jev": true
|
| 1267 |
+
},
|
| 1268 |
+
{
|
| 1269 |
+
"laya_0shot": 14.2,
|
| 1270 |
+
"laya_p50_ms": 271.4,
|
| 1271 |
+
"clm_0shot": 12.9,
|
| 1272 |
+
"clm_p50_ms": 88.2,
|
| 1273 |
+
"task": "ToxicChat",
|
| 1274 |
+
"metric": "toxic-class F1",
|
| 1275 |
+
"n_test": 5029,
|
| 1276 |
+
"ours_lora": 84.9,
|
| 1277 |
+
"ours_ci95": [
|
| 1278 |
+
81.9,
|
| 1279 |
+
87.6
|
| 1280 |
+
],
|
| 1281 |
+
"jev": 80.4,
|
| 1282 |
+
"wald_0shot": 73.4,
|
| 1283 |
+
"raw_qwen_0shot": 36.2,
|
| 1284 |
+
"raw_source": "verticals/toxicchat/runs/20260927-0045/report.json",
|
| 1285 |
+
"ours_minus_jev_ci95": [
|
| 1286 |
+
1.5,
|
| 1287 |
+
7.5
|
| 1288 |
+
],
|
| 1289 |
+
"train_usd": 0.2,
|
| 1290 |
+
"beats_jev": true
|
| 1291 |
+
},
|
| 1292 |
+
{
|
| 1293 |
+
"laya_0shot": null,
|
| 1294 |
+
"laya_p50_ms": null,
|
| 1295 |
+
"clm_0shot": 34.0,
|
| 1296 |
+
"clm_p50_ms": 1064.9,
|
| 1297 |
+
"task": "MetaTool",
|
| 1298 |
+
"metric": "accuracy",
|
| 1299 |
+
"n_test": 500,
|
| 1300 |
+
"ours_lora": 97.0,
|
| 1301 |
+
"ours_ci95": [
|
| 1302 |
+
95.4,
|
| 1303 |
+
98.4
|
| 1304 |
+
],
|
| 1305 |
+
"jev": 83.0,
|
| 1306 |
+
"wald_0shot": 80.4,
|
| 1307 |
+
"raw_qwen_0shot": 75.8,
|
| 1308 |
+
"raw_source": "verticals/metatool/runs/20260927-0228/report.json",
|
| 1309 |
+
"ours_minus_jev_ci95": [
|
| 1310 |
+
10.8,
|
| 1311 |
+
17.2
|
| 1312 |
+
],
|
| 1313 |
+
"train_usd": 0.79,
|
| 1314 |
+
"beats_jev": true
|
| 1315 |
+
},
|
| 1316 |
+
{
|
| 1317 |
+
"laya_0shot": null,
|
| 1318 |
+
"laya_p50_ms": null,
|
| 1319 |
+
"clm_0shot": 61.2,
|
| 1320 |
+
"clm_p50_ms": 1066.2,
|
| 1321 |
+
"task": "AndroidControl",
|
| 1322 |
+
"metric": "accuracy",
|
| 1323 |
+
"n_test": 500,
|
| 1324 |
+
"ours_lora": 81.6,
|
| 1325 |
+
"ours_ci95": [
|
| 1326 |
+
78.0,
|
| 1327 |
+
85.0
|
| 1328 |
+
],
|
| 1329 |
+
"jev": 73.4,
|
| 1330 |
+
"wald_0shot": 60.6,
|
| 1331 |
+
"raw_qwen_0shot": 62.0,
|
| 1332 |
+
"raw_source": "verticals/androidcontrol/runs/20260927-0245/report.json",
|
| 1333 |
+
"ours_minus_jev_ci95": [
|
| 1334 |
+
4.4,
|
| 1335 |
+
12.4
|
| 1336 |
+
],
|
| 1337 |
+
"train_usd": 1.48,
|
| 1338 |
+
"beats_jev": true
|
| 1339 |
+
},
|
| 1340 |
+
{
|
| 1341 |
+
"laya_0shot": 25.6,
|
| 1342 |
+
"laya_p50_ms": 992.6,
|
| 1343 |
+
"clm_0shot": 0.0,
|
| 1344 |
+
"clm_p50_ms": 1062.2,
|
| 1345 |
+
"task": "RouteLLM routing",
|
| 1346 |
+
"metric": "needs-strong F1",
|
| 1347 |
+
"n_test": 500,
|
| 1348 |
+
"ours_lora": 26.8,
|
| 1349 |
+
"ours_ci95": [
|
| 1350 |
+
18.3,
|
| 1351 |
+
35.4
|
| 1352 |
+
],
|
| 1353 |
+
"jev": 0.0,
|
| 1354 |
+
"wald_0shot": 11.3,
|
| 1355 |
+
"raw_qwen_0shot": 33.3,
|
| 1356 |
+
"raw_source": "verticals/model-routing/runs/20260927-0236/report.json",
|
| 1357 |
+
"ours_minus_jev_ci95": [
|
| 1358 |
+
18.3,
|
| 1359 |
+
35.4
|
| 1360 |
+
],
|
| 1361 |
+
"train_usd": 0.63,
|
| 1362 |
+
"beats_jev": true
|
| 1363 |
+
},
|
| 1364 |
+
{
|
| 1365 |
+
"laya_0shot": 13.3,
|
| 1366 |
+
"laya_p50_ms": 250.0,
|
| 1367 |
+
"clm_0shot": 61.7,
|
| 1368 |
+
"clm_p50_ms": 266.9,
|
| 1369 |
+
"task": "Agent-trajectory safety (hard)",
|
| 1370 |
+
"metric": "unsafe-class F1",
|
| 1371 |
+
"n_test": 500,
|
| 1372 |
+
"ours_lora": 72.2,
|
| 1373 |
+
"ours_ci95": [
|
| 1374 |
+
68.0,
|
| 1375 |
+
75.9
|
| 1376 |
+
],
|
| 1377 |
+
"jev": 47.0,
|
| 1378 |
+
"wald_0shot": 32.5,
|
| 1379 |
+
"raw_qwen_0shot": 70.1,
|
| 1380 |
+
"raw_source": "verticals/agent-trajectory-safety-hard/runs/20260927-0259/report.json",
|
| 1381 |
+
"ours_minus_jev_ci95": [
|
| 1382 |
+
18.3,
|
| 1383 |
+
32.2
|
| 1384 |
+
],
|
| 1385 |
+
"train_usd": 0.15,
|
| 1386 |
+
"beats_jev": true
|
| 1387 |
+
},
|
| 1388 |
+
{
|
| 1389 |
+
"laya_0shot": 2.1,
|
| 1390 |
+
"laya_p50_ms": 1020.9,
|
| 1391 |
+
"clm_0shot": 3.8,
|
| 1392 |
+
"clm_p50_ms": 1383.2,
|
| 1393 |
+
"task": "Mind2Web",
|
| 1394 |
+
"metric": "accuracy",
|
| 1395 |
+
"n_test": 480,
|
| 1396 |
+
"ours_lora": 51.9,
|
| 1397 |
+
"ours_ci95": [
|
| 1398 |
+
47.3,
|
| 1399 |
+
56.0
|
| 1400 |
+
],
|
| 1401 |
+
"jev": 48.5,
|
| 1402 |
+
"wald_0shot": 32.3,
|
| 1403 |
+
"raw_qwen_0shot": 9.8,
|
| 1404 |
+
"raw_source": "verticals/mind2web/runs/20260927-0119/report.json",
|
| 1405 |
+
"ours_minus_jev_ci95": [
|
| 1406 |
+
-1.2,
|
| 1407 |
+
7.9
|
| 1408 |
+
],
|
| 1409 |
+
"train_usd": 1.81,
|
| 1410 |
+
"beats_jev": false
|
| 1411 |
+
},
|
| 1412 |
+
{
|
| 1413 |
+
"laya_0shot": 56.6,
|
| 1414 |
+
"laya_p50_ms": 220.6,
|
| 1415 |
+
"clm_0shot": 30.6,
|
| 1416 |
+
"clm_p50_ms": 1021.4,
|
| 1417 |
+
"task": "Prompt injection",
|
| 1418 |
+
"metric": "injection-class F1",
|
| 1419 |
+
"n_test": 578,
|
| 1420 |
+
"ours_lora": 79.4,
|
| 1421 |
+
"ours_ci95": [
|
| 1422 |
+
73.5,
|
| 1423 |
+
84.5
|
| 1424 |
+
],
|
| 1425 |
+
"jev": 82.8,
|
| 1426 |
+
"wald_0shot": 71.3,
|
| 1427 |
+
"raw_qwen_0shot": 31.6,
|
| 1428 |
+
"raw_source": "verticals/prompt-injection/runs/20260927-0238/report.json",
|
| 1429 |
+
"ours_minus_jev_ci95": [
|
| 1430 |
+
-10.1,
|
| 1431 |
+
3.1
|
| 1432 |
+
],
|
| 1433 |
+
"train_usd": 0.12,
|
| 1434 |
+
"beats_jev": false
|
| 1435 |
+
},
|
| 1436 |
+
{
|
| 1437 |
+
"laya_0shot": 49.2,
|
| 1438 |
+
"laya_p50_ms": 547.9,
|
| 1439 |
+
"clm_0shot": 50.3,
|
| 1440 |
+
"clm_p50_ms": 602.6,
|
| 1441 |
+
"task": "RewardBench",
|
| 1442 |
+
"metric": "accuracy",
|
| 1443 |
+
"n_test": 1000,
|
| 1444 |
+
"ours_lora": 83.4,
|
| 1445 |
+
"ours_ci95": [
|
| 1446 |
+
81.0,
|
| 1447 |
+
85.7
|
| 1448 |
+
],
|
| 1449 |
+
"jev": 90.6,
|
| 1450 |
+
"wald_0shot": 83.7,
|
| 1451 |
+
"raw_qwen_0shot": 65.0,
|
| 1452 |
+
"raw_source": "verticals/rewardbench/runs/20260927-0048/report.json",
|
| 1453 |
+
"ours_minus_jev_ci95": [
|
| 1454 |
+
-9.4,
|
| 1455 |
+
-5.1
|
| 1456 |
+
],
|
| 1457 |
+
"train_usd": 1.49,
|
| 1458 |
+
"beats_jev": false
|
| 1459 |
+
},
|
| 1460 |
+
{
|
| 1461 |
+
"task": "COLD (Chinese)",
|
| 1462 |
+
"metric": "macro-F1",
|
| 1463 |
+
"n_test": 5323,
|
| 1464 |
+
"ours_lora": 84.1,
|
| 1465 |
+
"ours_ci95": [
|
| 1466 |
+
83.2,
|
| 1467 |
+
85.1
|
| 1468 |
+
],
|
| 1469 |
+
"jev": 75.3,
|
| 1470 |
+
"wald_0shot": 80.1,
|
| 1471 |
+
"raw_qwen_0shot": null,
|
| 1472 |
+
"raw_source": null,
|
| 1473 |
+
"laya_0shot": null,
|
| 1474 |
+
"laya_p50_ms": null,
|
| 1475 |
+
"clm_0shot": null,
|
| 1476 |
+
"clm_p50_ms": null,
|
| 1477 |
+
"ours_minus_jev_ci95": [
|
| 1478 |
+
7.7,
|
| 1479 |
+
10.0
|
| 1480 |
+
],
|
| 1481 |
+
"train_usd": 1.0,
|
| 1482 |
+
"beats_jev": true,
|
| 1483 |
+
"note": "LoRA on all 25,382 train labels averaged with the zero-shot base; ties the best published 83.7"
|
| 1484 |
+
}
|
| 1485 |
+
],
|
| 1486 |
+
"suite": [
|
| 1487 |
+
{
|
| 1488 |
+
"system": "Wald-4B v1.0 \u00b7 medium (packaged server)",
|
| 1489 |
+
"public231": 88.3,
|
| 1490 |
+
"xl_hidden": 63.2,
|
| 1491 |
+
"source": "public 231: packaged wald-serve smoke read of 021A0-f10 09-27 (ledger 197); XL: xl-bench leaderboard.md (hidden split)"
|
| 1492 |
+
},
|
| 1493 |
+
{
|
| 1494 |
+
"system": "Jev (API, jev-1.13.0)",
|
| 1495 |
+
"public231": 86.6,
|
| 1496 |
+
"xl_hidden": 73.0,
|
| 1497 |
+
"source": "results/external/jev-1.13.0; xl-bench #1"
|
| 1498 |
+
},
|
| 1499 |
+
{
|
| 1500 |
+
"system": "Cygnet (gemma-4-12B-it + shim)",
|
| 1501 |
+
"public231": null,
|
| 1502 |
+
"xl_hidden": 59.8,
|
| 1503 |
+
"source": "xl-bench #4"
|
| 1504 |
+
},
|
| 1505 |
+
{
|
| 1506 |
+
"system": "Laya (zero-shot, Jev's wire request)",
|
| 1507 |
+
"public231": 58.0,
|
| 1508 |
+
"xl_hidden": 13.4,
|
| 1509 |
+
"source": "results/external/laya; xl-bench #24"
|
| 1510 |
+
},
|
| 1511 |
+
{
|
| 1512 |
+
"system": "CLM-8B (zero-shot, same requests, deployment verified)",
|
| 1513 |
+
"public231": 39.0,
|
| 1514 |
+
"xl_hidden": 12.7,
|
| 1515 |
+
"source": "ledger 190 read (results/external/clm-8b); xl-bench #25"
|
| 1516 |
+
},
|
| 1517 |
+
{
|
| 1518 |
+
"system": "raw Qwen3.5-4B-Base (zero-shot, our readout)",
|
| 1519 |
+
"public231": 67.5,
|
| 1520 |
+
"xl_hidden": null,
|
| 1521 |
+
"source": "exported lineage summary (public one-pass 156 / 231)"
|
| 1522 |
+
}
|
| 1523 |
+
],
|
| 1524 |
+
"effort_serving": [
|
| 1525 |
+
{
|
| 1526 |
+
"effort": "none",
|
| 1527 |
+
"thinks": 0.0,
|
| 1528 |
+
"public231": 194,
|
| 1529 |
+
"xl_dev": 72.4,
|
| 1530 |
+
"processbench": 64.2,
|
| 1531 |
+
"p50_s": 0.051,
|
| 1532 |
+
"p95_s": 0.117,
|
| 1533 |
+
"usd_per_1k": 0.0184
|
| 1534 |
+
},
|
| 1535 |
+
{
|
| 1536 |
+
"effort": "low",
|
| 1537 |
+
"thinks": 1.2,
|
| 1538 |
+
"public231": 192,
|
| 1539 |
+
"xl_dev": 76.0,
|
| 1540 |
+
"processbench": 64.2,
|
| 1541 |
+
"p50_s": 0.048,
|
| 1542 |
+
"p95_s": 0.109,
|
| 1543 |
+
"usd_per_1k": 0.0194
|
| 1544 |
+
},
|
| 1545 |
+
{
|
| 1546 |
+
"effort": "medium",
|
| 1547 |
+
"thinks": 11.0,
|
| 1548 |
+
"public231": 202,
|
| 1549 |
+
"xl_dev": 81.1,
|
| 1550 |
+
"processbench": 66.7,
|
| 1551 |
+
"p50_s": 0.052,
|
| 1552 |
+
"p95_s": 1.08,
|
| 1553 |
+
"usd_per_1k": 0.0234
|
| 1554 |
+
},
|
| 1555 |
+
{
|
| 1556 |
+
"effort": "high",
|
| 1557 |
+
"thinks": 100.0,
|
| 1558 |
+
"public231": 194,
|
| 1559 |
+
"xl_dev": 85.8,
|
| 1560 |
+
"processbench": 83.4,
|
| 1561 |
+
"p50_s": 0.62,
|
| 1562 |
+
"p95_s": 1.7,
|
| 1563 |
+
"usd_per_1k": 0.0366
|
| 1564 |
+
}
|
| 1565 |
+
],
|
| 1566 |
+
"reference_api": {
|
| 1567 |
+
"name": "Jev API",
|
| 1568 |
+
"p50_s": 0.4,
|
| 1569 |
+
"p95_s": 0.83,
|
| 1570 |
+
"usd_per_1k": 0.04
|
| 1571 |
+
},
|
| 1572 |
+
"serving": {
|
| 1573 |
+
"rtxpro6000_fp8_usd_per_1k": 0.0073,
|
| 1574 |
+
"rtxpro6000_fp8_decisions_per_s": 114.9,
|
| 1575 |
+
"rtxpro6000_c1_p50_ms": 26,
|
| 1576 |
+
"multitenant_c32_usd_per_1k": 0.0067,
|
| 1577 |
+
"multitenant_c32_decisions_per_s": 70,
|
| 1578 |
+
"vertical_when2call_p50_ms": 56,
|
| 1579 |
+
"vertical_banking77_p50_ms": 176,
|
| 1580 |
+
"jev_when2call_p50_ms": 541,
|
| 1581 |
+
"jev_banking77_p50_ms": 702,
|
| 1582 |
+
"jev_usd_per_1k": 0.04
|
| 1583 |
+
}
|
| 1584 |
+
}
|
figures/di-areas.svg
ADDED
|
|
figures/di-vs-size.svg
ADDED
|
|
figures/latency-cost.svg
ADDED
|
|
figures/lora-cli-architecture.svg
ADDED
|
|
figures/verticals.svg
ADDED
|
|
generation_config.json
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_from_model_config": true,
|
| 3 |
+
"eos_token_id": 248044,
|
| 4 |
+
"transformers_version": "5.17.0",
|
| 5 |
+
"use_cache": true
|
| 6 |
+
}
|
model-files/serving.json
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"effort": "medium",
|
| 3 |
+
"prompt_format": "repeat_state_plain",
|
| 4 |
+
"max_model_len": 131072,
|
| 5 |
+
"temperature": "temperature.json"
|
| 6 |
+
}
|
model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:dabeb7bc3f43bf6ea7d20ecfdac8758b3945517a991809a5772988829598e893
|
| 3 |
+
size 8411558400
|
run.sh
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env bash
|
| 2 |
+
# One command, no Docker: create a Python 3.12 environment with vLLM 0.30.0 and wald-serve, then serve the weights on
|
| 3 |
+
# 0.0.0.0:${PORT:-8000} (POST /v1/systemone, GET /health). Needs a Linux CUDA machine and `uv`.
|
| 4 |
+
#
|
| 5 |
+
# ./run.sh /path/to/Wald-4B # declared policy from serving.json (effort medium)
|
| 6 |
+
# EFFORT=high ./run.sh /path/to/Wald-4B # another policy: none | low | medium | high | high-k<k>
|
| 7 |
+
set -euo pipefail
|
| 8 |
+
MODEL=${1:?usage: run.sh <weights dir>}
|
| 9 |
+
HERE=$(cd "$(dirname "$0")" && pwd)
|
| 10 |
+
VENV=${VENV:-$HERE/.venv}
|
| 11 |
+
[ -x "$VENV/bin/python" ] || { uv venv -p 3.12 "$VENV" && VIRTUAL_ENV="$VENV" uv pip install "vllm==0.30.0" "$HERE/server"; }
|
| 12 |
+
export VLLM_USE_FLASHINFER_SAMPLER=0 VLLM_NO_USAGE_STATS=1 TOKENIZERS_PARALLELISM=false HF_HUB_OFFLINE=1
|
| 13 |
+
exec "$VENV/bin/wald-serve" --model "$MODEL" --port "${PORT:-8000}" ${EFFORT:+--effort "$EFFORT"} --vllm-args "${VLLM_ARGS:-}"
|
server/README.md
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# wald-serve
|
| 2 |
+
|
| 3 |
+
The TypeSafe `POST /v1/systemone` server for Wald-4B: the option-letter readout over a vLLM OpenAI-compatible server,
|
| 4 |
+
the effort gate (none / low / medium / high / high-k), the knockout for more than 26 options, and the temperature table.
|
| 5 |
+
It needs only public packages: `pydantic`, plus `vllm==0.30.0` on the GPU machine.
|
| 6 |
+
|
| 7 |
+
```sh
|
| 8 |
+
pip install ".[vllm]" # on a Linux CUDA machine
|
| 9 |
+
wald-serve --model /path/to/Wald-4B --port 8000 # starts vLLM on the weights and serves /v1/systemone
|
| 10 |
+
```
|
| 11 |
+
|
| 12 |
+
Tests (no model, no GPU; a fake vLLM stands in): `pip install ".[test]" && cd tests && pytest -q`.
|
| 13 |
+
`tests/test_parity.py` compares prompts and answers byte for byte with the reference implementation the published
|
| 14 |
+
numbers were measured with; it is skipped unless that implementation is importable (WALD_PARITY_SRC set).
|
server/pyproject.toml
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[project]
|
| 2 |
+
name = "wald-serve"
|
| 3 |
+
version = "0.1.0"
|
| 4 |
+
description = "TypeSafe POST /v1/systemone server for Wald-4B: option-letter readout over vLLM, effort gate, knockout, calibration"
|
| 5 |
+
readme = "README.md"
|
| 6 |
+
license = { text = "Apache-2.0" }
|
| 7 |
+
requires-python = ">=3.10"
|
| 8 |
+
dependencies = ["pydantic>=2.5"]
|
| 9 |
+
|
| 10 |
+
[project.optional-dependencies]
|
| 11 |
+
vllm = ["vllm==0.30.0"]
|
| 12 |
+
test = ["pytest>=8"]
|
| 13 |
+
|
| 14 |
+
[project.scripts]
|
| 15 |
+
wald-serve = "wald_serve.server:main"
|
| 16 |
+
|
| 17 |
+
[build-system]
|
| 18 |
+
requires = ["setuptools>=68"]
|
| 19 |
+
build-backend = "setuptools.build_meta"
|
| 20 |
+
|
| 21 |
+
[tool.setuptools.packages.find]
|
| 22 |
+
where = ["src"]
|
server/src/wald_serve/__init__.py
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Wald-4B serving: TypeSafe `POST /v1/systemone` over vLLM (letter readout, effort gate, knockout, calibration)."""
|
| 2 |
+
__version__ = "0.1.0"
|
server/src/wald_serve/__main__.py
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from .server import main
|
| 2 |
+
|
| 3 |
+
raise SystemExit(main())
|
server/src/wald_serve/engine.py
ADDED
|
@@ -0,0 +1,285 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Reading one request: a letter read per question over a vLLM OpenAI-compatible server, the effort gate, the knockout
|
| 2 |
+
for more than 26 options, and the temperature table.
|
| 3 |
+
|
| 4 |
+
Per question (2..26 options):
|
| 5 |
+
1. one-pass read: the prompt's next-token distribution over the option letters, z_L = log(P("L") + P(" L")),
|
| 6 |
+
softmax over the question's letters (mode `A`);
|
| 7 |
+
2. if the effort policy's gate g > 0 and the one-pass max p < g: generate a thought after `Reasoning:` (at most
|
| 8 |
+
`budget` tokens, temperature 0.6 / top-p 0.95 / top-k 20, stop at `\\nAnswer`, seed = hash of the request and the
|
| 9 |
+
question id), then read the letters again after `\\nAnswer: (` (mode `B`). With k > 1 thoughts the answer is the
|
| 10 |
+
mean of the k post-thought distributions; thought 0 is the k = 1 thought;
|
| 11 |
+
3. the answer is tempered with the checkpoint's temperature table (bucket = question type x option count; table `A`
|
| 12 |
+
for one-pass answers, `B<budget>` for thought answers). Tempering changes confidence, never the argmax.
|
| 13 |
+
|
| 14 |
+
More than 26 options (knockout, mode `K`, one pass, no thought): the options are split, in their given order, into
|
| 15 |
+
ceil(n / 26) near-equal consecutive chunks read with the same prompt; the chunk winners are then read as one final
|
| 16 |
+
question; P(option) = P_final(its chunk) x P_chunk(option). This is one rule for every wide question.
|
| 17 |
+
|
| 18 |
+
Declared capacity limits: a prompt longer than max_model_len is refused with HTTP 422 "maximum context length" (never
|
| 19 |
+
truncated); a thought that would not fit falls back to the one-pass answer; at most 26 x 26 = 676 options per question
|
| 20 |
+
(the wire format allows 255).
|
| 21 |
+
"""
|
| 22 |
+
from __future__ import annotations
|
| 23 |
+
|
| 24 |
+
import hashlib
|
| 25 |
+
import json
|
| 26 |
+
import math
|
| 27 |
+
import time
|
| 28 |
+
import urllib.error
|
| 29 |
+
import urllib.request
|
| 30 |
+
from concurrent.futures import ThreadPoolExecutor
|
| 31 |
+
from pathlib import Path
|
| 32 |
+
|
| 33 |
+
from .prompt import CLOSE, CUE, LETTERS, STOP, THINK, question_prompt
|
| 34 |
+
from .wire import SystemOneRequest, to_record
|
| 35 |
+
|
| 36 |
+
BUCKETS = ((2, "2"), (4, "3-4"), (8, "5-8"), (math.inf, "9+"))
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
class Capacity(ValueError):
|
| 40 |
+
"""A declared capacity limit: answered with HTTP 422."""
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
# --- effort policies --------------------------------------------------------------------------------------------------
|
| 44 |
+
# gate: think when the one-pass max p < gate (0 = never, 1.01 = always); budget: max thought tokens; k: thoughts averaged.
|
| 45 |
+
POLICIES = {
|
| 46 |
+
"none": {"gate": 0.0, "budget": 512, "k": 1},
|
| 47 |
+
"low": {"gate": 0.5, "budget": 512, "k": 1},
|
| 48 |
+
"medium": {"gate": 0.7, "budget": 512, "k": 1},
|
| 49 |
+
"high": {"gate": 1.01, "budget": 512, "k": 1},
|
| 50 |
+
}
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
def policy(name: str) -> dict:
|
| 54 |
+
"""none / low / medium / high, or high-k<k> (always think, mean of k thoughts, 2 <= k <= 8)."""
|
| 55 |
+
if name in POLICIES:
|
| 56 |
+
return dict(POLICIES[name], name=name)
|
| 57 |
+
if name.startswith("high-k") and name[6:].isdigit() and 2 <= int(name[6:]) <= 8:
|
| 58 |
+
return {"gate": 1.01, "budget": 512, "k": int(name[6:]), "name": name}
|
| 59 |
+
raise ValueError(f"unknown effort {name!r}: none, low, medium, high or high-k<2..8>")
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
# --- temperature ------------------------------------------------------------------------------------------------------
|
| 63 |
+
def bucket_key(qtype: str, n: int) -> str:
|
| 64 |
+
return f"{qtype}|" + next(name for upper, name in BUCKETS if n <= upper)
|
| 65 |
+
|
| 66 |
+
|
| 67 |
+
def temperature(table, qtype: str, n: int) -> float:
|
| 68 |
+
if not table:
|
| 69 |
+
return 1.0
|
| 70 |
+
b = (table.get("buckets") or {}).get(bucket_key(qtype, n))
|
| 71 |
+
return float(b["temperature"]) if b else float(table.get("single", 1.0))
|
| 72 |
+
|
| 73 |
+
|
| 74 |
+
def temper(p, t: float):
|
| 75 |
+
if t == 1.0:
|
| 76 |
+
return list(p)
|
| 77 |
+
z = [math.log(max(x, 1e-12)) / t for x in p]
|
| 78 |
+
m = max(z)
|
| 79 |
+
e = [math.exp(v - m) for v in z]
|
| 80 |
+
s = sum(e)
|
| 81 |
+
return [v / s for v in e]
|
| 82 |
+
|
| 83 |
+
|
| 84 |
+
def load_tables(path, budget: int = 512) -> dict:
|
| 85 |
+
"""-> {"A": table, "B": table, "K": table}. A plain bucket table serves every mode; a gated table's thought entry is
|
| 86 |
+
B<budget> when present, else B."""
|
| 87 |
+
if not path:
|
| 88 |
+
return {}
|
| 89 |
+
t = json.loads(Path(path).read_text())
|
| 90 |
+
if "buckets" in t or "single" in t:
|
| 91 |
+
return {"A": t, "B": t, "K": t}
|
| 92 |
+
return {"A": t.get("A"), "B": t.get(f"B{budget}") or t.get("B"), "K": t.get("A")}
|
| 93 |
+
|
| 94 |
+
|
| 95 |
+
def hash_seed(s: str) -> int:
|
| 96 |
+
return int(hashlib.sha256(s.encode()).hexdigest()[:8], 16)
|
| 97 |
+
|
| 98 |
+
|
| 99 |
+
# --- vLLM client (stdlib; thread-safe, no shared mutable state after __init__) ----------------------------------------
|
| 100 |
+
class Client:
|
| 101 |
+
def __init__(self, endpoint: str, served: str, max_len: int, prompt_format: str = "plain", timeout: float = 900):
|
| 102 |
+
self.url, self.served, self.max_len, self.timeout = endpoint.rstrip("/"), served, max_len, timeout
|
| 103 |
+
self.prompt_format = prompt_format
|
| 104 |
+
self.letter_ids = []
|
| 105 |
+
for L in LETTERS:
|
| 106 |
+
ids = []
|
| 107 |
+
for v in (L, " " + L):
|
| 108 |
+
t = self.ids(v)
|
| 109 |
+
if len(t) == 1 and t[0] not in ids:
|
| 110 |
+
ids.append(t[0])
|
| 111 |
+
if not ids:
|
| 112 |
+
raise RuntimeError(f"no single-token encoding of letter {L!r}")
|
| 113 |
+
self.letter_ids.append(ids)
|
| 114 |
+
self.close = self.ids(CLOSE)
|
| 115 |
+
|
| 116 |
+
def post(self, path: str, body: dict, retries: int = 4):
|
| 117 |
+
last = None
|
| 118 |
+
for a in range(retries):
|
| 119 |
+
try:
|
| 120 |
+
req = urllib.request.Request(self.url + path, data=json.dumps(body).encode(), method="POST",
|
| 121 |
+
headers={"content-type": "application/json"})
|
| 122 |
+
with urllib.request.urlopen(req, timeout=self.timeout) as r:
|
| 123 |
+
return json.loads(r.read())
|
| 124 |
+
except urllib.error.HTTPError as e:
|
| 125 |
+
detail = e.read()[:400].decode("utf-8", "replace")
|
| 126 |
+
if e.code in (400, 413, 422):
|
| 127 |
+
if "maximum context length" in detail or "max_model_len" in detail or "too long" in detail:
|
| 128 |
+
raise Capacity(f"maximum context length: {detail[:300]}") from None
|
| 129 |
+
raise ValueError(f"{e.code}: {detail}") from None
|
| 130 |
+
last = RuntimeError(f"{e.code}: {detail}")
|
| 131 |
+
except (urllib.error.URLError, OSError) as e:
|
| 132 |
+
last = e
|
| 133 |
+
time.sleep(2 ** a)
|
| 134 |
+
raise RuntimeError(f"{path}: {last}")
|
| 135 |
+
|
| 136 |
+
def ids(self, text: str) -> list[int]:
|
| 137 |
+
return self.post("/tokenize", {"model": self.served, "prompt": text, "add_special_tokens": False})["tokens"]
|
| 138 |
+
|
| 139 |
+
def readout(self, ids: list[int], n: int):
|
| 140 |
+
"""-> (softmax over the first n letters, total letter mass)."""
|
| 141 |
+
if len(ids) + 1 > self.max_len:
|
| 142 |
+
raise Capacity(f"prompt of {len(ids)} tokens is longer than the maximum context length {self.max_len}")
|
| 143 |
+
want = [t for L in self.letter_ids[:n] for t in L]
|
| 144 |
+
body = {"model": self.served, "prompt": ids, "max_tokens": 1, "temperature": 0.0, "logprobs": 20,
|
| 145 |
+
"logprob_token_ids": want, "return_tokens_as_token_ids": True}
|
| 146 |
+
r = self.post("/v1/completions", body)
|
| 147 |
+
top = r["choices"][0]["logprobs"]["top_logprobs"][0]
|
| 148 |
+
lp = {int(k.split(":", 1)[1]): v for k, v in top.items() if k.startswith("token_id:") and v is not None and v > -9999}
|
| 149 |
+
floor = min(lp.values()) - 2.0 if lp else -30.0
|
| 150 |
+
z = [math.log(sum(math.exp(lp.get(t, floor)) for t in L)) for L in self.letter_ids[:n]]
|
| 151 |
+
m = max(z)
|
| 152 |
+
e = [math.exp(v - m) for v in z]
|
| 153 |
+
s = sum(e)
|
| 154 |
+
return [x / s for x in e], sum(math.exp(v) for v in z)
|
| 155 |
+
|
| 156 |
+
def generate(self, ids: list[int], max_tokens: int, seed: int, n: int = 1):
|
| 157 |
+
"""-> (text, completion tokens) for n = 1; (list of texts, total tokens) for n > 1."""
|
| 158 |
+
body = {"model": self.served, "prompt": ids, "max_tokens": max_tokens, "temperature": 0.6, "top_p": 0.95,
|
| 159 |
+
"top_k": 20, "seed": seed, "stop": [STOP], "include_stop_str_in_output": False}
|
| 160 |
+
if n > 1:
|
| 161 |
+
body["n"] = n
|
| 162 |
+
r = self.post("/v1/completions", body)
|
| 163 |
+
ntok = (r.get("usage") or {}).get("completion_tokens", 0)
|
| 164 |
+
if n > 1:
|
| 165 |
+
return [c.get("text") or "" for c in r["choices"]], ntok
|
| 166 |
+
c = r["choices"][0]
|
| 167 |
+
return c.get("text") or "", ntok
|
| 168 |
+
|
| 169 |
+
|
| 170 |
+
# --- one question -----------------------------------------------------------------------------------------------------
|
| 171 |
+
def chunks(n: int, size: int = 26) -> list[list[int]]:
|
| 172 |
+
k = math.ceil(n / size)
|
| 173 |
+
base, extra = divmod(n, k)
|
| 174 |
+
out, i = [], 0
|
| 175 |
+
for c in range(k):
|
| 176 |
+
m = base + (1 if c < extra else 0)
|
| 177 |
+
out.append(list(range(i, i + m)))
|
| 178 |
+
i += m
|
| 179 |
+
return out
|
| 180 |
+
|
| 181 |
+
|
| 182 |
+
def subset(q: dict, idx: list[int]) -> dict:
|
| 183 |
+
return {**q, "options": [q["options"][i] for i in idx], "option_texts": [q["option_texts"][i] for i in idx]}
|
| 184 |
+
|
| 185 |
+
|
| 186 |
+
def wide_read(n: int, read):
|
| 187 |
+
"""Knockout over n > 26 options, given read(option indices) -> probabilities over them."""
|
| 188 |
+
groups = chunks(n)
|
| 189 |
+
within, winners = [], []
|
| 190 |
+
for g in groups:
|
| 191 |
+
p = read(g)
|
| 192 |
+
within.append(p)
|
| 193 |
+
winners.append(g[max(range(len(g)), key=lambda j: p[j])])
|
| 194 |
+
top = read(winners)
|
| 195 |
+
out = [0.0] * n
|
| 196 |
+
for c, g in enumerate(groups):
|
| 197 |
+
for j, i in enumerate(g):
|
| 198 |
+
out[i] = top[c] * within[c][j]
|
| 199 |
+
s = sum(out)
|
| 200 |
+
return [x / s for x in out]
|
| 201 |
+
|
| 202 |
+
|
| 203 |
+
def read_question(cl: Client, state: str, q: dict, pol: dict, seed_text: str, usage: dict):
|
| 204 |
+
"""-> (probabilities in the question's option order, mode A | B | K)."""
|
| 205 |
+
n = len(q["options"])
|
| 206 |
+
if n <= len(LETTERS):
|
| 207 |
+
prompt = question_prompt(state, q, cl.prompt_format)
|
| 208 |
+
ida = cl.ids(prompt)
|
| 209 |
+
usage["input_tokens"] += len(ida)
|
| 210 |
+
p, _ = cl.readout(ida, n)
|
| 211 |
+
gate, budget, k = pol["gate"], pol["budget"], pol["k"]
|
| 212 |
+
if gate > 0 and n > 1 and max(p) < gate:
|
| 213 |
+
pre = cl.ids(prompt[: -len(CUE)] + THINK)
|
| 214 |
+
if len(pre) + budget + 64 <= cl.max_len: # else the one-pass answer stands
|
| 215 |
+
text, ntok = cl.generate(pre, budget, hash_seed(seed_text))
|
| 216 |
+
texts = [text]
|
| 217 |
+
usage["output_tokens"] += ntok
|
| 218 |
+
if k > 1:
|
| 219 |
+
more, ntok = cl.generate(pre, budget, hash_seed(seed_text + "#k"), n=k - 1)
|
| 220 |
+
texts += more
|
| 221 |
+
usage["output_tokens"] += ntok
|
| 222 |
+
reads = []
|
| 223 |
+
for text in texts:
|
| 224 |
+
body = text.strip()
|
| 225 |
+
ids = pre + (cl.ids(" " + body) if body else []) + cl.close
|
| 226 |
+
pb, _ = cl.readout(ids, n)
|
| 227 |
+
usage["input_tokens"] += len(ids)
|
| 228 |
+
reads.append(pb)
|
| 229 |
+
return [sum(r[i] for r in reads) / len(reads) for i in range(n)], "B"
|
| 230 |
+
return p, "A"
|
| 231 |
+
if len(chunks(n)) > len(LETTERS):
|
| 232 |
+
raise Capacity(f"{n} options: at most {len(LETTERS) ** 2} options per choice")
|
| 233 |
+
|
| 234 |
+
def read(idx):
|
| 235 |
+
ids = cl.ids(question_prompt(state, subset(q, idx), cl.prompt_format))
|
| 236 |
+
usage["input_tokens"] += len(ids)
|
| 237 |
+
return cl.readout(ids, len(idx))[0]
|
| 238 |
+
return wide_read(n, read), "K"
|
| 239 |
+
|
| 240 |
+
|
| 241 |
+
def seed_base(req: dict) -> str:
|
| 242 |
+
"""The thought seed covers the TypeSafe fields only (a per-request `effort` override does not change it)."""
|
| 243 |
+
return json.dumps({k: v for k, v in req.items() if k != "effort"}, sort_keys=True)
|
| 244 |
+
|
| 245 |
+
|
| 246 |
+
def answer(cl: Client, req: dict, pol: dict, tables: dict, workers: int = 8):
|
| 247 |
+
"""-> (answers in TypeSafe's wire format, usage). Questions are read concurrently (each question's read depends only
|
| 248 |
+
on the state, the question and its seed, so the order of reads does not change an answer)."""
|
| 249 |
+
rec, meta = to_record(SystemOneRequest.model_validate({k: v for k, v in req.items() if k != "effort"}))
|
| 250 |
+
base = seed_base(req)
|
| 251 |
+
jobs = []
|
| 252 |
+
for rq, m in zip(rec["questions"], meta):
|
| 253 |
+
q = {"id": m["id"], "type": rq["qtype"], "instructions": rq["instr"], "options": rq["options"],
|
| 254 |
+
"option_texts": rq["options"]}
|
| 255 |
+
jobs.append((q, m))
|
| 256 |
+
|
| 257 |
+
def one(job):
|
| 258 |
+
q, m = job
|
| 259 |
+
usage = {"input_tokens": 0, "output_tokens": 0}
|
| 260 |
+
p, mode = read_question(cl, rec["state"], q, pol, base + m["id"], usage)
|
| 261 |
+
return p, mode, usage
|
| 262 |
+
|
| 263 |
+
if workers > 1 and len(jobs) > 1:
|
| 264 |
+
with ThreadPoolExecutor(max_workers=min(workers, len(jobs))) as ex:
|
| 265 |
+
results = list(ex.map(one, jobs))
|
| 266 |
+
else:
|
| 267 |
+
results = [one(j) for j in jobs]
|
| 268 |
+
|
| 269 |
+
answers, usage = {}, {"input_tokens": 0, "output_tokens": 0}
|
| 270 |
+
for (q, m), (p, mode, u) in zip(jobs, results):
|
| 271 |
+
usage["input_tokens"] += u["input_tokens"]
|
| 272 |
+
usage["output_tokens"] += u["output_tokens"]
|
| 273 |
+
keys = m["keys"]
|
| 274 |
+
p = temper(p, temperature(tables.get(mode), m["type"], len(keys)))
|
| 275 |
+
j = max(range(len(p)), key=lambda i: p[i])
|
| 276 |
+
if m["type"] == "noul":
|
| 277 |
+
answers[m["id"]] = {"type": "noul", "noul": p[keys.index("true")], "mode": mode}
|
| 278 |
+
elif m["type"] == "choice":
|
| 279 |
+
K = len(p)
|
| 280 |
+
answers[m["id"]] = {"type": "choice", "choice": keys[j], "probabilities": dict(zip(keys, p)),
|
| 281 |
+
"confidence": 1.0 if K == 1 else (p[j] - 1 / K) / (1 - 1 / K), "mode": mode}
|
| 282 |
+
else:
|
| 283 |
+
answers[m["id"]] = {"type": "score", "score": sum(i * v for i, v in enumerate(p)),
|
| 284 |
+
"probabilities": {str(i): v for i, v in enumerate(p)}, "confidence": p[j], "mode": mode}
|
| 285 |
+
return answers, usage
|
server/src/wald_serve/prompt.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""The plain letter prompt the model reads, one prompt per question. No chat template, no special tokens:
|
| 2 |
+
|
| 3 |
+
State:
|
| 4 |
+
{state}
|
| 5 |
+
|
| 6 |
+
Question: {instructions}
|
| 7 |
+
(A) {option 1}
|
| 8 |
+
(B) {option 2}
|
| 9 |
+
Answer: (
|
| 10 |
+
|
| 11 |
+
The answer is read at the last position from the LM head's option-letter logits (engine.Client.readout). When the model
|
| 12 |
+
thinks, the final `Answer: (` is replaced by `Reasoning:`, a short thought is generated, and the letters are read again
|
| 13 |
+
after `\\nAnswer: (`.
|
| 14 |
+
|
| 15 |
+
Prompt formats (one general rewrite applied to every question, never chosen per benchmark):
|
| 16 |
+
plain the layout above
|
| 17 |
+
repeat_state_plain the state written twice, the second copy introduced by STATE_REPEAT. The sentence is taken
|
| 18 |
+
verbatim from featherless-ai/simple-jev @ dae340e (Apache-2.0; see NOTICE).
|
| 19 |
+
"""
|
| 20 |
+
from __future__ import annotations
|
| 21 |
+
|
| 22 |
+
import re
|
| 23 |
+
|
| 24 |
+
LETTERS = "ABCDEFGHIJKLMNOPQRSTUVWXYZ"
|
| 25 |
+
CUE = "Answer: ("
|
| 26 |
+
THINK = "Reasoning:"
|
| 27 |
+
CLOSE = "\nAnswer: ("
|
| 28 |
+
STOP = "\nAnswer"
|
| 29 |
+
STATE_TAIL = "Question:"
|
| 30 |
+
|
| 31 |
+
# featherless-ai/simple-jev @ dae340e, hf-server/hf_prompt_policies.py (Apache-2.0), verbatim.
|
| 32 |
+
STATE_REPEAT = ("\n\nRead the same context again before answering. This is a repeated copy, not additional events or "
|
| 33 |
+
"independent evidence:\n")
|
| 34 |
+
PROMPT_FORMATS = ("plain", "repeat_state_plain")
|
| 35 |
+
|
| 36 |
+
_SPECIAL = re.compile(r"<\|([A-Za-z0-9_]+)\|>") # caller text never forms a <|special|> token
|
| 37 |
+
_POSITIONAL = re.compile(r"[a-z]|o[0-9]+") # choice keys that only number the options
|
| 38 |
+
_SLUG = re.compile(r"[a-z0-9][a-z0-9_.-]{0,39}")
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
def plain(text) -> str:
|
| 42 |
+
return _SPECIAL.sub(r"<¦\1¦>", str(text))
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
def choice_keys(options) -> list[str]:
|
| 46 |
+
"""The options themselves when they are all distinct short slugs, else a, b, c, ... (o1, o2, ... past 26)."""
|
| 47 |
+
opts = [str(o) for o in options]
|
| 48 |
+
if opts and len(set(opts)) == len(opts) and all(isinstance(o, str) and _SLUG.fullmatch(o) for o in options):
|
| 49 |
+
return opts
|
| 50 |
+
return [chr(ord("a") + i) if len(opts) <= 26 else f"o{i + 1}" for i in range(len(opts))]
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
def raw_options(qtype: str, keys, texts) -> list[str]:
|
| 54 |
+
"""The option strings after their letter: a choice question whose keys are all positional (a, b, ... / o27) drops
|
| 55 |
+
the `a: ` prefix, since the letter replaces it."""
|
| 56 |
+
if qtype == "choice":
|
| 57 |
+
keys = list(keys)
|
| 58 |
+
if all(_POSITIONAL.fullmatch(k) for k in keys):
|
| 59 |
+
return [t[len(k) + 2:] if t.startswith(f"{k}: ") else t for k, t in zip(keys, texts)]
|
| 60 |
+
return list(texts)
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
def state_text(state: str) -> str:
|
| 64 |
+
"""`State:\\n{state}\\n\\nQuestion:` (just `Question:` for a blank state)."""
|
| 65 |
+
return (f"State:\n{plain(state)}\n\n" if str(state).strip() else "") + STATE_TAIL
|
| 66 |
+
|
| 67 |
+
|
| 68 |
+
def branch_text(q: dict) -> str:
|
| 69 |
+
"""The rest of one question's prompt after `Question:`. q = {"type", "instructions", "options", "option_texts"}."""
|
| 70 |
+
texts = [str(o) for o in q["option_texts"]]
|
| 71 |
+
if len(texts) > len(LETTERS):
|
| 72 |
+
raise ValueError(f"{len(texts)} options: one letter read handles at most {len(LETTERS)}")
|
| 73 |
+
keys = choice_keys(q["options"]) if q["type"] == "choice" else None
|
| 74 |
+
out = f" {plain(q['instructions'])}"
|
| 75 |
+
for j, t in enumerate(raw_options(q["type"], keys, texts)):
|
| 76 |
+
out += f"\n({LETTERS[j]}) {plain(t)}"
|
| 77 |
+
return out + "\n" + CUE
|
| 78 |
+
|
| 79 |
+
|
| 80 |
+
def format_prompt(prompt: str, fmt: str) -> str:
|
| 81 |
+
"""Apply a prompt format to one question's plain prompt."""
|
| 82 |
+
if fmt == "plain":
|
| 83 |
+
return prompt
|
| 84 |
+
if not prompt.endswith(CUE):
|
| 85 |
+
raise ValueError("prompt does not end with the answer cue")
|
| 86 |
+
if fmt == "repeat_state_plain":
|
| 87 |
+
if prompt.startswith("State:\n") and "\n\nQuestion:" in prompt:
|
| 88 |
+
i = prompt.index("\n\nQuestion:")
|
| 89 |
+
state = prompt[len("State:\n"):i]
|
| 90 |
+
return "State:\n" + state + STATE_REPEAT + state + prompt[i:]
|
| 91 |
+
return prompt # blank state: nothing to repeat
|
| 92 |
+
raise ValueError(f"unknown prompt format {fmt!r}; one of {PROMPT_FORMATS}")
|
| 93 |
+
|
| 94 |
+
|
| 95 |
+
def question_prompt(state: str, q: dict, fmt: str = "plain") -> str:
|
| 96 |
+
"""The one-pass prompt of one question under a rendered state."""
|
| 97 |
+
return format_prompt(state_text(state) + branch_text(q), fmt)
|
server/src/wald_serve/server.py
ADDED
|
@@ -0,0 +1,152 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""`wald-serve`: the TypeSafe `POST /v1/systemone` server for Wald-4B.
|
| 2 |
+
|
| 3 |
+
One command starts vLLM on the weights (a loopback-only sidecar) and this front-end in front of it:
|
| 4 |
+
|
| 5 |
+
wald-serve --model /path/to/Wald-4B --port 8000
|
| 6 |
+
|
| 7 |
+
or attaches to a vLLM server that is already running:
|
| 8 |
+
|
| 9 |
+
wald-serve --vllm http://127.0.0.1:8011 --served wald --temperature /path/to/temperature.json --port 8000
|
| 10 |
+
|
| 11 |
+
Defaults come from `<model>/serving.json` when present (`effort`, `prompt_format`, `max_model_len`), else the built-in
|
| 12 |
+
ones below; command-line flags override both. The response carries TypeSafe's answer keys plus `mode` (A = one pass,
|
| 13 |
+
B = after a thought, K = knockout) and `usage`.
|
| 14 |
+
|
| 15 |
+
Endpoints: POST /v1/systemone (also POST /), GET /health, GET /v1/models.
|
| 16 |
+
"""
|
| 17 |
+
from __future__ import annotations
|
| 18 |
+
|
| 19 |
+
import argparse
|
| 20 |
+
import atexit
|
| 21 |
+
import json
|
| 22 |
+
import os
|
| 23 |
+
import shlex
|
| 24 |
+
import signal
|
| 25 |
+
import subprocess
|
| 26 |
+
import sys
|
| 27 |
+
import time
|
| 28 |
+
import urllib.request
|
| 29 |
+
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
| 30 |
+
from pathlib import Path
|
| 31 |
+
|
| 32 |
+
from . import __version__
|
| 33 |
+
from .engine import Capacity, Client, answer, load_tables, policy
|
| 34 |
+
from .prompt import PROMPT_FORMATS
|
| 35 |
+
|
| 36 |
+
DEFAULTS = {"effort": "medium", "prompt_format": "plain", "max_model_len": 131072}
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
def read_serving(model_dir) -> dict:
|
| 40 |
+
p = Path(model_dir) / "serving.json" if model_dir else None
|
| 41 |
+
if p and p.is_file():
|
| 42 |
+
return {k: v for k, v in json.loads(p.read_text()).items() if k in DEFAULTS or k == "temperature"}
|
| 43 |
+
return {}
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
def launch_vllm(model: str, served: str, port: int, max_len: int, mem: float, extra: str) -> subprocess.Popen:
|
| 47 |
+
cmd = [sys.executable, "-m", "vllm.entrypoints.openai.api_server", "--model", model, "--served-model-name", served,
|
| 48 |
+
"--host", "127.0.0.1", "--port", str(port), "--max-model-len", str(max_len), "--gpu-memory-utilization",
|
| 49 |
+
str(mem), "--max-num-seqs", "256", "--seed", "0", *shlex.split(extra)]
|
| 50 |
+
print(json.dumps({"launching": cmd}), flush=True)
|
| 51 |
+
proc = subprocess.Popen(cmd, start_new_session=True)
|
| 52 |
+
|
| 53 |
+
def stop():
|
| 54 |
+
if proc.poll() is None:
|
| 55 |
+
os.killpg(proc.pid, signal.SIGTERM)
|
| 56 |
+
atexit.register(stop)
|
| 57 |
+
signal.signal(signal.SIGTERM, lambda *_: sys.exit(0))
|
| 58 |
+
url = f"http://127.0.0.1:{port}/health"
|
| 59 |
+
for _ in range(1200):
|
| 60 |
+
if proc.poll() is not None:
|
| 61 |
+
raise SystemExit(f"vLLM exited with code {proc.returncode}")
|
| 62 |
+
try:
|
| 63 |
+
with urllib.request.urlopen(url, timeout=5):
|
| 64 |
+
return proc
|
| 65 |
+
except OSError:
|
| 66 |
+
time.sleep(2)
|
| 67 |
+
raise SystemExit("vLLM did not become healthy within 40 minutes")
|
| 68 |
+
|
| 69 |
+
|
| 70 |
+
def make_handler(cl: Client, pol: dict, tables: dict, info: dict, workers: int):
|
| 71 |
+
class H(BaseHTTPRequestHandler):
|
| 72 |
+
protocol_version = "HTTP/1.1"
|
| 73 |
+
|
| 74 |
+
def log_message(self, *args):
|
| 75 |
+
pass
|
| 76 |
+
|
| 77 |
+
def send(self, code, obj):
|
| 78 |
+
b = json.dumps(obj).encode()
|
| 79 |
+
self.send_response(code)
|
| 80 |
+
self.send_header("content-type", "application/json")
|
| 81 |
+
self.send_header("content-length", str(len(b)))
|
| 82 |
+
self.end_headers()
|
| 83 |
+
self.wfile.write(b)
|
| 84 |
+
|
| 85 |
+
def do_GET(self):
|
| 86 |
+
if self.path.startswith("/v1/models"):
|
| 87 |
+
return self.send(200, {"object": "list", "data": [{"id": info["model"], "object": "model"}]})
|
| 88 |
+
self.send(200, {"ok": True, **info})
|
| 89 |
+
|
| 90 |
+
def do_POST(self):
|
| 91 |
+
try:
|
| 92 |
+
req = json.loads(self.rfile.read(int(self.headers.get("content-length", 0))))
|
| 93 |
+
p = policy(req["effort"]) if isinstance(req, dict) and req.get("effort") else pol
|
| 94 |
+
except Exception as e: # noqa: BLE001
|
| 95 |
+
return self.send(400, {"error": f"{type(e).__name__}: {e}"[:400]})
|
| 96 |
+
t0 = time.perf_counter()
|
| 97 |
+
try:
|
| 98 |
+
ans, usage = answer(cl, req, p, tables, workers)
|
| 99 |
+
except Capacity as e:
|
| 100 |
+
return self.send(422, {"error": str(e)[:400]})
|
| 101 |
+
except Exception as e: # noqa: BLE001
|
| 102 |
+
return self.send(400 if isinstance(e, ValueError) else 500, {"error": f"{type(e).__name__}: {e}"[:400]})
|
| 103 |
+
self.send(200, {"model": info["model"], "answers": ans, "usage": usage, "latency_s": time.perf_counter() - t0})
|
| 104 |
+
return H
|
| 105 |
+
|
| 106 |
+
|
| 107 |
+
def main(argv=None):
|
| 108 |
+
ap = argparse.ArgumentParser(prog="wald-serve", description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
| 109 |
+
src = ap.add_mutually_exclusive_group(required=True)
|
| 110 |
+
src.add_argument("--model", help="weights directory: start vLLM on it (serving.json / temperature.json read from it)")
|
| 111 |
+
src.add_argument("--vllm", help="URL of a running vLLM OpenAI-compatible server")
|
| 112 |
+
ap.add_argument("--served", default="wald", help="vLLM served model name")
|
| 113 |
+
ap.add_argument("--model-name", default="wald-4b", help="name reported in responses")
|
| 114 |
+
ap.add_argument("--effort", default=None, help="none | low | medium | high | high-k<k> (default: serving.json, else medium)")
|
| 115 |
+
ap.add_argument("--prompt-format", choices=PROMPT_FORMATS, default=None, help="default: serving.json, else plain")
|
| 116 |
+
ap.add_argument("--temperature", default=None, help="temperature table (default: <model>/temperature.json)")
|
| 117 |
+
ap.add_argument("--max-model-len", type=int, default=None, help="context limit in tokens (default: serving.json, else 131072)")
|
| 118 |
+
ap.add_argument("--question-workers", type=int, default=8, help="questions of one request read concurrently (1 = serial)")
|
| 119 |
+
ap.add_argument("--vllm-port", type=int, default=8011)
|
| 120 |
+
ap.add_argument("--gpu-memory-utilization", type=float, default=0.90)
|
| 121 |
+
ap.add_argument("--vllm-args", default="", help="extra `vllm serve` arguments, e.g. \"--quantization fp8\"")
|
| 122 |
+
ap.add_argument("--host", default="0.0.0.0")
|
| 123 |
+
ap.add_argument("--port", type=int, default=8000)
|
| 124 |
+
a = ap.parse_args(argv)
|
| 125 |
+
|
| 126 |
+
cfg = {**DEFAULTS, **read_serving(a.model)}
|
| 127 |
+
effort = a.effort or cfg["effort"]
|
| 128 |
+
fmt = a.prompt_format or cfg["prompt_format"]
|
| 129 |
+
max_len = a.max_model_len or int(cfg["max_model_len"])
|
| 130 |
+
pol = policy(effort)
|
| 131 |
+
temps = a.temperature or (str(Path(a.model) / cfg.get("temperature", "temperature.json")) if a.model else None)
|
| 132 |
+
if temps and not Path(temps).is_file():
|
| 133 |
+
raise SystemExit(f"temperature table not found: {temps}")
|
| 134 |
+
tables = load_tables(temps, pol["budget"])
|
| 135 |
+
|
| 136 |
+
if a.model:
|
| 137 |
+
launch_vllm(a.model, a.served, a.vllm_port, max_len, a.gpu_memory_utilization, a.vllm_args)
|
| 138 |
+
endpoint = f"http://127.0.0.1:{a.vllm_port}"
|
| 139 |
+
else:
|
| 140 |
+
endpoint = a.vllm
|
| 141 |
+
cl = Client(endpoint, a.served, max_len, prompt_format=fmt)
|
| 142 |
+
info = {"model": a.model_name, "version": __version__, "effort": pol["name"], "gate": pol["gate"],
|
| 143 |
+
"budget": pol["budget"], "think_k": pol["k"], "prompt_format": fmt, "max_model_len": max_len,
|
| 144 |
+
"wide": "knockout", "max_options": 676, "temperature": bool(tables)}
|
| 145 |
+
ThreadingHTTPServer.daemon_threads = True
|
| 146 |
+
httpd = ThreadingHTTPServer((a.host, a.port), make_handler(cl, pol, tables, info, a.question_workers))
|
| 147 |
+
print(json.dumps({"serving": info, "port": a.port}), flush=True)
|
| 148 |
+
httpd.serve_forever()
|
| 149 |
+
|
| 150 |
+
|
| 151 |
+
if __name__ == "__main__":
|
| 152 |
+
sys.exit(main())
|
server/src/wald_serve/wire.py
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""The TypeSafe `POST /v1/systemone` request shape and how a request becomes the text the model reads.
|
| 2 |
+
|
| 3 |
+
A request is a `state` (string, object, array or null) plus `questions` (id -> noul / choice / score). Every question is
|
| 4 |
+
answered with a probability for each of its options:
|
| 5 |
+
|
| 6 |
+
noul options [no, yes] answer {"type": "noul", "noul": p(yes)}
|
| 7 |
+
choice one option per criteria key answer {"type": "choice", "choice": key, "probabilities": {...}}
|
| 8 |
+
score one option per ordered level answer {"type": "score", "score": E[level], "probabilities": {...}}
|
| 9 |
+
|
| 10 |
+
`render` and `to_record` are adapted from Kev (github.com/jaredpalmer/kev, Apache-2.0; see NOTICE): the state and every
|
| 11 |
+
option are flattened to text exactly as the model saw them in training.
|
| 12 |
+
"""
|
| 13 |
+
from __future__ import annotations
|
| 14 |
+
|
| 15 |
+
from typing import Any, Literal, Union
|
| 16 |
+
|
| 17 |
+
from pydantic import BaseModel, Field, model_validator
|
| 18 |
+
|
| 19 |
+
JSONContent = Union[str, dict, list, int, float, bool, None]
|
| 20 |
+
MAX_OPTIONS = 255
|
| 21 |
+
|
| 22 |
+
|
| 23 |
+
class Noul(BaseModel):
|
| 24 |
+
type: Literal["noul"]
|
| 25 |
+
instructions: JSONContent = None
|
| 26 |
+
criteria: dict[str, JSONContent] | None = None
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
class Choice(BaseModel):
|
| 30 |
+
type: Literal["choice"]
|
| 31 |
+
instructions: JSONContent = None
|
| 32 |
+
criteria: dict[str, JSONContent]
|
| 33 |
+
|
| 34 |
+
@model_validator(mode="after")
|
| 35 |
+
def _check(self):
|
| 36 |
+
if not 1 <= len(self.criteria) <= MAX_OPTIONS:
|
| 37 |
+
raise ValueError(f"criteria must have 1..{MAX_OPTIONS} options")
|
| 38 |
+
return self
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
class Score(BaseModel):
|
| 42 |
+
type: Literal["score"]
|
| 43 |
+
instructions: JSONContent = None
|
| 44 |
+
criteria: list[JSONContent] = Field(min_length=1, max_length=MAX_OPTIONS)
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
Question = Union[Noul, Choice, Score]
|
| 48 |
+
|
| 49 |
+
|
| 50 |
+
class SystemOneRequest(BaseModel):
|
| 51 |
+
state: JSONContent
|
| 52 |
+
model: str = "wald"
|
| 53 |
+
questions: dict[str, Question] = Field(min_length=1)
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
def render(v: JSONContent, indent: int = 0) -> str:
|
| 57 |
+
"""Flatten str | object | array into text. Field names are kept as labels."""
|
| 58 |
+
pad = " " * indent
|
| 59 |
+
if v is None:
|
| 60 |
+
return ""
|
| 61 |
+
if isinstance(v, (str, int, float, bool)):
|
| 62 |
+
return str(v)
|
| 63 |
+
if isinstance(v, list):
|
| 64 |
+
return "\n".join(f"{pad}- {render(x, indent + 1).lstrip()}" for x in v)
|
| 65 |
+
return "\n".join(f"{pad}{k}:\n{render(x, indent + 1)}" if isinstance(x, (dict, list)) else f"{pad}{k}: {render(x)}"
|
| 66 |
+
for k, x in v.items())
|
| 67 |
+
|
| 68 |
+
|
| 69 |
+
def option_text(name: str, desc: JSONContent) -> str:
|
| 70 |
+
return name if desc is None or desc == "" else f"{name}: {render(desc)}"
|
| 71 |
+
|
| 72 |
+
|
| 73 |
+
def question_keys(qtype: str, criteria) -> list[str]:
|
| 74 |
+
"""The keys a question's probabilities are reported under, in option order."""
|
| 75 |
+
if qtype == "choice":
|
| 76 |
+
return list(criteria)
|
| 77 |
+
if qtype == "noul":
|
| 78 |
+
return ["false", "true"]
|
| 79 |
+
return [str(i) for i in range(len(criteria))]
|
| 80 |
+
|
| 81 |
+
|
| 82 |
+
def to_record(req: SystemOneRequest) -> tuple[dict[str, Any], list[dict[str, Any]]]:
|
| 83 |
+
"""-> ({"state": text, "questions": [{"instr", "options", "qtype", "keys"}]}, [{"id", "type", "keys"}])."""
|
| 84 |
+
qs, meta = [], []
|
| 85 |
+
for qid, q in req.questions.items():
|
| 86 |
+
m = {"id": qid, "type": q.type, "keys": question_keys(q.type, q.criteria)}
|
| 87 |
+
if q.type == "noul":
|
| 88 |
+
c = q.criteria or {}
|
| 89 |
+
opts = [option_text("no", c.get("false")), option_text("yes", c.get("true"))]
|
| 90 |
+
elif q.type == "choice":
|
| 91 |
+
opts = [option_text(k, v) for k, v in q.criteria.items()]
|
| 92 |
+
else:
|
| 93 |
+
opts = [render(x) for x in q.criteria]
|
| 94 |
+
qs.append({"instr": render(q.instructions), "options": opts, "qtype": q.type, "keys": m["keys"]})
|
| 95 |
+
meta.append(m)
|
| 96 |
+
return {"state": render(req.state), "questions": qs}, meta
|
server/tests/fakevllm.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""A stand-in for vLLM's OpenAI-compatible server, for tests without a model or a GPU.
|
| 2 |
+
|
| 3 |
+
Tokenizer: one token per character (so "A" is one token and " A" is two). Completions: the logprob of every requested
|
| 4 |
+
token id is a deterministic function of the prompt ids, so a read is reproducible; generation returns a short fixed
|
| 5 |
+
thought per seed. A prompt longer than `max_len` answers 400 "maximum context length", as vLLM does.
|
| 6 |
+
"""
|
| 7 |
+
from __future__ import annotations
|
| 8 |
+
|
| 9 |
+
import hashlib
|
| 10 |
+
import json
|
| 11 |
+
import threading
|
| 12 |
+
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
def _lp(prompt, tid):
|
| 16 |
+
h = hashlib.sha256(json.dumps([prompt[-64:], len(prompt), tid]).encode()).digest()
|
| 17 |
+
return -(int.from_bytes(h[:4], "big") % 4000) / 1000.0
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
class FakeVLLM:
|
| 21 |
+
def __init__(self, max_len=100_000):
|
| 22 |
+
self.max_len = max_len
|
| 23 |
+
self.calls = {"tokenize": 0, "read": 0, "generate": 0}
|
| 24 |
+
fake = self
|
| 25 |
+
|
| 26 |
+
class H(BaseHTTPRequestHandler):
|
| 27 |
+
protocol_version = "HTTP/1.1"
|
| 28 |
+
|
| 29 |
+
def log_message(self, *a):
|
| 30 |
+
pass
|
| 31 |
+
|
| 32 |
+
def send(self, code, obj):
|
| 33 |
+
b = json.dumps(obj).encode()
|
| 34 |
+
self.send_response(code)
|
| 35 |
+
self.send_header("content-type", "application/json")
|
| 36 |
+
self.send_header("content-length", str(len(b)))
|
| 37 |
+
self.end_headers()
|
| 38 |
+
self.wfile.write(b)
|
| 39 |
+
|
| 40 |
+
def do_GET(self):
|
| 41 |
+
self.send(200, {"ok": True})
|
| 42 |
+
|
| 43 |
+
def do_POST(self):
|
| 44 |
+
body = json.loads(self.rfile.read(int(self.headers["content-length"])))
|
| 45 |
+
if self.path == "/tokenize":
|
| 46 |
+
fake.calls["tokenize"] += 1
|
| 47 |
+
return self.send(200, {"tokens": [ord(c) for c in body["prompt"]]})
|
| 48 |
+
prompt = body["prompt"]
|
| 49 |
+
if len(prompt) + body["max_tokens"] > fake.max_len:
|
| 50 |
+
return self.send(400, {"message": f"This model's maximum context length is {fake.max_len} tokens."})
|
| 51 |
+
if body["max_tokens"] == 1:
|
| 52 |
+
fake.calls["read"] += 1
|
| 53 |
+
top = {f"token_id:{t}": _lp(prompt, t) for t in body["logprob_token_ids"]}
|
| 54 |
+
return self.send(200, {"choices": [{"logprobs": {"top_logprobs": [top]}}]})
|
| 55 |
+
fake.calls["generate"] += 1
|
| 56 |
+
n = body.get("n", 1)
|
| 57 |
+
texts = [f" thought {body['seed'] % 997} #{i}: the evidence points one way." for i in range(n)]
|
| 58 |
+
return self.send(200, {"choices": [{"text": t} for t in texts], "usage": {"completion_tokens": 9 * n}})
|
| 59 |
+
|
| 60 |
+
self.httpd = ThreadingHTTPServer(("127.0.0.1", 0), H)
|
| 61 |
+
self.httpd.daemon_threads = True
|
| 62 |
+
self.url = f"http://127.0.0.1:{self.httpd.server_address[1]}"
|
| 63 |
+
threading.Thread(target=self.httpd.serve_forever, daemon=True).start()
|
| 64 |
+
|
| 65 |
+
def stop(self):
|
| 66 |
+
self.httpd.shutdown()
|
server/tests/test_parity.py
ADDED
|
@@ -0,0 +1,102 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Parity with the reference implementation the Decision Index numbers were measured with.
|
| 2 |
+
|
| 3 |
+
Skipped unless WALD_PARITY_SRC points at a checkout that provides `eval.systemone_vllm` (with its `kev` and `midtrain`
|
| 4 |
+
imports on PYTHONPATH). Checks, on generated requests, that this package builds byte-identical prompts and returns
|
| 5 |
+
identical answers against the same (fake) vLLM for every effort policy and both prompt formats.
|
| 6 |
+
"""
|
| 7 |
+
from __future__ import annotations
|
| 8 |
+
|
| 9 |
+
import os
|
| 10 |
+
import random
|
| 11 |
+
|
| 12 |
+
import pytest
|
| 13 |
+
|
| 14 |
+
from fakevllm import FakeVLLM
|
| 15 |
+
from wald_serve import engine
|
| 16 |
+
from wald_serve.prompt import question_prompt
|
| 17 |
+
from wald_serve.wire import SystemOneRequest, to_record
|
| 18 |
+
|
| 19 |
+
pytestmark = pytest.mark.skipif(not os.environ.get("WALD_PARITY_SRC"), reason="reference implementation not available")
|
| 20 |
+
|
| 21 |
+
WORDS = "the a order refund card payment late missing charged twice store policy item box user tool call ask".split()
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
def text(rng, n):
|
| 25 |
+
return " ".join(rng.choice(WORDS) for _ in range(n))
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
def gen_request(rng):
|
| 29 |
+
kind = rng.random()
|
| 30 |
+
if kind < 0.25:
|
| 31 |
+
state = ""
|
| 32 |
+
elif kind < 0.6:
|
| 33 |
+
state = text(rng, rng.randint(3, 60))
|
| 34 |
+
else:
|
| 35 |
+
state = {"ticket": text(rng, 20), "meta": {"n": rng.randint(1, 9), "tags": [text(rng, 2), text(rng, 3)]}}
|
| 36 |
+
qs = {}
|
| 37 |
+
for i in range(rng.randint(1, 5)):
|
| 38 |
+
t = rng.choice(["choice", "choice", "noul", "score"])
|
| 39 |
+
if t == "noul":
|
| 40 |
+
qs[f"q{i}"] = {"type": "noul", "instructions": text(rng, 6)}
|
| 41 |
+
elif t == "score":
|
| 42 |
+
qs[f"q{i}"] = {"type": "score", "instructions": text(rng, 5), "criteria": [text(rng, 2) for _ in range(rng.randint(2, 7))]}
|
| 43 |
+
else:
|
| 44 |
+
n = rng.choice([2, 3, 4, 9, 26, 27, 53, 77])
|
| 45 |
+
style = rng.random()
|
| 46 |
+
if style < 0.33:
|
| 47 |
+
crit = {chr(97 + j) if n <= 26 else f"o{j + 1}": text(rng, 3) for j in range(n)}
|
| 48 |
+
elif style < 0.66:
|
| 49 |
+
crit = {f"label_{j}": None for j in range(n)}
|
| 50 |
+
else:
|
| 51 |
+
crit = {f"Option {j}": text(rng, 4) for j in range(n)}
|
| 52 |
+
qs[f"q{i}"] = {"type": "choice", "instructions": text(rng, 7), "criteria": crit}
|
| 53 |
+
return {"state": state, "questions": qs}
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
@pytest.fixture(scope="module")
|
| 57 |
+
def fake():
|
| 58 |
+
f = FakeVLLM()
|
| 59 |
+
yield f
|
| 60 |
+
f.stop()
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
def test_prompts_identical():
|
| 64 |
+
from eval.systemone_vllm import format_prompt as ref_format
|
| 65 |
+
from kev.api import SystemOneRequest as RefReq, to_record as ref_record
|
| 66 |
+
from midtrain.letter import question_prompt as ref_prompt
|
| 67 |
+
rng = random.Random(7)
|
| 68 |
+
n = 0
|
| 69 |
+
for _ in range(300):
|
| 70 |
+
req = gen_request(rng)
|
| 71 |
+
rec, _ = to_record(SystemOneRequest.model_validate(req))
|
| 72 |
+
rrec, _ = ref_record(RefReq.model_validate(req))
|
| 73 |
+
assert rec["state"] == rrec["state"]
|
| 74 |
+
for rq, rrq in zip(rec["questions"], rrec["questions"]):
|
| 75 |
+
q = {"type": rq["qtype"], "instructions": rq["instr"], "options": rq["options"], "option_texts": rq["options"]}
|
| 76 |
+
if len(q["options"]) > 26:
|
| 77 |
+
continue
|
| 78 |
+
for fmt in ("plain", "repeat_state_plain"):
|
| 79 |
+
assert question_prompt(rec["state"], q, fmt) == ref_format(ref_prompt(rrec["state"], q, "paren", "letter"), fmt)
|
| 80 |
+
n += 1
|
| 81 |
+
assert n > 500
|
| 82 |
+
|
| 83 |
+
|
| 84 |
+
@pytest.mark.parametrize("effort,gate,k", [("none", 0.0, 1), ("low", 0.5, 1), ("medium", 0.7, 1), ("high", 1.01, 1),
|
| 85 |
+
("high-k4", 1.01, 4)])
|
| 86 |
+
@pytest.mark.parametrize("fmt", ["plain", "repeat_state_plain"])
|
| 87 |
+
def test_answers_identical(fake, effort, gate, k, fmt):
|
| 88 |
+
from eval import systemone_vllm as ref
|
| 89 |
+
table = {"A": {"single": 1.4, "buckets": {"choice|2": {"temperature": 1.1}, "noul|2": {"temperature": 1.3}}},
|
| 90 |
+
"B512": {"single": 1.8, "buckets": {"choice|3-4": {"temperature": 2.2}}}}
|
| 91 |
+
tables = {"A": table["A"], "B": table["B512"], "K": table["A"]}
|
| 92 |
+
rcl = ref.Client(fake.url, "wald", 100_000)
|
| 93 |
+
rcl.wide, rcl.template, rcl.prompt_format, rcl.topk = "knockout", "paren", fmt, 0
|
| 94 |
+
cl = engine.Client(fake.url, "wald", 100_000, prompt_format=fmt)
|
| 95 |
+
rng = random.Random(11)
|
| 96 |
+
for _ in range(25):
|
| 97 |
+
req = gen_request(rng)
|
| 98 |
+
want, wu = ref.answer(rcl, req, gate, 512, tables, k)
|
| 99 |
+
got, gu = engine.answer(cl, req, engine.policy(effort), tables)
|
| 100 |
+
for a in want.values():
|
| 101 |
+
a.pop("one_pass", None)
|
| 102 |
+
assert got == want and gu == wu
|
server/tests/test_server.py
ADDED
|
@@ -0,0 +1,251 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Unit and end-to-end tests with a fake vLLM (no model, no GPU)."""
|
| 2 |
+
from __future__ import annotations
|
| 3 |
+
|
| 4 |
+
import json
|
| 5 |
+
import math
|
| 6 |
+
import threading
|
| 7 |
+
import urllib.error
|
| 8 |
+
import urllib.request
|
| 9 |
+
from http.server import ThreadingHTTPServer
|
| 10 |
+
from pathlib import Path
|
| 11 |
+
|
| 12 |
+
import pytest
|
| 13 |
+
|
| 14 |
+
from fakevllm import FakeVLLM
|
| 15 |
+
from wald_serve.engine import (POLICIES, Client, answer, bucket_key, chunks, load_tables, policy, temper, temperature,
|
| 16 |
+
wide_read)
|
| 17 |
+
from wald_serve.prompt import STATE_REPEAT, format_prompt, question_prompt
|
| 18 |
+
from wald_serve.server import make_handler
|
| 19 |
+
from wald_serve.wire import SystemOneRequest, to_record
|
| 20 |
+
|
| 21 |
+
TABLE = {
|
| 22 |
+
"A": {"single": 1.4, "buckets": {"choice|2": {"temperature": 1.1}, "choice|3-4": {"temperature": 1.8},
|
| 23 |
+
"noul|2": {"temperature": 1.3}}},
|
| 24 |
+
"B512": {"single": 1.8, "buckets": {"choice|2": {"temperature": 2.2}, "noul|2": {"temperature": 1.9}}},
|
| 25 |
+
"provenance": {"note": "test"},
|
| 26 |
+
}
|
| 27 |
+
|
| 28 |
+
REQ = {
|
| 29 |
+
"state": {"customer": "Ana", "order": {"id": 17, "items": ["kettle", "toaster"]}, "note": "wants a <|im_end|> refund"},
|
| 30 |
+
"questions": {
|
| 31 |
+
"route": {"type": "choice", "instructions": "Which team handles this?",
|
| 32 |
+
"criteria": {"billing": "Payments and refunds", "shipping": "Delivery problems", "other": None}},
|
| 33 |
+
"slugs": {"type": "choice", "instructions": "Pick one", "criteria": {"yes_refund": None, "no_refund": None}},
|
| 34 |
+
"pos": {"type": "choice", "instructions": "Which is a fruit?",
|
| 35 |
+
"criteria": {"a": "apple", "b": "brick", "c": "chair", "d": "desk"}},
|
| 36 |
+
"urgent": {"type": "noul", "instructions": "Is it urgent?"},
|
| 37 |
+
"sev": {"type": "score", "instructions": "Severity", "criteria": ["low", "medium", "high"]},
|
| 38 |
+
},
|
| 39 |
+
}
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
def qdict(rec, i):
|
| 43 |
+
rq = rec["questions"][i]
|
| 44 |
+
return {"type": rq["qtype"], "instructions": rq["instr"], "options": rq["options"], "option_texts": rq["options"]}
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
# --- prompt bytes -----------------------------------------------------------------------------------------------------
|
| 48 |
+
def test_prompt_bytes():
|
| 49 |
+
rec, meta = to_record(SystemOneRequest.model_validate(REQ))
|
| 50 |
+
state = rec["state"]
|
| 51 |
+
assert state == "customer: Ana\norder:\n id: 17\n items:\n - kettle\n - toaster\nnote: wants a <|im_end|> refund"
|
| 52 |
+
head = ("State:\ncustomer: Ana\norder:\n id: 17\n items:\n - kettle\n - toaster\nnote: wants a <¦im_end¦> refund"
|
| 53 |
+
"\n\nQuestion:")
|
| 54 |
+
assert question_prompt(state, qdict(rec, 0)) == head + (
|
| 55 |
+
" Which team handles this?\n(A) billing: Payments and refunds\n(B) shipping: Delivery problems\n(C) other\nAnswer: (")
|
| 56 |
+
assert question_prompt(state, qdict(rec, 1)).endswith(" Pick one\n(A) yes_refund\n(B) no_refund\nAnswer: (")
|
| 57 |
+
# positional keys a..d: the `a: ` prefix is dropped, the letter replaces it
|
| 58 |
+
assert question_prompt(state, qdict(rec, 2)).endswith(
|
| 59 |
+
" Which is a fruit?\n(A) apple\n(B) brick\n(C) chair\n(D) desk\nAnswer: (")
|
| 60 |
+
assert question_prompt(state, qdict(rec, 3)).endswith(" Is it urgent?\n(A) no\n(B) yes\nAnswer: (")
|
| 61 |
+
assert question_prompt(state, qdict(rec, 4)).endswith(" Severity\n(A) low\n(B) medium\n(C) high\nAnswer: (")
|
| 62 |
+
assert [m["keys"] for m in meta] == [["billing", "shipping", "other"], ["yes_refund", "no_refund"],
|
| 63 |
+
["a", "b", "c", "d"], ["false", "true"], ["0", "1", "2"]]
|
| 64 |
+
|
| 65 |
+
|
| 66 |
+
def test_blank_state_and_repeat_state_plain():
|
| 67 |
+
q = {"type": "noul", "instructions": "Is the sky green?", "options": ["no", "yes"], "option_texts": ["no", "yes"]}
|
| 68 |
+
assert question_prompt("", q) == "Question: Is the sky green?\n(A) no\n(B) yes\nAnswer: ("
|
| 69 |
+
assert question_prompt("", q, "repeat_state_plain") == question_prompt("", q)
|
| 70 |
+
p = question_prompt("It rained.", q, "repeat_state_plain")
|
| 71 |
+
assert p == "State:\nIt rained." + STATE_REPEAT + "It rained.\n\nQuestion: Is the sky green?\n(A) no\n(B) yes\nAnswer: ("
|
| 72 |
+
assert STATE_REPEAT.startswith("\n\nRead the same context again before answering.")
|
| 73 |
+
with pytest.raises(ValueError):
|
| 74 |
+
format_prompt("State:\nx\n\nQuestion: q\n(A) a\nAnswer: (", "strict")
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
def test_wire_validation():
|
| 78 |
+
with pytest.raises(Exception):
|
| 79 |
+
SystemOneRequest.model_validate({"state": "", "questions": {}})
|
| 80 |
+
with pytest.raises(Exception):
|
| 81 |
+
SystemOneRequest.model_validate({"state": "", "questions": {"q": {"type": "choice", "criteria": {}}}})
|
| 82 |
+
big = {f"o{i}": None for i in range(256)}
|
| 83 |
+
with pytest.raises(Exception):
|
| 84 |
+
SystemOneRequest.model_validate({"state": "", "questions": {"q": {"type": "choice", "criteria": big}}})
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
# --- numerics ---------------------------------------------------------------------------------------------------------
|
| 88 |
+
def test_chunks_and_knockout():
|
| 89 |
+
assert chunks(26) == [list(range(26))]
|
| 90 |
+
assert [len(c) for c in chunks(77)] == [26, 26, 25]
|
| 91 |
+
assert [len(c) for c in chunks(151)] == [26, 25, 25, 25, 25, 25]
|
| 92 |
+
assert sum(len(c) for c in chunks(255)) == 255
|
| 93 |
+
|
| 94 |
+
def read(idx): # a fixed preference for lower indices
|
| 95 |
+
z = [math.exp(-i / 10) for i in idx]
|
| 96 |
+
s = sum(z)
|
| 97 |
+
return [v / s for v in z]
|
| 98 |
+
p = wide_read(77, read)
|
| 99 |
+
assert len(p) == 77 and abs(sum(p) - 1) < 1e-12 and max(range(77), key=lambda i: p[i]) == 0
|
| 100 |
+
|
| 101 |
+
|
| 102 |
+
def test_temperature():
|
| 103 |
+
tables = json.loads(json.dumps(TABLE))
|
| 104 |
+
assert bucket_key("choice", 4) == "choice|3-4" and bucket_key("choice", 77) == "choice|9+"
|
| 105 |
+
assert temperature(tables["A"], "choice", 2) == 1.1 and temperature(tables["A"], "score", 7) == 1.4
|
| 106 |
+
p = [0.6, 0.3, 0.1]
|
| 107 |
+
q = temper(p, 1.8)
|
| 108 |
+
assert abs(sum(q) - 1) < 1e-12 and q.index(max(q)) == 0 and max(q) < 0.6
|
| 109 |
+
|
| 110 |
+
|
| 111 |
+
def test_policies():
|
| 112 |
+
assert policy("medium") == {"gate": 0.7, "budget": 512, "k": 1, "name": "medium"}
|
| 113 |
+
assert policy("high-k4")["k"] == 4 and policy("high-k4")["gate"] > 1
|
| 114 |
+
assert set(POLICIES) == {"none", "low", "medium", "high"}
|
| 115 |
+
for bad in ("max", "high-k1", "high-k9", "high-kx"):
|
| 116 |
+
with pytest.raises(ValueError):
|
| 117 |
+
policy(bad)
|
| 118 |
+
|
| 119 |
+
|
| 120 |
+
# --- end to end over HTTP with the fake vLLM ---------------------------------------------------------------------------
|
| 121 |
+
@pytest.fixture(scope="module")
|
| 122 |
+
def fake():
|
| 123 |
+
f = FakeVLLM(max_len=100_000)
|
| 124 |
+
yield f
|
| 125 |
+
f.stop()
|
| 126 |
+
|
| 127 |
+
|
| 128 |
+
def serve(cl, effort="medium", tables=None, workers=8):
|
| 129 |
+
info = {"model": "wald-4b", "effort": effort}
|
| 130 |
+
httpd = ThreadingHTTPServer(("127.0.0.1", 0), make_handler(cl, policy(effort), tables or {}, info, workers))
|
| 131 |
+
httpd.daemon_threads = True
|
| 132 |
+
threading.Thread(target=httpd.serve_forever, daemon=True).start()
|
| 133 |
+
return httpd, f"http://127.0.0.1:{httpd.server_address[1]}"
|
| 134 |
+
|
| 135 |
+
|
| 136 |
+
def post(url, body):
|
| 137 |
+
req = urllib.request.Request(url + "/v1/systemone", data=json.dumps(body).encode(), method="POST",
|
| 138 |
+
headers={"content-type": "application/json"})
|
| 139 |
+
try:
|
| 140 |
+
with urllib.request.urlopen(req, timeout=60) as r:
|
| 141 |
+
return r.status, json.loads(r.read())
|
| 142 |
+
except urllib.error.HTTPError as e:
|
| 143 |
+
return e.code, json.loads(e.read())
|
| 144 |
+
|
| 145 |
+
|
| 146 |
+
def wide_request(n=77):
|
| 147 |
+
crit = {f"intent_{i:03d}": f"customer intent number {i}" for i in range(n)}
|
| 148 |
+
return {"state": "I was charged twice for one card payment.",
|
| 149 |
+
"questions": {"intent": {"type": "choice", "instructions": "Classify the banking intent.", "criteria": crit}}}
|
| 150 |
+
|
| 151 |
+
|
| 152 |
+
def check_wire(req, resp):
|
| 153 |
+
"""The Decision Index kit's validation: every question answered, a finite probability per option, sum 1 +- 0.01,
|
| 154 |
+
the choice among the options."""
|
| 155 |
+
assert set(resp["answers"]) == set(req["questions"])
|
| 156 |
+
for qid, q in req["questions"].items():
|
| 157 |
+
a = resp["answers"][qid]
|
| 158 |
+
assert a["type"] == q["type"]
|
| 159 |
+
if q["type"] == "noul":
|
| 160 |
+
assert 0.0 <= a["noul"] <= 1.0
|
| 161 |
+
continue
|
| 162 |
+
keys = list(q["criteria"]) if q["type"] == "choice" else [str(i) for i in range(len(q["criteria"]))]
|
| 163 |
+
assert list(a["probabilities"]) == keys
|
| 164 |
+
assert all(math.isfinite(v) and v >= 0 for v in a["probabilities"].values())
|
| 165 |
+
assert abs(sum(a["probabilities"].values()) - 1) < 0.01
|
| 166 |
+
if q["type"] == "choice":
|
| 167 |
+
assert a["choice"] in keys and a["probabilities"][a["choice"]] == max(a["probabilities"].values())
|
| 168 |
+
|
| 169 |
+
|
| 170 |
+
def test_http_end_to_end(fake, tmp_path):
|
| 171 |
+
Path(tmp_path / "t.json").write_text(json.dumps(TABLE))
|
| 172 |
+
tables = load_tables(tmp_path / "t.json", 512)
|
| 173 |
+
cl = Client(fake.url, "wald", 100_000)
|
| 174 |
+
httpd, url = serve(cl, "medium", tables)
|
| 175 |
+
try:
|
| 176 |
+
for req in (REQ, wide_request(77), wide_request(151), wide_request(255)):
|
| 177 |
+
code, resp = post(url, req)
|
| 178 |
+
assert code == 200, resp
|
| 179 |
+
check_wire(req, resp)
|
| 180 |
+
assert resp["usage"]["input_tokens"] > 0
|
| 181 |
+
code, resp = post(url, wide_request(77))
|
| 182 |
+
assert resp["answers"]["intent"]["mode"] == "K"
|
| 183 |
+
with urllib.request.urlopen(url + "/health") as r:
|
| 184 |
+
assert json.loads(r.read())["ok"] is True
|
| 185 |
+
finally:
|
| 186 |
+
httpd.shutdown()
|
| 187 |
+
|
| 188 |
+
|
| 189 |
+
def test_gate_modes(fake):
|
| 190 |
+
cl = Client(fake.url, "wald", 100_000)
|
| 191 |
+
one, u0 = answer(cl, REQ, policy("none"), {})
|
| 192 |
+
assert all(a["mode"] == "A" for a in one.values()) and u0["output_tokens"] == 0
|
| 193 |
+
high, uh = answer(cl, REQ, policy("high"), {})
|
| 194 |
+
assert all(a["mode"] == "B" for a in high.values()) and uh["output_tokens"] > 0
|
| 195 |
+
med, _ = answer(cl, REQ, policy("medium"), {})
|
| 196 |
+
for qid, a in one.items():
|
| 197 |
+
pmax = a["noul"] if a["type"] == "noul" else max(a["probabilities"].values())
|
| 198 |
+
if a["type"] == "noul":
|
| 199 |
+
pmax = max(pmax, 1 - pmax)
|
| 200 |
+
assert med[qid]["mode"] == ("B" if pmax < 0.7 else "A")
|
| 201 |
+
if med[qid]["mode"] == "B": # medium's thought is high's thought (same seed)
|
| 202 |
+
assert med[qid] == high[qid]
|
| 203 |
+
k4, uk = answer(cl, REQ, policy("high-k4"), {})
|
| 204 |
+
assert all(a["mode"] == "B" for a in k4.values()) and uk["output_tokens"] == 4 * uh["output_tokens"]
|
| 205 |
+
|
| 206 |
+
|
| 207 |
+
def test_workers_do_not_change_answers(fake):
|
| 208 |
+
cl = Client(fake.url, "wald", 100_000)
|
| 209 |
+
a1, u1 = answer(cl, REQ, policy("high"), {}, workers=1)
|
| 210 |
+
a8, u8 = answer(cl, REQ, policy("high"), {}, workers=8)
|
| 211 |
+
assert a1 == a8 and u1 == u8
|
| 212 |
+
|
| 213 |
+
|
| 214 |
+
def test_effort_override_keeps_seed(fake):
|
| 215 |
+
cl = Client(fake.url, "wald", 100_000)
|
| 216 |
+
a, _ = answer(cl, REQ, policy("high"), {})
|
| 217 |
+
b, _ = answer(cl, {**REQ, "effort": "high"}, policy("high"), {})
|
| 218 |
+
assert a == b
|
| 219 |
+
|
| 220 |
+
|
| 221 |
+
def test_prompt_format_changes_prompt_only(fake):
|
| 222 |
+
plain = Client(fake.url, "wald", 100_000)
|
| 223 |
+
rsp = Client(fake.url, "wald", 100_000, prompt_format="repeat_state_plain")
|
| 224 |
+
a, ua = answer(plain, REQ, policy("none"), {})
|
| 225 |
+
b, ub = answer(rsp, REQ, policy("none"), {})
|
| 226 |
+
assert ub["input_tokens"] > ua["input_tokens"]
|
| 227 |
+
check_wire(REQ, {"answers": b})
|
| 228 |
+
|
| 229 |
+
|
| 230 |
+
def test_capacity_is_422(fake):
|
| 231 |
+
cl = Client(fake.url, "wald", 300) # declared limit 300 "tokens" (characters in the fake)
|
| 232 |
+
httpd, url = serve(cl, "none")
|
| 233 |
+
try:
|
| 234 |
+
code, resp = post(url, {"state": "x " * 400, "questions": {"q": {"type": "noul", "instructions": "ok?"}}})
|
| 235 |
+
assert code == 422 and "maximum context length" in resp["error"]
|
| 236 |
+
code, resp = post(url, {"state": "short", "questions": {"q": {"type": "noul", "instructions": "ok?"}}})
|
| 237 |
+
assert code == 200
|
| 238 |
+
code, resp = post(url, {"state": "short", "questions": {}})
|
| 239 |
+
assert code == 400
|
| 240 |
+
code, resp = post(url, {**REQ, "effort": "extreme"})
|
| 241 |
+
assert code == 400
|
| 242 |
+
finally:
|
| 243 |
+
httpd.shutdown()
|
| 244 |
+
|
| 245 |
+
|
| 246 |
+
def test_thought_that_does_not_fit_falls_back(fake):
|
| 247 |
+
q = {"state": "y " * 60, "questions": {"q": {"type": "choice", "instructions": "pick",
|
| 248 |
+
"criteria": {"a": "one", "b": "two"}}}}
|
| 249 |
+
cl = Client(fake.url, "wald", 400) # the prompt fits, prompt + 512-token budget does not
|
| 250 |
+
a, u = answer(cl, q, policy("high"), {})
|
| 251 |
+
assert a["q"]["mode"] == "A" and u["output_tokens"] == 0
|
serving.json
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"effort": "medium",
|
| 3 |
+
"prompt_format": "repeat_state_plain",
|
| 4 |
+
"max_model_len": 131072,
|
| 5 |
+
"temperature": "temperature.json"
|
| 6 |
+
}
|
temperature.json
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"A": {
|
| 3 |
+
"single": 1.6817928305074288,
|
| 4 |
+
"buckets": {
|
| 5 |
+
"choice|2": {
|
| 6 |
+
"temperature": 1.5157165665103978,
|
| 7 |
+
"n": 37,
|
| 8 |
+
"fitted": true
|
| 9 |
+
},
|
| 10 |
+
"choice|3-4": {
|
| 11 |
+
"temperature": 1.9318726578496908,
|
| 12 |
+
"n": 69,
|
| 13 |
+
"fitted": true
|
| 14 |
+
},
|
| 15 |
+
"choice|5-8": {
|
| 16 |
+
"temperature": 1.6817928305074288,
|
| 17 |
+
"n": 17,
|
| 18 |
+
"fitted": false
|
| 19 |
+
},
|
| 20 |
+
"choice|9+": {
|
| 21 |
+
"temperature": 1.3660402567543957,
|
| 22 |
+
"n": 27,
|
| 23 |
+
"fitted": true
|
| 24 |
+
},
|
| 25 |
+
"noul|2": {
|
| 26 |
+
"temperature": 1.9999999999999998,
|
| 27 |
+
"n": 107,
|
| 28 |
+
"fitted": true
|
| 29 |
+
},
|
| 30 |
+
"score|5-8": {
|
| 31 |
+
"temperature": 1.6817928305074288,
|
| 32 |
+
"n": 18,
|
| 33 |
+
"fitted": false
|
| 34 |
+
}
|
| 35 |
+
},
|
| 36 |
+
"min_rows": 20,
|
| 37 |
+
"rows": 275,
|
| 38 |
+
"readout": "A"
|
| 39 |
+
},
|
| 40 |
+
"B256": {
|
| 41 |
+
"single": 4.756828460010884,
|
| 42 |
+
"buckets": {
|
| 43 |
+
"choice|2": {
|
| 44 |
+
"temperature": 6.498019170849887,
|
| 45 |
+
"n": 37,
|
| 46 |
+
"fitted": true
|
| 47 |
+
},
|
| 48 |
+
"choice|3-4": {
|
| 49 |
+
"temperature": 4.0,
|
| 50 |
+
"n": 69,
|
| 51 |
+
"fitted": true
|
| 52 |
+
},
|
| 53 |
+
"choice|5-8": {
|
| 54 |
+
"temperature": 4.756828460010884,
|
| 55 |
+
"n": 17,
|
| 56 |
+
"fitted": false
|
| 57 |
+
},
|
| 58 |
+
"choice|9+": {
|
| 59 |
+
"temperature": 3.863745315699382,
|
| 60 |
+
"n": 27,
|
| 61 |
+
"fitted": true
|
| 62 |
+
},
|
| 63 |
+
"noul|2": {
|
| 64 |
+
"temperature": 5.856342783782502,
|
| 65 |
+
"n": 107,
|
| 66 |
+
"fitted": true
|
| 67 |
+
},
|
| 68 |
+
"score|5-8": {
|
| 69 |
+
"temperature": 4.756828460010884,
|
| 70 |
+
"n": 18,
|
| 71 |
+
"fitted": false
|
| 72 |
+
}
|
| 73 |
+
},
|
| 74 |
+
"min_rows": 20,
|
| 75 |
+
"rows": 275,
|
| 76 |
+
"readout": "B256"
|
| 77 |
+
},
|
| 78 |
+
"B512": {
|
| 79 |
+
"single": 4.924577653379665,
|
| 80 |
+
"buckets": {
|
| 81 |
+
"choice|2": {
|
| 82 |
+
"temperature": 6.727171322029713,
|
| 83 |
+
"n": 37,
|
| 84 |
+
"fitted": true
|
| 85 |
+
},
|
| 86 |
+
"choice|3-4": {
|
| 87 |
+
"temperature": 4.43827788827138,
|
| 88 |
+
"n": 69,
|
| 89 |
+
"fitted": true
|
| 90 |
+
},
|
| 91 |
+
"choice|5-8": {
|
| 92 |
+
"temperature": 4.924577653379665,
|
| 93 |
+
"n": 17,
|
| 94 |
+
"fitted": false
|
| 95 |
+
},
|
| 96 |
+
"choice|9+": {
|
| 97 |
+
"temperature": 3.863745315699382,
|
| 98 |
+
"n": 27,
|
| 99 |
+
"fitted": true
|
| 100 |
+
},
|
| 101 |
+
"noul|2": {
|
| 102 |
+
"temperature": 5.856342783782502,
|
| 103 |
+
"n": 107,
|
| 104 |
+
"fitted": true
|
| 105 |
+
},
|
| 106 |
+
"score|5-8": {
|
| 107 |
+
"temperature": 4.924577653379665,
|
| 108 |
+
"n": 18,
|
| 109 |
+
"fitted": false
|
| 110 |
+
}
|
| 111 |
+
},
|
| 112 |
+
"min_rows": 20,
|
| 113 |
+
"rows": 275,
|
| 114 |
+
"readout": "B512"
|
| 115 |
+
},
|
| 116 |
+
"B": {
|
| 117 |
+
"single": 4.924577653379665,
|
| 118 |
+
"buckets": {
|
| 119 |
+
"choice|2": {
|
| 120 |
+
"temperature": 6.727171322029713,
|
| 121 |
+
"n": 37,
|
| 122 |
+
"fitted": true
|
| 123 |
+
},
|
| 124 |
+
"choice|3-4": {
|
| 125 |
+
"temperature": 4.43827788827138,
|
| 126 |
+
"n": 69,
|
| 127 |
+
"fitted": true
|
| 128 |
+
},
|
| 129 |
+
"choice|5-8": {
|
| 130 |
+
"temperature": 4.924577653379665,
|
| 131 |
+
"n": 17,
|
| 132 |
+
"fitted": false
|
| 133 |
+
},
|
| 134 |
+
"choice|9+": {
|
| 135 |
+
"temperature": 3.863745315699382,
|
| 136 |
+
"n": 27,
|
| 137 |
+
"fitted": true
|
| 138 |
+
},
|
| 139 |
+
"noul|2": {
|
| 140 |
+
"temperature": 5.856342783782502,
|
| 141 |
+
"n": 107,
|
| 142 |
+
"fitted": true
|
| 143 |
+
},
|
| 144 |
+
"score|5-8": {
|
| 145 |
+
"temperature": 4.924577653379665,
|
| 146 |
+
"n": 18,
|
| 147 |
+
"fitted": false
|
| 148 |
+
}
|
| 149 |
+
},
|
| 150 |
+
"min_rows": 20,
|
| 151 |
+
"rows": 275,
|
| 152 |
+
"readout": "B512"
|
| 153 |
+
}
|
| 154 |
+
}
|
tokenizer.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:bd53432f0de26d67a83b634040d4f043053da4ecd0c759e7f4b24ad4f8bb9a81
|
| 3 |
+
size 19989509
|
tokenizer_config.json
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"audio_bos_token": "<|audio_start|>",
|
| 4 |
+
"audio_eos_token": "<|audio_end|>",
|
| 5 |
+
"audio_token": "<|audio_pad|>",
|
| 6 |
+
"backend": "tokenizers",
|
| 7 |
+
"bos_token": null,
|
| 8 |
+
"clean_up_tokenization_spaces": false,
|
| 9 |
+
"eos_token": "<|endoftext|>",
|
| 10 |
+
"errors": "replace",
|
| 11 |
+
"image_token": "<|image_pad|>",
|
| 12 |
+
"is_local": true,
|
| 13 |
+
"local_files_only": false,
|
| 14 |
+
"model_max_length": 262144,
|
| 15 |
+
"model_specific_special_tokens": {
|
| 16 |
+
"audio_bos_token": "<|audio_start|>",
|
| 17 |
+
"audio_eos_token": "<|audio_end|>",
|
| 18 |
+
"audio_token": "<|audio_pad|>",
|
| 19 |
+
"image_token": "<|image_pad|>",
|
| 20 |
+
"video_token": "<|video_pad|>",
|
| 21 |
+
"vision_bos_token": "<|vision_start|>",
|
| 22 |
+
"vision_eos_token": "<|vision_end|>"
|
| 23 |
+
},
|
| 24 |
+
"pad_token": "<|endoftext|>",
|
| 25 |
+
"pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
| 26 |
+
"split_special_tokens": false,
|
| 27 |
+
"tokenizer_class": "Qwen2Tokenizer",
|
| 28 |
+
"unk_token": null,
|
| 29 |
+
"video_token": "<|video_pad|>",
|
| 30 |
+
"vision_bos_token": "<|vision_start|>",
|
| 31 |
+
"vision_eos_token": "<|vision_end|>"
|
| 32 |
+
}
|
tools/export_contamination.py
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Export contamination/trained-on-di-ids.json: every Decision Index item id found in the training data of the
|
| 2 |
+
submitted checkpoint's whole lineage, with the training component(s) it was found in and why it is there.
|
| 3 |
+
|
| 4 |
+
python tools/export_contamination.py --src <research checkout>
|
| 5 |
+
|
| 6 |
+
Inputs (relative to --src; committed scan outputs, ids and counts only, no item text):
|
| 7 |
+
results/decision-index/contamination-verified.json strict-verified hits of the first scan (stage-1 corpus)
|
| 8 |
+
trainer/data/blocklists/filters/<dataset>.di-scan.json full-blocklist scans of each training dataset
|
| 9 |
+
"""
|
| 10 |
+
from __future__ import annotations
|
| 11 |
+
|
| 12 |
+
import argparse
|
| 13 |
+
import collections
|
| 14 |
+
import json
|
| 15 |
+
from pathlib import Path
|
| 16 |
+
|
| 17 |
+
ROOT = Path(__file__).resolve().parents[1]
|
| 18 |
+
|
| 19 |
+
# lineage component -> the scan that covers it (the stage-1 corpus scan covers the verified stage-1 hits as well)
|
| 20 |
+
COMPONENTS = {
|
| 21 |
+
"stage 1 (full-parameter decision training corpus)": "corpus-c11-train",
|
| 22 |
+
"stage 2 (LoRA refinement data)": "finetune-f3-stage2",
|
| 23 |
+
"stage 3 (short-thought distillation data)": "think-paren-v1",
|
| 24 |
+
}
|
| 25 |
+
VERIFIED_STAGE1 = "c9"
|
| 26 |
+
# Stage 4 (the DI-targeted LoRA) and the RL arm's data were filtered with the full blocklist before training: 0 hits.
|
| 27 |
+
|
| 28 |
+
WHY = {
|
| 29 |
+
"4": ("BANKING77", "scored", "train-split duplicate: the benchmark's public train split contains the same sentence as the test item"),
|
| 30 |
+
"5": ("CLINC150+OOS", "scored", "train-split duplicate: the public train split contains the same sentence as the test item"),
|
| 31 |
+
"28": ("WinoGrande", "scored", "train-split duplicate: the public train split contains the same sentence"),
|
| 32 |
+
"26": ("ARC-Easy", "display only", "train-split duplicate (ARC train / MMLU auxiliary_train), the board's known duplicates"),
|
| 33 |
+
"27": ("ARC-Challenge", "display only", "train-split duplicate (ARC train / MMLU auxiliary_train), the board's known duplicates"),
|
| 34 |
+
"24": ("MMLU", "display only", "question also present in MMLU auxiliary_train / other public QA sets"),
|
| 35 |
+
"2": ("ToolRet", "scored", "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data"),
|
| 36 |
+
"33": ("SATA-Bench", "scored", "shared upstream source: reading passage from RACE, which is part of MMLU auxiliary_train"),
|
| 37 |
+
"36": ("BRIGHT", "scored", "shared upstream source: a public programming problem statement (LeetCode)"),
|
| 38 |
+
"57": ("MMLU-Pro", "scored", "shared upstream source: MMLU-Pro includes problems from MATH / TheoremQA; the same problem is in our math data"),
|
| 39 |
+
"12": ("ANLI", "scored", "shared upstream text: the premise text also appears in another public dataset in our data"),
|
| 40 |
+
"6": ("RouterBench", "not scored in 0.2.1", "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data"),
|
| 41 |
+
}
|
| 42 |
+
|
| 43 |
+
|
| 44 |
+
def bench(i: str) -> str:
|
| 45 |
+
return i.split(":")[1] if i.startswith("candidates-v3:") else i.split(":")[0]
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
def main():
|
| 49 |
+
ap = argparse.ArgumentParser()
|
| 50 |
+
ap.add_argument("--src", required=True)
|
| 51 |
+
a = ap.parse_args()
|
| 52 |
+
src = Path(a.src)
|
| 53 |
+
where = collections.defaultdict(set)
|
| 54 |
+
ver = json.loads((src / "results/decision-index/contamination-verified.json").read_text())
|
| 55 |
+
for ids in ver["run_ids"].get(VERIFIED_STAGE1, {}).values():
|
| 56 |
+
for i in ids:
|
| 57 |
+
where[i].add("stage 1 (full-parameter decision training corpus)")
|
| 58 |
+
scans = {}
|
| 59 |
+
for comp, name in COMPONENTS.items():
|
| 60 |
+
d = json.loads((src / f"trainer/data/blocklists/filters/{name}.di-scan.json").read_text())
|
| 61 |
+
scans[comp] = {"strict_hits_in_trained_files": d["strict_total"], "blocklist_sha256": d["blocklist"]["strict_sha256"],
|
| 62 |
+
"suite_fingerprints_sha256": d["blocklist"]["suite_sha256"],
|
| 63 |
+
"files": {f: {"lines": v["lines"], "sha256": v["sha256"], "strict": v["strict"]} for f, v in d["files"].items()}}
|
| 64 |
+
for i in d["di_items"]:
|
| 65 |
+
where[i].add(comp)
|
| 66 |
+
items = []
|
| 67 |
+
for i in sorted(where, key=lambda x: (int(bench(x)), x)):
|
| 68 |
+
b = bench(i)
|
| 69 |
+
name, scored, why = WHY[b]
|
| 70 |
+
items.append({"id": i, "benchmark_id": int(b), "benchmark": name, "scored_in_0_2_1": scored, "why": why,
|
| 71 |
+
"found_in": sorted(where[i])})
|
| 72 |
+
per = collections.Counter((it["benchmark_id"], it["benchmark"], it["scored_in_0_2_1"]) for it in items)
|
| 73 |
+
out = {"schema": "wald/di-trained-on/1",
|
| 74 |
+
"what": "Decision Index 0.2 / 0.2.1 item ids present in the training data of the submitted checkpoint's lineage "
|
| 75 |
+
"(strict match: the whole state in one training record, plus >= 50 % of option text where options are the "
|
| 76 |
+
"item's content). Count every one of them as trained on.",
|
| 77 |
+
"total": len(items),
|
| 78 |
+
"per_benchmark": [{"benchmark_id": k[0], "benchmark": k[1], "scored_in_0_2_1": k[2], "items": n} for k, n in sorted(per.items())],
|
| 79 |
+
"scans": scans,
|
| 80 |
+
"stage_4_targeted_lora": "stage-4 data (Home-appliance generator rows; iSarcasmEval / API-Bank / ContractNLI / VAST / "
|
| 81 |
+
"NLI4CT / ACOS / RAGTruth train splits; replay) was checked against the full blocklist and the "
|
| 82 |
+
"sample rows before training: 0 hits after filtering",
|
| 83 |
+
"items": items}
|
| 84 |
+
(ROOT / "contamination").mkdir(exist_ok=True)
|
| 85 |
+
(ROOT / "contamination/trained-on-di-ids.json").write_text(json.dumps(out, indent=1) + "\n")
|
| 86 |
+
print(json.dumps({"total": len(items), "per_benchmark": out["per_benchmark"]}))
|
| 87 |
+
|
| 88 |
+
|
| 89 |
+
if __name__ == "__main__":
|
| 90 |
+
main()
|
tools/export_figure_data.py
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Export the numbers behind figures/*.svg and the comparison table into figures/data.json.
|
| 2 |
+
|
| 3 |
+
python tools/export_figure_data.py --src <research checkout> --summary <exported summary.json>
|
| 4 |
+
|
| 5 |
+
Reads only committed result files of the research checkout (paths below are relative to --src):
|
| 6 |
+
results/decision-index/board-live.json the live Decision Index 0.2.1 board (official full-suite numbers)
|
| 7 |
+
results/decision-index/boot021-*.json our paired-bootstrap reads on the 6,948-request stratified sample
|
| 8 |
+
verticals/<task>/runs/*/report.json raw Qwen3.5-4B-Base zero-shot on each vertical's fixed test set
|
| 9 |
+
and the exported summary (verticals: ours + LoRA / Jev / Wald zero-shot; effort table; serving cost and latency).
|
| 10 |
+
|
| 11 |
+
A read that is not committed yet is taken from PENDING (a number relayed from the run, marked "reported") and is
|
| 12 |
+
replaced automatically once its bootstrap file exists. Nothing here is recomputed: the script copies and labels.
|
| 13 |
+
"""
|
| 14 |
+
from __future__ import annotations
|
| 15 |
+
|
| 16 |
+
import argparse
|
| 17 |
+
import glob
|
| 18 |
+
import json
|
| 19 |
+
import re
|
| 20 |
+
from pathlib import Path
|
| 21 |
+
|
| 22 |
+
ROOT = Path(__file__).resolve().parents[1]
|
| 23 |
+
|
| 24 |
+
# Our Decision Index rows. boot = (bootstrap file, model key) once committed; reported = relayed number until then.
|
| 25 |
+
WALD_ROWS = [ # one row: v1.0 = 021A0-f10 at the declared policy (medium, prompt format repeat_state_plain)
|
| 26 |
+
{"id": "v1.0", "label": "Wald-4B v1.0 · effort medium", "boot": ("boot021-0927-021A0-f10.json", "arrmed"),
|
| 27 |
+
"reported": {"index": 53.91, "blocklist_adjusted": 53.62, "board_rank": 6},
|
| 28 |
+
"reported_areas_file": "figures/reported-areas-v1.0.json"},
|
| 29 |
+
]
|
| 30 |
+
EFFORT_KEYS = {"none": "g4one", "low": "g4low", "medium": "g4med", "high": "g4high", "high-k4": "g4k4"}
|
| 31 |
+
QWEN27 = ("boot021-bigbase-0927.json", "q27paren")
|
| 32 |
+
JEV_SAMPLE = ("boot021-0927.json", "jev")
|
| 33 |
+
BOARD_NAMES = {"decider_4b": "Decider 4B", "decider_35b": "Decider 35B-A3B"}
|
| 34 |
+
SUITE = [ # JevBench public 231 (accuracy %) and XL hidden split (XL-Int), zero-shot, same requests
|
| 35 |
+
{"system": "Wald-4B v1.0 · medium (packaged server)", "public231": round(100 * 204 / 231, 1), "xl_hidden": 63.2,
|
| 36 |
+
"source": "public 231: packaged wald-serve smoke read of 021A0-f10 09-27 (ledger 197); XL: xl-bench leaderboard.md (hidden split)"},
|
| 37 |
+
{"system": "Jev (API, jev-1.13.0)", "public231": 86.6, "xl_hidden": 73.0, "source": "results/external/jev-1.13.0; xl-bench #1"},
|
| 38 |
+
{"system": "Cygnet (gemma-4-12B-it + shim)", "public231": None, "xl_hidden": 59.8, "source": "xl-bench #4"},
|
| 39 |
+
{"system": "Laya (zero-shot, Jev's wire request)", "public231": 58.0, "xl_hidden": 13.4, "source": "results/external/laya; xl-bench #24"},
|
| 40 |
+
{"system": "CLM-8B (zero-shot, same requests, deployment verified)", "public231": 39.0, "xl_hidden": 12.7,
|
| 41 |
+
"source": "ledger 190 read (results/external/clm-8b); xl-bench #25"},
|
| 42 |
+
{"system": "raw Qwen3.5-4B-Base (zero-shot, our readout)", "public231": round(100 * 156 / 231, 1), "xl_hidden": None,
|
| 43 |
+
"source": "exported lineage summary (public one-pass 156 / 231)"},
|
| 44 |
+
]
|
| 45 |
+
NC_TASKS = {"WebLINX intent"}
|
| 46 |
+
COLD_PAIRED_CI = [7.7, 10.0] # paired bootstrap, docs: buyer-demo COLD v3 full test (+8.8)
|
| 47 |
+
COLD_USD = 1.0 # three COLD arms shared about 1 GPU-hour (about $1) on one RTX PRO 6000
|
| 48 |
+
VERTICAL_TASKS = { # summary task name -> vertical directory
|
| 49 |
+
"When2Call": "when2call", "BANKING77": "banking77", "SGD intent": "sgd-intent", "ToxicChat": "toxicchat",
|
| 50 |
+
"MetaTool": "metatool", "AndroidControl": "androidcontrol", "WebLINX intent": "weblinx-intent",
|
| 51 |
+
"RouteLLM routing": "model-routing", "Agent-trajectory safety (hard)": "agent-trajectory-safety-hard",
|
| 52 |
+
"Mind2Web": "mind2web", "Prompt injection": "prompt-injection", "RewardBench": "rewardbench",
|
| 53 |
+
}
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
def size_b(base_model: str | None):
|
| 57 |
+
"""Total parameters in billions read from a base model name (35B-A3B -> 35, E4B -> 4); None when absent."""
|
| 58 |
+
if not base_model:
|
| 59 |
+
return None
|
| 60 |
+
m = re.search(r"(?<![A-Za-z0-9.])E?(\d+(?:\.\d+)?)B(?:-A\d+B)?", base_model.split("/")[-1])
|
| 61 |
+
return float(m.group(1)) if m else None
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
def boot(src: Path, f: str, key: str):
|
| 65 |
+
p = src / "results/decision-index" / f
|
| 66 |
+
if not p.is_file():
|
| 67 |
+
return None
|
| 68 |
+
m = json.loads(p.read_text())["models"].get(key)
|
| 69 |
+
if not m:
|
| 70 |
+
return None
|
| 71 |
+
return {"index": m["balanced_skill"], "ci95": m["ci"]["balanced_skill"], "areas": m["areas"],
|
| 72 |
+
"source": f"results/decision-index/{f}#{key}"}
|
| 73 |
+
|
| 74 |
+
|
| 75 |
+
def external_vertical(src: Path, task_dir: str, prefix: str):
|
| 76 |
+
"""A zero-shot external system's read on a vertical (system id starting with `prefix`), with its p50 latency."""
|
| 77 |
+
for rep in sorted(glob.glob(str(src / "verticals" / task_dir / "runs/*/report.json"))):
|
| 78 |
+
for s in json.loads(Path(rep).read_text())["systems"]:
|
| 79 |
+
if s["id"] == prefix or (s["id"].startswith(prefix) and "multilingual" not in s["id"]):
|
| 80 |
+
return {"value": round(100 * s["metric"], 1), "p50_ms": s.get("p50_ms"), "source": str(Path(rep).relative_to(src))}
|
| 81 |
+
return None
|
| 82 |
+
|
| 83 |
+
|
| 84 |
+
def raw_qwen_vertical(src: Path, task_dir: str):
|
| 85 |
+
for rep in sorted(glob.glob(str(src / "verticals" / task_dir / "runs/*/report.json"))):
|
| 86 |
+
for s in json.loads(Path(rep).read_text())["systems"]:
|
| 87 |
+
if s["id"] == "hosted-qwen35-4b-raw" or (s["kind"] == "base" and "(raw)" in s["label"] and "chat" not in s["label"]):
|
| 88 |
+
return {"value": round(100 * s["metric"], 1), "source": str(Path(rep).relative_to(src))}
|
| 89 |
+
return None
|
| 90 |
+
|
| 91 |
+
|
| 92 |
+
def main():
|
| 93 |
+
ap = argparse.ArgumentParser()
|
| 94 |
+
ap.add_argument("--src", required=True)
|
| 95 |
+
ap.add_argument("--summary", required=True)
|
| 96 |
+
a = ap.parse_args()
|
| 97 |
+
src = Path(a.src)
|
| 98 |
+
summ = json.loads(Path(a.summary).read_text())
|
| 99 |
+
board = json.loads((src / "results/decision-index/board-live.json").read_text())
|
| 100 |
+
|
| 101 |
+
entries = sorted(board["entries"], key=lambda e: -e["index"])
|
| 102 |
+
out = {"board": {"edition": board["edition"], "generated_utc": board["generated_utc"],
|
| 103 |
+
"jev": {"index": board["jev"]["index"], "areas": board["jev"]["areas"]},
|
| 104 |
+
"entries": [{"rank": i + 1, "name": e["name"], "size_b": size_b(e["base_model"]),
|
| 105 |
+
"base_model": e["base_model"] if size_b(e["base_model"]) else None,
|
| 106 |
+
"kind": e["kind"], "index": e["index"], "areas": e["areas"]} for i, e in enumerate(entries)]}}
|
| 107 |
+
wald = []
|
| 108 |
+
for r in WALD_ROWS:
|
| 109 |
+
got = boot(src, *r["boot"])
|
| 110 |
+
row = {"id": r["id"], "label": r["label"], "size_b": 4.0}
|
| 111 |
+
if got:
|
| 112 |
+
row.update(got, status="committed")
|
| 113 |
+
elif r.get("reported"):
|
| 114 |
+
row.update(r["reported"], status="reported (bootstrap not committed yet)")
|
| 115 |
+
ra = ROOT / r["reported_areas_file"]
|
| 116 |
+
if ra.is_file():
|
| 117 |
+
row["areas"] = json.loads(ra.read_text())["areas"]
|
| 118 |
+
else:
|
| 119 |
+
continue
|
| 120 |
+
wald.append(row)
|
| 121 |
+
out["wald_di"] = wald
|
| 122 |
+
out["effort_di"] = {k: boot(src, "boot021-0927-015G0-f4.json", v) for k, v in EFFORT_KEYS.items()}
|
| 123 |
+
out["qwen27_paren"] = boot(src, *QWEN27)
|
| 124 |
+
out["jev_sample"] = boot(src, *JEV_SAMPLE)
|
| 125 |
+
out["board_named"] = {k: next(({"rank": e["rank"], "index": e["index"], "base_model": e["base_model"]}
|
| 126 |
+
for e in out["board"]["entries"] if e["name"] == n), None) for k, n in BOARD_NAMES.items()}
|
| 127 |
+
out["board_4_5"] = [out["board"]["entries"][3], out["board"]["entries"][4]]
|
| 128 |
+
|
| 129 |
+
verts = []
|
| 130 |
+
for v in summ["verticals"]:
|
| 131 |
+
d = VERTICAL_TASKS.get(v["task"])
|
| 132 |
+
raw = raw_qwen_vertical(src, d) if d else None
|
| 133 |
+
laya = external_vertical(src, d, "hosted-laya") if d else None
|
| 134 |
+
clm = external_vertical(src, d, "hosted-clm") if d else None
|
| 135 |
+
verts.append({"laya_0shot": laya and laya["value"], "laya_p50_ms": laya and laya["p50_ms"],
|
| 136 |
+
"clm_0shot": clm and clm["value"], "clm_p50_ms": clm and clm["p50_ms"],
|
| 137 |
+
"task": v["task"], "metric": v["metric"], "n_test": v["n_test"], "ours_lora": v["lora_all"],
|
| 138 |
+
"ours_ci95": v.get("lora_all_ci95"), "jev": v["jev"], "wald_0shot": v["base"],
|
| 139 |
+
"raw_qwen_0shot": raw["value"] if raw else None, "raw_source": raw["source"] if raw else None,
|
| 140 |
+
"ours_minus_jev_ci95": v.get("all_minus_jev_ci95"), "train_usd": v.get("train_usd"),
|
| 141 |
+
"beats_jev": v.get("beats_jev")})
|
| 142 |
+
# NC-licensed task data: internal only, not on the public card
|
| 143 |
+
verts = [v for v in verts if v["task"] not in NC_TASKS]
|
| 144 |
+
# COLD (Chinese offensive language), full test set, from the committed buyer-demo read
|
| 145 |
+
cold = json.loads((src / "results/buyer/cold-v3/full-test/summary.json").read_text())["systems"]
|
| 146 |
+
f1 = lambda k: round(100 * cold[k]["full5323"]["macro_f1"]["value"], 1)
|
| 147 |
+
ci = [round(100 * x, 1) for x in cold["01900-f8+base"]["full5323"]["macro_f1"]["ci95"]]
|
| 148 |
+
verts.append({"task": "COLD (Chinese)", "metric": "macro-F1", "n_test": 5323, "ours_lora": f1("01900-f8+base"),
|
| 149 |
+
"ours_ci95": ci, "jev": f1("jev"), "wald_0shot": f1("015D0-f4"), "raw_qwen_0shot": None, "raw_source": None,
|
| 150 |
+
"laya_0shot": None, "laya_p50_ms": None, "clm_0shot": None, "clm_p50_ms": None,
|
| 151 |
+
"ours_minus_jev_ci95": COLD_PAIRED_CI, "train_usd": COLD_USD, "beats_jev": True,
|
| 152 |
+
"note": "LoRA on all 25,382 train labels averaged with the zero-shot base; ties the best published 83.7"})
|
| 153 |
+
out["verticals"] = verts
|
| 154 |
+
# same-request suite reads (JevBench public 231 accuracy %, XL-Intelligence) from committed reports / notes
|
| 155 |
+
out["suite"] = SUITE
|
| 156 |
+
out["effort_serving"] = summ["effort"]
|
| 157 |
+
out["reference_api"] = summ["reference_api"]
|
| 158 |
+
out["serving"] = summ["serving"]
|
| 159 |
+
(ROOT / "figures").mkdir(exist_ok=True)
|
| 160 |
+
(ROOT / "figures/data.json").write_text(json.dumps(out, indent=1) + "\n")
|
| 161 |
+
print(json.dumps({"wald_di": [(w["id"], w["index"], w["status"]) for w in wald],
|
| 162 |
+
"verticals_with_raw": sum(v["raw_qwen_0shot"] is not None for v in verts)}))
|
| 163 |
+
|
| 164 |
+
|
| 165 |
+
if __name__ == "__main__":
|
| 166 |
+
main()
|
tools/make_figures.py
ADDED
|
@@ -0,0 +1,299 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Render figures/*.svg from figures/data.json (written by tools/export_figure_data.py). Standard library only:
|
| 2 |
+
|
| 3 |
+
python tools/make_figures.py
|
| 4 |
+
|
| 5 |
+
Every SVG carries light and dark colors (prefers-color-scheme), a <title> tooltip on each mark and direct labels, so
|
| 6 |
+
identity never depends on color alone. This script only draws the numbers in data.json.
|
| 7 |
+
"""
|
| 8 |
+
from __future__ import annotations
|
| 9 |
+
|
| 10 |
+
import json
|
| 11 |
+
import math
|
| 12 |
+
from html import escape
|
| 13 |
+
from pathlib import Path
|
| 14 |
+
|
| 15 |
+
ROOT = Path(__file__).resolve().parents[1]
|
| 16 |
+
D = json.loads((ROOT / "figures/data.json").read_text())
|
| 17 |
+
OUT = ROOT / "figures"
|
| 18 |
+
FONT = "system-ui, -apple-system, 'Segoe UI', Roboto, Helvetica, Arial, sans-serif"
|
| 19 |
+
AREAS = [("knowledge", "Knowledge"), ("language", "Language"), ("retrieval", "Retrieval"), ("tools", "Tools"), ("arts", "Arts")]
|
| 20 |
+
|
| 21 |
+
STYLE = """
|
| 22 |
+
<style>
|
| 23 |
+
svg { --surface:#fcfcfb; --ink:#0b0b0b; --ink2:#52514e; --muted:#8a8984; --grid:#e6e5e1; --line:#c9c8c2;
|
| 24 |
+
--s1:#2a78d6; --s2:#eb6834; --s3:#1baf7a; --s4:#8b6fd6; --laya:#c2369b; --clm:#9a7b2f; --dot:#b9b8b2; --hl:#e8f0fb; }
|
| 25 |
+
@media (prefers-color-scheme: dark) {
|
| 26 |
+
svg { --surface:#1a1a19; --ink:#ffffff; --ink2:#c3c2b7; --muted:#8f8e86; --grid:#2e2e2b; --line:#4a4a45;
|
| 27 |
+
--s1:#3987e5; --s2:#d95926; --s3:#199e70; --s4:#9d86e8; --laya:#e05cb8; --clm:#c9a24a; --dot:#5c5b56; --hl:#1d2a3b; }
|
| 28 |
+
}
|
| 29 |
+
text { font-family: FONT; fill: var(--ink); }
|
| 30 |
+
.t2 { fill: var(--ink2); } .mu { fill: var(--muted); }
|
| 31 |
+
.h1 { font-size: 16px; font-weight: 600; } .sm { font-size: 11px; } .xs { font-size: 10px; } .b { font-weight: 600; }
|
| 32 |
+
.grid { stroke: var(--grid); stroke-width: 1; } .axis { stroke: var(--line); stroke-width: 1; }
|
| 33 |
+
</style>""".replace("FONT", FONT)
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def svg(w, h, body, title, desc):
|
| 37 |
+
return (f'<svg xmlns="http://www.w3.org/2000/svg" width="{w}" height="{h}" viewBox="0 0 {w} {h}" role="img" '
|
| 38 |
+
f'aria-labelledby="t d">\n<title id="t">{escape(title)}</title>\n<desc id="d">{escape(desc)}</desc>\n'
|
| 39 |
+
f'{STYLE}\n<rect width="{w}" height="{h}" fill="var(--surface)"/>\n' + "\n".join(body) + "\n</svg>\n")
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
def text(x, y, s, cls="sm", anchor="start"):
|
| 43 |
+
return f'<text x="{x:.1f}" y="{y:.1f}" class="{cls}" text-anchor="{anchor}">{escape(str(s))}</text>'
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
FAINT = ("--laya", "--clm") # weak external baselines: shown, but pale so they do not draw the eye
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
def op(var):
|
| 50 |
+
return ' fill-opacity="0.38"' if var in FAINT else ""
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
def legend(body, x, y, items, step=190):
|
| 54 |
+
for name, var in items:
|
| 55 |
+
body.append(f'<rect x="{x}" y="{y - 10}" width="12" height="12" rx="2" fill="var({var})"{op(var)}/>')
|
| 56 |
+
body.append(text(x + 18, y, name, "sm mu" if var in FAINT else "sm"))
|
| 57 |
+
x += step
|
| 58 |
+
|
| 59 |
+
|
| 60 |
+
# --- 1. Decision Index vs model size ----------------------------------------------------------------------------------
|
| 61 |
+
def di_vs_size():
|
| 62 |
+
W, H, L, T, PW, PH = 1000, 520, 70, 90, 640, 360
|
| 63 |
+
xmin, xmax, ymin, ymax = math.log10(1.5), math.log10(45), 20, 60
|
| 64 |
+
sx = lambda b: L + PW * (math.log10(b) - xmin) / (xmax - xmin)
|
| 65 |
+
sy = lambda v: T + PH - PH * (v - ymin) / (ymax - ymin)
|
| 66 |
+
body = [text(24, 30, "Decision Index 0.2.1 vs model size: a 4B model among 12–35B entries", "h1"),
|
| 67 |
+
text(24, 50, "Grey: every board entry with a parameter count in its base model name (official full-suite scores). "
|
| 68 |
+
"Blue: Wald-4B v1.0 on our 6,948-request stratified sample", "xs t2"),
|
| 69 |
+
text(24, 63, "(Jev scores 57.19 on that sample vs 57.89 on the board, so the sample reads about 0.7 low). "
|
| 70 |
+
"Hollow = reported from a run whose result file is not committed yet.", "xs t2")]
|
| 71 |
+
for v in range(ymin, ymax + 1, 10):
|
| 72 |
+
body.append(f'<line x1="{L}" y1="{sy(v):.1f}" x2="{L + PW}" y2="{sy(v):.1f}" class="grid"/>')
|
| 73 |
+
body.append(text(L - 8, sy(v) + 4, v, "xs mu", "end"))
|
| 74 |
+
for b in (2, 4, 8, 12, 27, 35):
|
| 75 |
+
body.append(f'<line x1="{sx(b):.1f}" y1="{T}" x2="{sx(b):.1f}" y2="{T + PH}" class="grid"/>')
|
| 76 |
+
body.append(text(sx(b), T + PH + 16, f"{b}B", "xs mu", "middle"))
|
| 77 |
+
body.append(text(L + PW / 2, T + PH + 34, "parameters of the base model (log scale; MoE = total)", "xs mu", "middle"))
|
| 78 |
+
body.append(text(L - 50, T - 10, "index", "xs mu"))
|
| 79 |
+
jev = D["board"]["jev"]["index"]
|
| 80 |
+
body.append(f'<g><title>Jev (hosted API, size undisclosed): {jev}</title><line x1="{L}" y1="{sy(jev):.1f}" x2="{L + PW}" '
|
| 81 |
+
f'y2="{sy(jev):.1f}" stroke="var(--s2)" stroke-width="1.5" stroke-dasharray="6 4"/></g>')
|
| 82 |
+
body.append(text(L + 6, sy(jev) - 5, f"Jev {jev} (hosted API, size undisclosed)", "xs"))
|
| 83 |
+
named = {e["name"] for e in D["board"]["entries"][:5]} | {"Decider 4B", "Decider 35B-A3B", "Hopper", "JevK5"}
|
| 84 |
+
labels = []
|
| 85 |
+
for e in D["board"]["entries"]:
|
| 86 |
+
if not e["size_b"]:
|
| 87 |
+
continue
|
| 88 |
+
x, y = sx(e["size_b"]) + (sum(map(ord, e["name"])) % 7 - 3), sy(e["index"])
|
| 89 |
+
body.append(f'<g><title>#{e["rank"]} {escape(e["name"])} ({escape(e["base_model"] or "")}): {e["index"]}</title>'
|
| 90 |
+
f'<circle cx="{x:.1f}" cy="{y:.1f}" r="4" fill="var(--dot)"/></g>')
|
| 91 |
+
if e["name"] in named:
|
| 92 |
+
labels.append([y, x, f"#{e['rank']} {e['name']} {e['index']}"])
|
| 93 |
+
labels.sort()
|
| 94 |
+
for i in range(1, len(labels)): # keep labels in the same column at least 12 px apart
|
| 95 |
+
if abs(labels[i][1] - labels[i - 1][1]) < 80:
|
| 96 |
+
labels[i][0] = max(labels[i][0], labels[i - 1][0] + 12)
|
| 97 |
+
for y, x, s_ in labels:
|
| 98 |
+
body.append(text(x + 8, y + 3, s_, "xs t2"))
|
| 99 |
+
for i, w in enumerate(D["wald_di"]):
|
| 100 |
+
x, y = sx(4) + 14, sy(w["index"])
|
| 101 |
+
filled = w["status"] == "committed"
|
| 102 |
+
body.append(f'<g><title>{escape(w["label"])}: {w["index"]} ({w["status"]})</title><circle cx="{x:.1f}" cy="{y:.1f}" r="6" '
|
| 103 |
+
f'fill="{"var(--s1)" if filled else "var(--surface)"}" stroke="var(--s1)" stroke-width="2.5"/></g>')
|
| 104 |
+
body.append(text(x + 10, y + 4, f"{w['label']} {w['index']}", "sm b" if i == 0 else "xs"))
|
| 105 |
+
body.append(f'<line x1="{L}" y1="{T}" x2="{L}" y2="{T + PH}" class="axis"/>')
|
| 106 |
+
return svg(W, H, body, "Decision Index vs model size", "Wald-4B against every sized board entry; Jev as a dashed line.")
|
| 107 |
+
|
| 108 |
+
|
| 109 |
+
# --- 2. Decision Index areas --------------------------------------------------------------------------------------------
|
| 110 |
+
def di_areas():
|
| 111 |
+
w = D["wald_di"][0]
|
| 112 |
+
rows = [("Jev (board)", D["board"]["jev"]["areas"], "--s2")]
|
| 113 |
+
if w.get("areas"):
|
| 114 |
+
rows.insert(0, (w["label"] + ("" if w["status"] == "committed" else " (reported)"), w["areas"], "--s1"))
|
| 115 |
+
for e, var in zip(D["board_4_5"], ("--s3", "--s4")):
|
| 116 |
+
rows.append((f"#{e['rank']} {e['name']} (board)", e["areas"], var))
|
| 117 |
+
W, L, PW, GH, BAR = 900, 110, 600, 78, 13
|
| 118 |
+
T = 146
|
| 119 |
+
H = T + len(AREAS) * GH + 40
|
| 120 |
+
sx = lambda v: L + PW * v
|
| 121 |
+
body = [text(24, 30, "Decision Index 0.2.1 by area (chance-corrected skill, 0–1)", "h1"),
|
| 122 |
+
text(24, 50, "Wald-4B v1.0 (4B) vs Jev and the board's #4 / #5 entries (27B). Board rows are official full-suite "
|
| 123 |
+
"numbers; the Wald row is our stratified-sample read." if w.get("areas") else
|
| 124 |
+
"Wald-4B v1.0 areas: pending its committed read. Board rows are official full-suite numbers.", "xs t2")]
|
| 125 |
+
for j, r in enumerate(rows):
|
| 126 |
+
body.append(f'<rect x="24" y="{66 + j * 16}" width="12" height="12" rx="2" fill="var({r[2]})"/>')
|
| 127 |
+
body.append(text(42, 76 + j * 16, r[0]))
|
| 128 |
+
for v in (0, 0.2, 0.4, 0.6, 0.8):
|
| 129 |
+
body.append(f'<line x1="{sx(v):.1f}" y1="{T - 6}" x2="{sx(v):.1f}" y2="{H - 30}" class="grid"/>')
|
| 130 |
+
body.append(text(sx(v), H - 16, f"{v:.1f}", "xs mu", "middle"))
|
| 131 |
+
for i, (aid, aname) in enumerate(AREAS):
|
| 132 |
+
y = T + i * GH
|
| 133 |
+
body.append(text(L - 10, y + 30, aname, "sm", "end"))
|
| 134 |
+
for j, (name, areas, var) in enumerate(rows):
|
| 135 |
+
v = areas[aid]
|
| 136 |
+
yy = y + j * (BAR + 3)
|
| 137 |
+
body.append(f'<g><title>{escape(name)} · {aname}: {v:.3f}</title><rect x="{L}" y="{yy}" width="{max(sx(v) - L, 1):.1f}" '
|
| 138 |
+
f'height="{BAR}" rx="2" fill="var({var})"/></g>')
|
| 139 |
+
body.append(text(sx(v) + 4, yy + 10, f"{v:.2f}", "xs"))
|
| 140 |
+
body.append(f'<line x1="{L}" y1="{T - 6}" x2="{L}" y2="{H - 30}" class="axis"/>')
|
| 141 |
+
return svg(W, H, body, "Decision Index by area", "Five area skills for Wald-4B, Jev and the board's #4 and #5.")
|
| 142 |
+
|
| 143 |
+
|
| 144 |
+
# --- 3. verticals --------------------------------------------------------------------------------------------------------
|
| 145 |
+
def verticals():
|
| 146 |
+
rows = sorted(D["verticals"], key=lambda r: -(r["ours_lora"] - r["jev"]))
|
| 147 |
+
L, PW, GH, BAR = 230, 440, 80, 10
|
| 148 |
+
W, T = L + PW + 330, 132
|
| 149 |
+
H = T + len(rows) * GH + 50
|
| 150 |
+
sx = lambda v: L + PW * v / 100
|
| 151 |
+
wins = sum(bool(r["beats_jev"]) for r in rows)
|
| 152 |
+
lo_c, hi_c = min(r["train_usd"] for r in rows), max(r["train_usd"] for r in rows)
|
| 153 |
+
body = [text(24, 30, f"Quick LoRA (< $2, < 2 GPU-h per task) on Wald-4B beats Jev on {wins} of {len(rows)} tasks", "h1"),
|
| 154 |
+
text(24, 50, f"Same fixed test items and byte-identical requests for every system. Ours = one LoRA on the task's "
|
| 155 |
+
f"labels, ${lo_c:.2f}–${hi_c:.2f} of GPU time each; the others are zero-shot. Jev cannot be fine-tuned.", "xs t2")]
|
| 156 |
+
series = [("ours_lora", "Wald-4B v0.9 + LoRA (< $2)", "--s1"), ("jev", "Jev (API, zero-shot)", "--s2"),
|
| 157 |
+
("wald_0shot", "Wald-4B v0.9 zero-shot", "--s3"), ("raw_qwen_0shot", "raw Qwen3.5-4B-Base zero-shot", "--dot"),
|
| 158 |
+
("laya_0shot", "Laya 421M (zero-shot, Jev's wire request)", "--laya"),
|
| 159 |
+
("clm_0shot", "CLM-8B (zero-shot, deployment verified)", "--clm")]
|
| 160 |
+
legend(body, 24, 76, [(n, v) for _, n, v in series[:3]], step=330)
|
| 161 |
+
legend(body, 24, 94, [(n, v) for _, n, v in series[3:]], step=330)
|
| 162 |
+
missing = [n.split(" (")[0] for k, n, _ in series if all(r[k] is None for r in rows)]
|
| 163 |
+
if missing:
|
| 164 |
+
body.append(text(24, 112, f"Not shown (no committed read yet): {', '.join(missing)}. A task without a bar for a system "
|
| 165 |
+
"has no read of it.", "xs mu"))
|
| 166 |
+
for v in range(0, 101, 20):
|
| 167 |
+
body.append(f'<line x1="{sx(v):.1f}" y1="{T - 8}" x2="{sx(v):.1f}" y2="{H - 40}" class="grid"/>')
|
| 168 |
+
body.append(text(sx(v), H - 26, v, "xs mu", "middle"))
|
| 169 |
+
body.append(text(L + PW / 2, H - 10, "task metric (%)", "xs mu", "middle"))
|
| 170 |
+
body.append(text(W - 270, T - 14, "ours − Jev [95 % CI] · LoRA cost", "xs t2"))
|
| 171 |
+
for i, r in enumerate(rows):
|
| 172 |
+
y = T + i * GH
|
| 173 |
+
body.append(text(L - 10, y + 16, r["task"], "sm", "end"))
|
| 174 |
+
body.append(text(L - 10, y + 30, f"{r['metric']} · n = {r['n_test']:,}", "xs mu", "end"))
|
| 175 |
+
for j, (key, name, var) in enumerate(series):
|
| 176 |
+
v = r[key]
|
| 177 |
+
if v is None:
|
| 178 |
+
continue
|
| 179 |
+
yy = y + j * (BAR + 2)
|
| 180 |
+
body.append(f'<g><title>{escape(r["task"])}: {escape(name)} {v:.1f}</title><rect x="{L}" y="{yy}" '
|
| 181 |
+
f'width="{max(sx(v) - L, 1.5):.1f}" height="{BAR}" rx="2" fill="var({var})"{op(var)}/></g>')
|
| 182 |
+
if j == 0:
|
| 183 |
+
body.append(text(sx(v) + 4, yy + 9, f"{v:.1f}", "xs"))
|
| 184 |
+
lo, hi = r["ours_minus_jev_ci95"]
|
| 185 |
+
d = r["ours_lora"] - r["jev"]
|
| 186 |
+
mark = "▲" if r["beats_jev"] else ("▼" if hi < 0 else "n.s.")
|
| 187 |
+
body.append(text(W - 270, y + 18, f"{d:+.1f} [{lo:+.1f}, {hi:+.1f}] {mark} · ${r['train_usd']:.2f}", "sm"))
|
| 188 |
+
body.append(f'<line x1="{L}" y1="{T - 8}" x2="{L}" y2="{H - 40}" class="axis"/>')
|
| 189 |
+
return svg(W, H, body, "Verticals: quick LoRA vs Jev vs zero-shot", "Grouped bars for 12 tasks with paired differences to Jev.")
|
| 190 |
+
|
| 191 |
+
|
| 192 |
+
# --- 4. effort -----------------------------------------------------------------------------------------------------------
|
| 193 |
+
def effort():
|
| 194 |
+
E, DI, api = D["effort_serving"], D["effort_di"], D["reference_api"]
|
| 195 |
+
W, H = 1060, 420
|
| 196 |
+
body = [text(24, 30, "Effort: think only when unsure", "h1"),
|
| 197 |
+
text(24, 50, "Left: Decision Index 0.2.1 (stratified sample, 95 % CI) of one always-think read re-scored per "
|
| 198 |
+
"policy (the next-generation base). Right: accuracy on our eval banks and serial latency per decision "
|
| 199 |
+
"(Wald-4B, one H100).", "xs t2")]
|
| 200 |
+
L, T, PW, PH = 70, 90, 360, 250
|
| 201 |
+
keys = list(DI)
|
| 202 |
+
xs = {k: L + 30 + i * (PW - 60) / (len(keys) - 1) for i, k in enumerate(keys)}
|
| 203 |
+
lo, hi = 38, 54
|
| 204 |
+
sy = lambda v: T + PH - PH * (v - lo) / (hi - lo)
|
| 205 |
+
for v in range(lo, hi + 1, 4):
|
| 206 |
+
body.append(f'<line x1="{L}" y1="{sy(v):.1f}" x2="{L + PW}" y2="{sy(v):.1f}" class="grid"/>')
|
| 207 |
+
body.append(text(L - 8, sy(v) + 4, v, "xs mu", "end"))
|
| 208 |
+
body.append(text(L - 50, T - 10, "Decision Index", "xs mu"))
|
| 209 |
+
pts = []
|
| 210 |
+
for k in keys:
|
| 211 |
+
r = DI[k]
|
| 212 |
+
if not r:
|
| 213 |
+
continue
|
| 214 |
+
x, (a, b) = xs[k], r["ci95"]
|
| 215 |
+
body.append(f'<line x1="{x:.1f}" y1="{sy(a):.1f}" x2="{x:.1f}" y2="{sy(b):.1f}" stroke="var(--s1)" stroke-width="2"/>')
|
| 216 |
+
body.append(f'<g><title>effort {k}: {r["index"]} [{a}, {b}]</title><circle cx="{x:.1f}" cy="{sy(r["index"]):.1f}" r="5" '
|
| 217 |
+
f'fill="{"var(--s1)" if k != "medium" else "var(--s2)"}" stroke="var(--surface)" stroke-width="2"/></g>')
|
| 218 |
+
body.append(text(x + 8, sy(r["index"]) + 4, f"{r['index']:.1f}", "xs"))
|
| 219 |
+
body.append(text(x, T + PH + 18, k, "sm b" if k == "medium" else "sm", "middle"))
|
| 220 |
+
pts.append((x, sy(r["index"])))
|
| 221 |
+
body.append(f'<polyline points="{" ".join(f"{x:.1f},{y:.1f}" for x, y in pts)}" fill="none" stroke="var(--s1)" '
|
| 222 |
+
f'stroke-width="1" stroke-dasharray="3 3"/>')
|
| 223 |
+
body.append(text(L, T + PH + 36, "declared policy: medium (orange) · high-k4 = mean of 4 thoughts", "xs mu"))
|
| 224 |
+
# right panel: accuracy (XL dev) vs latency, one point per effort
|
| 225 |
+
R0, RT, RW, RH = 580, 90, 380, 250
|
| 226 |
+
lmin, lmax = math.log10(0.03), math.log10(3.0)
|
| 227 |
+
lx = lambda s: R0 + RW * (math.log10(s) - lmin) / (lmax - lmin)
|
| 228 |
+
alo, ahi = 70, 90
|
| 229 |
+
ay = lambda v: RT + RH - RH * (v - alo) / (ahi - alo)
|
| 230 |
+
for s in (0.03, 0.1, 0.3, 1, 3):
|
| 231 |
+
body.append(f'<line x1="{lx(s):.1f}" y1="{RT}" x2="{lx(s):.1f}" y2="{RT + RH}" class="grid"/>')
|
| 232 |
+
body.append(text(lx(s), RT + RH + 16, f"{s:g} s", "xs mu", "middle"))
|
| 233 |
+
for v in range(alo, ahi + 1, 5):
|
| 234 |
+
body.append(f'<line x1="{R0}" y1="{ay(v):.1f}" x2="{R0 + RW}" y2="{ay(v):.1f}" class="grid"/>')
|
| 235 |
+
body.append(text(R0 - 8, ay(v) + 4, v, "xs mu", "end"))
|
| 236 |
+
body.append(text(R0 - 40, RT - 10, "XL dev accuracy %", "xs mu"))
|
| 237 |
+
body.append(text(R0 + RW / 2, RT + RH + 34, "serial latency per decision, p50 (dot) to p95 (ring), log scale", "xs mu", "middle"))
|
| 238 |
+
for e in E:
|
| 239 |
+
y, x1, x2 = ay(e["xl_dev"]), lx(e["p50_s"]), lx(e["p95_s"])
|
| 240 |
+
body.append(f'<g><title>effort {e["effort"]}: XL dev {e["xl_dev"]}, p50 {e["p50_s"]} s, p95 {e["p95_s"]} s, thinks on '
|
| 241 |
+
f'{e["thinks"]} %</title><line x1="{x1:.1f}" y1="{y:.1f}" x2="{x2:.1f}" y2="{y:.1f}" stroke="var(--s1)" stroke-width="2"/>'
|
| 242 |
+
f'<circle cx="{x1:.1f}" cy="{y:.1f}" r="5" fill="var(--s1)" stroke="var(--surface)" stroke-width="2"/>'
|
| 243 |
+
f'<circle cx="{x2:.1f}" cy="{y:.1f}" r="5" fill="var(--surface)" stroke="var(--s1)" stroke-width="2"/></g>')
|
| 244 |
+
body.append(text(x1 - 8, y + 4, f"{e['effort']} ({e['thinks']:g} % think)", "xs", "end"))
|
| 245 |
+
xa = lx(api["p50_s"])
|
| 246 |
+
body.append(f'<line x1="{xa:.1f}" y1="{RT}" x2="{xa:.1f}" y2="{RT + RH}" stroke="var(--s2)" stroke-width="1.5" stroke-dasharray="5 4"/>')
|
| 247 |
+
body.append(text(xa + 5, RT + 12, f"Jev API p50 {api['p50_s']:g} s", "xs t2"))
|
| 248 |
+
return svg(W, H, body, "Effort vs accuracy vs latency", "Index rises with effort; median latency stays near one pass up to medium.")
|
| 249 |
+
|
| 250 |
+
|
| 251 |
+
# --- 5. latency and cost vs Jev ----------------------------------------------------------------------------------------
|
| 252 |
+
def latency_cost():
|
| 253 |
+
S, api = D["serving"], D["reference_api"]
|
| 254 |
+
W, H = 1000, 380
|
| 255 |
+
body = [text(24, 30, "Latency and cost per decision vs Jev", "h1"),
|
| 256 |
+
text(24, 50, "Left: median serial latency (ms). Right: $ per 1,000 decisions. Wald-4B on one RTX PRO 6000 "
|
| 257 |
+
"(self-hosted GPU-hour cost at measured throughput); Jev at API list price.", "xs t2")]
|
| 258 |
+
vt = {r["task"]: r for r in D["verticals"]}
|
| 259 |
+
lat = [("one question, one pass (c = 1)", S["rtxpro6000_c1_p50_ms"], None),
|
| 260 |
+
("When2Call request", S["vertical_when2call_p50_ms"], S["jev_when2call_p50_ms"]),
|
| 261 |
+
("BANKING77 request (77 options)", S["vertical_banking77_p50_ms"], S["jev_banking77_p50_ms"])]
|
| 262 |
+
L, T, PW = 220, 100, 250
|
| 263 |
+
mx = 1100
|
| 264 |
+
for i, (name, w, j) in enumerate(lat):
|
| 265 |
+
y = T + i * 70
|
| 266 |
+
body.append(text(L - 10, y + 14, name, "sm", "end"))
|
| 267 |
+
key = "When2Call" if "When2Call" in name else ("BANKING77" if "BANKING77" in name else None)
|
| 268 |
+
laya = vt[key]["laya_p50_ms"] if key else None
|
| 269 |
+
clm = vt[key]["clm_p50_ms"] if key else None
|
| 270 |
+
for k, (v, var, who) in enumerate(((w, "--s1", "Wald-4B"), (j, "--s2", "Jev"), (laya and round(laya), "--laya", "Laya"),
|
| 271 |
+
(clm and round(clm), "--clm", "CLM-8B"))):
|
| 272 |
+
if v is None:
|
| 273 |
+
continue
|
| 274 |
+
yy = y + k * 16
|
| 275 |
+
body.append(f'<g><title>{escape(name)}: {escape(str(who))} {v} ms</title><rect x="{L}" y="{yy}" width="{max(PW * v / mx, 1.5):.1f}" '
|
| 276 |
+
f'height="13" rx="2" fill="var({var})"{op(var)}/></g>')
|
| 277 |
+
body.append(text(L + PW * v / mx + 4, yy + 10, f"{who} {v:g} ms", "xs mu" if var in FAINT else "xs"))
|
| 278 |
+
R0 = 740
|
| 279 |
+
costs = [("Wald-4B fp8, batched", S["rtxpro6000_fp8_usd_per_1k"], "--s1"),
|
| 280 |
+
("Wald-4B multi-tenant, c = 32", S["multitenant_c32_usd_per_1k"], "--s1"),
|
| 281 |
+
("Jev API", api["usd_per_1k"], "--s2")]
|
| 282 |
+
for i, (name, v, var) in enumerate(costs):
|
| 283 |
+
y = T + i * 40
|
| 284 |
+
body.append(text(R0 - 10, y + 11, name, "sm", "end"))
|
| 285 |
+
body.append(f'<g><title>{escape(name)}: ${v} per 1,000 decisions</title><rect x="{R0}" y="{y}" width="{200 * v / 0.045:.1f}" '
|
| 286 |
+
f'height="14" rx="2" fill="var({var})"/></g>')
|
| 287 |
+
body.append(text(R0 + 200 * v / 0.045 + 4, y + 11, f"${v:.4f}", "xs"))
|
| 288 |
+
ratio = api["usd_per_1k"] / S["rtxpro6000_fp8_usd_per_1k"]
|
| 289 |
+
body.append(text(R0 - 150, T + 140, f"Jev costs about {ratio:.1f}× more per decision.", "sm b"))
|
| 290 |
+
body.append(text(24, H - 14, "Laya: its own server, as measured in the same vertical runs. CLM-8B: latency and cost not "
|
| 291 |
+
"committed yet. Not every system was measured on the same GPU; see the model card.", "xs mu"))
|
| 292 |
+
return svg(W, H, body, "Latency and cost vs Jev", "Median latency and dollars per 1,000 decisions, Wald-4B vs Jev.")
|
| 293 |
+
|
| 294 |
+
|
| 295 |
+
if __name__ == "__main__":
|
| 296 |
+
for name, fn in (("di-vs-size", di_vs_size), ("di-areas", di_areas), ("verticals", verticals),
|
| 297 |
+
("latency-cost", latency_cost)):
|
| 298 |
+
(OUT / f"{name}.svg").write_text(fn())
|
| 299 |
+
print("wrote", OUT / f"{name}.svg")
|
tools/prepare_model_dir.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Prepare a weights directory for release: copy the checkpoint's own temperature table without its provenance field,
|
| 2 |
+
write serving.json (declared policy and prompt format), and write MANIFEST.json with the sha256 of every file.
|
| 3 |
+
|
| 4 |
+
python tools/prepare_model_dir.py WEIGHTS_DIR --temperature path/to/temperature.json --prompt-format plain
|
| 5 |
+
"""
|
| 6 |
+
import argparse, hashlib, json
|
| 7 |
+
from pathlib import Path
|
| 8 |
+
|
| 9 |
+
ap = argparse.ArgumentParser()
|
| 10 |
+
ap.add_argument("weights"); ap.add_argument("--temperature", required=True)
|
| 11 |
+
ap.add_argument("--prompt-format", choices=("plain", "repeat_state_plain"), required=True)
|
| 12 |
+
ap.add_argument("--effort", default="medium"); ap.add_argument("--max-model-len", type=int, default=131072)
|
| 13 |
+
a = ap.parse_args()
|
| 14 |
+
w = Path(a.weights)
|
| 15 |
+
t = json.loads(Path(a.temperature).read_text())
|
| 16 |
+
t = {k: v for k, v in t.items() if k in ("A", "B", "B256", "B512", "buckets", "single")}
|
| 17 |
+
for k, v in t.items(): # keep only the fields the server reads
|
| 18 |
+
if isinstance(v, dict):
|
| 19 |
+
t[k] = {kk: vv for kk, vv in v.items() if kk in ("single", "buckets", "min_rows", "rows", "readout")}
|
| 20 |
+
(w / "temperature.json").write_text(json.dumps(t, indent=1) + "\n")
|
| 21 |
+
(w / "serving.json").write_text(json.dumps({"effort": a.effort, "prompt_format": a.prompt_format,
|
| 22 |
+
"max_model_len": a.max_model_len, "temperature": "temperature.json"}, indent=1) + "\n")
|
| 23 |
+
man = {}
|
| 24 |
+
for f in sorted(w.rglob("*")):
|
| 25 |
+
if f.is_file() and f.name != "MANIFEST.json" and ".cache" not in f.parts:
|
| 26 |
+
h = hashlib.sha256()
|
| 27 |
+
with open(f, "rb") as fh:
|
| 28 |
+
for b in iter(lambda: fh.read(1 << 24), b""): h.update(b)
|
| 29 |
+
man[str(f.relative_to(w))] = {"bytes": f.stat().st_size, "sha256": h.hexdigest()}
|
| 30 |
+
(w / "MANIFEST.json").write_text(json.dumps(man, indent=1) + "\n")
|
| 31 |
+
print(json.dumps({"files": len(man), "serving": json.loads((w / "serving.json").read_text())}))
|
tools/secret_scan.py
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Scan this repository for secrets and private infrastructure before any push. Prints variable NAMES only, never values.
|
| 2 |
+
|
| 3 |
+
python tools/secret_scan.py --env /path/to/.env [--extra-pattern REGEX ...]
|
| 4 |
+
Exit code 1 on any hit."""
|
| 5 |
+
import argparse, os, re, sys
|
| 6 |
+
|
| 7 |
+
ap = argparse.ArgumentParser(); ap.add_argument("--env", required=True); ap.add_argument("--extra-pattern", action="append", default=[])
|
| 8 |
+
a = ap.parse_args()
|
| 9 |
+
root = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
| 10 |
+
pairs = []
|
| 11 |
+
for line in open(a.env):
|
| 12 |
+
m = re.match(r'\s*(?:export\s+)?([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(.*)$', line)
|
| 13 |
+
if m: pairs.append((m.group(1), m.group(2).strip().strip('"\'')))
|
| 14 |
+
files = [os.path.join(d, f) for d, ds, fs in os.walk(root) for f in fs
|
| 15 |
+
if not any(p in d.split(os.sep) for p in (".git", "__pycache__", ".venv", ".pytest_cache"))
|
| 16 |
+
and os.path.abspath(os.path.join(d, f)) != os.path.abspath(__file__)]
|
| 17 |
+
pat = re.compile("|".join([r"/Users/", r"/root/", r"/private/tmp", r"autodl", r"hf_[A-Za-z0-9]{30,}", r"sk-[A-Za-z0-9]{20,}",
|
| 18 |
+
r"ghp_[A-Za-z0-9]{20,}", r"AKIA[0-9A-Z]{16}", r"BEGIN [A-Z ]*PRIVATE KEY"] + a.extra_pattern), re.I)
|
| 19 |
+
hits = 0
|
| 20 |
+
for f in files:
|
| 21 |
+
t = open(f, errors="ignore").read(); rel = os.path.relpath(f, root)
|
| 22 |
+
for name, val in pairs:
|
| 23 |
+
if name in t: print(f"env NAME {name} in {rel}"); hits += 1
|
| 24 |
+
if len(val) >= 8 and val in t: print(f"env VALUE of {name} in {rel}"); hits += 1
|
| 25 |
+
for i, line in enumerate(t.splitlines(), 1):
|
| 26 |
+
for m in pat.finditer(line): print(f"pattern {m.group(0)!r} at {rel}:{i}"); hits += 1
|
| 27 |
+
print(f"{len(pairs)} env variables, {len(files)} files, {hits} hits")
|
| 28 |
+
sys.exit(1 if hits else 0)
|