Wald-4B / model-info.json
Harry19081's picture
main = Wald-Q4B v1.2 (02600-f19, robustness release): weights from v1.2-release, v1.2 serving.json (effort none) and runbook; card: v1.2 on main, v1.1 at tag v1.1
981b91b verified
Raw History Blame Contribute Delete
6.15 kB
{
"name": "Wald-Q4B",
"version": "v1.2",
"aliases": [
"Wald-4B"
],
"checkpoint": "02600-f19",
"description": "Open-weight 4B decision model that returns a calibrated probability for every option through a Jev-compatible POST /v1/systemone API; one pass or optional thinking; self-hosted. Independent, not affiliated with TypeSafe AI. v1.2 is the robustness release: v1.1 plus one merged LoRA stage against distracting and adversarial text.",
"base_model": "Qwen/Qwen3.5-4B-Base",
"model_url": "https://huggingface.co/org2ai/Wald-4B",
"weights_tag": "v1.2",
"weight_precision": "BF16",
"parameter_scale": "4B",
"license": "Apache-2.0 (weights and code); see PROVENANCE.md for source-data usage limitations",
"use_cases": [
"tool selection",
"agent routing",
"classification",
"clarification decisions"
],
"endpoint": "POST /v1/systemone",
"input_fields": [
"state",
"questions",
"effort"
],
"output": "per-option probabilities",
"default_effort": "none",
"efforts": {
"none": {
"gate": 0,
"generated_thoughts": 0
},
"low": {
"gate": 0.5,
"max_thought_tokens": 512
},
"medium": {
"gate": 0.7,
"max_thought_tokens": 512
},
"high": {
"gate": 1.01,
"max_thought_tokens": 512
},
"high-k2..high-k8": {
"thoughts": "2..8",
"max_tokens_per_thought": 512
}
},
"thinking_scope": "2–26 options and sufficient context; wider sets use grouped readout; insufficient thought space retains initial answer",
"evaluation": {
"jevadvbench": {
"questions": 812,
"attack_types": 9,
"effort": "none",
"mean_flip_rate_pct": 4.6,
"v1_1_mean_flip_rate_pct": 9.2,
"jev_1_13_mean_flip_rate_pct": 6.1,
"paired_diff_vs_v1_1_pp": [
-4.6,
-5.5,
-3.6
],
"clean_accuracy_143_human_reviewed_pct": 76.2,
"v1_1_clean_accuracy_pct": 79.0,
"clean_paired_diff_vs_v1_1_pp": [
-2.8,
-6.2,
-0.6
],
"harness": "JevAdvBench/JevAdvBench@3218e05 request bytes and analysis code",
"status": "self-run; not submitted"
},
"jevbench_public": {
"items": 231,
"effort": "none",
"correct": 204,
"accuracy": 0.8831,
"ece": 0.045,
"brier": 0.191,
"hardware": "NVIDIA RTX 5090 32 GB",
"harness": "fstandhartinger/jevbench@9ec6f15a (jevbench.cli, typesafe adapter, serial, loopback)",
"status": "self-scored; public items were a development scoreboard, not held out; not submitted",
"thinking_check": {
"medium_correct": 198,
"high_correct": 198,
"runs_each": 1
}
},
"decision_index_sample": {
"edition": "0.2.1",
"requests": 6948,
"read": "one pass",
"balanced_skill": 50.03,
"v1_1_balanced_skill": 49.76,
"paired_diff": [
0.27,
-0.36,
1.05
],
"status": "sample only; no complete-suite run for v1.2"
},
"details": "https://huggingface.co/org2ai/Wald-4B/blob/v1.2/evaluation/v1.2/summary.json"
},
"documentation_languages": [
"en",
"zh"
],
"docs": {
"en": "README.md",
"zh": "docs/readmes/README.zh.md",
"serving": "RUNBOOK.md",
"sources": "PROVENANCE.md",
"evaluation_notes": "CONTAMINATION.md",
"api": "docs/api.md",
"citation": "CITATION.cff",
"llms": "llms.txt"
},
"affiliation": "Independent; not affiliated with TypeSafe AI",
"evaluations": [
{
"note": "v1.1 results (not this revision's weights)",
"decision_index_complete_suite": {
"benchmark": "Decision Index",
"edition": "0.2.1",
"scope": "complete suite",
"score": 54.59,
"metric": "balanced_skill",
"requests": 150317,
"ok": 150317,
"effort": "high",
"hardware": "NVIDIA RTX PRO 6000 96 GB",
"status": "author-run; maintainer validation pending",
"submission": "https://github.com/apolinario/decision-index/pull/30"
},
"jevbench_public": {
"benchmark": "JevBench",
"split": "public",
"items": 231,
"effort": "none",
"correct": 203,
"accuracy": 0.8788,
"ece": 0.041,
"brier": 0.188,
"latency_p50_ms": 33,
"latency_p95_ms": 168,
"hardware": "NVIDIA RTX PRO 6000 96 GB",
"harness": "fstandhartinger/jevbench@9ec6f15a (jevbench.cli, typesafe adapter, serial, loopback)",
"status": "self-scored; public items were a development scoreboard, not held out; leaderboard measurement requested",
"request": "https://github.com/fstandhartinger/jevbench/issues/146"
}
}
],
"links": {
"model": "https://huggingface.co/org2ai/Wald-4B",
"github": "https://github.com/org2AI/wald-4b",
"results_dataset": "https://huggingface.co/datasets/org2ai/Wald-Q4B-decision-index-results",
"decision_index_submission": "https://github.com/apolinario/decision-index/pull/30",
"jevbench_request": "https://github.com/fstandhartinger/jevbench/issues/146",
"api": "https://huggingface.co/org2ai/Wald-4B/blob/main/docs/api.md",
"evaluation_summary": "https://huggingface.co/org2ai/Wald-4B/blob/v1.2/evaluation/v1.2/summary.json"
},
"weights_sha256": {
"model-00001-of-00002.safetensors": "0b15067f769e7388bd598aabafa2ea3211a3a0c4e49e47d07c4f65c65ffc25dd",
"model-00002-of-00002.safetensors": "7cff102b314edeb3dc5abfa7063f72ad3e888884da237601239dc3582074120a"
},
"parent_release": {
"version": "v1.1",
"checkpoint": "022D0-f7",
"tag": "v1.1",
"revision": "50f94ecd5d6e7e8459e9c198ef811bfaba12a3fe"
},
"thinking_note": "v1.2 is a one-pass model. On the JevBench public set it scored 198/231 with medium and with high (one run each) against 204/231 with none. Thinking efforts are evaluated on v1.1.",
"training_data_note": "perturbed copies of v1.1 training questions; inserted texts written by Claude Haiku (Anthropic) from our templates; targets are v1.1 answers on the clean questions; no JevAdvBench text"
}