diff --git "a/contamination/trained-on-di-ids.json" "b/contamination/trained-on-di-ids.json" new file mode 100644--- /dev/null +++ "b/contamination/trained-on-di-ids.json" @@ -0,0 +1,3862 @@ +{ + "schema": "wald/di-trained-on/1", + "what": "Decision Index 0.2 / 0.2.1 item ids present in the training data of the submitted checkpoint's lineage (strict match: the whole state in one training record, plus >= 50 % of option text where options are the item's content). Count every one of them as trained on.", + "total": 363, + "per_benchmark": [ + { + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "items": 67 + }, + { + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "items": 49 + }, + { + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "items": 44 + }, + { + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "items": 170 + }, + { + "benchmark_id": 12, + "benchmark": "ANLI", + "scored_in_0_2_1": "scored", + "items": 3 + }, + { + "benchmark_id": 24, + "benchmark": "MMLU", + "scored_in_0_2_1": "display only", + "items": 6 + }, + { + "benchmark_id": 26, + "benchmark": "ARC-Easy", + "scored_in_0_2_1": "display only", + "items": 7 + }, + { + "benchmark_id": 27, + "benchmark": "ARC-Challenge", + "scored_in_0_2_1": "display only", + "items": 4 + }, + { + "benchmark_id": 28, + "benchmark": "WinoGrande", + "scored_in_0_2_1": "scored", + "items": 2 + }, + { + "benchmark_id": 33, + "benchmark": "SATA-Bench", + "scored_in_0_2_1": "scored", + "items": 4 + }, + { + "benchmark_id": 36, + "benchmark": "BRIGHT", + "scored_in_0_2_1": "scored", + "items": 1 + }, + { + "benchmark_id": 57, + "benchmark": "MMLU-Pro", + "scored_in_0_2_1": "scored", + "items": 6 + } + ], + "scans": { + "stage 1 (full-parameter decision training corpus)": { + "strict_hits_in_trained_files": 177, + "blocklist_sha256": "a9dc4133464334273a6326c125744beaf6e44d45eba1d25fae5b61225e1ad4ba", + "suite_fingerprints_sha256": "df92bd3ceed2f57df7d25e219a22e18521097db625403b5c3de90226b809b2a5", + "files": { + "train-00000.jsonl": { + "lines": 92013, + "sha256": "ed48083526b4b063ee8be3bc921b018ef1ec63d2c308e03d8ee57574c292431b", + "strict": 36 + }, + "train-00001.jsonl": { + "lines": 92531, + "sha256": "ac12385f728387d46cb97a9323c42be6e5d5f32ef83f0a257efbf9f8db980bc4", + "strict": 44 + }, + "train-00002.jsonl": { + "lines": 92089, + "sha256": "10bbbe85daaafef4d575abc3d12b3738d1484e93caf64129ffbba2b98413f86a", + "strict": 44 + }, + "train-00003.jsonl": { + "lines": 92440, + "sha256": "3a883ad81670c1b51dd583e1eaf346e4abe4336cfcd59a15dd1227f831e33052", + "strict": 53 + }, + "train-00004.jsonl": { + "lines": 2900, + "sha256": "8b3e1b34ab30c8746f8858047aa31dd39f42a4bdaebe46786e19e9762aa02e9a", + "strict": 0 + } + } + }, + "stage 2 (LoRA refinement data)": { + "strict_hits_in_trained_files": 42, + "blocklist_sha256": "a9dc4133464334273a6326c125744beaf6e44d45eba1d25fae5b61225e1ad4ba", + "suite_fingerprints_sha256": "df92bd3ceed2f57df7d25e219a22e18521097db625403b5c3de90226b809b2a5", + "files": { + "train-f3base-00000.jsonl": { + "lines": 18441, + "sha256": "9a882f66a815d27bf31264fc19a9680e04bf59340d0b275d87d017925a46a870", + "strict": 36 + }, + "train-replay-00000.jsonl": { + "lines": 6746, + "sha256": "b236f01529a76d45dbcf5a049b629fc8c9da2f30eab6e14b615360b248dc608c", + "strict": 6 + }, + "train-teacher-00000.jsonl": { + "lines": 63, + "sha256": "ae4ce672535b2fa9a11af0df8f2866e61a2cdbe80042d1f5607778d60e0af98c", + "strict": 0 + } + } + }, + "stage 3 (short-thought distillation data)": { + "strict_hits_in_trained_files": 158, + "blocklist_sha256": "a9dc4133464334273a6326c125744beaf6e44d45eba1d25fae5b61225e1ad4ba", + "suite_fingerprints_sha256": "df92bd3ceed2f57df7d25e219a22e18521097db625403b5c3de90226b809b2a5", + "files": { + "train-final.jsonl": { + "lines": 33361, + "sha256": "ecaaf022dd5376e6823d2c5f8acd9602e3cc76a06fb582a72cb3287edb311d9e", + "strict": 79 + }, + "train.jsonl": { + "lines": 33361, + "sha256": "846caf99110bff0d8c26e35c43aee35da72d44b33b26599b1fb53d9dee762a0d", + "strict": 79 + } + } + } + }, + "stage_4_targeted_lora": "stage-4 data (Home-appliance generator rows; iSarcasmEval / API-Bank / ContractNLI / VAST / NLI4CT / ACOS / RAGTruth train splits; replay) was checked against the full blocklist and the sample rows before training: 0 hits after filtering", + "items": [ + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:craft-math-algebra:craft_Math_algebra_query_170:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_109:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_128:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_12:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_13:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_191:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_194:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_194:chunk1", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_213:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_217:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_232:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_232:chunk1", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_234:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_238:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_241:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_257:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_299:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_307:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_323:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_340:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_343:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_343:chunk1", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_343:chunk2", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_34:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_357:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_358:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_363:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_378:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_37:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_386:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_392:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_395:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_406:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_406:chunk1", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_420:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_422:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_439:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_45:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_516:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_516:chunk1", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_570:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_575:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_57:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_57:chunk1", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_580:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_59:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_601:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_603:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_632:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_65:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_667:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_687:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_705:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_705:chunk1", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_734:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_742:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_766:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_781:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_796:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_85:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_86:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_888:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_905:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_90:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_91:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_91:chunk1", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "2:ToolRet-retrieval:ToolRet-retrieval:toolace:toolACE_query_980:chunk0", + "benchmark_id": 2, + "benchmark": "ToolRet", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: ToolRet aggregates user queries of public tool-calling train sets (ToolACE, Glaive) that are in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1197", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1198", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1246", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1289", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1322", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1332", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1342", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1432", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1437", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1474", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1573", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1616", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1687", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1735", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1817", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1821", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1936", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1946", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:1993", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2083", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2139", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2145", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2149", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2254", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2298", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2331", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2381", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2436", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2441", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2466", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2476", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:250", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:256", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2601", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2605", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2755", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:2821", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:3012", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:3070", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:3074", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:347", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:391", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:450", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:460", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:510", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:888", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:890", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:891", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "4:BANKING77:BANKING77:test:893", + "benchmark_id": 4, + "benchmark": "BANKING77", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the benchmark's public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:1055", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:1126", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:1132", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:1215", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:1233", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:1252", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:1332", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:1333", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:1487", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:1505", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:1518", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:1809", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:1922", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:2234", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:2236", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:2243", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:2246", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:2372", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:2446", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:2658", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:274", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:2752", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3316", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3321", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3629", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3630", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3631", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3637", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3643", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3644", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3645", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3652", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3658", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3667", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3679", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3705", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3808", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:3865", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:4281", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:4457", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:659", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:671", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:70", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "5:CLINC150+OOS:CLINC150+OOS:test:930", + "benchmark_id": 5, + "benchmark": "CLINC150+OOS", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence as the test item", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:10002", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:10008", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:10026", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:10027", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:10060", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:10071", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:10180", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:10345", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:10350", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:10548", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:3624", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:3645", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:3661", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:3795", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4005", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4168", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4223", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4296", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4367", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4388", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4422", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4434", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4438", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4497", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4558", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4769", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4807", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:4829", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:5056", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:5116", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:5153", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:5192", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:5208", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:5227", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:5240", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:5244", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:5326", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:5451", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:5480", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:5501", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:5597", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:6265", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:6566", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:6659", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:6699", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:6733", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:6743", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:6795", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:6872", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:6916", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:7172", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:7209", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:7270", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:7524", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:7566", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:7705", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:7964", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8017", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8023", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8079", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8130", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8131", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8302", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8421", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8525", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8634", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8648", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8691", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8718", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8766", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8807", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8830", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8913", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:8943", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:9255", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:9256", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:9292", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:9363", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:9411", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:9431", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:9439", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:9484", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:9620", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:9651", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-0shot:RouterBench-0shot:9801", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:10016", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:10022", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:10040", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:10041", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:10074", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:10085", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:10194", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:10359", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:10364", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:10562", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:3638", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:3659", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:3675", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:3809", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4019", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4182", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4237", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4310", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4381", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4402", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4436", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4448", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4452", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4511", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4572", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4783", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4821", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:4843", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:5070", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:5130", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:5167", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:5206", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:5222", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:5241", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:5254", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:5258", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:5340", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:5465", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:5494", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:5515", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:5611", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:6279", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:6580", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:6673", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:6713", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:6747", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:6757", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:6809", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:6886", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:6930", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:7186", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:7223", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:7284", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:7538", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:7580", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:7719", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:7978", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8031", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8037", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8093", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8144", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8145", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8316", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8435", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8539", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8648", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8662", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8705", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8732", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8780", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8821", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8844", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8927", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:8957", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:9269", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:9270", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:9306", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:9377", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:9425", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:9445", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:9453", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:9498", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:9634", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:9665", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "6:RouterBench-5shot:RouterBench-5shot:9815", + "benchmark_id": 6, + "benchmark": "RouterBench", + "scored_in_0_2_1": "not scored in 0.2.1", + "why": "shared upstream source: RouterBench prompts embed GSM8K / MMLU questions that are in our math and QA data", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "12:ANLI:ANLI:test_r2:4bb4236d-e277-443c-9a83-05aa3e1f36ab", + "benchmark_id": 12, + "benchmark": "ANLI", + "scored_in_0_2_1": "scored", + "why": "shared upstream text: the premise text also appears in another public dataset in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "12:ANLI:ANLI:test_r2:5c674fb6-d204-46ff-8ab1-19b483d97f14", + "benchmark_id": 12, + "benchmark": "ANLI", + "scored_in_0_2_1": "scored", + "why": "shared upstream text: the premise text also appears in another public dataset in our data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "12:ANLI:ANLI:test_r3:af91dd15-20c2-4428-9bdf-8eb8eb1ea180", + "benchmark_id": 12, + "benchmark": "ANLI", + "scored_in_0_2_1": "scored", + "why": "shared upstream text: the premise text also appears in another public dataset in our data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "24:MMLU:MMLU:test:4250", + "benchmark_id": 24, + "benchmark": "MMLU", + "scored_in_0_2_1": "display only", + "why": "question also present in MMLU auxiliary_train / other public QA sets", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "24:MMLU:MMLU:test:4261", + "benchmark_id": 24, + "benchmark": "MMLU", + "scored_in_0_2_1": "display only", + "why": "question also present in MMLU auxiliary_train / other public QA sets", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "24:MMLU:MMLU:test:4274", + "benchmark_id": 24, + "benchmark": "MMLU", + "scored_in_0_2_1": "display only", + "why": "question also present in MMLU auxiliary_train / other public QA sets", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "24:MMLU:MMLU:test:4281", + "benchmark_id": 24, + "benchmark": "MMLU", + "scored_in_0_2_1": "display only", + "why": "question also present in MMLU auxiliary_train / other public QA sets", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "24:MMLU:MMLU:test:4419", + "benchmark_id": 24, + "benchmark": "MMLU", + "scored_in_0_2_1": "display only", + "why": "question also present in MMLU auxiliary_train / other public QA sets", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "24:MMLU:MMLU:test:7920", + "benchmark_id": 24, + "benchmark": "MMLU", + "scored_in_0_2_1": "display only", + "why": "question also present in MMLU auxiliary_train / other public QA sets", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "26:ARC-Easy:ARC-Easy:test:CSZ20770", + "benchmark_id": 26, + "benchmark": "ARC-Easy", + "scored_in_0_2_1": "display only", + "why": "train-split duplicate (ARC train / MMLU auxiliary_train), the board's known duplicates", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "26:ARC-Easy:ARC-Easy:test:CSZ30768", + "benchmark_id": 26, + "benchmark": "ARC-Easy", + "scored_in_0_2_1": "display only", + "why": "train-split duplicate (ARC train / MMLU auxiliary_train), the board's known duplicates", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "26:ARC-Easy:ARC-Easy:test:LEAP__4_10224", + "benchmark_id": 26, + "benchmark": "ARC-Easy", + "scored_in_0_2_1": "display only", + "why": "train-split duplicate (ARC train / MMLU auxiliary_train), the board's known duplicates", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "26:ARC-Easy:ARC-Easy:test:MDSA_2013_8_7", + "benchmark_id": 26, + "benchmark": "ARC-Easy", + "scored_in_0_2_1": "display only", + "why": "train-split duplicate (ARC train / MMLU auxiliary_train), the board's known duplicates", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "26:ARC-Easy:ARC-Easy:test:Mercury_7172813", + "benchmark_id": 26, + "benchmark": "ARC-Easy", + "scored_in_0_2_1": "display only", + "why": "train-split duplicate (ARC train / MMLU auxiliary_train), the board's known duplicates", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "26:ARC-Easy:ARC-Easy:test:NYSEDREGENTS_2012_4_20", + "benchmark_id": 26, + "benchmark": "ARC-Easy", + "scored_in_0_2_1": "display only", + "why": "train-split duplicate (ARC train / MMLU auxiliary_train), the board's known duplicates", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "26:ARC-Easy:ARC-Easy:test:NYSEDREGENTS_2012_8_18", + "benchmark_id": 26, + "benchmark": "ARC-Easy", + "scored_in_0_2_1": "display only", + "why": "train-split duplicate (ARC train / MMLU auxiliary_train), the board's known duplicates", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "27:ARC-Challenge:ARC-Challenge:test:CSZ_2009_8_CSZ20740", + "benchmark_id": 27, + "benchmark": "ARC-Challenge", + "scored_in_0_2_1": "display only", + "why": "train-split duplicate (ARC train / MMLU auxiliary_train), the board's known duplicates", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "27:ARC-Challenge:ARC-Challenge:test:Mercury_406136", + "benchmark_id": 27, + "benchmark": "ARC-Challenge", + "scored_in_0_2_1": "display only", + "why": "train-split duplicate (ARC train / MMLU auxiliary_train), the board's known duplicates", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "27:ARC-Challenge:ARC-Challenge:test:Mercury_SC_416167", + "benchmark_id": 27, + "benchmark": "ARC-Challenge", + "scored_in_0_2_1": "display only", + "why": "train-split duplicate (ARC train / MMLU auxiliary_train), the board's known duplicates", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "27:ARC-Challenge:ARC-Challenge:test:NYSEDREGENTS_2015_4_29", + "benchmark_id": 27, + "benchmark": "ARC-Challenge", + "scored_in_0_2_1": "display only", + "why": "train-split duplicate (ARC train / MMLU auxiliary_train), the board's known duplicates", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "28:WinoGrande:WinoGrande:validation:1193", + "benchmark_id": 28, + "benchmark": "WinoGrande", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "28:WinoGrande:WinoGrande:validation:876", + "benchmark_id": 28, + "benchmark": "WinoGrande", + "scored_in_0_2_1": "scored", + "why": "train-split duplicate: the public train split contains the same sentence", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "33:SATA-Bench:SATA-Bench:test:195", + "benchmark_id": 33, + "benchmark": "SATA-Bench", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: reading passage from RACE, which is part of MMLU auxiliary_train", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "33:SATA-Bench:SATA-Bench:test:29", + "benchmark_id": 33, + "benchmark": "SATA-Bench", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: reading passage from RACE, which is part of MMLU auxiliary_train", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "33:SATA-Bench:SATA-Bench:test:295", + "benchmark_id": 33, + "benchmark": "SATA-Bench", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: reading passage from RACE, which is part of MMLU auxiliary_train", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "33:SATA-Bench:SATA-Bench:test:30", + "benchmark_id": 33, + "benchmark": "SATA-Bench", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: reading passage from RACE, which is part of MMLU auxiliary_train", + "found_in": [ + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "36:BRIGHT-retrieval:BRIGHT-retrieval:leetcode:140:chunk0", + "benchmark_id": 36, + "benchmark": "BRIGHT", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: a public programming problem statement (LeetCode)", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "candidates-v3:57:5457", + "benchmark_id": 57, + "benchmark": "MMLU-Pro", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: MMLU-Pro includes problems from MATH / TheoremQA; the same problem is in our math data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "candidates-v3:57:8161", + "benchmark_id": 57, + "benchmark": "MMLU-Pro", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: MMLU-Pro includes problems from MATH / TheoremQA; the same problem is in our math data", + "found_in": [ + "stage 2 (LoRA refinement data)" + ] + }, + { + "id": "candidates-v3:57:8164", + "benchmark_id": 57, + "benchmark": "MMLU-Pro", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: MMLU-Pro includes problems from MATH / TheoremQA; the same problem is in our math data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "candidates-v3:57:8273", + "benchmark_id": 57, + "benchmark": "MMLU-Pro", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: MMLU-Pro includes problems from MATH / TheoremQA; the same problem is in our math data", + "found_in": [ + "stage 1 (full-parameter decision training corpus)" + ] + }, + { + "id": "candidates-v3:57:8713", + "benchmark_id": 57, + "benchmark": "MMLU-Pro", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: MMLU-Pro includes problems from MATH / TheoremQA; the same problem is in our math data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + }, + { + "id": "candidates-v3:57:8721", + "benchmark_id": 57, + "benchmark": "MMLU-Pro", + "scored_in_0_2_1": "scored", + "why": "shared upstream source: MMLU-Pro includes problems from MATH / TheoremQA; the same problem is in our math data", + "found_in": [ + "stage 2 (LoRA refinement data)", + "stage 3 (short-thought distillation data)" + ] + } + ] +}