Banaxi-Tech commited on
Commit
34ffe11
Β·
verified Β·
1 Parent(s): 3820e2a

Fix Specialists

Browse files

Fix Specialists and Half Specialists by classifying a model as specialist if the Arithmark 2 score is more than 30% higher in points than Hellaswag. This fixes models like Atom 2.7 While Also Fixing models like Nexus Erebus 135M that train on alot of synthetic arithmetic and almost no actual web corpus that didn't get flagged as specialist before. For models that actually perform like this there is a exclusion list.

Files changed (1) hide show
  1. index.html +33 -34
index.html CHANGED
@@ -57,14 +57,14 @@
57
  font-weight: 400;
58
  letter-spacing: -0.2px;
59
  }
60
-
61
  .hero-sub a {
62
  color: var(--accent);
63
  text-decoration: none;
64
  border-bottom: 1px solid transparent;
65
  transition: border-color 0.15s;
66
  }
67
-
68
  .hero-sub a:hover {
69
  border-bottom-color: var(--accent);
70
  }
@@ -929,12 +929,12 @@
929
  To support our work and help us keep this leaderboard up to date, please consider giving the space a like and follow!
930
  </p>
931
  </section>
932
-
933
  <!-- HIGHLIGHTS -->
934
  <section id="highlights" style="padding-top:20px;">
935
  <div class="insight-grid" id="insight-grid"></div>
936
  </section>
937
-
938
  <!-- NOTABLE NEW RELEASES -->
939
  <section id="new-releases" style="padding-top:40px;">
940
  <div class="release-head">
@@ -1248,12 +1248,12 @@
1248
  { name: 'BananaMind-2-Medium-Chat', org: 'bananamind', params: 49559552, paramsDisplay: '49.6M', arc: 43.43, hellaswag: 31.71, piqa: 60.66, arcChall: 24.40, arithmark2: 29.92, links: { card: 'https://huggingface.co/BananaMind/BananaMind-2-Medium-Chat' } },
1249
  { name: 'Photon-2.0-1M', org: 'atomixlabs', params: 1049728, paramsDisplay: '1.0M', arc: 30.05, hellaswag: 28.17, piqa: 53.92, arcChall: 22.53, arithmark2: 26.64, links: { card: 'https://huggingface.co/AtomixLabs/Photon-2.0-1M' } },
1250
  { name: 'Nexus-Erebus-135M', org: 'ideoalabs', params: 134000000, paramsDisplay: '135M', arc: 48.61, hellaswag: 30.61, piqa: 64.53, arcChall: 25.43, arithmark2: 66.68, links: { card: 'https://huggingface.co/MaliosDark/Nexus-Erebus-135M' } },
1251
- ];
1252
 
1253
  // ═══════════════════════════════════════════════════════════════
1254
  // ═══ STATE
1255
  // ═══════════════════════════════════════════════════════════════
1256
-
1257
  const BENCHMARKS = ['arc', 'hellaswag', 'piqa', 'arcChall', 'arithmark2'];
1258
  const METRICS = [
1259
  { key: 'avg', label: 'Avg', fullLabel: 'Average Score' },
@@ -1277,8 +1277,16 @@
1277
  let activeBenchmark = 'avg';
1278
  let chartInstances = {};
1279
  const ORG_MOVEMENT_CUTOFF_TYPE = 'orgMovementCutoff';
1280
- const SPECIALIST_MIN_FIELD_Z = 3.5;
1281
- const SPECIALIST_MIN_Z_GAP_TO_NEXT_METRIC = 2.5;
 
 
 
 
 
 
 
 
1282
 
1283
  function isBelowOrgMovementCutoff(model) {
1284
  const cutoffIndex = MODELS.findIndex(m => m.type === ORG_MOVEMENT_CUTOFF_TYPE);
@@ -1321,37 +1329,28 @@
1321
  return item ? (full ? item.fullLabel : item.label) : metric;
1322
  }
1323
 
1324
- function getSpecialistMetric(model, models = getModelEntries(MODELS)) {
1325
- const zScores = BENCHMARKS.map(metric => {
1326
- const score = model[metric];
1327
- if (score === null || score === undefined) return null;
1328
-
1329
- const fieldScores = getModelEntries(models)
1330
- .map(m => m[metric])
1331
- .filter(v => v !== null && v !== undefined);
1332
- if (fieldScores.length < 2) return null;
1333
-
1334
- const mean = fieldScores.reduce((sum, v) => sum + v, 0) / fieldScores.length;
1335
- const std = getStdDev(fieldScores);
1336
- if (!std) return null;
1337
 
1338
- return { metric, z: (score - mean) / std, score };
1339
- }).filter(Boolean).sort((a, b) => b.z - a.z);
 
 
1340
 
1341
- if (zScores.length < 2) return null;
 
 
1342
 
1343
- const [top, next] = zScores;
1344
- return top.z >= SPECIALIST_MIN_FIELD_Z && (top.z - next.z) >= SPECIALIST_MIN_Z_GAP_TO_NEXT_METRIC
1345
- ? top.metric
1346
- : null;
1347
- }
1348
 
1349
- function isSpecialistModel(model, models = getFilteredModels()) {
1350
- return Boolean(getSpecialistMetric(model));
1351
- }
1352
  function getModelEntries(models = MODELS) {
1353
- return models.filter(m => !m.type);
1354
- }
1355
 
1356
  function getFilteredModels(models = MODELS) {
1357
  let arr = getModelEntries(models);
 
57
  font-weight: 400;
58
  letter-spacing: -0.2px;
59
  }
60
+
61
  .hero-sub a {
62
  color: var(--accent);
63
  text-decoration: none;
64
  border-bottom: 1px solid transparent;
65
  transition: border-color 0.15s;
66
  }
67
+
68
  .hero-sub a:hover {
69
  border-bottom-color: var(--accent);
70
  }
 
929
  To support our work and help us keep this leaderboard up to date, please consider giving the space a like and follow!
930
  </p>
931
  </section>
932
+
933
  <!-- HIGHLIGHTS -->
934
  <section id="highlights" style="padding-top:20px;">
935
  <div class="insight-grid" id="insight-grid"></div>
936
  </section>
937
+
938
  <!-- NOTABLE NEW RELEASES -->
939
  <section id="new-releases" style="padding-top:40px;">
940
  <div class="release-head">
 
1248
  { name: 'BananaMind-2-Medium-Chat', org: 'bananamind', params: 49559552, paramsDisplay: '49.6M', arc: 43.43, hellaswag: 31.71, piqa: 60.66, arcChall: 24.40, arithmark2: 29.92, links: { card: 'https://huggingface.co/BananaMind/BananaMind-2-Medium-Chat' } },
1249
  { name: 'Photon-2.0-1M', org: 'atomixlabs', params: 1049728, paramsDisplay: '1.0M', arc: 30.05, hellaswag: 28.17, piqa: 53.92, arcChall: 22.53, arithmark2: 26.64, links: { card: 'https://huggingface.co/AtomixLabs/Photon-2.0-1M' } },
1250
  { name: 'Nexus-Erebus-135M', org: 'ideoalabs', params: 134000000, paramsDisplay: '135M', arc: 48.61, hellaswag: 30.61, piqa: 64.53, arcChall: 25.43, arithmark2: 66.68, links: { card: 'https://huggingface.co/MaliosDark/Nexus-Erebus-135M' } },
1251
+ ];
1252
 
1253
  // ═══════════════════════════════════════════════════════════════
1254
  // ═══ STATE
1255
  // ═══════════════════════════════════════════════════════════════
1256
+
1257
  const BENCHMARKS = ['arc', 'hellaswag', 'piqa', 'arcChall', 'arithmark2'];
1258
  const METRICS = [
1259
  { key: 'avg', label: 'Avg', fullLabel: 'Average Score' },
 
1277
  let activeBenchmark = 'avg';
1278
  let chartInstances = {};
1279
  const ORG_MOVEMENT_CUTOFF_TYPE = 'orgMovementCutoff';
1280
+ // 'relative' β†’ ArithMark-2 > HellaSwag * 1.30 (30% higher)
1281
+ // 'points' β†’ ArithMark-2 - HellaSwag > 30 (30 percentage points higher)
1282
+ const SPECIALIST_MODE = 'points';
1283
+ const SPECIALIST_MARGIN = 30;
1284
+ // Models that must never be flagged as specialists, regardless of the rule.
1285
+ // Names must match the `name` field in MODELS exactly.
1286
+ const SPECIALIST_EXCLUSIONS = new Set([
1287
+ // 'MobileLLM-R1-140M-base',
1288
+ // 'Quark-135M',
1289
+ ]);
1290
 
1291
  function isBelowOrgMovementCutoff(model) {
1292
  const cutoffIndex = MODELS.findIndex(m => m.type === ORG_MOVEMENT_CUTOFF_TYPE);
 
1329
  return item ? (full ? item.fullLabel : item.label) : metric;
1330
  }
1331
 
1332
+ function getSpecialistMetric(model) {
1333
+ if (!model || model.type) return null;
1334
+ if (SPECIALIST_EXCLUSIONS.has(model.name)) return null;
 
 
 
 
 
 
 
 
 
 
1335
 
1336
+ const arith = model.arithmark2;
1337
+ const hella = model.hellaswag;
1338
+ if (arith === null || arith === undefined) return null;
1339
+ if (hella === null || hella === undefined || hella <= 0) return null;
1340
 
1341
+ const flagged = SPECIALIST_MODE === 'relative'
1342
+ ? arith > hella * (1 + SPECIALIST_MARGIN / 100)
1343
+ : (arith - hella) > SPECIALIST_MARGIN;
1344
 
1345
+ return flagged ? 'arithmark2' : null;
1346
+ }
 
 
 
1347
 
1348
+ function isSpecialistModel(model) {
1349
+ return Boolean(getSpecialistMetric(model));
1350
+ }
1351
  function getModelEntries(models = MODELS) {
1352
+ return models.filter(m => !m.type);
1353
+ }
1354
 
1355
  function getFilteredModels(models = MODELS) {
1356
  let arr = getModelEntries(models);