evijit HF Staff commited on
Commit
d8c2856
·
1 Parent(s): dd0b4fc

Differentiate audience modes and tighten eval navigation

Browse files
app/evals/page.tsx CHANGED
@@ -2,13 +2,14 @@
2
 
3
  import { useCallback, useEffect, useMemo, useRef, useState } from "react"
4
  import { useRouter } from "next/navigation"
5
- import { ArrowLeft, Search, X } from "lucide-react"
6
 
7
  import { useAudienceMode } from "@/components/audience-mode-provider"
8
  import { ListPagination } from "@/components/list-pagination"
9
  import { Navigation } from "@/components/navigation"
10
  import { PageHeader } from "@/components/page-header"
11
  import { Button } from "@/components/ui/button"
 
12
  import { Input } from "@/components/ui/input"
13
  import type { EvalHierarchy } from "@/lib/backend-artifacts"
14
  import type { BenchmarkCard, CategoryType } from "@/lib/benchmark-schema"
@@ -262,6 +263,7 @@ interface EvalBrowserNode {
262
  domains: string[]
263
  dataType?: string
264
  license?: string
 
265
  modelsCount: number
266
  metricCount: number
267
  topScore?: number
@@ -353,6 +355,71 @@ function formatCompactScore(value: number | undefined) {
353
  return value.toFixed(value >= 100 ? 0 : 2)
354
  }
355
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
356
  function getNodeCard(
357
  benchmarkCards: Record<string, BenchmarkCard>,
358
  ...candidates: Array<string | undefined>
@@ -452,6 +519,7 @@ function mapHierarchyCategory(value: string | undefined | null): CategoryType {
452
  export default function EvalsPage() {
453
  const { mode } = useAudienceMode()
454
  const router = useRouter()
 
455
 
456
  const [summaries, setSummaries] = useState<BenchmarkEvalListItem[]>([])
457
  const [benchmarkCards, setBenchmarkCards] = useState<Record<string, BenchmarkCard>>({})
@@ -464,6 +532,7 @@ export default function EvalsPage() {
464
  const [selectedNodeKind, setSelectedNodeKind] = useState<EvalBrowserNodeKind | null>(null)
465
  const [currentNodeId, setCurrentNodeId] = useState<string | null>(null)
466
  const [page, setPage] = useState(1)
 
467
  const pendingHistoryActionRef = useRef<"push" | "replace">("replace")
468
 
469
  useEffect(() => {
@@ -635,6 +704,7 @@ export default function EvalsPage() {
635
  domains: Array.from(new Set(domains.flatMap((domain) => normalizeDomainList(domain)))),
636
  dataType: card?.benchmark_details?.data_type,
637
  license: card?.ethical_and_legal_considerations?.data_licensing,
 
638
  modelsCount: stats.modelsCount,
639
  metricCount: stats.metricCount,
640
  topScore: stats.topScore,
@@ -655,9 +725,9 @@ export default function EvalsPage() {
655
  metrics?: Array<{ key: string; display_name: string }>
656
  }>,
657
  scopeKeys: string[]
658
- ): EvalBrowserNode["matrixPreview"] | null => {
659
  if (benchmarks.length < 2) {
660
- return null
661
  }
662
 
663
  const metricLabels = new Set<string>()
@@ -665,7 +735,7 @@ export default function EvalsPage() {
665
 
666
  for (const benchmark of benchmarks) {
667
  if ((benchmark.slices?.length ?? 0) > 0 || (benchmark.metrics?.length ?? 0) !== 1) {
668
- return null
669
  }
670
 
671
  const metric = benchmark.metrics?.[0]
@@ -679,7 +749,7 @@ export default function EvalsPage() {
679
  }
680
 
681
  if (metricLabels.size !== 1) {
682
- return null
683
  }
684
 
685
  return {
@@ -747,6 +817,10 @@ export default function EvalsPage() {
747
  const card = summary?.benchmark_card ?? getNodeCard(benchmarkCards, ...cardCandidates)
748
  const childSlices = slices.filter((slice) => !isSameHierarchyKey(slice.key, benchmarkKey))
749
  const drilldownSlices = childSlices.filter((slice) => (slice.metrics?.length ?? 0) > 1)
 
 
 
 
750
  const isParentRollupBenchmark =
751
  Boolean(parentId) && scopeKeys.some((scopeKey) => isSameHierarchyKey(scopeKey, benchmarkKey))
752
 
@@ -775,9 +849,14 @@ export default function EvalsPage() {
775
  domains,
776
  summaries: summary ? [summary] : [],
777
  card,
778
- href: drilldownSlices.length === 0 && summary
779
- ? `/evals/${summary.evaluation_id}`
780
- : undefined,
 
 
 
 
 
781
  scopeKeys,
782
  descriptionFallback: `Browse the {label} benchmark and its lower-level breakdowns.`,
783
  })
@@ -874,6 +953,8 @@ export default function EvalsPage() {
874
  const suiteBenchmarks = (composite.benchmarks ?? []).filter((benchmark) => !isSameHierarchyKey(benchmark.key, composite.key))
875
  const suiteMatrixPreview = buildSingleMetricMatrixPreview(suiteBenchmarks, suiteScopeKeys)
876
  const rollupSummary = pickSummaryForKey(summariesWithCards, composite.key, suiteScopeKeys)
 
 
877
 
878
  buildNode({
879
  id: suiteId,
@@ -886,7 +967,13 @@ export default function EvalsPage() {
886
  summaries: suiteSummaries,
887
  card: suiteCard,
888
  sourceLabel: suiteLabel,
889
- href: suiteMatrixPreview && rollupSummary ? `/evals/${rollupSummary.evaluation_id}` : undefined,
 
 
 
 
 
 
890
  scopeKeys: suiteScopeKeys,
891
  matrixPreview: suiteMatrixPreview,
892
  descriptionFallback: `Browse the {label} suite and then open its benchmark children.`,
@@ -928,8 +1015,12 @@ export default function EvalsPage() {
928
  }
929
 
930
  const suiteNode = nodes.get(suiteId)
931
- if (suiteNode && suiteNode.childIds.length === 0 && rollupSummary) {
932
- suiteNode.href = `/evals/${rollupSummary.evaluation_id}`
 
 
 
 
933
  }
934
  }
935
 
@@ -979,6 +1070,8 @@ export default function EvalsPage() {
979
  const visibleBenchmarks = suiteBenchmarks.filter((benchmark) => !isSameHierarchyKey(benchmark.key, suiteKey))
980
  const suiteMatrixPreview = buildSingleMetricMatrixPreview(visibleBenchmarks, suiteScopeKeys)
981
  const rollupSummary = pickSummaryForKey(summariesWithCards, suiteKey, suiteScopeKeys)
 
 
982
  const suiteSummaries = summariesWithCards.filter((summary) => {
983
  const familyScope = getSummaryScopeKey(summary.benchmark_family_key ?? summary.composite_benchmark_key)
984
  return familyScope === getSummaryScopeKey(suiteKey)
@@ -1002,7 +1095,13 @@ export default function EvalsPage() {
1002
  family.key
1003
  ),
1004
  sourceLabel: suiteLabel,
1005
- href: suiteMatrixPreview && rollupSummary ? `/evals/${rollupSummary.evaluation_id}` : undefined,
 
 
 
 
 
 
1006
  scopeKeys: suiteScopeKeys,
1007
  matrixPreview: suiteMatrixPreview,
1008
  descriptionFallback: `Browse the {label} suite and then open its benchmark children.`,
@@ -1044,8 +1143,12 @@ export default function EvalsPage() {
1044
  }
1045
 
1046
  const suiteNode = nodes.get(suiteId)
1047
- if (suiteNode && suiteNode.childIds.length === 0 && rollupSummary) {
1048
- suiteNode.href = `/evals/${rollupSummary.evaluation_id}`
 
 
 
 
1049
  }
1050
 
1051
  continue
@@ -1250,6 +1353,12 @@ export default function EvalsPage() {
1250
  }
1251
  }, [currentLevelKinds, selectedNodeKind])
1252
 
 
 
 
 
 
 
1253
  const handleNodeOpen = useCallback(
1254
  (node: EvalBrowserNode) => {
1255
  if (node.childIds.length > 0) {
@@ -1455,116 +1564,150 @@ export default function EvalsPage() {
1455
  </div>
1456
  </div>
1457
 
1458
- {currentLevelKinds.length > 1 && (
1459
- <div className="mt-4 space-y-1.5">
1460
- <div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-stone-500 dark:text-stone-400">
1461
- Granularity
1462
- </div>
1463
- <div className="flex flex-wrap items-center gap-1.5">
1464
- <button
1465
- type="button"
1466
- onClick={() => setSelectedNodeKind(null)}
1467
- className={cn(
1468
- "shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
1469
- selectedNodeKind === null
1470
- ? "border-stone-950 bg-stone-950 text-stone-50 dark:border-stone-100 dark:bg-stone-100 dark:text-stone-950"
1471
- : "border-stone-200/80 bg-stone-50/80 text-stone-600 hover:bg-stone-100 dark:border-stone-700/80 dark:bg-stone-900/70 dark:text-stone-300 dark:hover:bg-stone-800"
 
 
 
 
 
 
 
 
 
1472
  )}
1473
- >
1474
- All
1475
- </button>
1476
- {currentLevelKinds.map((kind) => (
1477
- <button
1478
- key={kind}
1479
- type="button"
1480
- onClick={() => setSelectedNodeKind(selectedNodeKind === kind ? null : kind)}
1481
- className={cn(
1482
- "shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
1483
- selectedNodeKind === kind
1484
- ? "border-sky-300 bg-sky-50 text-sky-800 dark:border-sky-800 dark:bg-sky-950/50 dark:text-sky-200"
1485
- : "border-stone-200/80 bg-white text-stone-600 hover:bg-stone-50 dark:border-stone-700/80 dark:bg-stone-900 dark:text-stone-300 dark:hover:bg-stone-800"
1486
- )}
1487
- >
1488
- {getBrowserNodeKindLabel(kind)}
1489
- </button>
1490
- ))}
1491
- </div>
1492
- </div>
1493
- )}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1494
 
1495
- {allDomains.length > 0 && (
1496
- <div className="mt-4 space-y-1.5">
1497
- <div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-stone-500 dark:text-stone-400">
1498
- Domain
1499
- </div>
1500
- <div className="flex flex-wrap items-center gap-1.5">
1501
- <button
1502
- type="button"
1503
- onClick={() => setSelectedDomain(null)}
1504
- className={cn(
1505
- "shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
1506
- selectedDomain === null
1507
- ? "border-stone-950 bg-stone-950 text-stone-50 dark:border-stone-100 dark:bg-stone-100 dark:text-stone-950"
1508
- : "border-stone-200/80 bg-stone-50/80 text-stone-600 hover:bg-stone-100 dark:border-stone-700/80 dark:bg-stone-900/70 dark:text-stone-300 dark:hover:bg-stone-800"
1509
- )}
1510
- >
1511
- All
1512
- </button>
1513
- {allDomains.map((domain) => (
1514
- <button
1515
- key={domain}
1516
- type="button"
1517
- onClick={() => setSelectedDomain(selectedDomain === domain ? null : domain)}
1518
- className={cn(
1519
- "shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
1520
- selectedDomain === domain
1521
- ? "border-sky-300 bg-sky-50 text-sky-800 dark:border-sky-800 dark:bg-sky-950/50 dark:text-sky-200"
1522
- : "border-stone-200/80 bg-white text-stone-600 hover:bg-stone-50 dark:border-stone-700/80 dark:bg-stone-900 dark:text-stone-300 dark:hover:bg-stone-800"
1523
- )}
1524
- >
1525
- {domain}
1526
- </button>
1527
- ))}
1528
- </div>
1529
- </div>
1530
- )}
1531
 
1532
- {allCategories.length > 0 && (
1533
- <div className="mt-4 space-y-1.5">
1534
- <div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-stone-500 dark:text-stone-400">
1535
- Category
1536
- </div>
1537
- <div className="flex flex-wrap items-center gap-1.5">
1538
- <button
1539
- type="button"
1540
- onClick={() => setSelectedCategory(null)}
1541
- className={cn(
1542
- "shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
1543
- selectedCategory === null
1544
- ? "border-stone-950 bg-stone-950 text-stone-50 dark:border-stone-100 dark:bg-stone-100 dark:text-stone-950"
1545
- : "border-stone-200/80 bg-stone-50/80 text-stone-600 hover:bg-stone-100 dark:border-stone-700/80 dark:bg-stone-900/70 dark:text-stone-300 dark:hover:bg-stone-800"
1546
- )}
1547
- >
1548
- All
1549
- </button>
1550
- {allCategories.map((category) => (
1551
- <button
1552
- key={category}
1553
- type="button"
1554
- onClick={() => setSelectedCategory(selectedCategory === category ? null : category)}
1555
- className={cn(
1556
- "shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
1557
- selectedCategory === category
1558
- ? `${getCategoryColor(category as CategoryType)} border`
1559
- : "border-stone-200/80 bg-white text-stone-600 hover:bg-stone-50 dark:border-stone-700/80 dark:bg-stone-900 dark:text-stone-300 dark:hover:bg-stone-800"
1560
- )}
1561
- >
1562
- {category}
1563
- </button>
1564
- ))}
 
 
 
1565
  </div>
1566
- </div>
1567
- )}
1568
  </section>
1569
 
1570
  {filtered.length === 0 ? (
@@ -1580,6 +1723,8 @@ export default function EvalsPage() {
1580
  const isNavigable = node.childIds.length > 0 || Boolean(node.href)
1581
  const actionLabel = node.childIds.length > 0 ? "Open level" : node.matrixPreview ? "View rollup" : "View benchmark"
1582
  const kindLabel = getBrowserNodeKindLabel(node.kind)
 
 
1583
 
1584
  return (
1585
  <button
@@ -1655,6 +1800,84 @@ export default function EvalsPage() {
1655
  </p>
1656
  )}
1657
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1658
  {node.matrixPreview && (
1659
  <div className="mb-4 overflow-hidden rounded-[1.2rem] border border-stone-200/80 bg-stone-50/85 dark:border-stone-800/80 dark:bg-stone-900/85">
1660
  <div className="space-y-2 bg-white/92 px-3 py-3 dark:bg-stone-950/92">
@@ -1679,18 +1902,18 @@ export default function EvalsPage() {
1679
  <div className="mb-4 grid gap-px overflow-hidden rounded-[1.2rem] border border-stone-200/80 bg-stone-200/80 dark:border-stone-800/80 dark:bg-stone-800/80 sm:grid-cols-2 xl:grid-cols-3">
1680
  {topScoreLabel !== "—" && (
1681
  <div className="bg-white/90 px-3 py-3 dark:bg-stone-950/90">
1682
- <div className="text-[10px] font-semibold uppercase tracking-[0.16em] text-stone-500 dark:text-stone-400">Top score</div>
1683
  <div className="mt-1 text-sm font-semibold text-stone-900 dark:text-stone-100">{topScoreLabel}</div>
1684
  </div>
1685
  )}
1686
  <div className="bg-white/90 px-3 py-3 dark:bg-stone-950/90">
1687
- <div className="text-[10px] font-semibold uppercase tracking-[0.16em] text-stone-500 dark:text-stone-400">Source</div>
1688
  <div className="mt-1 truncate text-sm font-semibold text-stone-900 dark:text-stone-100">
1689
  {node.sourceLabel}
1690
  </div>
1691
  </div>
1692
  <div className="bg-white/90 px-3 py-3 dark:bg-stone-950/90">
1693
- <div className="text-[10px] font-semibold uppercase tracking-[0.16em] text-stone-500 dark:text-stone-400">Instance data</div>
1694
  <div className="mt-1 text-sm font-semibold text-stone-900 dark:text-stone-100">
1695
  {node.instanceDataLabel}
1696
  </div>
 
2
 
3
  import { useCallback, useEffect, useMemo, useRef, useState } from "react"
4
  import { useRouter } from "next/navigation"
5
+ import { ArrowLeft, ChevronDown, Search, SlidersHorizontal, X } from "lucide-react"
6
 
7
  import { useAudienceMode } from "@/components/audience-mode-provider"
8
  import { ListPagination } from "@/components/list-pagination"
9
  import { Navigation } from "@/components/navigation"
10
  import { PageHeader } from "@/components/page-header"
11
  import { Button } from "@/components/ui/button"
12
+ import { Collapsible, CollapsibleContent, CollapsibleTrigger } from "@/components/ui/collapsible"
13
  import { Input } from "@/components/ui/input"
14
  import type { EvalHierarchy } from "@/lib/backend-artifacts"
15
  import type { BenchmarkCard, CategoryType } from "@/lib/benchmark-schema"
 
263
  domains: string[]
264
  dataType?: string
265
  license?: string
266
+ card?: BenchmarkCard
267
  modelsCount: number
268
  metricCount: number
269
  topScore?: number
 
355
  return value.toFixed(value >= 100 ? 0 : 2)
356
  }
357
 
358
+ function hasConcreteText(value: string | undefined | null) {
359
+ if (!value) return false
360
+ const normalized = value.trim().toLowerCase()
361
+ return Boolean(normalized) && normalized !== "not specified" && normalized !== "unknown"
362
+ }
363
+
364
+ function firstConcreteListValue(value: string[] | string | undefined | null) {
365
+ if (Array.isArray(value)) {
366
+ return value.find((entry) => hasConcreteText(entry))
367
+ }
368
+
369
+ return hasConcreteText(value) ? value : undefined
370
+ }
371
+
372
+ function countConcreteListValues(value: string[] | undefined | null) {
373
+ return (value ?? []).filter((entry) => hasConcreteText(entry)).length
374
+ }
375
+
376
+ function getNodePolicySummary(node: EvalBrowserNode) {
377
+ const card = node.card
378
+ if (!card) return null
379
+
380
+ const riskCount = card.possible_risks?.length ?? 0
381
+ const reportingGapCount =
382
+ card.missing_fields?.filter(
383
+ (field) => field.startsWith("methodology") || field.startsWith("purpose_and_intended_users")
384
+ ).length ?? 0
385
+
386
+ return {
387
+ goal: hasConcreteText(card.purpose_and_intended_users?.goal)
388
+ ? card.purpose_and_intended_users.goal
389
+ : undefined,
390
+ limitations: hasConcreteText(card.purpose_and_intended_users?.limitations)
391
+ ? card.purpose_and_intended_users.limitations
392
+ : undefined,
393
+ audience: firstConcreteListValue(card.purpose_and_intended_users?.audience),
394
+ compliance: hasConcreteText(card.ethical_and_legal_considerations?.compliance_with_regulations)
395
+ ? card.ethical_and_legal_considerations.compliance_with_regulations
396
+ : undefined,
397
+ riskCount,
398
+ reportingGapCount,
399
+ }
400
+ }
401
+
402
+ function getNodeResearchSummary(node: EvalBrowserNode) {
403
+ const card = node.card
404
+ if (!card) return null
405
+
406
+ const similarBenchmarks = Array.isArray(card.benchmark_details?.similar_benchmarks)
407
+ ? card.benchmark_details.similar_benchmarks
408
+ : card.benchmark_details?.similar_benchmarks
409
+ ? [card.benchmark_details.similar_benchmarks]
410
+ : []
411
+
412
+ return {
413
+ methodsCount: countConcreteListValues(card.methodology?.methods),
414
+ metricsCount: countConcreteListValues(card.methodology?.metrics),
415
+ similarCount: countConcreteListValues(similarBenchmarks),
416
+ interpretation: hasConcreteText(card.methodology?.interpretation)
417
+ ? card.methodology.interpretation
418
+ : undefined,
419
+ missingMethodCount: card.missing_fields?.filter((field) => field.startsWith("methodology")).length ?? 0,
420
+ }
421
+ }
422
+
423
  function getNodeCard(
424
  benchmarkCards: Record<string, BenchmarkCard>,
425
  ...candidates: Array<string | undefined>
 
519
  export default function EvalsPage() {
520
  const { mode } = useAudienceMode()
521
  const router = useRouter()
522
+ const isResearchView = mode === "research"
523
 
524
  const [summaries, setSummaries] = useState<BenchmarkEvalListItem[]>([])
525
  const [benchmarkCards, setBenchmarkCards] = useState<Record<string, BenchmarkCard>>({})
 
532
  const [selectedNodeKind, setSelectedNodeKind] = useState<EvalBrowserNodeKind | null>(null)
533
  const [currentNodeId, setCurrentNodeId] = useState<string | null>(null)
534
  const [page, setPage] = useState(1)
535
+ const [filtersOpen, setFiltersOpen] = useState(false)
536
  const pendingHistoryActionRef = useRef<"push" | "replace">("replace")
537
 
538
  useEffect(() => {
 
704
  domains: Array.from(new Set(domains.flatMap((domain) => normalizeDomainList(domain)))),
705
  dataType: card?.benchmark_details?.data_type,
706
  license: card?.ethical_and_legal_considerations?.data_licensing,
707
+ card,
708
  modelsCount: stats.modelsCount,
709
  metricCount: stats.metricCount,
710
  topScore: stats.topScore,
 
725
  metrics?: Array<{ key: string; display_name: string }>
726
  }>,
727
  scopeKeys: string[]
728
+ ): EvalBrowserNode["matrixPreview"] | undefined => {
729
  if (benchmarks.length < 2) {
730
+ return undefined
731
  }
732
 
733
  const metricLabels = new Set<string>()
 
735
 
736
  for (const benchmark of benchmarks) {
737
  if ((benchmark.slices?.length ?? 0) > 0 || (benchmark.metrics?.length ?? 0) !== 1) {
738
+ return undefined
739
  }
740
 
741
  const metric = benchmark.metrics?.[0]
 
749
  }
750
 
751
  if (metricLabels.size !== 1) {
752
+ return undefined
753
  }
754
 
755
  return {
 
817
  const card = summary?.benchmark_card ?? getNodeCard(benchmarkCards, ...cardCandidates)
818
  const childSlices = slices.filter((slice) => !isSameHierarchyKey(slice.key, benchmarkKey))
819
  const drilldownSlices = childSlices.filter((slice) => (slice.metrics?.length ?? 0) > 1)
820
+ const fallbackSummary =
821
+ !summary && metrics.length > 0
822
+ ? scopeKeys.map((scopeKey) => pickSummaryForKey(summariesWithCards, scopeKey, scopeKeys)).find(Boolean)
823
+ : undefined
824
  const isParentRollupBenchmark =
825
  Boolean(parentId) && scopeKeys.some((scopeKey) => isSameHierarchyKey(scopeKey, benchmarkKey))
826
 
 
849
  domains,
850
  summaries: summary ? [summary] : [],
851
  card,
852
+ href:
853
+ drilldownSlices.length === 0
854
+ ? summary
855
+ ? `/evals/${summary.evaluation_id}`
856
+ : fallbackSummary
857
+ ? `/evals/${fallbackSummary.evaluation_id}`
858
+ : undefined
859
+ : undefined,
860
  scopeKeys,
861
  descriptionFallback: `Browse the {label} benchmark and its lower-level breakdowns.`,
862
  })
 
953
  const suiteBenchmarks = (composite.benchmarks ?? []).filter((benchmark) => !isSameHierarchyKey(benchmark.key, composite.key))
954
  const suiteMatrixPreview = buildSingleMetricMatrixPreview(suiteBenchmarks, suiteScopeKeys)
955
  const rollupSummary = pickSummaryForKey(summariesWithCards, composite.key, suiteScopeKeys)
956
+ const hasSuiteRollup = Boolean(rollupBenchmark && rollupSummary)
957
+ const syntheticMatrixEvalId = suiteMatrixPreview && !hasSuiteRollup ? `matrix__${composite.key}` : undefined
958
 
959
  buildNode({
960
  id: suiteId,
 
967
  summaries: suiteSummaries,
968
  card: suiteCard,
969
  sourceLabel: suiteLabel,
970
+ href: suiteMatrixPreview
971
+ ? hasSuiteRollup
972
+ ? `/evals/${rollupSummary.evaluation_id}`
973
+ : syntheticMatrixEvalId
974
+ ? `/evals/${syntheticMatrixEvalId}`
975
+ : undefined
976
+ : undefined,
977
  scopeKeys: suiteScopeKeys,
978
  matrixPreview: suiteMatrixPreview,
979
  descriptionFallback: `Browse the {label} suite and then open its benchmark children.`,
 
1015
  }
1016
 
1017
  const suiteNode = nodes.get(suiteId)
1018
+ if (suiteNode && suiteNode.childIds.length === 0) {
1019
+ if (hasSuiteRollup) {
1020
+ suiteNode.href = `/evals/${rollupSummary.evaluation_id}`
1021
+ } else if (syntheticMatrixEvalId) {
1022
+ suiteNode.href = `/evals/${syntheticMatrixEvalId}`
1023
+ }
1024
  }
1025
  }
1026
 
 
1070
  const visibleBenchmarks = suiteBenchmarks.filter((benchmark) => !isSameHierarchyKey(benchmark.key, suiteKey))
1071
  const suiteMatrixPreview = buildSingleMetricMatrixPreview(visibleBenchmarks, suiteScopeKeys)
1072
  const rollupSummary = pickSummaryForKey(summariesWithCards, suiteKey, suiteScopeKeys)
1073
+ const hasSuiteRollup = Boolean(rollupBenchmark && rollupSummary)
1074
+ const syntheticMatrixEvalId = suiteMatrixPreview && !hasSuiteRollup ? `matrix__${suiteKey}` : undefined
1075
  const suiteSummaries = summariesWithCards.filter((summary) => {
1076
  const familyScope = getSummaryScopeKey(summary.benchmark_family_key ?? summary.composite_benchmark_key)
1077
  return familyScope === getSummaryScopeKey(suiteKey)
 
1095
  family.key
1096
  ),
1097
  sourceLabel: suiteLabel,
1098
+ href: suiteMatrixPreview
1099
+ ? hasSuiteRollup
1100
+ ? `/evals/${rollupSummary.evaluation_id}`
1101
+ : syntheticMatrixEvalId
1102
+ ? `/evals/${syntheticMatrixEvalId}`
1103
+ : undefined
1104
+ : undefined,
1105
  scopeKeys: suiteScopeKeys,
1106
  matrixPreview: suiteMatrixPreview,
1107
  descriptionFallback: `Browse the {label} suite and then open its benchmark children.`,
 
1143
  }
1144
 
1145
  const suiteNode = nodes.get(suiteId)
1146
+ if (suiteNode && suiteNode.childIds.length === 0) {
1147
+ if (hasSuiteRollup) {
1148
+ suiteNode.href = `/evals/${rollupSummary.evaluation_id}`
1149
+ } else if (syntheticMatrixEvalId) {
1150
+ suiteNode.href = `/evals/${syntheticMatrixEvalId}`
1151
+ }
1152
  }
1153
 
1154
  continue
 
1353
  }
1354
  }, [currentLevelKinds, selectedNodeKind])
1355
 
1356
+ useEffect(() => {
1357
+ if (activeFilterCount > 0) {
1358
+ setFiltersOpen(true)
1359
+ }
1360
+ }, [activeFilterCount])
1361
+
1362
  const handleNodeOpen = useCallback(
1363
  (node: EvalBrowserNode) => {
1364
  if (node.childIds.length > 0) {
 
1564
  </div>
1565
  </div>
1566
 
1567
+ <Collapsible
1568
+ open={filtersOpen}
1569
+ onOpenChange={setFiltersOpen}
1570
+ className="mt-4 rounded-[1.35rem] border border-stone-200/80 bg-white/70 dark:border-stone-800/80 dark:bg-stone-950/60"
1571
+ >
1572
+ <CollapsibleTrigger asChild>
1573
+ <button type="button" className="flex w-full items-center justify-between gap-3 px-4 py-3 text-left">
1574
+ <div className="min-w-0">
1575
+ <div className="flex items-center gap-2 text-[11px] font-semibold uppercase tracking-[0.2em] text-stone-500 dark:text-stone-400">
1576
+ <SlidersHorizontal className="h-3.5 w-3.5" />
1577
+ Refine this list
1578
+ </div>
1579
+ <p className="mt-1 text-sm text-stone-600 dark:text-stone-300">
1580
+ {activeFilterCount > 0
1581
+ ? `${activeFilterCount} filter${activeFilterCount === 1 ? "" : "s"} active`
1582
+ : "Open filters only when you need to narrow by node type, domain tags, or category."}
1583
+ </p>
1584
+ </div>
1585
+ <div className="flex items-center gap-2">
1586
+ {activeFilterCount > 0 && (
1587
+ <span className="rounded-full border border-stone-200/80 bg-stone-100/80 px-2.5 py-1 text-[11px] font-medium text-stone-700 dark:border-stone-700/80 dark:bg-stone-900/80 dark:text-stone-200">
1588
+ {activeFilterCount} active
1589
+ </span>
1590
  )}
1591
+ <ChevronDown className={cn("h-4 w-4 text-stone-500 transition-transform dark:text-stone-400", filtersOpen && "rotate-180")} />
1592
+ </div>
1593
+ </button>
1594
+ </CollapsibleTrigger>
1595
+
1596
+ <CollapsibleContent>
1597
+ <div className="border-t border-stone-200/80 px-4 pb-4 pt-4 dark:border-stone-800/80">
1598
+ {currentLevelKinds.length > 1 && (
1599
+ <div className="space-y-1.5">
1600
+ <div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-stone-500 dark:text-stone-400">
1601
+ Node type
1602
+ </div>
1603
+ <div className="flex flex-wrap items-center gap-1.5">
1604
+ <button
1605
+ type="button"
1606
+ onClick={() => setSelectedNodeKind(null)}
1607
+ className={cn(
1608
+ "shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
1609
+ selectedNodeKind === null
1610
+ ? "border-stone-950 bg-stone-950 text-stone-50 dark:border-stone-100 dark:bg-stone-100 dark:text-stone-950"
1611
+ : "border-stone-200/80 bg-stone-50/80 text-stone-600 hover:bg-stone-100 dark:border-stone-700/80 dark:bg-stone-900/70 dark:text-stone-300 dark:hover:bg-stone-800"
1612
+ )}
1613
+ >
1614
+ All
1615
+ </button>
1616
+ {currentLevelKinds.map((kind) => (
1617
+ <button
1618
+ key={kind}
1619
+ type="button"
1620
+ onClick={() => setSelectedNodeKind(selectedNodeKind === kind ? null : kind)}
1621
+ className={cn(
1622
+ "shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
1623
+ selectedNodeKind === kind
1624
+ ? "border-sky-300 bg-sky-50 text-sky-800 dark:border-sky-800 dark:bg-sky-950/50 dark:text-sky-200"
1625
+ : "border-stone-200/80 bg-white text-stone-600 hover:bg-stone-50 dark:border-stone-700/80 dark:bg-stone-900 dark:text-stone-300 dark:hover:bg-stone-800"
1626
+ )}
1627
+ >
1628
+ {getBrowserNodeKindLabel(kind)}
1629
+ </button>
1630
+ ))}
1631
+ </div>
1632
+ </div>
1633
+ )}
1634
 
1635
+ {allDomains.length > 0 && (
1636
+ <div className="mt-4 space-y-1.5">
1637
+ <div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-stone-500 dark:text-stone-400">
1638
+ Domain tags
1639
+ </div>
1640
+ <div className="flex max-h-40 flex-wrap items-center gap-1.5 overflow-y-auto pr-1">
1641
+ <button
1642
+ type="button"
1643
+ onClick={() => setSelectedDomain(null)}
1644
+ className={cn(
1645
+ "shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
1646
+ selectedDomain === null
1647
+ ? "border-stone-950 bg-stone-950 text-stone-50 dark:border-stone-100 dark:bg-stone-100 dark:text-stone-950"
1648
+ : "border-stone-200/80 bg-stone-50/80 text-stone-600 hover:bg-stone-100 dark:border-stone-700/80 dark:bg-stone-900/70 dark:text-stone-300 dark:hover:bg-stone-800"
1649
+ )}
1650
+ >
1651
+ All
1652
+ </button>
1653
+ {allDomains.map((domain) => (
1654
+ <button
1655
+ key={domain}
1656
+ type="button"
1657
+ onClick={() => setSelectedDomain(selectedDomain === domain ? null : domain)}
1658
+ className={cn(
1659
+ "shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
1660
+ selectedDomain === domain
1661
+ ? "border-sky-300 bg-sky-50 text-sky-800 dark:border-sky-800 dark:bg-sky-950/50 dark:text-sky-200"
1662
+ : "border-stone-200/80 bg-white text-stone-600 hover:bg-stone-50 dark:border-stone-700/80 dark:bg-stone-900 dark:text-stone-300 dark:hover:bg-stone-800"
1663
+ )}
1664
+ >
1665
+ {domain}
1666
+ </button>
1667
+ ))}
1668
+ </div>
1669
+ </div>
1670
+ )}
1671
 
1672
+ {allCategories.length > 0 && (
1673
+ <div className="mt-4 space-y-1.5">
1674
+ <div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-stone-500 dark:text-stone-400">
1675
+ Category
1676
+ </div>
1677
+ <div className="flex flex-wrap items-center gap-1.5">
1678
+ <button
1679
+ type="button"
1680
+ onClick={() => setSelectedCategory(null)}
1681
+ className={cn(
1682
+ "shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
1683
+ selectedCategory === null
1684
+ ? "border-stone-950 bg-stone-950 text-stone-50 dark:border-stone-100 dark:bg-stone-100 dark:text-stone-950"
1685
+ : "border-stone-200/80 bg-stone-50/80 text-stone-600 hover:bg-stone-100 dark:border-stone-700/80 dark:bg-stone-900/70 dark:text-stone-300 dark:hover:bg-stone-800"
1686
+ )}
1687
+ >
1688
+ All
1689
+ </button>
1690
+ {allCategories.map((category) => (
1691
+ <button
1692
+ key={category}
1693
+ type="button"
1694
+ onClick={() => setSelectedCategory(selectedCategory === category ? null : category)}
1695
+ className={cn(
1696
+ "shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
1697
+ selectedCategory === category
1698
+ ? `${getCategoryColor(category as CategoryType)} border`
1699
+ : "border-stone-200/80 bg-white text-stone-600 hover:bg-stone-50 dark:border-stone-700/80 dark:bg-stone-900 dark:text-stone-300 dark:hover:bg-stone-800"
1700
+ )}
1701
+ >
1702
+ {category}
1703
+ </button>
1704
+ ))}
1705
+ </div>
1706
+ </div>
1707
+ )}
1708
  </div>
1709
+ </CollapsibleContent>
1710
+ </Collapsible>
1711
  </section>
1712
 
1713
  {filtered.length === 0 ? (
 
1723
  const isNavigable = node.childIds.length > 0 || Boolean(node.href)
1724
  const actionLabel = node.childIds.length > 0 ? "Open level" : node.matrixPreview ? "View rollup" : "View benchmark"
1725
  const kindLabel = getBrowserNodeKindLabel(node.kind)
1726
+ const policySummary = getNodePolicySummary(node)
1727
+ const researchSummary = getNodeResearchSummary(node)
1728
 
1729
  return (
1730
  <button
 
1800
  </p>
1801
  )}
1802
 
1803
+ {!isResearchView && policySummary && (
1804
+ <div className="mb-4 space-y-2 rounded-[1.1rem] border border-amber-200/80 bg-amber-50/70 p-3 dark:border-amber-900/50 dark:bg-amber-950/20">
1805
+ <div className="text-[10px] font-semibold uppercase tracking-[0.18em] text-amber-800 dark:text-amber-200">
1806
+ Policy notes
1807
+ </div>
1808
+ {policySummary.goal && (
1809
+ <p className="text-sm leading-6 text-stone-700 dark:text-stone-200">
1810
+ <span className="font-semibold">What it measures: </span>
1811
+ {policySummary.goal}
1812
+ </p>
1813
+ )}
1814
+ {policySummary.limitations && (
1815
+ <p className="text-sm leading-6 text-stone-700 dark:text-stone-200">
1816
+ <span className="font-semibold">Main caveat: </span>
1817
+ {policySummary.limitations}
1818
+ </p>
1819
+ )}
1820
+ <div className="flex flex-wrap gap-2 text-[11px] text-stone-700 dark:text-stone-200">
1821
+ {policySummary.audience && (
1822
+ <span className="rounded-full border border-amber-200/80 bg-white/80 px-2.5 py-1 dark:border-amber-900/50 dark:bg-stone-950/70">
1823
+ Intended for {policySummary.audience}
1824
+ </span>
1825
+ )}
1826
+ {policySummary.compliance && (
1827
+ <span className="rounded-full border border-amber-200/80 bg-white/80 px-2.5 py-1 dark:border-amber-900/50 dark:bg-stone-950/70">
1828
+ Regulation note documented
1829
+ </span>
1830
+ )}
1831
+ {policySummary.riskCount > 0 && (
1832
+ <span className="rounded-full border border-amber-200/80 bg-white/80 px-2.5 py-1 dark:border-amber-900/50 dark:bg-stone-950/70">
1833
+ {policySummary.riskCount} risk note{policySummary.riskCount === 1 ? "" : "s"}
1834
+ </span>
1835
+ )}
1836
+ {policySummary.reportingGapCount > 0 && (
1837
+ <span className="rounded-full border border-rose-200/80 bg-white/80 px-2.5 py-1 text-rose-700 dark:border-rose-900/50 dark:bg-stone-950/70 dark:text-rose-200">
1838
+ {policySummary.reportingGapCount} missing reporting field{policySummary.reportingGapCount === 1 ? "" : "s"}
1839
+ </span>
1840
+ )}
1841
+ </div>
1842
+ </div>
1843
+ )}
1844
+
1845
+ {isResearchView && researchSummary && (
1846
+ <div className="mb-4 space-y-2 rounded-[1.1rem] border border-sky-200/80 bg-sky-50/70 p-3 dark:border-sky-900/50 dark:bg-sky-950/20">
1847
+ <div className="text-[10px] font-semibold uppercase tracking-[0.18em] text-sky-800 dark:text-sky-200">
1848
+ Research notes
1849
+ </div>
1850
+ {researchSummary.interpretation && (
1851
+ <p className="text-sm leading-6 text-stone-700 dark:text-stone-200">
1852
+ <span className="font-semibold">Score interpretation: </span>
1853
+ {researchSummary.interpretation}
1854
+ </p>
1855
+ )}
1856
+ <div className="flex flex-wrap gap-2 text-[11px] text-stone-700 dark:text-stone-200">
1857
+ {researchSummary.methodsCount > 0 && (
1858
+ <span className="rounded-full border border-sky-200/80 bg-white/80 px-2.5 py-1 dark:border-sky-900/50 dark:bg-stone-950/70">
1859
+ {researchSummary.methodsCount} method note{researchSummary.methodsCount === 1 ? "" : "s"}
1860
+ </span>
1861
+ )}
1862
+ {researchSummary.metricsCount > 0 && (
1863
+ <span className="rounded-full border border-sky-200/80 bg-white/80 px-2.5 py-1 dark:border-sky-900/50 dark:bg-stone-950/70">
1864
+ {researchSummary.metricsCount} documented metric{researchSummary.metricsCount === 1 ? "" : "s"}
1865
+ </span>
1866
+ )}
1867
+ {researchSummary.similarCount > 0 && (
1868
+ <span className="rounded-full border border-sky-200/80 bg-white/80 px-2.5 py-1 dark:border-sky-900/50 dark:bg-stone-950/70">
1869
+ {researchSummary.similarCount} related benchmark{researchSummary.similarCount === 1 ? "" : "s"}
1870
+ </span>
1871
+ )}
1872
+ {researchSummary.missingMethodCount > 0 && (
1873
+ <span className="rounded-full border border-rose-200/80 bg-white/80 px-2.5 py-1 text-rose-700 dark:border-rose-900/50 dark:bg-stone-950/70 dark:text-rose-200">
1874
+ {researchSummary.missingMethodCount} missing method field{researchSummary.missingMethodCount === 1 ? "" : "s"}
1875
+ </span>
1876
+ )}
1877
+ </div>
1878
+ </div>
1879
+ )}
1880
+
1881
  {node.matrixPreview && (
1882
  <div className="mb-4 overflow-hidden rounded-[1.2rem] border border-stone-200/80 bg-stone-50/85 dark:border-stone-800/80 dark:bg-stone-900/85">
1883
  <div className="space-y-2 bg-white/92 px-3 py-3 dark:bg-stone-950/92">
 
1902
  <div className="mb-4 grid gap-px overflow-hidden rounded-[1.2rem] border border-stone-200/80 bg-stone-200/80 dark:border-stone-800/80 dark:bg-stone-800/80 sm:grid-cols-2 xl:grid-cols-3">
1903
  {topScoreLabel !== "—" && (
1904
  <div className="bg-white/90 px-3 py-3 dark:bg-stone-950/90">
1905
+ <div className="text-[10px] font-semibold uppercase tracking-[0.16em] text-stone-500 dark:text-stone-400">Reported score</div>
1906
  <div className="mt-1 text-sm font-semibold text-stone-900 dark:text-stone-100">{topScoreLabel}</div>
1907
  </div>
1908
  )}
1909
  <div className="bg-white/90 px-3 py-3 dark:bg-stone-950/90">
1910
+ <div className="text-[10px] font-semibold uppercase tracking-[0.16em] text-stone-500 dark:text-stone-400">Dataset or record</div>
1911
  <div className="mt-1 truncate text-sm font-semibold text-stone-900 dark:text-stone-100">
1912
  {node.sourceLabel}
1913
  </div>
1914
  </div>
1915
  <div className="bg-white/90 px-3 py-3 dark:bg-stone-950/90">
1916
+ <div className="text-[10px] font-semibold uppercase tracking-[0.16em] text-stone-500 dark:text-stone-400">Linked instances</div>
1917
  <div className="mt-1 text-sm font-semibold text-stone-900 dark:text-stone-100">
1918
  {node.instanceDataLabel}
1919
  </div>
app/page.tsx CHANGED
@@ -57,7 +57,7 @@ export default async function HomePage() {
57
  Public reporting for AI evaluations.
58
  </h1>
59
  <p className="max-w-3xl text-lg leading-8 text-muted-foreground sm:text-xl">
60
- Start from the question you have: which models show broad evidence, which benchmarks have thin reporting, and where the public record is still incomplete.
61
  </p>
62
  </div>
63
 
@@ -86,20 +86,20 @@ export default async function HomePage() {
86
  <div className="grid md:grid-cols-3 md:divide-x md:divide-border/60">
87
  <SignalCard
88
  icon={<Database className="h-4 w-4" />}
89
- title="Coverage"
90
- body="Benchmark breadth, setup details, and provenance stay visible."
91
  tone="sky"
92
  />
93
  <SignalCard
94
  icon={<Scale className="h-4 w-4" />}
95
- title="Comparability"
96
- body="Configuration gaps and evaluator relationships remain part of the reading."
97
  tone="amber"
98
  />
99
  <SignalCard
100
  icon={<BookOpenText className="h-4 w-4" />}
101
- title="Reading modes"
102
- body="Research and policy views shift emphasis without hiding the record."
103
  tone="emerald"
104
  />
105
  </div>
@@ -113,7 +113,7 @@ export default async function HomePage() {
113
  Start from a reader question
114
  </div>
115
  <div className="mt-1 text-sm leading-6 text-muted-foreground">
116
- The site is strongest when you treat it as an evidence reader, not a leaderboard.
117
  </div>
118
  </div>
119
  <Database className="h-4 w-4 text-muted-foreground" />
@@ -121,27 +121,27 @@ export default async function HomePage() {
121
 
122
  <div className="mt-5 space-y-3">
123
  <InquiryRow
124
- title="Which models have broad public evidence?"
125
- body="Use the models view to scan benchmark breadth, reported results, and where coverage is still thin."
126
  href="/models"
127
  />
128
  <InquiryRow
129
  title="How is one benchmark being reported?"
130
- body="Use the evaluations view to inspect methodology context, score spread, and missing configuration details."
131
  href="/evals"
132
  />
133
  <InquiryRow
134
- title="Who is publishing the evidence?"
135
- body="Group models by developer or open a developer page to see which organizations are reporting the most."
136
  href="/developers"
137
  />
138
  </div>
139
 
140
  <div className="mt-5 grid gap-3 border-t border-border/60 pt-5 sm:grid-cols-2">
141
  <QuietStat label="Models" value={models.length.toString()} detail="Tracked in the current corpus" tone="amber" />
142
- <QuietStat label="Evaluations" value={evalSummaries.length.toString()} detail="Benchmark views with linked details" tone="sky" />
143
  <QuietStat label="Developers" value={developerCount.toString()} detail="Organizations represented" tone="emerald" />
144
- <QuietStat label="Reported results" value={totalReportedResults.toLocaleString()} detail={`Avg ${avgBenchmarksPerModel.toFixed(1)} benchmark suites per model`} tone="slate" />
145
  </div>
146
  </aside>
147
  </div>
@@ -151,20 +151,20 @@ export default async function HomePage() {
151
  <RoutePanel
152
  href="/models"
153
  icon={<Database className="h-4 w-4" />}
154
- title="Model-first reading"
155
- body="See the reported benchmark footprint of a model, including where evidence is broad, narrow, or missing."
156
  />
157
  <RoutePanel
158
  href="/evals"
159
  icon={<BookOpenText className="h-4 w-4" />}
160
- title="Benchmark-first reading"
161
- body="Inspect how a benchmark is reported across models, with room to compare slices, setups, and sources."
162
  />
163
  <RoutePanel
164
  href="/about"
165
  icon={<Scale className="h-4 w-4" />}
166
- title="Project framing"
167
- body="Read why this reporting format exists and how it supports both research and policy reading modes."
168
  />
169
  </div>
170
 
@@ -200,7 +200,7 @@ function InquiryRow({
200
  href: string
201
  }) {
202
  return (
203
- <Link href={href} className="group rounded-[1.3rem] border border-border/70 bg-muted/10 p-4 transition-colors hover:bg-muted/20">
204
  <div className="flex items-start justify-between gap-3">
205
  <div>
206
  <div className="text-sm font-semibold text-foreground">{title}</div>
 
57
  Public reporting for AI evaluations.
58
  </h1>
59
  <p className="max-w-3xl text-lg leading-8 text-muted-foreground sm:text-xl">
60
+ Start from the question you have: which models have more reported benchmarks, which benchmarks are sparsely reported, and where the record is still incomplete.
61
  </p>
62
  </div>
63
 
 
86
  <div className="grid md:grid-cols-3 md:divide-x md:divide-border/60">
87
  <SignalCard
88
  icon={<Database className="h-4 w-4" />}
89
+ title="Reported benchmarks"
90
+ body="See which benchmarks, settings, and sources are actually documented."
91
  tone="sky"
92
  />
93
  <SignalCard
94
  icon={<Scale className="h-4 w-4" />}
95
+ title="Comparison context"
96
+ body="Configuration gaps and evaluator relationships stay attached to each record."
97
  tone="amber"
98
  />
99
  <SignalCard
100
  icon={<BookOpenText className="h-4 w-4" />}
101
+ title="Reader modes"
102
+ body="Research and policy views prioritize different fields without hiding the same record."
103
  tone="emerald"
104
  />
105
  </div>
 
113
  Start from a reader question
114
  </div>
115
  <div className="mt-1 text-sm leading-6 text-muted-foreground">
116
+ The site works best when you inspect reported records, not just the ranking order.
117
  </div>
118
  </div>
119
  <Database className="h-4 w-4 text-muted-foreground" />
 
121
 
122
  <div className="mt-5 space-y-3">
123
  <InquiryRow
124
+ title="Which models have the most reported benchmarks?"
125
+ body="Use the models view to scan reported benchmarks, result counts, and where reporting is sparse."
126
  href="/models"
127
  />
128
  <InquiryRow
129
  title="How is one benchmark being reported?"
130
+ body="Use the evaluations view to inspect benchmark notes, score ranges, and missing setup details."
131
  href="/evals"
132
  />
133
  <InquiryRow
134
+ title="Which organizations are publishing results?"
135
+ body="Open the developer pages to see which organizations appear most often in the reported records."
136
  href="/developers"
137
  />
138
  </div>
139
 
140
  <div className="mt-5 grid gap-3 border-t border-border/60 pt-5 sm:grid-cols-2">
141
  <QuietStat label="Models" value={models.length.toString()} detail="Tracked in the current corpus" tone="amber" />
142
+ <QuietStat label="Evaluations" value={evalSummaries.length.toString()} detail="Benchmark records with linked details" tone="sky" />
143
  <QuietStat label="Developers" value={developerCount.toString()} detail="Organizations represented" tone="emerald" />
144
+ <QuietStat label="Reported results" value={totalReportedResults.toLocaleString()} detail={`Avg ${avgBenchmarksPerModel.toFixed(1)} reported benchmarks per model`} tone="slate" />
145
  </div>
146
  </aside>
147
  </div>
 
151
  <RoutePanel
152
  href="/models"
153
  icon={<Database className="h-4 w-4" />}
154
+ title="Model records"
155
+ body="See which benchmarks are reported for a model and where reporting is missing or thin."
156
  />
157
  <RoutePanel
158
  href="/evals"
159
  icon={<BookOpenText className="h-4 w-4" />}
160
+ title="Benchmark records"
161
+ body="Inspect how one benchmark is reported across models, including slices, setup notes, and sources."
162
  />
163
  <RoutePanel
164
  href="/about"
165
  icon={<Scale className="h-4 w-4" />}
166
+ title="Project notes"
167
+ body="Read why this reporting format exists and how the same records support research and policy use."
168
  />
169
  </div>
170
 
 
200
  href: string
201
  }) {
202
  return (
203
+ <Link href={href} className="group block rounded-[1.3rem] border border-border/70 bg-muted/10 p-4 transition-colors hover:bg-muted/20">
204
  <div className="flex items-start justify-between gap-3">
205
  <div>
206
  <div className="text-sm font-semibold text-foreground">{title}</div>
components/eval-detail.tsx CHANGED
@@ -300,6 +300,7 @@ export function EvalDetail({ summary }: EvalDetailProps) {
300
  const hasMultiMetricLeaderboard =
301
  (summary.leaderboard_metrics?.length ?? 0) > 1 &&
302
  (summary.leaderboard_rows?.length ?? 0) > 0
 
303
  const [expandedRows, setExpandedRows] = useState<Record<string, boolean>>({})
304
  const [leaderboardPage, setLeaderboardPage] = useState(1)
305
  const [minParamStep, setMinParamStep] = useState(0)
@@ -409,203 +410,198 @@ export function EvalDetail({ summary }: EvalDetailProps) {
409
  return (
410
  <div className="space-y-6">
411
  <Card className="overflow-hidden">
412
- <CardContent className="space-y-5 p-5 sm:p-6">
413
- <div className="flex flex-col gap-5 xl:flex-row xl:items-start xl:justify-between">
414
- <div className="space-y-3">
415
- <div className="flex flex-wrap items-center gap-2">
416
- <Badge variant="outline" className="border-border/60 bg-background/80 text-[11px] uppercase tracking-[0.18em]">
417
- {summary.is_aggregated ? "Merged Benchmark" : "Single Benchmark"}
418
- </Badge>
419
- {summary.is_aggregated ? (
 
 
 
 
420
  <Badge variant="secondary" className="font-normal">
421
- {summary.aggregate_sources?.length ?? 0} composite benchmarks
422
  </Badge>
423
- ) : (
424
  <Badge variant="secondary" className="font-normal">
425
- Composite: {summary.composite_benchmark_name}
 
 
426
  </Badge>
427
- )}
428
- <Badge variant="secondary" className="font-normal capitalize">
429
- {summary.metric_config.score_type}
430
- </Badge>
431
- <Badge variant="secondary" className="font-normal">
432
- {summary.metric_config.lower_is_better ? "Lower is better" : "Higher is better"}
433
- </Badge>
434
- {summary.tags?.languages && summary.tags.languages.length > 0 && (
435
- <Badge variant="secondary" className="font-normal">
436
- {summary.tags.languages.join(", ")}
437
- </Badge>
438
- )}
439
- </div>
440
-
441
- <div className="space-y-1">
442
- <div className="text-2xl font-semibold tracking-tight sm:text-[1.9rem]">{summary.evaluation_name}</div>
443
- <p className="max-w-3xl text-sm leading-6 text-muted-foreground">
444
- {summary.metric_config.evaluation_description}
445
- </p>
446
  </div>
447
-
448
- {!isResearchView && (
449
- <p className="max-w-3xl text-sm leading-6 text-muted-foreground">
450
- {`${summary.benchmark_card?.purpose_and_intended_users?.goal ?? "This benchmark provides a public-facing capability signal."} Scores should be read alongside benchmark scope, metric definitions, and the source dataset context.`}
451
- </p>
452
  )}
453
- </div>
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
454
 
455
- <div className="grid w-full gap-3 grid-cols-2 xl:grid-cols-4">
456
- <div className="rounded-2xl border border-sky-200/80 bg-sky-50/80 px-4 py-3 dark:border-sky-900/40 dark:bg-sky-950/20">
457
- <div className="text-[11px] font-semibold uppercase tracking-[0.18em] text-sky-700 dark:text-sky-200">Models</div>
458
- <div className="mt-1 text-2xl font-semibold text-sky-950 dark:text-sky-50">{summary.models_count}</div>
459
- </div>
460
- <div className="rounded-2xl border border-border/70 bg-muted/20 px-4 py-3">
461
- <div className="text-[11px] font-semibold uppercase tracking-[0.18em] text-muted-foreground">
462
- {hasMultiMetricLeaderboard ? "Metrics" : isResearchView ? "Avg score" : "Metrics"}
463
- </div>
464
- <div className="mt-1 text-2xl font-semibold">
465
- {hasMultiMetricLeaderboard ? summary.metrics_count ?? summary.leaderboard_metrics?.length ?? 1 : isResearchView ? avgScoreLabel : summary.metrics_count ?? 1}
466
- </div>
467
- </div>
468
- <div className="rounded-2xl border border-emerald-200/80 bg-emerald-50/80 px-4 py-3 dark:border-emerald-900/40 dark:bg-emerald-950/20">
469
- <div className="text-[11px] font-semibold uppercase tracking-[0.18em] text-emerald-700 dark:text-emerald-200">
470
- {hasMultiMetricLeaderboard || !isResearchView ? "Source dataset" : "Top model"}
471
- </div>
472
- <div className="mt-1 text-sm font-semibold text-emerald-950 dark:text-emerald-50">
473
- {hasMultiMetricLeaderboard || !isResearchView
474
- ? sourceDatasetLabel
475
- : summary.best_model?.name ?? "Unknown"}
476
  </div>
477
- {!hasMultiMetricLeaderboard && isResearchView && summary.best_model && (
478
- <div className="mt-1 text-xs text-emerald-700/80 dark:text-emerald-200/80">
479
- {formatRawScore(summary.best_model.score, summary.metric_config.unit)}
 
 
480
  </div>
481
- )}
482
- </div>
483
- <div className="rounded-2xl border border-amber-200/80 bg-amber-50/80 px-4 py-3 dark:border-amber-900/40 dark:bg-amber-950/20">
484
- <div className="text-[11px] font-semibold uppercase tracking-[0.18em] text-amber-700 dark:text-amber-200">
485
- {hasMultiMetricLeaderboard || !isResearchView ? "Instance data" : "Bottom model"}
486
- </div>
487
- <div className="mt-1 text-sm font-semibold text-amber-950 dark:text-amber-50">
488
- {hasMultiMetricLeaderboard || !isResearchView
489
- ? instanceDataLabel
490
- : summary.worst_model?.name ?? "Unknown"}
491
- </div>
492
- {!hasMultiMetricLeaderboard && isResearchView && summary.worst_model && (
493
- <div className="mt-1 text-xs text-amber-700/80 dark:text-amber-200/80">
494
- {formatRawScore(summary.worst_model.score, summary.metric_config.unit)}
495
  </div>
496
- )}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
497
  </div>
498
- </div>
499
- </div>
500
 
501
- <div className="rounded-[1.5rem] border bg-muted/10 p-4">
502
- <div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-muted-foreground">
503
- {isResearchView ? "Metric specification" : "Reading context"}
504
- </div>
505
- <dl className="mt-3 grid gap-x-6 gap-y-3 text-sm sm:grid-cols-2 xl:grid-cols-5">
506
- <div>
507
- <dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
508
- Composite benchmark
509
- </dt>
510
- <dd className="mt-1 break-words font-medium">
511
- {summary.is_aggregated
512
- ? summary.aggregate_sources?.map((source) => source.composite_benchmark_name).join(", ") || "Multiple composite benchmarks"
513
- : summary.composite_benchmark_name}
514
- </dd>
515
- </div>
516
- <div>
517
- <dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
518
- {isResearchView ? "Single benchmark ID" : "What this covers"}
519
- </dt>
520
- <dd className="mt-1 break-words font-medium">
521
- {isResearchView
522
- ? summary.evaluation_id
523
- : summary.is_aggregated
524
- ? summary.metric_config.evaluation_description
525
- : summary.metric_config.evaluation_description}
526
- </dd>
527
- </div>
528
- <div>
529
- <dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
530
- {isResearchView ? "Score scale" : "How to read scores"}
531
- </dt>
532
- <dd className="mt-1 font-medium">
533
- {isResearchView
534
- ? `${summary.metric_config.min_score ?? 0} - ${summary.metric_config.max_score ?? 1}`
535
- : scoreDirectionLabel}
536
- </dd>
537
- </div>
538
- {summary.tags?.domains && summary.tags.domains.length > 0 && (
539
- <div>
540
- <dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">Domain coverage</dt>
541
- <dd className="mt-1 font-medium capitalize">
542
- {summary.tags.domains.slice(0, 2).join(", ")}
543
- {summary.tags.domains.length > 2 ? ` +${summary.tags.domains.length - 2} more` : ""}
544
- </dd>
545
  </div>
546
- )}
547
- <div>
548
- <dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
549
- {isResearchView ? "Source dataset" : "Instance data"}
550
- </dt>
551
- <dd className="mt-1 font-medium">
552
- {isResearchView ? sourceDatasetLabel : instanceDataLabel}
553
- </dd>
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
554
  </div>
555
- </dl>
556
- </div>
557
- </CardContent>
558
- </Card>
559
 
560
- {!hasMultiMetricLeaderboard && (summary.root_metrics?.length || summary.subtasks?.length) ? (
561
- <Card className="overflow-hidden">
562
- <CardHeader className="border-b bg-muted/10">
563
- <CardTitle className="text-xl">Benchmark structure</CardTitle>
564
- <CardDescription>
565
- Benchmark-level summary metrics and benchmark breakdowns are shown as separate sections.
566
- </CardDescription>
567
- </CardHeader>
568
- <CardContent className="space-y-6 p-5 sm:p-6">
569
- {summary.root_metrics && summary.root_metrics.length > 0 && (
570
- <section className="space-y-3">
571
- <div>
572
- <div className="text-sm font-semibold">Benchmark-level metrics</div>
573
- <div className="text-xs text-muted-foreground">
574
- Benchmark summary metrics used in this evaluation view.
575
  </div>
576
- </div>
577
- <div className="flex flex-wrap gap-2">
578
- {summary.root_metrics.map((metric) => (
579
- <span
580
- key={metric.metric_summary_id}
581
- className="rounded-full border border-border/70 bg-background px-3 py-1.5 text-xs font-medium"
582
- title={metric.canonical_display_name || metric.display_name}
583
- >
584
- {getCompactMetricLabel(metric.display_name)}
585
- {typeof metric.top_score === "number" ? ` · ${formatRawScore(metric.top_score, metric.unit)}` : ""}
586
- </span>
587
- ))}
588
- </div>
589
- </section>
590
- )}
591
 
592
- {summary.subtasks && summary.subtasks.length > 0 && (
593
- <section className="space-y-3">
594
- <div>
595
- <div className="text-sm font-semibold">Subtask breakdown</div>
596
- </div>
597
- <div className="grid gap-3 lg:grid-cols-2">
598
- {summary.subtasks.map((subtask) => (
599
- <div key={subtask.subtask_key} className="rounded-2xl border bg-background p-4">
600
- <div className="font-semibold">{subtask.display_name || subtask.subtask_name}</div>
601
- <div className="mt-1 text-xs text-muted-foreground" title={subtask.canonical_display_name || subtask.display_name}>
602
- {subtask.canonical_display_name || subtask.display_name}
603
  </div>
604
- <div className="mt-3 flex flex-wrap gap-2">
605
- {subtask.metrics.map((metric) => (
606
  <span
607
  key={metric.metric_summary_id}
608
- className="rounded-full border border-border/70 bg-muted/20 px-2.5 py-1 text-[11px] font-medium"
609
  title={metric.canonical_display_name || metric.display_name}
610
  >
611
  {getCompactMetricLabel(metric.display_name)}
@@ -614,18 +610,50 @@ export function EvalDetail({ summary }: EvalDetailProps) {
614
  ))}
615
  </div>
616
  </div>
617
- ))}
618
- </div>
619
- </section>
620
- )}
621
- </CardContent>
622
- </Card>
623
- ) : null}
624
 
625
- {/* Policy: benchmark context BEFORE the leaderboard (context first, numbers second) */}
626
- {!isResearchView && summary.benchmark_card && (
627
- <BenchmarkCardPanel card={summary.benchmark_card} isResearchView={false} defaultRisksOpen />
628
- )}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
629
 
630
  {hasMultiMetricLeaderboard ? (
631
  <MultiMetricLeaderboard summary={summary} isResearchView={isResearchView} />
@@ -1135,11 +1163,6 @@ export function EvalDetail({ summary }: EvalDetailProps) {
1135
  </CardContent>
1136
  </Card>
1137
  )}
1138
-
1139
- {/* Research: benchmark card details AFTER the leaderboard, collapsed by default */}
1140
- {isResearchView && summary.benchmark_card && (
1141
- <ResearchBenchmarkCardCollapsible card={summary.benchmark_card} />
1142
- )}
1143
  </div>
1144
  )
1145
  }
@@ -1433,8 +1456,8 @@ function MultiMetricLeaderboard({
1433
  </div>
1434
  <CardDescription>
1435
  {isResearchView
1436
- ? "Each column is a reported benchmark metric. Distinct measures stay separate instead of collapsing into a single raw score."
1437
- : "Each column is a separately reported metric or subtask metric so the benchmark can be read without flattening unlike measures into one number."}
1438
  </CardDescription>
1439
  </div>
1440
 
@@ -1446,8 +1469,8 @@ function MultiMetricLeaderboard({
1446
  </Badge>
1447
  <Badge variant="outline">
1448
  {visibleMetrics.length === leaderboardMetrics.length
1449
- ? `${leaderboardMetrics.length} metrics`
1450
- : `${visibleMetrics.length} of ${leaderboardMetrics.length} metrics`}
1451
  </Badge>
1452
  {hasParameterData && (numericMinParams != null || numericMaxParams != null) && (
1453
  <Badge variant="outline">
@@ -1462,7 +1485,7 @@ function MultiMetricLeaderboard({
1462
  </Button>
1463
  </DropdownMenuTrigger>
1464
  <DropdownMenuContent align="end" className="w-80">
1465
- <DropdownMenuLabel>Visible metric columns</DropdownMenuLabel>
1466
  <DropdownMenuItem onSelect={() => setVisibleMetricKeys(allMetricKeys)}>
1467
  Show all
1468
  </DropdownMenuItem>
@@ -1638,7 +1661,7 @@ function MultiMetricLeaderboard({
1638
  onClick={() => handleSort("coverage")}
1639
  className="w-full text-right font-semibold transition-colors hover:text-primary"
1640
  >
1641
- Coverage{getSortIndicator("coverage")}
1642
  </button>
1643
  </TableHead>
1644
  {visibleMetrics.map((metric) => (
@@ -1773,8 +1796,18 @@ function MultiMetricLeaderboard({
1773
  )
1774
  }
1775
 
1776
- function ResearchBenchmarkCardCollapsible({ card }: { card: BenchmarkCard }) {
1777
- const [open, setOpen] = useState(false)
 
 
 
 
 
 
 
 
 
 
1778
  return (
1779
  <Collapsible open={open} onOpenChange={setOpen}>
1780
  <CollapsibleTrigger asChild>
@@ -1797,7 +1830,11 @@ function ResearchBenchmarkCardCollapsible({ card }: { card: BenchmarkCard }) {
1797
  </button>
1798
  </CollapsibleTrigger>
1799
  <CollapsibleContent className="mt-2">
1800
- <BenchmarkCardPanel card={card} isResearchView defaultRisksOpen={false} />
 
 
 
 
1801
  </CollapsibleContent>
1802
  </Collapsible>
1803
  )
 
300
  const hasMultiMetricLeaderboard =
301
  (summary.leaderboard_metrics?.length ?? 0) > 1 &&
302
  (summary.leaderboard_rows?.length ?? 0) > 0
303
+ const [overviewOpen, setOverviewOpen] = useState(true)
304
  const [expandedRows, setExpandedRows] = useState<Record<string, boolean>>({})
305
  const [leaderboardPage, setLeaderboardPage] = useState(1)
306
  const [minParamStep, setMinParamStep] = useState(0)
 
410
  return (
411
  <div className="space-y-6">
412
  <Card className="overflow-hidden">
413
+ <Collapsible open={overviewOpen} onOpenChange={setOverviewOpen}>
414
+ <CollapsibleTrigger asChild>
415
+ <button
416
+ type="button"
417
+ className="flex w-full items-center justify-between gap-4 border-b bg-muted/10 px-4 py-3 text-left transition-colors hover:bg-muted/15 sm:px-5"
418
+ >
419
+ <div className="min-w-0 space-y-1">
420
+ <div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-muted-foreground">
421
+ {isResearchView ? "Benchmark overview" : "Reading overview"}
422
+ </div>
423
+ <div className="flex flex-wrap items-center gap-2">
424
+ <span className="text-base font-semibold tracking-tight sm:text-lg">{summary.evaluation_name}</span>
425
  <Badge variant="secondary" className="font-normal">
426
+ {summary.models_count} models
427
  </Badge>
 
428
  <Badge variant="secondary" className="font-normal">
429
+ {hasMultiMetricLeaderboard
430
+ ? `${summary.metrics_count ?? summary.leaderboard_metrics?.length ?? 1} measures`
431
+ : `${summary.metrics_count ?? 1} ${(summary.metrics_count ?? 1) === 1 ? "measure" : "measures"}`}
432
  </Badge>
433
+ </div>
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
434
  </div>
435
+ {overviewOpen ? (
436
+ <ChevronUp className="h-4 w-4 shrink-0 text-muted-foreground" />
437
+ ) : (
438
+ <ChevronDown className="h-4 w-4 shrink-0 text-muted-foreground" />
 
439
  )}
440
+ </button>
441
+ </CollapsibleTrigger>
442
+
443
+ <CollapsibleContent>
444
+ <CardContent className="space-y-4 p-4 sm:p-5">
445
+ <div className="flex flex-col gap-4 xl:flex-row xl:items-start xl:justify-between">
446
+ <div className="space-y-2.5">
447
+ <div className="flex flex-wrap items-center gap-2">
448
+ <Badge variant="outline" className="border-border/60 bg-background/80 text-[11px] uppercase tracking-[0.18em]">
449
+ {summary.is_aggregated ? "Merged Benchmark" : "Single Benchmark"}
450
+ </Badge>
451
+ {summary.is_aggregated ? (
452
+ <Badge variant="secondary" className="font-normal">
453
+ {summary.aggregate_sources?.length ?? 0} composite benchmarks
454
+ </Badge>
455
+ ) : (
456
+ <Badge variant="secondary" className="font-normal">
457
+ Composite: {summary.composite_benchmark_name}
458
+ </Badge>
459
+ )}
460
+ <Badge variant="secondary" className="font-normal capitalize">
461
+ {summary.metric_config.score_type}
462
+ </Badge>
463
+ <Badge variant="secondary" className="font-normal">
464
+ {summary.metric_config.lower_is_better ? "Lower is better" : "Higher is better"}
465
+ </Badge>
466
+ {summary.tags?.languages && summary.tags.languages.length > 0 && (
467
+ <Badge variant="secondary" className="font-normal">
468
+ {summary.tags.languages.join(", ")}
469
+ </Badge>
470
+ )}
471
+ </div>
472
 
473
+ <p className="max-w-3xl text-sm leading-6 text-muted-foreground">
474
+ {summary.metric_config.evaluation_description}
475
+ </p>
476
+
477
+ {!isResearchView && (
478
+ <p className="max-w-3xl text-sm leading-6 text-muted-foreground">
479
+ {`${summary.benchmark_card?.purpose_and_intended_users?.goal ?? "This benchmark reports a capability result."} Scores should be read alongside benchmark scope, metric definitions, and the source dataset context.`}
480
+ </p>
481
+ )}
 
 
 
 
 
 
 
 
 
 
 
 
482
  </div>
483
+
484
+ <div className="grid w-full grid-cols-2 gap-2 xl:grid-cols-4">
485
+ <div className="rounded-xl border border-sky-200/80 bg-sky-50/80 px-3 py-2.5 dark:border-sky-900/40 dark:bg-sky-950/20">
486
+ <div className="text-[11px] font-semibold uppercase tracking-[0.18em] text-sky-700 dark:text-sky-200">Models</div>
487
+ <div className="mt-1 text-xl font-semibold text-sky-950 dark:text-sky-50">{summary.models_count}</div>
488
  </div>
489
+ <div className="rounded-xl border border-border/70 bg-muted/20 px-3 py-2.5">
490
+ <div className="text-[11px] font-semibold uppercase tracking-[0.18em] text-muted-foreground">
491
+ {hasMultiMetricLeaderboard ? "Measures" : isResearchView ? "Avg score" : "Measures"}
492
+ </div>
493
+ <div className="mt-1 text-xl font-semibold">
494
+ {hasMultiMetricLeaderboard ? summary.metrics_count ?? summary.leaderboard_metrics?.length ?? 1 : isResearchView ? avgScoreLabel : summary.metrics_count ?? 1}
495
+ </div>
 
 
 
 
 
 
 
496
  </div>
497
+ <div className="rounded-xl border border-emerald-200/80 bg-emerald-50/80 px-3 py-2.5 dark:border-emerald-900/40 dark:bg-emerald-950/20">
498
+ <div className="text-[11px] font-semibold uppercase tracking-[0.18em] text-emerald-700 dark:text-emerald-200">
499
+ {hasMultiMetricLeaderboard || !isResearchView ? "Source dataset" : "Top model"}
500
+ </div>
501
+ <div className="mt-1 text-sm font-semibold text-emerald-950 dark:text-emerald-50">
502
+ {hasMultiMetricLeaderboard || !isResearchView
503
+ ? sourceDatasetLabel
504
+ : summary.best_model?.name ?? "Unknown"}
505
+ </div>
506
+ {!hasMultiMetricLeaderboard && isResearchView && summary.best_model && (
507
+ <div className="mt-1 text-xs text-emerald-700/80 dark:text-emerald-200/80">
508
+ {formatRawScore(summary.best_model.score, summary.metric_config.unit)}
509
+ </div>
510
+ )}
511
+ </div>
512
+ <div className="rounded-xl border border-amber-200/80 bg-amber-50/80 px-3 py-2.5 dark:border-amber-900/40 dark:bg-amber-950/20">
513
+ <div className="text-[11px] font-semibold uppercase tracking-[0.18em] text-amber-700 dark:text-amber-200">
514
+ {hasMultiMetricLeaderboard || !isResearchView ? "Instance data" : "Bottom model"}
515
+ </div>
516
+ <div className="mt-1 text-sm font-semibold text-amber-950 dark:text-amber-50">
517
+ {hasMultiMetricLeaderboard || !isResearchView
518
+ ? instanceDataLabel
519
+ : summary.worst_model?.name ?? "Unknown"}
520
+ </div>
521
+ {!hasMultiMetricLeaderboard && isResearchView && summary.worst_model && (
522
+ <div className="mt-1 text-xs text-amber-700/80 dark:text-amber-200/80">
523
+ {formatRawScore(summary.worst_model.score, summary.metric_config.unit)}
524
+ </div>
525
+ )}
526
+ </div>
527
+ </div>
528
  </div>
 
 
529
 
530
+ <div className="rounded-2xl border bg-muted/10 p-3.5">
531
+ <div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-muted-foreground">
532
+ {isResearchView ? "Metric specification" : "Reading context"}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
533
  </div>
534
+ <dl className="mt-3 grid gap-x-5 gap-y-3 text-sm sm:grid-cols-2 xl:grid-cols-5">
535
+ <div>
536
+ <dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
537
+ Composite benchmark
538
+ </dt>
539
+ <dd className="mt-1 break-words font-medium">
540
+ {summary.is_aggregated
541
+ ? summary.aggregate_sources?.map((source) => source.composite_benchmark_name).join(", ") || "Multiple composite benchmarks"
542
+ : summary.composite_benchmark_name}
543
+ </dd>
544
+ </div>
545
+ <div>
546
+ <dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
547
+ {isResearchView ? "Single benchmark ID" : "What this covers"}
548
+ </dt>
549
+ <dd className="mt-1 break-words font-medium">
550
+ {isResearchView ? summary.evaluation_id : summary.metric_config.evaluation_description}
551
+ </dd>
552
+ </div>
553
+ <div>
554
+ <dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
555
+ {isResearchView ? "Score scale" : "How to read scores"}
556
+ </dt>
557
+ <dd className="mt-1 font-medium">
558
+ {isResearchView
559
+ ? `${summary.metric_config.min_score ?? 0} - ${summary.metric_config.max_score ?? 1}`
560
+ : scoreDirectionLabel}
561
+ </dd>
562
+ </div>
563
+ {summary.tags?.domains && summary.tags.domains.length > 0 && (
564
+ <div>
565
+ <dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">Domain tags</dt>
566
+ <dd className="mt-1 font-medium capitalize">
567
+ {summary.tags.domains.slice(0, 2).join(", ")}
568
+ {summary.tags.domains.length > 2 ? ` +${summary.tags.domains.length - 2} more` : ""}
569
+ </dd>
570
+ </div>
571
+ )}
572
+ <div>
573
+ <dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
574
+ {isResearchView ? "Source dataset" : "Instance data"}
575
+ </dt>
576
+ <dd className="mt-1 font-medium">
577
+ {isResearchView ? sourceDatasetLabel : instanceDataLabel}
578
+ </dd>
579
+ </div>
580
+ </dl>
581
  </div>
 
 
 
 
582
 
583
+ {!hasMultiMetricLeaderboard && (summary.root_metrics?.length || summary.subtasks?.length) ? (
584
+ <section className="rounded-2xl border bg-muted/5 p-3.5">
585
+ <div className="space-y-1">
586
+ <div className="text-sm font-semibold">Benchmark structure</div>
587
+ <div className="text-xs text-muted-foreground">
588
+ Benchmark-level summary metrics and subtask slices grouped in one compact section.
589
+ </div>
 
 
 
 
 
 
 
 
590
  </div>
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
591
 
592
+ {summary.root_metrics && summary.root_metrics.length > 0 && (
593
+ <div className="mt-4 space-y-2.5">
594
+ <div>
595
+ <div className="text-sm font-semibold">Benchmark-level metrics</div>
596
+ <div className="text-xs text-muted-foreground">
597
+ Benchmark summary metrics used in this evaluation view.
598
+ </div>
 
 
 
 
599
  </div>
600
+ <div className="flex flex-wrap gap-2">
601
+ {summary.root_metrics.map((metric) => (
602
  <span
603
  key={metric.metric_summary_id}
604
+ className="rounded-full border border-border/70 bg-background px-3 py-1.5 text-xs font-medium"
605
  title={metric.canonical_display_name || metric.display_name}
606
  >
607
  {getCompactMetricLabel(metric.display_name)}
 
610
  ))}
611
  </div>
612
  </div>
613
+ )}
 
 
 
 
 
 
614
 
615
+ {summary.subtasks && summary.subtasks.length > 0 && (
616
+ <div className="mt-4 space-y-2.5">
617
+ <div className="text-sm font-semibold">Subtask breakdown</div>
618
+ <div className="grid gap-3 lg:grid-cols-2">
619
+ {summary.subtasks.map((subtask) => (
620
+ <div key={subtask.subtask_key} className="rounded-xl border bg-background p-3.5">
621
+ <div className="font-semibold">{subtask.display_name || subtask.subtask_name}</div>
622
+ <div className="mt-1 text-xs text-muted-foreground" title={subtask.canonical_display_name || subtask.display_name}>
623
+ {subtask.canonical_display_name || subtask.display_name}
624
+ </div>
625
+ <div className="mt-3 flex flex-wrap gap-2">
626
+ {subtask.metrics.map((metric) => (
627
+ <span
628
+ key={metric.metric_summary_id}
629
+ className="rounded-full border border-border/70 bg-muted/20 px-2.5 py-1 text-[11px] font-medium"
630
+ title={metric.canonical_display_name || metric.display_name}
631
+ >
632
+ {getCompactMetricLabel(metric.display_name)}
633
+ {typeof metric.top_score === "number" ? ` · ${formatRawScore(metric.top_score, metric.unit)}` : ""}
634
+ </span>
635
+ ))}
636
+ </div>
637
+ </div>
638
+ ))}
639
+ </div>
640
+ </div>
641
+ )}
642
+ </section>
643
+ ) : null}
644
+
645
+ {summary.benchmark_card && (
646
+ <BenchmarkCardCollapsible
647
+ card={summary.benchmark_card}
648
+ isResearchView={isResearchView}
649
+ defaultOpen
650
+ defaultRisksOpen={!isResearchView}
651
+ />
652
+ )}
653
+ </CardContent>
654
+ </CollapsibleContent>
655
+ </Collapsible>
656
+ </Card>
657
 
658
  {hasMultiMetricLeaderboard ? (
659
  <MultiMetricLeaderboard summary={summary} isResearchView={isResearchView} />
 
1163
  </CardContent>
1164
  </Card>
1165
  )}
 
 
 
 
 
1166
  </div>
1167
  )
1168
  }
 
1456
  </div>
1457
  <CardDescription>
1458
  {isResearchView
1459
+ ? "Each column is a reported benchmark measure. Distinct measures stay separate instead of collapsing into a single raw score."
1460
+ : "Each column is a separately reported measure so the benchmark can be read without flattening different results into one number."}
1461
  </CardDescription>
1462
  </div>
1463
 
 
1469
  </Badge>
1470
  <Badge variant="outline">
1471
  {visibleMetrics.length === leaderboardMetrics.length
1472
+ ? `${leaderboardMetrics.length} measures`
1473
+ : `${visibleMetrics.length} of ${leaderboardMetrics.length} measures`}
1474
  </Badge>
1475
  {hasParameterData && (numericMinParams != null || numericMaxParams != null) && (
1476
  <Badge variant="outline">
 
1485
  </Button>
1486
  </DropdownMenuTrigger>
1487
  <DropdownMenuContent align="end" className="w-80">
1488
+ <DropdownMenuLabel>Visible measure columns</DropdownMenuLabel>
1489
  <DropdownMenuItem onSelect={() => setVisibleMetricKeys(allMetricKeys)}>
1490
  Show all
1491
  </DropdownMenuItem>
 
1661
  onClick={() => handleSort("coverage")}
1662
  className="w-full text-right font-semibold transition-colors hover:text-primary"
1663
  >
1664
+ Measures present{getSortIndicator("coverage")}
1665
  </button>
1666
  </TableHead>
1667
  {visibleMetrics.map((metric) => (
 
1796
  )
1797
  }
1798
 
1799
+ function BenchmarkCardCollapsible({
1800
+ card,
1801
+ isResearchView,
1802
+ defaultOpen = true,
1803
+ defaultRisksOpen = false,
1804
+ }: {
1805
+ card: BenchmarkCard
1806
+ isResearchView: boolean
1807
+ defaultOpen?: boolean
1808
+ defaultRisksOpen?: boolean
1809
+ }) {
1810
+ const [open, setOpen] = useState(defaultOpen)
1811
  return (
1812
  <Collapsible open={open} onOpenChange={setOpen}>
1813
  <CollapsibleTrigger asChild>
 
1830
  </button>
1831
  </CollapsibleTrigger>
1832
  <CollapsibleContent className="mt-2">
1833
+ <BenchmarkCardPanel
1834
+ card={card}
1835
+ isResearchView={isResearchView}
1836
+ defaultRisksOpen={defaultRisksOpen}
1837
+ />
1838
  </CollapsibleContent>
1839
  </Collapsible>
1840
  )
components/page-loading-state.tsx ADDED
@@ -0,0 +1,120 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ "use client"
2
+
3
+ import { useEffect, useMemo, useState } from "react"
4
+
5
+ import { cn } from "@/lib/utils"
6
+
7
+ export interface PageLoadingStage {
8
+ label: string
9
+ done: boolean
10
+ }
11
+
12
+ interface PageLoadingStateProps {
13
+ title: string
14
+ description?: string
15
+ stages: PageLoadingStage[]
16
+ className?: string
17
+ }
18
+
19
+ function clamp(value: number, min: number, max: number) {
20
+ return Math.min(max, Math.max(min, value))
21
+ }
22
+
23
+ function getTargetProgress(completedStages: number, totalStages: number) {
24
+ const start = 12
25
+ const ceiling = 94
26
+
27
+ if (totalStages <= 0) {
28
+ return start
29
+ }
30
+
31
+ const stageWidth = (ceiling - start) / totalStages
32
+ const hardTarget = start + completedStages * stageWidth
33
+
34
+ if (completedStages >= totalStages) {
35
+ return 100
36
+ }
37
+
38
+ return Math.min(ceiling, hardTarget + Math.min(stageWidth * 0.35, 12))
39
+ }
40
+
41
+ export function PageLoadingState({ title, description, stages, className }: PageLoadingStateProps) {
42
+ const safeStages = stages.length > 0 ? stages : [{ label: "Loading", done: false }]
43
+
44
+ const { completedStages, currentStageLabel, targetProgress } = useMemo(() => {
45
+ const completed = safeStages.filter((stage) => stage.done).length
46
+ const currentStage = safeStages.find((stage) => !stage.done)?.label ?? "Finalizing"
47
+
48
+ return {
49
+ completedStages: completed,
50
+ currentStageLabel: currentStage,
51
+ targetProgress: getTargetProgress(completed, safeStages.length),
52
+ }
53
+ }, [safeStages])
54
+
55
+ const [displayProgress, setDisplayProgress] = useState(() => clamp(targetProgress, 0, 100))
56
+
57
+ useEffect(() => {
58
+ setDisplayProgress((current) => {
59
+ if (targetProgress < current) {
60
+ return clamp(targetProgress, 0, 100)
61
+ }
62
+
63
+ return current
64
+ })
65
+ }, [targetProgress])
66
+
67
+ useEffect(() => {
68
+ if (Math.abs(displayProgress - targetProgress) < 0.5) {
69
+ if (displayProgress !== targetProgress) {
70
+ setDisplayProgress(targetProgress)
71
+ }
72
+ return
73
+ }
74
+
75
+ const interval = window.setInterval(() => {
76
+ setDisplayProgress((current) => {
77
+ const difference = targetProgress - current
78
+ if (Math.abs(difference) < 0.5) {
79
+ return targetProgress
80
+ }
81
+
82
+ const increment =
83
+ difference > 18 ? 6 : difference > 10 ? 4 : difference > 4 ? 2 : 1
84
+
85
+ return clamp(current + increment, 0, targetProgress)
86
+ })
87
+ }, 110)
88
+
89
+ return () => {
90
+ window.clearInterval(interval)
91
+ }
92
+ }, [displayProgress, targetProgress])
93
+
94
+ const progressStyle = {
95
+ background: `conic-gradient(from 180deg, hsl(var(--primary)) 0deg ${displayProgress * 3.6}deg, hsl(var(--border)) ${displayProgress * 3.6}deg 360deg)`,
96
+ }
97
+
98
+ return (
99
+ <div className={cn("flex min-h-[20rem] items-center justify-center px-4", className)}>
100
+ <section className="flex w-full max-w-md flex-col items-center gap-5 rounded-[1.75rem] border border-border/60 bg-background/95 px-6 py-8 text-center shadow-[0_24px_70px_-52px_rgba(15,23,42,0.5)] backdrop-blur">
101
+ <div className="relative flex h-28 w-28 items-center justify-center rounded-full" style={progressStyle}>
102
+ <div className="flex h-[5.4rem] w-[5.4rem] items-center justify-center rounded-full border border-border/60 bg-background text-2xl font-semibold tracking-tight text-foreground tabular-nums">
103
+ {Math.round(displayProgress)}%
104
+ </div>
105
+ </div>
106
+
107
+ <div className="space-y-1.5">
108
+ <h2 className="text-xl font-semibold tracking-tight text-foreground sm:text-2xl">{title}</h2>
109
+ {description ? (
110
+ <p className="text-sm leading-6 text-muted-foreground">{description}</p>
111
+ ) : null}
112
+ </div>
113
+
114
+ <p className="text-[11px] font-medium uppercase tracking-[0.22em] text-muted-foreground">
115
+ {currentStageLabel} · {completedStages}/{safeStages.length} ready
116
+ </p>
117
+ </section>
118
+ </div>
119
+ )
120
+ }
data/benchmarks/ace.json DELETED
@@ -1,120 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "anthropic/Opus 4.1",
5
- "name": "Opus 4.1",
6
- "developer": "Anthropic",
7
- "scores": {
8
- "Overall Score": 0.4,
9
- "Gaming Score": 0.318
10
- }
11
- },
12
- {
13
- "model_id": "anthropic/Opus 4.5",
14
- "name": "Opus 4.5",
15
- "developer": "Anthropic",
16
- "scores": {
17
- "Overall Score": 0.478,
18
- "Gaming Score": 0.391
19
- }
20
- },
21
- {
22
- "model_id": "anthropic/Sonnet 4.5",
23
- "name": "Sonnet 4.5",
24
- "developer": "Anthropic",
25
- "scores": {
26
- "Overall Score": 0.44,
27
- "Gaming Score": 0.373
28
- }
29
- },
30
- {
31
- "model_id": "google/Gemini 2.5 Flash",
32
- "name": "Gemini 2.5 Flash",
33
- "developer": "Google",
34
- "scores": {
35
- "Overall Score": 0.38,
36
- "Gaming Score": 0.284
37
- }
38
- },
39
- {
40
- "model_id": "google/Gemini 2.5 Pro",
41
- "name": "Gemini 2.5 Pro",
42
- "developer": "Google",
43
- "scores": {
44
- "Overall Score": 0.4,
45
- "Gaming Score": 0.285
46
- }
47
- },
48
- {
49
- "model_id": "google/Gemini 3 Flash",
50
- "name": "Gemini 3 Flash",
51
- "developer": "Google",
52
- "scores": {
53
- "Gaming Score": 0.415
54
- }
55
- },
56
- {
57
- "model_id": "google/Gemini 3 Pro",
58
- "name": "Gemini 3 Pro",
59
- "developer": "Google",
60
- "scores": {
61
- "Overall Score": 0.47,
62
- "Gaming Score": 0.509
63
- }
64
- },
65
- {
66
- "model_id": "openai/GPT 5",
67
- "name": "GPT 5",
68
- "developer": "OpenAI",
69
- "scores": {
70
- "Overall Score": 0.561,
71
- "DIY Score": 0.55,
72
- "Food Score": 0.7,
73
- "Gaming Score": 0.575
74
- }
75
- },
76
- {
77
- "model_id": "openai/GPT 5.1",
78
- "name": "GPT 5.1",
79
- "developer": "OpenAI",
80
- "scores": {
81
- "Overall Score": 0.551,
82
- "DIY Score": 0.56,
83
- "Gaming Score": 0.61,
84
- "Shopping Score": 0.45
85
- }
86
- },
87
- {
88
- "model_id": "openai/GPT 5.2",
89
- "name": "GPT 5.2",
90
- "developer": "OpenAI",
91
- "scores": {
92
- "Overall Score": 0.515,
93
- "Food Score": 0.65,
94
- "Gaming Score": 0.578
95
- }
96
- },
97
- {
98
- "model_id": "openai/o3",
99
- "name": "o3",
100
- "developer": "OpenAI",
101
- "scores": {
102
- "Overall Score": 0.529,
103
- "Gaming Score": 0.585,
104
- "Shopping Score": 0.45
105
- }
106
- },
107
- {
108
- "model_id": "openai/o3 Pro",
109
- "name": "o3 Pro",
110
- "developer": "OpenAI",
111
- "scores": {
112
- "Overall Score": 0.552,
113
- "DIY Score": 0.54,
114
- "Food Score": 0.6,
115
- "Gaming Score": 0.613,
116
- "Shopping Score": 0.45
117
- }
118
- }
119
- ]
120
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/apex-agents.json DELETED
@@ -1,218 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "anthropic/Opus 4.5",
5
- "name": "Opus 4.5",
6
- "developer": "Anthropic",
7
- "scores": {
8
- "Overall Pass@1": 0.184,
9
- "Overall Pass@8": 0.34,
10
- "Overall Mean Score": 0.348,
11
- "Investment Banking Pass@1": 0.216,
12
- "Management Consulting Pass@1": 0.132,
13
- "Corporate Law Pass@1": 0.202,
14
- "Corporate Lawyer Mean Score": 0.471
15
- }
16
- },
17
- {
18
- "model_id": "anthropic/Opus 4.6",
19
- "name": "Opus 4.6",
20
- "developer": "Anthropic",
21
- "scores": {
22
- "Overall Pass@1": 0.298,
23
- "Corporate Lawyer Mean Score": 0.502
24
- }
25
- },
26
- {
27
- "model_id": "applied-compute/Applied Compute: Small",
28
- "name": "Applied Compute: Small",
29
- "developer": "applied-compute",
30
- "scores": {
31
- "Overall Pass@1": 0.23,
32
- "Overall Mean Score": 0.401,
33
- "Corporate Law Pass@1": 0.266,
34
- "Corporate Lawyer Mean Score": 0.548
35
- }
36
- },
37
- {
38
- "model_id": "google/Gemini 3 Flash",
39
- "name": "Gemini 3 Flash",
40
- "developer": "Google",
41
- "scores": {
42
- "Overall Pass@1": 0.24,
43
- "Overall Pass@8": 0.367,
44
- "Overall Mean Score": 0.395,
45
- "Investment Banking Pass@1": 0.267,
46
- "Management Consulting Pass@1": 0.193,
47
- "Corporate Law Pass@1": 0.259,
48
- "Corporate Lawyer Mean Score": 0.524
49
- }
50
- },
51
- {
52
- "model_id": "google/Gemini 3 Pro",
53
- "name": "Gemini 3 Pro",
54
- "developer": "Google",
55
- "scores": {
56
- "Overall Pass@1": 0.184,
57
- "Overall Pass@8": 0.373,
58
- "Overall Mean Score": 0.341,
59
- "Investment Banking Pass@1": 0.188,
60
- "Management Consulting Pass@1": 0.124,
61
- "Corporate Law Pass@1": 0.239,
62
- "Corporate Lawyer Mean Score": 0.487
63
- }
64
- },
65
- {
66
- "model_id": "google/Gemini 3.1 Pro",
67
- "name": "Gemini 3.1 Pro",
68
- "developer": "Google",
69
- "scores": {
70
- "Overall Pass@1": 0.335,
71
- "Corporate Lawyer Mean Score": 0.494
72
- }
73
- },
74
- {
75
- "model_id": "minimax/Minimax-2.5",
76
- "name": "Minimax-2.5",
77
- "developer": "minimax",
78
- "scores": {
79
- "Corporate Lawyer Mean Score": 0.339
80
- }
81
- },
82
- {
83
- "model_id": "moonshot/Kimi K2 Thinking",
84
- "name": "Kimi K2 Thinking",
85
- "developer": "moonshot",
86
- "scores": {
87
- "Overall Pass@1": 0.04,
88
- "Overall Pass@8": 0.144,
89
- "Overall Mean Score": 0.115,
90
- "Investment Banking Pass@1": 0.012,
91
- "Management Consulting Pass@1": 0.029,
92
- "Corporate Law Pass@1": 0.08,
93
- "Corporate Lawyer Mean Score": 0.223
94
- }
95
- },
96
- {
97
- "model_id": "moonshot/Kimi K2.5",
98
- "name": "Kimi K2.5",
99
- "developer": "moonshot",
100
- "scores": {
101
- "Corporate Lawyer Mean Score": 0.402
102
- }
103
- },
104
- {
105
- "model_id": "openai/GPT 5",
106
- "name": "GPT 5",
107
- "developer": "OpenAI",
108
- "scores": {
109
- "Overall Pass@1": 0.183,
110
- "Overall Pass@8": 0.31,
111
- "Overall Mean Score": 0.329,
112
- "Investment Banking Pass@1": 0.273,
113
- "Management Consulting Pass@1": 0.123,
114
- "Corporate Law Pass@1": 0.153,
115
- "Corporate Lawyer Mean Score": 0.382
116
- }
117
- },
118
- {
119
- "model_id": "openai/GPT 5 Codex",
120
- "name": "GPT 5 Codex",
121
- "developer": "OpenAI",
122
- "scores": {
123
- "Corporate Lawyer Mean Score": 0.362
124
- }
125
- },
126
- {
127
- "model_id": "openai/GPT 5.1",
128
- "name": "GPT 5.1",
129
- "developer": "OpenAI",
130
- "scores": {
131
- "Corporate Lawyer Mean Score": 0.376
132
- }
133
- },
134
- {
135
- "model_id": "openai/GPT 5.1 Codex",
136
- "name": "GPT 5.1 Codex",
137
- "developer": "OpenAI",
138
- "scores": {
139
- "Corporate Lawyer Mean Score": 0.366
140
- }
141
- },
142
- {
143
- "model_id": "openai/GPT 5.2",
144
- "name": "GPT 5.2",
145
- "developer": "OpenAI",
146
- "scores": {
147
- "Overall Pass@1": 0.23,
148
- "Overall Pass@8": 0.4,
149
- "Overall Mean Score": 0.387,
150
- "Investment Banking Pass@1": 0.273,
151
- "Management Consulting Pass@1": 0.227,
152
- "Corporate Law Pass@1": 0.189,
153
- "Corporate Lawyer Mean Score": 0.443
154
- }
155
- },
156
- {
157
- "model_id": "openai/GPT 5.2 Codex",
158
- "name": "GPT 5.2 Codex",
159
- "developer": "OpenAI",
160
- "scores": {
161
- "Overall Pass@1": 0.276,
162
- "Corporate Lawyer Mean Score": 0.394
163
- }
164
- },
165
- {
166
- "model_id": "openai/GPT 5.3 Codex",
167
- "name": "GPT 5.3 Codex",
168
- "developer": "OpenAI",
169
- "scores": {
170
- "Overall Pass@1": 0.317
171
- }
172
- },
173
- {
174
- "model_id": "openai/GPT OSS 120B",
175
- "name": "GPT OSS 120B",
176
- "developer": "OpenAI",
177
- "scores": {
178
- "Overall Pass@1": 0.047,
179
- "Overall Pass@8": 0.115,
180
- "Overall Mean Score": 0.145,
181
- "Investment Banking Pass@1": 0.027,
182
- "Management Consulting Pass@1": 0.035,
183
- "Corporate Law Pass@1": 0.078,
184
- "Corporate Lawyer Mean Score": 0.269
185
- }
186
- },
187
- {
188
- "model_id": "xai/Grok 4",
189
- "name": "Grok 4",
190
- "developer": "xAI",
191
- "scores": {
192
- "Overall Pass@1": 0.152,
193
- "Overall Pass@8": 0.329,
194
- "Overall Mean Score": 0.303,
195
- "Investment Banking Pass@1": 0.17,
196
- "Management Consulting Pass@1": 0.12,
197
- "Corporate Law Pass@1": 0.165,
198
- "Corporate Lawyer Mean Score": 0.41
199
- }
200
- },
201
- {
202
- "model_id": "zhipu/GLM 4.6",
203
- "name": "GLM 4.6",
204
- "developer": "zhipu",
205
- "scores": {
206
- "Corporate Lawyer Mean Score": 0.196
207
- }
208
- },
209
- {
210
- "model_id": "zhipu/GLM 4.7",
211
- "name": "GLM 4.7",
212
- "developer": "zhipu",
213
- "scores": {
214
- "Corporate Lawyer Mean Score": 0.147
215
- }
216
- }
217
- ]
218
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/apex-v1.json DELETED
@@ -1,93 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "anthropic/Opus 4.5",
5
- "name": "Opus 4.5",
6
- "developer": "Anthropic",
7
- "scores": {
8
- "Medicine (MD) Score": 0.65
9
- }
10
- },
11
- {
12
- "model_id": "google/Gemini 2.5 Flash",
13
- "name": "Gemini 2.5 Flash",
14
- "developer": "Google",
15
- "scores": {
16
- "Overall Score": 0.604
17
- }
18
- },
19
- {
20
- "model_id": "google/Gemini 3 Flash",
21
- "name": "Gemini 3 Flash",
22
- "developer": "Google",
23
- "scores": {
24
- "Overall Score": 0.64,
25
- "Consulting Score": 0.64
26
- }
27
- },
28
- {
29
- "model_id": "google/Gemini 3 Pro",
30
- "name": "Gemini 3 Pro",
31
- "developer": "Google",
32
- "scores": {
33
- "Overall Score": 0.643,
34
- "Consulting Score": 0.64,
35
- "Investment Banking Score": 0.63
36
- }
37
- },
38
- {
39
- "model_id": "openai/GPT 4o",
40
- "name": "GPT 4o",
41
- "developer": "OpenAI",
42
- "scores": {
43
- "Overall Score": 0.359
44
- }
45
- },
46
- {
47
- "model_id": "openai/GPT 5",
48
- "name": "GPT 5",
49
- "developer": "OpenAI",
50
- "scores": {
51
- "Overall Score": 0.67,
52
- "Big Law Score": 0.78,
53
- "Medicine (MD) Score": 0.66,
54
- "Investment Banking Score": 0.61
55
- }
56
- },
57
- {
58
- "model_id": "openai/GPT 5.1",
59
- "name": "GPT 5.1",
60
- "developer": "OpenAI",
61
- "scores": {
62
- "Big Law Score": 0.77
63
- }
64
- },
65
- {
66
- "model_id": "openai/GPT 5.2 Pro",
67
- "name": "GPT 5.2 Pro",
68
- "developer": "OpenAI",
69
- "scores": {
70
- "Overall Score": 0.668,
71
- "Consulting Score": 0.64,
72
- "Medicine (MD) Score": 0.65,
73
- "Investment Banking Score": 0.64
74
- }
75
- },
76
- {
77
- "model_id": "openai/o3",
78
- "name": "o3",
79
- "developer": "OpenAI",
80
- "scores": {
81
- "Big Law Score": 0.76
82
- }
83
- },
84
- {
85
- "model_id": "xai/Grok 4",
86
- "name": "Grok 4",
87
- "developer": "xAI",
88
- "scores": {
89
- "Overall Score": 0.635
90
- }
91
- }
92
- ]
93
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/appworld_test_normal.json DELETED
@@ -1,28 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "anthropic/claude-opus-4-5",
5
- "name": "claude-opus-4-5",
6
- "developer": "Anthropic",
7
- "scores": {
8
- "appworld/test_normal": 0.68
9
- }
10
- },
11
- {
12
- "model_id": "google/gemini-3-pro-preview",
13
- "name": "gemini-3-pro-preview",
14
- "developer": "Google",
15
- "scores": {
16
- "appworld/test_normal": 0.505
17
- }
18
- },
19
- {
20
- "model_id": "openai/gpt-5.2-2025-12-11",
21
- "name": "gpt-5.2-2025-12-11",
22
- "developer": "OpenAI",
23
- "scores": {
24
- "appworld/test_normal": 0.0
25
- }
26
- }
27
- ]
28
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/bfcl.json DELETED
The diff for this file is too large to render. See raw diff
 
data/benchmarks/browsecompplus.json DELETED
@@ -1,28 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "anthropic/claude-opus-4-5",
5
- "name": "claude-opus-4-5",
6
- "developer": "Anthropic",
7
- "scores": {
8
- "browsecompplus": 0.61
9
- }
10
- },
11
- {
12
- "model_id": "google/gemini-3-pro-preview",
13
- "name": "gemini-3-pro-preview",
14
- "developer": "Google",
15
- "scores": {
16
- "browsecompplus": 0.48
17
- }
18
- },
19
- {
20
- "model_id": "openai/gpt-5.2-2025-12-11",
21
- "name": "gpt-5.2-2025-12-11",
22
- "developer": "OpenAI",
23
- "scores": {
24
- "browsecompplus": 0.26
25
- }
26
- }
27
- ]
28
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/global-mmlu-lite.json DELETED
@@ -1,706 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "alibaba/qwen3-235b-a22b-instruct-2507",
5
- "name": "qwen3-235b-a22b-instruct-2507",
6
- "developer": "alibaba",
7
- "scores": {
8
- "Global MMLU Lite": 0.8798,
9
- "Culturally Sensitive": 0.8522,
10
- "Culturally Agnostic": 0.9075,
11
- "Arabic": 0.88,
12
- "English": 0.89,
13
- "Bengali": 0.8875,
14
- "German": 0.885,
15
- "French": 0.88,
16
- "Hindi": 0.8775,
17
- "Indonesian": 0.88,
18
- "Italian": 0.88,
19
- "Japanese": 0.88,
20
- "Korean": 0.875,
21
- "Portuguese": 0.8875,
22
- "Spanish": 0.875,
23
- "Swahili": 0.87,
24
- "Yoruba": 0.8725,
25
- "Chinese": 0.8775,
26
- "Burmese": 0.88
27
- }
28
- },
29
- {
30
- "model_id": "anthropic/claude-3-5-haiku-20241022",
31
- "name": "Claude 3.5 Haiku 20241022",
32
- "developer": "Anthropic",
33
- "scores": {
34
- "Global MMLU Lite": 0.6114,
35
- "Culturally Sensitive": 0.5834,
36
- "Culturally Agnostic": 0.6394,
37
- "Arabic": 0.695,
38
- "English": 0.485,
39
- "Bengali": 0.675,
40
- "German": 0.565,
41
- "French": 0.61,
42
- "Hindi": 0.6575,
43
- "Indonesian": 0.5475,
44
- "Italian": 0.48,
45
- "Japanese": 0.655,
46
- "Korean": 0.6575,
47
- "Portuguese": 0.5225,
48
- "Spanish": 0.485,
49
- "Swahili": 0.69,
50
- "Yoruba": 0.6675,
51
- "Chinese": 0.69,
52
- "Burmese": 0.7
53
- }
54
- },
55
- {
56
- "model_id": "anthropic/claude-3-7-sonnet-20250219",
57
- "name": "claude-3-7-sonnet-20250219",
58
- "developer": "Anthropic",
59
- "scores": {
60
- "Global MMLU Lite": 0.8078,
61
- "Culturally Sensitive": 0.7794,
62
- "Culturally Agnostic": 0.8362,
63
- "Arabic": 0.7925,
64
- "English": 0.7625,
65
- "Bengali": 0.825,
66
- "German": 0.8125,
67
- "French": 0.7675,
68
- "Hindi": 0.805,
69
- "Indonesian": 0.8175,
70
- "Italian": 0.8225,
71
- "Japanese": 0.8425,
72
- "Korean": 0.83,
73
- "Portuguese": 0.77,
74
- "Spanish": 0.8075,
75
- "Swahili": 0.8125,
76
- "Yoruba": 0.81,
77
- "Chinese": 0.835,
78
- "Burmese": 0.8125
79
- }
80
- },
81
- {
82
- "model_id": "anthropic/claude-opus-4-1-20250805",
83
- "name": "claude-opus-4-1-20250805",
84
- "developer": "Anthropic",
85
- "scores": {
86
- "Global MMLU Lite": 0.943,
87
- "Culturally Sensitive": 0.9331,
88
- "Culturally Agnostic": 0.9528,
89
- "Arabic": 0.945,
90
- "English": 0.9475,
91
- "Bengali": 0.9425,
92
- "German": 0.94,
93
- "French": 0.945,
94
- "Hindi": 0.9475,
95
- "Indonesian": 0.9425,
96
- "Italian": 0.94,
97
- "Japanese": 0.94,
98
- "Korean": 0.95,
99
- "Portuguese": 0.945,
100
- "Spanish": 0.945,
101
- "Swahili": 0.93,
102
- "Yoruba": 0.9375,
103
- "Chinese": 0.945,
104
- "Burmese": 0.945
105
- }
106
- },
107
- {
108
- "model_id": "anthropic/claude-sonnet-4-20250514",
109
- "name": "claude-sonnet-4-20250514",
110
- "developer": "Anthropic",
111
- "scores": {
112
- "Global MMLU Lite": 0.9058,
113
- "Culturally Sensitive": 0.8913,
114
- "Culturally Agnostic": 0.9203,
115
- "Arabic": 0.9125,
116
- "English": 0.905,
117
- "Bengali": 0.9075,
118
- "German": 0.9125,
119
- "French": 0.91,
120
- "Hindi": 0.9,
121
- "Indonesian": 0.9025,
122
- "Italian": 0.9075,
123
- "Japanese": 0.9,
124
- "Korean": 0.9125,
125
- "Portuguese": 0.91,
126
- "Spanish": 0.9075,
127
- "Swahili": 0.8975,
128
- "Yoruba": 0.8975,
129
- "Chinese": 0.9175,
130
- "Burmese": 0.8925
131
- }
132
- },
133
- {
134
- "model_id": "cohere/aya-expanse-32b",
135
- "name": "aya-expanse-32b",
136
- "developer": "cohere",
137
- "scores": {
138
- "Global MMLU Lite": 0.7353,
139
- "Culturally Sensitive": 0.6891,
140
- "Culturally Agnostic": 0.7815,
141
- "Arabic": 0.7425,
142
- "English": 0.7544,
143
- "Bengali": 0.7343,
144
- "German": 0.7425,
145
- "French": 0.7325,
146
- "Hindi": 0.7375,
147
- "Indonesian": 0.7594,
148
- "Italian": 0.7305,
149
- "Japanese": 0.7419,
150
- "Korean": 0.7525,
151
- "Portuguese": 0.7544,
152
- "Spanish": 0.7362,
153
- "Swahili": 0.7071,
154
- "Yoruba": 0.6942,
155
- "Chinese": 0.743,
156
- "Burmese": 0.7025
157
- }
158
- },
159
- {
160
- "model_id": "cohere/command-a-03-2025",
161
- "name": "command-a-03-2025",
162
- "developer": "cohere",
163
- "scores": {
164
- "Global MMLU Lite": 0.8385,
165
- "Culturally Sensitive": 0.7993,
166
- "Culturally Agnostic": 0.8778,
167
- "Arabic": 0.8425,
168
- "English": 0.855,
169
- "Bengali": 0.8225,
170
- "German": 0.8425,
171
- "French": 0.8375,
172
- "Hindi": 0.8421,
173
- "Indonesian": 0.8546,
174
- "Italian": 0.8375,
175
- "Japanese": 0.845,
176
- "Korean": 0.85,
177
- "Portuguese": 0.84,
178
- "Spanish": 0.8525,
179
- "Swahili": 0.8275,
180
- "Yoruba": 0.815,
181
- "Chinese": 0.835,
182
- "Burmese": 0.8175
183
- }
184
- },
185
- {
186
- "model_id": "deepseek/deepseek-r1-0528",
187
- "name": "deepseek-r1-0528",
188
- "developer": "deepseek",
189
- "scores": {
190
- "Global MMLU Lite": 0.6744,
191
- "Culturally Sensitive": 0.6672,
192
- "Culturally Agnostic": 0.6816,
193
- "Arabic": 0.6825,
194
- "English": 0.715,
195
- "Bengali": 0.655,
196
- "German": 0.6375,
197
- "French": 0.6925,
198
- "Hindi": 0.6475,
199
- "Indonesian": 0.655,
200
- "Italian": 0.6775,
201
- "Japanese": 0.7725,
202
- "Korean": 0.6575,
203
- "Portuguese": 0.635,
204
- "Spanish": 0.7175,
205
- "Swahili": 0.6775,
206
- "Yoruba": 0.77,
207
- "Chinese": 0.5075,
208
- "Burmese": 0.69
209
- }
210
- },
211
- {
212
- "model_id": "deepseek/deepseek-v3.1",
213
- "name": "deepseek-v3.1",
214
- "developer": "deepseek",
215
- "scores": {
216
- "Global MMLU Lite": 0.8044,
217
- "Culturally Sensitive": 0.7793,
218
- "Culturally Agnostic": 0.8295,
219
- "Arabic": 0.805,
220
- "English": 0.825,
221
- "Bengali": 0.8157,
222
- "German": 0.7925,
223
- "French": 0.8175,
224
- "Hindi": 0.7569,
225
- "Indonesian": 0.7764,
226
- "Italian": 0.8075,
227
- "Japanese": 0.8312,
228
- "Korean": 0.8125,
229
- "Portuguese": 0.8246,
230
- "Spanish": 0.8125,
231
- "Swahili": 0.801,
232
- "Yoruba": 0.7831,
233
- "Chinese": 0.8161,
234
- "Burmese": 0.7925
235
- }
236
- },
237
- {
238
- "model_id": "google/gemini-2.5-flash",
239
- "name": "Gemini 2.5 Flash",
240
- "developer": "Google",
241
- "scores": {
242
- "Global MMLU Lite": 0.9145,
243
- "Culturally Sensitive": 0.9,
244
- "Culturally Agnostic": 0.9291,
245
- "Arabic": 0.9125,
246
- "English": 0.9325,
247
- "Bengali": 0.91,
248
- "German": 0.9025,
249
- "French": 0.91,
250
- "Hindi": 0.925,
251
- "Indonesian": 0.9075,
252
- "Italian": 0.9225,
253
- "Japanese": 0.9125,
254
- "Korean": 0.915,
255
- "Portuguese": 0.9125,
256
- "Spanish": 0.9175,
257
- "Swahili": 0.915,
258
- "Yoruba": 0.9075,
259
- "Chinese": 0.915,
260
- "Burmese": 0.915
261
- }
262
- },
263
- {
264
- "model_id": "google/gemini-2.5-flash-preview-05-20",
265
- "name": "gemini-2.5-flash-preview-05-20",
266
- "developer": "Google",
267
- "scores": {
268
- "Global MMLU Lite": 0.9092,
269
- "Culturally Sensitive": 0.8925,
270
- "Culturally Agnostic": 0.9259,
271
- "Arabic": 0.905,
272
- "English": 0.9225,
273
- "Bengali": 0.91,
274
- "German": 0.905,
275
- "French": 0.925,
276
- "Hindi": 0.9125,
277
- "Indonesian": 0.9075,
278
- "Italian": 0.89,
279
- "Japanese": 0.9125,
280
- "Korean": 0.9075,
281
- "Portuguese": 0.915,
282
- "Spanish": 0.915,
283
- "Swahili": 0.905,
284
- "Yoruba": 0.8825,
285
- "Chinese": 0.93,
286
- "Burmese": 0.9025
287
- }
288
- },
289
- {
290
- "model_id": "google/gemini-2.5-pro",
291
- "name": "Gemini 2.5 Pro",
292
- "developer": "Google",
293
- "scores": {
294
- "Global MMLU Lite": 0.9323,
295
- "Culturally Sensitive": 0.9241,
296
- "Culturally Agnostic": 0.9406,
297
- "Arabic": 0.9475,
298
- "English": 0.9275,
299
- "Bengali": 0.9275,
300
- "German": 0.93,
301
- "French": 0.9425,
302
- "Hindi": 0.9275,
303
- "Indonesian": 0.925,
304
- "Italian": 0.935,
305
- "Japanese": 0.9375,
306
- "Korean": 0.9275,
307
- "Portuguese": 0.93,
308
- "Spanish": 0.94,
309
- "Swahili": 0.9375,
310
- "Yoruba": 0.925,
311
- "Chinese": 0.9275,
312
- "Burmese": 0.93
313
- }
314
- },
315
- {
316
- "model_id": "google/gemini-3-pro-preview",
317
- "name": "gemini-3-pro-preview",
318
- "developer": "Google",
319
- "scores": {
320
- "Global MMLU Lite": 0.9453,
321
- "Culturally Sensitive": 0.9397,
322
- "Culturally Agnostic": 0.9509,
323
- "Arabic": 0.9475,
324
- "English": 0.9425,
325
- "Bengali": 0.9425,
326
- "German": 0.94,
327
- "French": 0.9575,
328
- "Hindi": 0.9425,
329
- "Indonesian": 0.955,
330
- "Italian": 0.955,
331
- "Japanese": 0.94,
332
- "Korean": 0.94,
333
- "Portuguese": 0.9425,
334
- "Spanish": 0.9475,
335
- "Swahili": 0.94,
336
- "Yoruba": 0.9425,
337
- "Chinese": 0.9475,
338
- "Burmese": 0.9425
339
- }
340
- },
341
- {
342
- "model_id": "google/gemma-3-27b-it",
343
- "name": "gemma-3-27b-it",
344
- "developer": "Google",
345
- "scores": {
346
- "Global MMLU Lite": 0.763,
347
- "Culturally Sensitive": 0.7528,
348
- "Culturally Agnostic": 0.7733,
349
- "Arabic": 0.78,
350
- "English": 0.7337,
351
- "Bengali": 0.75,
352
- "German": 0.775,
353
- "French": 0.7481,
354
- "Hindi": 0.7335,
355
- "Indonesian": 0.7563,
356
- "Italian": 0.75,
357
- "Japanese": 0.7925,
358
- "Korean": 0.798,
359
- "Portuguese": 0.7481,
360
- "Spanish": 0.7494,
361
- "Swahili": 0.785,
362
- "Yoruba": 0.7444,
363
- "Chinese": 0.7925,
364
- "Burmese": 0.7719
365
- }
366
- },
367
- {
368
- "model_id": "google/gemma-3-4b-it",
369
- "name": "gemma-3-4b-it",
370
- "developer": "Google",
371
- "scores": {
372
- "Global MMLU Lite": 0.6511,
373
- "Culturally Sensitive": 0.6116,
374
- "Culturally Agnostic": 0.6906,
375
- "Arabic": 0.6525,
376
- "English": 0.67,
377
- "Bengali": 0.68,
378
- "German": 0.6525,
379
- "French": 0.6575,
380
- "Hindi": 0.6475,
381
- "Indonesian": 0.6775,
382
- "Italian": 0.6675,
383
- "Japanese": 0.6325,
384
- "Korean": 0.66,
385
- "Portuguese": 0.68,
386
- "Spanish": 0.6725,
387
- "Swahili": 0.6075,
388
- "Yoruba": 0.5825,
389
- "Chinese": 0.6475,
390
- "Burmese": 0.63
391
- }
392
- },
393
- {
394
- "model_id": "ibm/granite-4.0-h-small",
395
- "name": "granite-4.0-h-small",
396
- "developer": "ibm",
397
- "scores": {
398
- "Global MMLU Lite": 0.7503,
399
- "Culturally Sensitive": 0.7182,
400
- "Culturally Agnostic": 0.7826,
401
- "Arabic": 0.7613,
402
- "English": 0.77,
403
- "Bengali": 0.7613,
404
- "German": 0.755,
405
- "French": 0.7594,
406
- "Hindi": 0.7575,
407
- "Indonesian": 0.7614,
408
- "Italian": 0.7525,
409
- "Japanese": 0.7406,
410
- "Korean": 0.7525,
411
- "Portuguese": 0.757,
412
- "Spanish": 0.7638,
413
- "Swahili": 0.7318,
414
- "Yoruba": 0.6921,
415
- "Chinese": 0.7475,
416
- "Burmese": 0.7419
417
- }
418
- },
419
- {
420
- "model_id": "mistralai/mistral-medium-3",
421
- "name": "mistral-medium-3",
422
- "developer": "mistralai",
423
- "scores": {
424
- "Global MMLU Lite": 0.5511,
425
- "Culturally Sensitive": 0.5391,
426
- "Culturally Agnostic": 0.5631,
427
- "Arabic": 0.455,
428
- "English": 0.38,
429
- "Bengali": 0.5175,
430
- "German": 0.4775,
431
- "French": 0.41,
432
- "Hindi": 0.555,
433
- "Indonesian": 0.515,
434
- "Italian": 0.535,
435
- "Japanese": 0.58,
436
- "Korean": 0.595,
437
- "Portuguese": 0.5175,
438
- "Spanish": 0.5375,
439
- "Swahili": 0.7075,
440
- "Yoruba": 0.7675,
441
- "Chinese": 0.535,
442
- "Burmese": 0.7325
443
- }
444
- },
445
- {
446
- "model_id": "mistralai/mistral-small-2503",
447
- "name": "mistral-small-2503",
448
- "developer": "mistralai",
449
- "scores": {
450
- "Global MMLU Lite": 0.7852,
451
- "Culturally Sensitive": 0.7537,
452
- "Culturally Agnostic": 0.8166,
453
- "Arabic": 0.7875,
454
- "English": 0.8,
455
- "Bengali": 0.7725,
456
- "German": 0.7975,
457
- "French": 0.8,
458
- "Hindi": 0.795,
459
- "Indonesian": 0.785,
460
- "Italian": 0.805,
461
- "Japanese": 0.77,
462
- "Korean": 0.79,
463
- "Portuguese": 0.7925,
464
- "Spanish": 0.7825,
465
- "Swahili": 0.775,
466
- "Yoruba": 0.735,
467
- "Chinese": 0.7925,
468
- "Burmese": 0.7825
469
- }
470
- },
471
- {
472
- "model_id": "openai/gpt-4.1-2025-04-14",
473
- "name": "gpt-4.1-2025-04-14",
474
- "developer": "OpenAI",
475
- "scores": {
476
- "Global MMLU Lite": 0.8755,
477
- "Culturally Sensitive": 0.8541,
478
- "Culturally Agnostic": 0.8969,
479
- "Arabic": 0.88,
480
- "English": 0.8825,
481
- "Bengali": 0.8625,
482
- "German": 0.875,
483
- "French": 0.8875,
484
- "Hindi": 0.8775,
485
- "Indonesian": 0.885,
486
- "Italian": 0.88,
487
- "Japanese": 0.8725,
488
- "Korean": 0.87,
489
- "Portuguese": 0.875,
490
- "Spanish": 0.885,
491
- "Swahili": 0.8725,
492
- "Yoruba": 0.875,
493
- "Chinese": 0.87,
494
- "Burmese": 0.8575
495
- }
496
- },
497
- {
498
- "model_id": "openai/gpt-5-2025-08-07",
499
- "name": "gpt-5-2025-08-07",
500
- "developer": "OpenAI",
501
- "scores": {
502
- "Global MMLU Lite": 0.8895,
503
- "Culturally Sensitive": 0.8913,
504
- "Culturally Agnostic": 0.8878,
505
- "Arabic": 0.8925,
506
- "English": 0.8725,
507
- "Bengali": 0.9,
508
- "German": 0.91,
509
- "French": 0.9075,
510
- "Hindi": 0.865,
511
- "Indonesian": 0.795,
512
- "Italian": 0.9075,
513
- "Japanese": 0.8875,
514
- "Korean": 0.915,
515
- "Portuguese": 0.8875,
516
- "Spanish": 0.905,
517
- "Swahili": 0.865,
518
- "Yoruba": 0.9125,
519
- "Chinese": 0.895,
520
- "Burmese": 0.915
521
- }
522
- },
523
- {
524
- "model_id": "openai/o3-mini-2025-01-31",
525
- "name": "o3-mini-2025-01-31",
526
- "developer": "OpenAI",
527
- "scores": {
528
- "Global MMLU Lite": 0.78,
529
- "Culturally Sensitive": 0.765,
530
- "Culturally Agnostic": 0.795,
531
- "Arabic": 0.7725,
532
- "English": 0.8025,
533
- "Bengali": 0.77,
534
- "German": 0.7525,
535
- "French": 0.74,
536
- "Hindi": 0.7525,
537
- "Indonesian": 0.7425,
538
- "Italian": 0.8,
539
- "Japanese": 0.81,
540
- "Korean": 0.8075,
541
- "Portuguese": 0.7975,
542
- "Spanish": 0.775,
543
- "Swahili": 0.765,
544
- "Yoruba": 0.7725,
545
- "Chinese": 0.8125,
546
- "Burmese": 0.8075
547
- }
548
- },
549
- {
550
- "model_id": "openai/o4-mini-2025-04-16",
551
- "name": "o4-mini-2025-04-16",
552
- "developer": "OpenAI",
553
- "scores": {
554
- "Global MMLU Lite": 0.8705,
555
- "Culturally Sensitive": 0.8503,
556
- "Culturally Agnostic": 0.8906,
557
- "Arabic": 0.865,
558
- "English": 0.8675,
559
- "Bengali": 0.8875,
560
- "German": 0.8775,
561
- "French": 0.87,
562
- "Hindi": 0.87,
563
- "Indonesian": 0.8675,
564
- "Italian": 0.855,
565
- "Japanese": 0.885,
566
- "Korean": 0.88,
567
- "Portuguese": 0.88,
568
- "Spanish": 0.855,
569
- "Swahili": 0.8525,
570
- "Yoruba": 0.8525,
571
- "Chinese": 0.89,
572
- "Burmese": 0.8725
573
- }
574
- },
575
- {
576
- "model_id": "unknown/aya-expanse-32b",
577
- "name": "aya-expanse-32b",
578
- "developer": "unknown",
579
- "scores": {
580
- "Global MMLU Lite": 0.7353,
581
- "Culturally Sensitive": 0.6891,
582
- "Culturally Agnostic": 0.7815,
583
- "Arabic": 0.7425,
584
- "English": 0.7544,
585
- "Bengali": 0.7343,
586
- "German": 0.7425,
587
- "French": 0.7325,
588
- "Hindi": 0.7375,
589
- "Indonesian": 0.7594,
590
- "Italian": 0.7305,
591
- "Japanese": 0.7419,
592
- "Korean": 0.7525,
593
- "Portuguese": 0.7544,
594
- "Spanish": 0.7362,
595
- "Swahili": 0.7071,
596
- "Yoruba": 0.6942,
597
- "Chinese": 0.743,
598
- "Burmese": 0.7025
599
- }
600
- },
601
- {
602
- "model_id": "unknown/granite-4.0-h-small",
603
- "name": "granite-4.0-h-small",
604
- "developer": "unknown",
605
- "scores": {
606
- "Global MMLU Lite": 0.7503,
607
- "Culturally Sensitive": 0.7182,
608
- "Culturally Agnostic": 0.7826,
609
- "Arabic": 0.7613,
610
- "English": 0.77,
611
- "Bengali": 0.7613,
612
- "German": 0.755,
613
- "French": 0.7594,
614
- "Hindi": 0.7575,
615
- "Indonesian": 0.7614,
616
- "Italian": 0.7525,
617
- "Japanese": 0.7406,
618
- "Korean": 0.7525,
619
- "Portuguese": 0.757,
620
- "Spanish": 0.7638,
621
- "Swahili": 0.7318,
622
- "Yoruba": 0.6921,
623
- "Chinese": 0.7475,
624
- "Burmese": 0.7419
625
- }
626
- },
627
- {
628
- "model_id": "unknown/o4-mini-2025-04-16",
629
- "name": "o4-mini-2025-04-16",
630
- "developer": "unknown",
631
- "scores": {
632
- "Global MMLU Lite": 0.8705,
633
- "Culturally Sensitive": 0.8503,
634
- "Culturally Agnostic": 0.8906,
635
- "Arabic": 0.865,
636
- "English": 0.8675,
637
- "Bengali": 0.8875,
638
- "German": 0.8775,
639
- "French": 0.87,
640
- "Hindi": 0.87,
641
- "Indonesian": 0.8675,
642
- "Italian": 0.855,
643
- "Japanese": 0.885,
644
- "Korean": 0.88,
645
- "Portuguese": 0.88,
646
- "Spanish": 0.855,
647
- "Swahili": 0.8525,
648
- "Yoruba": 0.8525,
649
- "Chinese": 0.89,
650
- "Burmese": 0.8725
651
- }
652
- },
653
- {
654
- "model_id": "xai/grok-3-mini",
655
- "name": "grok-3-mini",
656
- "developer": "xAI",
657
- "scores": {
658
- "Global MMLU Lite": 0.673,
659
- "Culturally Sensitive": 0.6717,
660
- "Culturally Agnostic": 0.6743,
661
- "Arabic": 0.755,
662
- "English": 0.5075,
663
- "Bengali": 0.7355,
664
- "German": 0.6591,
665
- "French": 0.485,
666
- "Hindi": 0.56,
667
- "Indonesian": 0.725,
668
- "Italian": 0.696,
669
- "Japanese": 0.6575,
670
- "Korean": 0.7325,
671
- "Portuguese": 0.6275,
672
- "Spanish": 0.61,
673
- "Swahili": 0.7625,
674
- "Yoruba": 0.8296,
675
- "Chinese": 0.5564,
676
- "Burmese": 0.8693
677
- }
678
- },
679
- {
680
- "model_id": "xai/grok-4-0709",
681
- "name": "grok-4-0709",
682
- "developer": "xAI",
683
- "scores": {
684
- "Global MMLU Lite": 0.8881,
685
- "Culturally Sensitive": 0.8862,
686
- "Culturally Agnostic": 0.89,
687
- "Arabic": 0.885,
688
- "English": 0.905,
689
- "Bengali": 0.8925,
690
- "German": 0.8725,
691
- "French": 0.875,
692
- "Hindi": 0.8675,
693
- "Indonesian": 0.89,
694
- "Italian": 0.9025,
695
- "Japanese": 0.87,
696
- "Korean": 0.895,
697
- "Portuguese": 0.8725,
698
- "Spanish": 0.9075,
699
- "Swahili": 0.91,
700
- "Yoruba": 0.905,
701
- "Chinese": 0.8525,
702
- "Burmese": 0.9075
703
- }
704
- }
705
- ]
706
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/helm_capabilities.json DELETED
@@ -1,1026 +0,0 @@
1
- {
2
- "benchmark_cards": {
3
- "Omni-MATH": {
4
- "benchmark_details": {
5
- "name": "Omni-MATH",
6
- "overview": "Omni-MATH is a benchmark designed to measure the mathematical reasoning capabilities of large language models at the Olympiad competition level. It comprises 4,428 challenging problems, categorized into over 33 sub-domains and more than 10 distinct difficulty levels. It was created because existing mathematical benchmarks are nearing saturation and are inadequate for assessing models on truly difficult tasks.",
7
- "data_type": "text",
8
- "domains": [
9
- "math",
10
- "olympiads"
11
- ],
12
- "languages": [
13
- "English"
14
- ],
15
- "similar_benchmarks": [
16
- "GSM8K",
17
- "MATH"
18
- ],
19
- "resources": [
20
- "https://arxiv.org/abs/2410.07985",
21
- "https://huggingface.co/datasets/KbsdJames/Omni-MATH",
22
- "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
23
- ]
24
- },
25
- "purpose_and_intended_users": {
26
- "goal": "To provide a challenging benchmark for assessing and advancing the mathematical reasoning capabilities of large language models at the Olympiad and competition level, as existing benchmarks have become less challenging. It enables a nuanced analysis of performance across various mathematical disciplines and complexity levels.",
27
- "audience": [
28
- "Researchers evaluating large language models"
29
- ],
30
- "tasks": [
31
- "Solving Olympiad-level mathematical problems",
32
- "Solving competition-level mathematical problems",
33
- "Rule-based evaluation on a filtered subset of problems (Omni-MATH-Rule)"
34
- ],
35
- "limitations": "Some problems are not suitable for reliable rule-based evaluation, leading to a filtered 'testable' set and an 'untestable' set.",
36
- "out_of_scope_uses": [
37
- "Not specified"
38
- ]
39
- },
40
- "data": {
41
- "source": "The data is sourced from contest pages, the AoPS Wiki, and the AoPS Forum. Initial raw counts from these sources were filtered to create the final dataset.",
42
- "size": "4,428 problems. A subset of 2,821 problems (Omni-MATH-Rule) is suitable for rule-based evaluation, and a separate subset of 100 samples was used for meta-evaluation.",
43
- "format": "JSON",
44
- "annotation": "Problems underwent rigorous human annotation by a team including PhD and Master's students. Each problem was reviewed by two annotators for cross-validation, achieving high agreement rates."
45
- },
46
- "methodology": {
47
- "methods": [
48
- "Models are evaluated by generating solutions to the mathematical problems.",
49
- "Evaluation is conducted using both model-based judgment (e.g., GPT-4o or the open-source Omni-Judge) and rule-based evaluation for a specific subset of problems (Omni-MATH-Rule)."
50
- ],
51
- "metrics": [
52
- "Accuracy (Acc)"
53
- ],
54
- "calculation": "The overall score is the accuracy percentage, calculated as the number of correct solutions divided by the total number of problems.",
55
- "interpretation": "Higher accuracy indicates stronger performance. The benchmark is considered challenging, as even advanced models achieve only moderate accuracy.",
56
- "baseline_results": "Paper baselines (accuracy on full Omni-MATH / Omni-MATH-Rule subset): o1-mini (60.54% / 62.2%), o1-preview (52.55% / 51.7%), qwen2.5-MATH-72b-Instruct (36.20% / 35.7%), qwen2.5-MATH-7b-Instruct (33.22% / 32.3%), GPT-4o (30.49% / 29.2%), NuminaMATH-72b-cot (28.45% / 27.1%), DeepseekMATH-7b-RL (16.12% / 14.9%). EEE results: OLMo 2 32B Instruct March 2025 scored 0.161 (16.1%).",
57
- "validation": "For the Omni-MATH-Rule subset, two PhD students annotated problems to determine suitability for rule-based evaluation, achieving 98% cross-validation accuracy. A meta-evaluation compared model-generated judgments against human gold-standard annotations on a 100-sample subset to assess reliability."
58
- },
59
- "ethical_and_legal_considerations": {
60
- "privacy_and_anonymity": "Not specified",
61
- "data_licensing": "Apache License 2.0",
62
- "consent_procedures": "Not specified",
63
- "compliance_with_regulations": "Not specified"
64
- },
65
- "possible_risks": [
66
- {
67
- "category": "Over- or under-reliance",
68
- "description": [
69
- "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
70
- ],
71
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
72
- },
73
- {
74
- "category": "Unrepresentative data",
75
- "description": [
76
- "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
77
- ],
78
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
79
- },
80
- {
81
- "category": "Data bias",
82
- "description": [
83
- "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
84
- ],
85
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
86
- },
87
- {
88
- "category": "Lack of data transparency",
89
- "description": [
90
- "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
91
- ],
92
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
93
- },
94
- {
95
- "category": "Improper usage",
96
- "description": [
97
- "Improper usage occurs when a model is used for a purpose that it was not originally designed for."
98
- ],
99
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html"
100
- }
101
- ],
102
- "flagged_fields": {},
103
- "missing_fields": [
104
- "purpose_and_intended_users.out_of_scope_uses",
105
- "ethical_and_legal_considerations.privacy_and_anonymity",
106
- "ethical_and_legal_considerations.consent_procedures",
107
- "ethical_and_legal_considerations.compliance_with_regulations"
108
- ],
109
- "card_info": {
110
- "created_at": "2026-03-17T13:34:44.331592",
111
- "llm": "deepseek-ai/DeepSeek-V3.2"
112
- }
113
- },
114
- "WildBench": {
115
- "benchmark_details": {
116
- "name": "WildBench",
117
- "overview": "WildBench is a benchmark that measures the performance of large language models on challenging, open-ended tasks derived from real-world user queries. It consists of 1,024 tasks selected from human-chatbot conversation logs. Its distinctive features include using a natural distribution of real user queries and employing automated evaluation with two novel metrics, WB-Reward and WB-Score, which utilize task-specific checklists and structured explanations.",
118
- "data_type": "tabular, text",
119
- "domains": [
120
- "Info Seeking",
121
- "Math & Data",
122
- "Reasoning & Planning",
123
- "Creative Tasks"
124
- ],
125
- "languages": [
126
- "English"
127
- ],
128
- "similar_benchmarks": [
129
- "AlpacaEval",
130
- "ArenaHard",
131
- "MT-bench",
132
- "Chatbot Arena"
133
- ],
134
- "resources": [
135
- "https://arxiv.org/abs/2406.04770",
136
- "https://huggingface.co/datasets/allenai/WildBench",
137
- "https://huggingface.co/spaces/allenai/WildBench",
138
- "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
139
- ]
140
- },
141
- "purpose_and_intended_users": {
142
- "goal": "To provide a realistic, automated, and cost-effective evaluation framework for large language models that reflects their capabilities on challenging, real-world user tasks. It aims to be contamination-resilient and dynamically updated.",
143
- "audience": [
144
- "Researchers and practitioners evaluating large language models"
145
- ],
146
- "tasks": [
147
- "Open-ended text generation in response to diverse user queries"
148
- ],
149
- "limitations": "The benchmark acknowledges a common issue in LLM-as-a-judge evaluations: bias towards longer outputs. It introduces a length-penalty method to mitigate this.",
150
- "out_of_scope_uses": [
151
- "Not specified"
152
- ]
153
- },
154
- "data": {
155
- "source": "The data is derived from real user-chatbot conversation logs collected by the AI2 WildChat project. Over one million logs were filtered to create the benchmark.",
156
- "size": "The benchmark contains 1,024 tasks. A separate 'v2-hard' configuration contains 256 examples. The dataset falls into the size category of 1K to 10K examples.",
157
- "format": "The data is stored in Parquet format.",
158
- "annotation": "Human annotation was used for quality control. GPT-4-Turbo assisted by summarizing query intents to help human reviewers filter out nonsensical tasks. Human reviewers manually assessed tasks to ensure they were challenging, diverse, and that associated checklist questions were clear. The final set of tasks was retained after this process. Each example includes fine-grained annotations such as task types and checklists for evaluating response quality."
159
- },
160
- "methodology": {
161
- "methods": [
162
- "Models are evaluated automatically using LLM-as-a-judge, with GPT-4-turbo as the evaluator. The evaluation is not fine-tuning-based and appears to be zero-shot or prompted.",
163
- "The evaluation uses length-penalized pairwise comparisons and individual scoring to prevent bias towards longer outputs."
164
- ],
165
- "metrics": [
166
- "WB-Reward (for pairwise comparisons)",
167
- "WB-Score (for individual scoring)",
168
- "WB-Elo (for leaderboard ranking, merging WB-Reward and WB-Score)"
169
- ],
170
- "calculation": "WB-Reward compares a test model against three baseline models of varying performance, producing outcomes of 'much better', 'slightly better', 'tie', 'slightly worse', or 'much worse'. A length penalty converts 'slightly better/worse' to 'tie' if the winner's response is excessively longer. WB-Score evaluates individual model outputs. WB-Elo merges the pairwise comparisons from WB-Reward and WB-Score to perform Elo rating updates.",
171
- "interpretation": "Higher scores on WB-Reward and WB-Score indicate better performance. Strong performance is validated by a high correlation with human-voted Elo ratings from Chatbot Arena, with WB-Reward achieving a Pearson correlation of 0.98 and WB-Score reaching 0.95.",
172
- "baseline_results": "Paper baselines: The paper reports results for 40 LLMs but does not list specific scores. It notes that using Haiku as a baseline yields the best correlation with human ratings. EEE results: OLMo 2 32B Instruct March 2025 scored 0.7340 on WildBench.",
173
- "validation": "The evaluation metrics are validated by their strong correlation with human-voted Elo ratings from Chatbot Arena on hard tasks. Checklist questions used in the evaluation were manually verified for clarity and relevance."
174
- },
175
- "ethical_and_legal_considerations": {
176
- "privacy_and_anonymity": "Not specified",
177
- "data_licensing": "The dataset is made available under a CC BY license, intended for research and educational use in accordance with AI2's Responsible Use Guidelines.",
178
- "consent_procedures": "Not specified",
179
- "compliance_with_regulations": "Not specified"
180
- },
181
- "possible_risks": [
182
- {
183
- "category": "Over- or under-reliance",
184
- "description": [
185
- "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
186
- ],
187
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
188
- },
189
- {
190
- "category": "Unrepresentative data",
191
- "description": [
192
- "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
193
- ],
194
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
195
- },
196
- {
197
- "category": "Data bias",
198
- "description": [
199
- "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
200
- ],
201
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
202
- },
203
- {
204
- "category": "Data contamination",
205
- "description": [
206
- "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation."
207
- ],
208
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html"
209
- },
210
- {
211
- "category": "Lack of data transparency",
212
- "description": [
213
- "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
214
- ],
215
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
216
- }
217
- ],
218
- "flagged_fields": {},
219
- "missing_fields": [
220
- "purpose_and_intended_users.out_of_scope_uses",
221
- "ethical_and_legal_considerations.privacy_and_anonymity",
222
- "ethical_and_legal_considerations.consent_procedures",
223
- "ethical_and_legal_considerations.compliance_with_regulations"
224
- ],
225
- "card_info": {
226
- "created_at": "2026-03-17T13:56:24.159440",
227
- "llm": "deepseek-ai/DeepSeek-V3.2"
228
- }
229
- }
230
- },
231
- "models": [
232
- {
233
- "model_id": "allenai/OLMo-2-1124-7B-Instruct",
234
- "name": "OLMo 2 7B Instruct November 2024",
235
- "developer": "allenai",
236
- "scores": {
237
- "Mean score": 0.405,
238
- "MMLU-Pro": 0.292,
239
- "GPQA": 0.296,
240
- "IFEval": 0.693,
241
- "WildBench": 0.628,
242
- "Omni-MATH": 0.116
243
- }
244
- },
245
- {
246
- "model_id": "allenai/OLMoE-1B-7B-0125-Instruct",
247
- "name": "OLMoE 1B-7B Instruct January 2025",
248
- "developer": "allenai",
249
- "scores": {
250
- "Mean score": 0.332,
251
- "MMLU-Pro": 0.169,
252
- "GPQA": 0.22,
253
- "IFEval": 0.628,
254
- "WildBench": 0.551,
255
- "Omni-MATH": 0.093
256
- }
257
- },
258
- {
259
- "model_id": "allenai/olmo-2-0325-32b-instruct",
260
- "name": "OLMo 2 32B Instruct March 2025",
261
- "developer": "allenai",
262
- "scores": {
263
- "Mean score": 0.475,
264
- "MMLU-Pro": 0.414,
265
- "GPQA": 0.287,
266
- "IFEval": 0.78,
267
- "WildBench": 0.734,
268
- "Omni-MATH": 0.161
269
- }
270
- },
271
- {
272
- "model_id": "allenai/olmo-2-1124-13b-instruct",
273
- "name": "OLMo 2 13B Instruct November 2024",
274
- "developer": "allenai",
275
- "scores": {
276
- "Mean score": 0.44,
277
- "MMLU-Pro": 0.31,
278
- "GPQA": 0.316,
279
- "IFEval": 0.73,
280
- "WildBench": 0.689,
281
- "Omni-MATH": 0.156
282
- }
283
- },
284
- {
285
- "model_id": "amazon/nova-lite-v1:0",
286
- "name": "Amazon Nova Lite",
287
- "developer": "amazon",
288
- "scores": {
289
- "Mean score": 0.551,
290
- "MMLU-Pro": 0.6,
291
- "GPQA": 0.397,
292
- "IFEval": 0.776,
293
- "WildBench": 0.75,
294
- "Omni-MATH": 0.233
295
- }
296
- },
297
- {
298
- "model_id": "amazon/nova-micro-v1:0",
299
- "name": "Amazon Nova Micro",
300
- "developer": "amazon",
301
- "scores": {
302
- "Mean score": 0.522,
303
- "MMLU-Pro": 0.511,
304
- "GPQA": 0.383,
305
- "IFEval": 0.76,
306
- "WildBench": 0.743,
307
- "Omni-MATH": 0.214
308
- }
309
- },
310
- {
311
- "model_id": "amazon/nova-premier-v1:0",
312
- "name": "Amazon Nova Premier",
313
- "developer": "amazon",
314
- "scores": {
315
- "Mean score": 0.637,
316
- "MMLU-Pro": 0.726,
317
- "GPQA": 0.518,
318
- "IFEval": 0.803,
319
- "WildBench": 0.788,
320
- "Omni-MATH": 0.35
321
- }
322
- },
323
- {
324
- "model_id": "amazon/nova-pro-v1:0",
325
- "name": "Amazon Nova Pro",
326
- "developer": "amazon",
327
- "scores": {
328
- "Mean score": 0.591,
329
- "MMLU-Pro": 0.673,
330
- "GPQA": 0.446,
331
- "IFEval": 0.815,
332
- "WildBench": 0.777,
333
- "Omni-MATH": 0.242
334
- }
335
- },
336
- {
337
- "model_id": "anthropic/claude-3-5-haiku-20241022",
338
- "name": "Claude 3.5 Haiku 20241022",
339
- "developer": "Anthropic",
340
- "scores": {
341
- "Mean score": 0.549,
342
- "MMLU-Pro": 0.605,
343
- "GPQA": 0.363,
344
- "IFEval": 0.792,
345
- "WildBench": 0.76,
346
- "Omni-MATH": 0.224
347
- }
348
- },
349
- {
350
- "model_id": "anthropic/claude-3-5-sonnet-20241022",
351
- "name": "Claude 3.5 Sonnet 20241022",
352
- "developer": "Anthropic",
353
- "scores": {
354
- "Mean score": 0.653,
355
- "MMLU-Pro": 0.777,
356
- "GPQA": 0.565,
357
- "IFEval": 0.856,
358
- "WildBench": 0.792,
359
- "Omni-MATH": 0.276
360
- }
361
- },
362
- {
363
- "model_id": "anthropic/claude-3-7-sonnet-20250219",
364
- "name": "claude-3-7-sonnet-20250219",
365
- "developer": "Anthropic",
366
- "scores": {
367
- "Mean score": 0.674,
368
- "MMLU-Pro": 0.784,
369
- "GPQA": 0.608,
370
- "IFEval": 0.834,
371
- "WildBench": 0.814,
372
- "Omni-MATH": 0.33
373
- }
374
- },
375
- {
376
- "model_id": "anthropic/claude-opus-4-20250514",
377
- "name": "Claude 4 Opus 20250514",
378
- "developer": "Anthropic",
379
- "scores": {
380
- "Mean score": 0.757,
381
- "MMLU-Pro": 0.859,
382
- "GPQA": 0.666,
383
- "IFEval": 0.918,
384
- "WildBench": 0.833,
385
- "Omni-MATH": 0.511
386
- }
387
- },
388
- {
389
- "model_id": "anthropic/claude-opus-4-20250514-thinking-10k",
390
- "name": "Claude 4 Opus 20250514, extended thinking",
391
- "developer": "Anthropic",
392
- "scores": {
393
- "Mean score": 0.78,
394
- "MMLU-Pro": 0.875,
395
- "GPQA": 0.709,
396
- "IFEval": 0.849,
397
- "WildBench": 0.852,
398
- "Omni-MATH": 0.616
399
- }
400
- },
401
- {
402
- "model_id": "anthropic/claude-sonnet-4-20250514",
403
- "name": "claude-sonnet-4-20250514",
404
- "developer": "Anthropic",
405
- "scores": {
406
- "Mean score": 0.733,
407
- "MMLU-Pro": 0.843,
408
- "GPQA": 0.643,
409
- "IFEval": 0.839,
410
- "WildBench": 0.825,
411
- "Omni-MATH": 0.512
412
- }
413
- },
414
- {
415
- "model_id": "anthropic/claude-sonnet-4-20250514-thinking-10k",
416
- "name": "Claude 4 Sonnet 20250514, extended thinking",
417
- "developer": "Anthropic",
418
- "scores": {
419
- "Mean score": 0.766,
420
- "MMLU-Pro": 0.843,
421
- "GPQA": 0.706,
422
- "IFEval": 0.84,
423
- "WildBench": 0.838,
424
- "Omni-MATH": 0.602
425
- }
426
- },
427
- {
428
- "model_id": "deepseek-ai/deepseek-r1-0528",
429
- "name": "DeepSeek-R1-0528",
430
- "developer": "deepseek-ai",
431
- "scores": {
432
- "Mean score": 0.699,
433
- "MMLU-Pro": 0.793,
434
- "GPQA": 0.666,
435
- "IFEval": 0.784,
436
- "WildBench": 0.828,
437
- "Omni-MATH": 0.424
438
- }
439
- },
440
- {
441
- "model_id": "deepseek-ai/deepseek-v3",
442
- "name": "DeepSeek v3",
443
- "developer": "deepseek-ai",
444
- "scores": {
445
- "Mean score": 0.665,
446
- "MMLU-Pro": 0.723,
447
- "GPQA": 0.538,
448
- "IFEval": 0.832,
449
- "WildBench": 0.831,
450
- "Omni-MATH": 0.403
451
- }
452
- },
453
- {
454
- "model_id": "google/gemini-1.5-flash-002",
455
- "name": "Gemini 1.5 Flash 002",
456
- "developer": "Google",
457
- "scores": {
458
- "Mean score": 0.609,
459
- "MMLU-Pro": 0.678,
460
- "GPQA": 0.437,
461
- "IFEval": 0.831,
462
- "WildBench": 0.792,
463
- "Omni-MATH": 0.305
464
- }
465
- },
466
- {
467
- "model_id": "google/gemini-1.5-pro-002",
468
- "name": "Gemini 1.5 Pro 002",
469
- "developer": "Google",
470
- "scores": {
471
- "Mean score": 0.657,
472
- "MMLU-Pro": 0.737,
473
- "GPQA": 0.534,
474
- "IFEval": 0.837,
475
- "WildBench": 0.813,
476
- "Omni-MATH": 0.364
477
- }
478
- },
479
- {
480
- "model_id": "google/gemini-2.0-flash-001",
481
- "name": "Gemini 2.0 Flash",
482
- "developer": "Google",
483
- "scores": {
484
- "Mean score": 0.679,
485
- "MMLU-Pro": 0.737,
486
- "GPQA": 0.556,
487
- "IFEval": 0.841,
488
- "WildBench": 0.8,
489
- "Omni-MATH": 0.459
490
- }
491
- },
492
- {
493
- "model_id": "google/gemini-2.0-flash-lite-preview-02-05",
494
- "name": "Gemini 2.0 Flash Lite 02-05 preview",
495
- "developer": "Google",
496
- "scores": {
497
- "Mean score": 0.642,
498
- "MMLU-Pro": 0.72,
499
- "GPQA": 0.5,
500
- "IFEval": 0.824,
501
- "WildBench": 0.79,
502
- "Omni-MATH": 0.374
503
- }
504
- },
505
- {
506
- "model_id": "google/gemini-2.5-flash-lite",
507
- "name": "Gemini 2.5 Flash-Lite",
508
- "developer": "Google",
509
- "scores": {
510
- "Mean score": 0.591,
511
- "MMLU-Pro": 0.537,
512
- "GPQA": 0.309,
513
- "IFEval": 0.81,
514
- "WildBench": 0.818,
515
- "Omni-MATH": 0.48
516
- }
517
- },
518
- {
519
- "model_id": "google/gemini-2.5-flash-preview-04-17",
520
- "name": "Gemini 2.5 Flash 04-17 preview",
521
- "developer": "Google",
522
- "scores": {
523
- "Mean score": 0.626,
524
- "MMLU-Pro": 0.639,
525
- "GPQA": 0.39,
526
- "IFEval": 0.898,
527
- "WildBench": 0.817,
528
- "Omni-MATH": 0.384
529
- }
530
- },
531
- {
532
- "model_id": "google/gemini-2.5-pro-preview-03-25",
533
- "name": "Gemini 2.5 Pro 03-25 preview",
534
- "developer": "Google",
535
- "scores": {
536
- "Mean score": 0.745,
537
- "MMLU-Pro": 0.863,
538
- "GPQA": 0.749,
539
- "IFEval": 0.84,
540
- "WildBench": 0.857,
541
- "Omni-MATH": 0.416
542
- }
543
- },
544
- {
545
- "model_id": "ibm/granite-3.3-8b-instruct",
546
- "name": "IBM Granite 3.3 8B Instruct",
547
- "developer": "ibm",
548
- "scores": {
549
- "Mean score": 0.463,
550
- "MMLU-Pro": 0.343,
551
- "GPQA": 0.325,
552
- "IFEval": 0.729,
553
- "WildBench": 0.741,
554
- "Omni-MATH": 0.176
555
- }
556
- },
557
- {
558
- "model_id": "marin-community/marin-8b-instruct",
559
- "name": "Marin 8B Instruct",
560
- "developer": "marin-community",
561
- "scores": {
562
- "Mean score": 0.325,
563
- "MMLU-Pro": 0.188,
564
- "GPQA": 0.168,
565
- "IFEval": 0.632,
566
- "WildBench": 0.477,
567
- "Omni-MATH": 0.16
568
- }
569
- },
570
- {
571
- "model_id": "meta/llama-3.1-405b-instruct-turbo",
572
- "name": "Llama 3.1 Instruct Turbo 405B",
573
- "developer": "Meta",
574
- "scores": {
575
- "Mean score": 0.618,
576
- "MMLU-Pro": 0.723,
577
- "GPQA": 0.522,
578
- "IFEval": 0.811,
579
- "WildBench": 0.783,
580
- "Omni-MATH": 0.249
581
- }
582
- },
583
- {
584
- "model_id": "meta/llama-3.1-70b-instruct-turbo",
585
- "name": "Llama 3.1 Instruct Turbo 70B",
586
- "developer": "Meta",
587
- "scores": {
588
- "Mean score": 0.574,
589
- "MMLU-Pro": 0.653,
590
- "GPQA": 0.426,
591
- "IFEval": 0.821,
592
- "WildBench": 0.758,
593
- "Omni-MATH": 0.21
594
- }
595
- },
596
- {
597
- "model_id": "meta/llama-3.1-8b-instruct-turbo",
598
- "name": "Llama 3.1 Instruct Turbo 8B",
599
- "developer": "Meta",
600
- "scores": {
601
- "Mean score": 0.444,
602
- "MMLU-Pro": 0.406,
603
- "GPQA": 0.247,
604
- "IFEval": 0.743,
605
- "WildBench": 0.686,
606
- "Omni-MATH": 0.137
607
- }
608
- },
609
- {
610
- "model_id": "meta/llama-4-maverick-17b-128e-instruct-fp8",
611
- "name": "Llama 4 Maverick 17Bx128E Instruct FP8",
612
- "developer": "Meta",
613
- "scores": {
614
- "Mean score": 0.718,
615
- "MMLU-Pro": 0.81,
616
- "GPQA": 0.65,
617
- "IFEval": 0.908,
618
- "WildBench": 0.8,
619
- "Omni-MATH": 0.422
620
- }
621
- },
622
- {
623
- "model_id": "meta/llama-4-scout-17b-16e-instruct",
624
- "name": "Llama 4 Scout 17Bx16E Instruct",
625
- "developer": "Meta",
626
- "scores": {
627
- "Mean score": 0.644,
628
- "MMLU-Pro": 0.742,
629
- "GPQA": 0.507,
630
- "IFEval": 0.818,
631
- "WildBench": 0.779,
632
- "Omni-MATH": 0.373
633
- }
634
- },
635
- {
636
- "model_id": "mistralai/Mixtral-8x22B-Instruct-v0.1",
637
- "name": "Mixtral-8x22B-Instruct-v0.1",
638
- "developer": "mistralai",
639
- "scores": {
640
- "Mean score": 0.478,
641
- "MMLU-Pro": 0.46,
642
- "GPQA": 0.334,
643
- "IFEval": 0.724,
644
- "WildBench": 0.711,
645
- "Omni-MATH": 0.163
646
- }
647
- },
648
- {
649
- "model_id": "mistralai/Mixtral-8x7B-Instruct-v0.1",
650
- "name": "Mixtral-8x7B-Instruct-v0.1",
651
- "developer": "mistralai",
652
- "scores": {
653
- "Mean score": 0.397,
654
- "MMLU-Pro": 0.335,
655
- "GPQA": 0.296,
656
- "IFEval": 0.575,
657
- "WildBench": 0.673,
658
- "Omni-MATH": 0.105
659
- }
660
- },
661
- {
662
- "model_id": "mistralai/mistral-7b-instruct-v0.3",
663
- "name": "Mistral Instruct v0.3 7B",
664
- "developer": "mistralai",
665
- "scores": {
666
- "Mean score": 0.376,
667
- "MMLU-Pro": 0.277,
668
- "GPQA": 0.303,
669
- "IFEval": 0.567,
670
- "WildBench": 0.66,
671
- "Omni-MATH": 0.072
672
- }
673
- },
674
- {
675
- "model_id": "mistralai/mistral-large-2411",
676
- "name": "Mistral Large 2411",
677
- "developer": "mistralai",
678
- "scores": {
679
- "Mean score": 0.598,
680
- "MMLU-Pro": 0.599,
681
- "GPQA": 0.435,
682
- "IFEval": 0.876,
683
- "WildBench": 0.801,
684
- "Omni-MATH": 0.281
685
- }
686
- },
687
- {
688
- "model_id": "mistralai/mistral-small-2503",
689
- "name": "mistral-small-2503",
690
- "developer": "mistralai",
691
- "scores": {
692
- "Mean score": 0.558,
693
- "MMLU-Pro": 0.61,
694
- "GPQA": 0.392,
695
- "IFEval": 0.75,
696
- "WildBench": 0.788,
697
- "Omni-MATH": 0.248
698
- }
699
- },
700
- {
701
- "model_id": "moonshotai/kimi-k2-instruct",
702
- "name": "Kimi K2 Instruct",
703
- "developer": "moonshotai",
704
- "scores": {
705
- "Mean score": 0.768,
706
- "MMLU-Pro": 0.819,
707
- "GPQA": 0.652,
708
- "IFEval": 0.85,
709
- "WildBench": 0.862,
710
- "Omni-MATH": 0.654
711
- }
712
- },
713
- {
714
- "model_id": "openai/gpt-4.1-2025-04-14",
715
- "name": "gpt-4.1-2025-04-14",
716
- "developer": "OpenAI",
717
- "scores": {
718
- "Mean score": 0.727,
719
- "MMLU-Pro": 0.811,
720
- "GPQA": 0.659,
721
- "IFEval": 0.838,
722
- "WildBench": 0.854,
723
- "Omni-MATH": 0.471
724
- }
725
- },
726
- {
727
- "model_id": "openai/gpt-4.1-mini-2025-04-14",
728
- "name": "GPT-4.1 mini 2025-04-14",
729
- "developer": "OpenAI",
730
- "scores": {
731
- "Mean score": 0.726,
732
- "MMLU-Pro": 0.783,
733
- "GPQA": 0.614,
734
- "IFEval": 0.904,
735
- "WildBench": 0.838,
736
- "Omni-MATH": 0.491
737
- }
738
- },
739
- {
740
- "model_id": "openai/gpt-4.1-nano-2025-04-14",
741
- "name": "GPT-4.1 nano 2025-04-14",
742
- "developer": "OpenAI",
743
- "scores": {
744
- "Mean score": 0.616,
745
- "MMLU-Pro": 0.55,
746
- "GPQA": 0.507,
747
- "IFEval": 0.843,
748
- "WildBench": 0.811,
749
- "Omni-MATH": 0.367
750
- }
751
- },
752
- {
753
- "model_id": "openai/gpt-4o-2024-11-20",
754
- "name": "GPT-4o 2024-11-20",
755
- "developer": "OpenAI",
756
- "scores": {
757
- "Mean score": 0.634,
758
- "MMLU-Pro": 0.713,
759
- "GPQA": 0.52,
760
- "IFEval": 0.817,
761
- "WildBench": 0.828,
762
- "Omni-MATH": 0.293
763
- }
764
- },
765
- {
766
- "model_id": "openai/gpt-4o-mini-2024-07-18",
767
- "name": "GPT-4o mini 2024-07-18",
768
- "developer": "OpenAI",
769
- "scores": {
770
- "Mean score": 0.565,
771
- "MMLU-Pro": 0.603,
772
- "GPQA": 0.368,
773
- "IFEval": 0.782,
774
- "WildBench": 0.791,
775
- "Omni-MATH": 0.28
776
- }
777
- },
778
- {
779
- "model_id": "openai/gpt-5-2025-08-07",
780
- "name": "gpt-5-2025-08-07",
781
- "developer": "OpenAI",
782
- "scores": {
783
- "Mean score": 0.807,
784
- "MMLU-Pro": 0.863,
785
- "GPQA": 0.791,
786
- "IFEval": 0.875,
787
- "WildBench": 0.857,
788
- "Omni-MATH": 0.647
789
- }
790
- },
791
- {
792
- "model_id": "openai/gpt-5-mini-2025-08-07",
793
- "name": "GPT-5 mini 2025-08-07",
794
- "developer": "OpenAI",
795
- "scores": {
796
- "Mean score": 0.819,
797
- "MMLU-Pro": 0.835,
798
- "GPQA": 0.756,
799
- "IFEval": 0.927,
800
- "WildBench": 0.855,
801
- "Omni-MATH": 0.722
802
- }
803
- },
804
- {
805
- "model_id": "openai/gpt-5-nano-2025-08-07",
806
- "name": "GPT-5 nano 2025-08-07",
807
- "developer": "OpenAI",
808
- "scores": {
809
- "Mean score": 0.748,
810
- "MMLU-Pro": 0.778,
811
- "GPQA": 0.679,
812
- "IFEval": 0.932,
813
- "WildBench": 0.806,
814
- "Omni-MATH": 0.547
815
- }
816
- },
817
- {
818
- "model_id": "openai/gpt-oss-120b",
819
- "name": "GPT-OSS-120B",
820
- "developer": "OpenAI",
821
- "scores": {
822
- "Mean score": 0.77,
823
- "MMLU-Pro": 0.795,
824
- "GPQA": 0.684,
825
- "IFEval": 0.836,
826
- "WildBench": 0.845,
827
- "Omni-MATH": 0.688
828
- }
829
- },
830
- {
831
- "model_id": "openai/gpt-oss-20b",
832
- "name": "GPT-OSS-20B",
833
- "developer": "OpenAI",
834
- "scores": {
835
- "Mean score": 0.674,
836
- "MMLU-Pro": 0.74,
837
- "GPQA": 0.594,
838
- "IFEval": 0.732,
839
- "WildBench": 0.737,
840
- "Omni-MATH": 0.565
841
- }
842
- },
843
- {
844
- "model_id": "openai/o3-2025-04-16",
845
- "name": "o3-2025-04-16",
846
- "developer": "OpenAI",
847
- "scores": {
848
- "Mean score": 0.811,
849
- "MMLU-Pro": 0.859,
850
- "GPQA": 0.753,
851
- "IFEval": 0.869,
852
- "WildBench": 0.861,
853
- "Omni-MATH": 0.714
854
- }
855
- },
856
- {
857
- "model_id": "openai/o4-mini-2025-04-16",
858
- "name": "o4-mini-2025-04-16",
859
- "developer": "OpenAI",
860
- "scores": {
861
- "Mean score": 0.812,
862
- "MMLU-Pro": 0.82,
863
- "GPQA": 0.735,
864
- "IFEval": 0.929,
865
- "WildBench": 0.854,
866
- "Omni-MATH": 0.72
867
- }
868
- },
869
- {
870
- "model_id": "qwen/qwen2.5-72b-instruct-turbo",
871
- "name": "Qwen2.5 Instruct Turbo 72B",
872
- "developer": "qwen",
873
- "scores": {
874
- "Mean score": 0.599,
875
- "MMLU-Pro": 0.631,
876
- "GPQA": 0.426,
877
- "IFEval": 0.806,
878
- "WildBench": 0.802,
879
- "Omni-MATH": 0.33
880
- }
881
- },
882
- {
883
- "model_id": "qwen/qwen2.5-7b-instruct-turbo",
884
- "name": "Qwen2.5 Instruct Turbo 7B",
885
- "developer": "qwen",
886
- "scores": {
887
- "Mean score": 0.529,
888
- "MMLU-Pro": 0.539,
889
- "GPQA": 0.341,
890
- "IFEval": 0.741,
891
- "WildBench": 0.731,
892
- "Omni-MATH": 0.294
893
- }
894
- },
895
- {
896
- "model_id": "qwen/qwen3-235b-a22b-fp8-tput",
897
- "name": "Qwen3 235B A22B FP8 Throughput",
898
- "developer": "qwen",
899
- "scores": {
900
- "Mean score": 0.726,
901
- "MMLU-Pro": 0.817,
902
- "GPQA": 0.623,
903
- "IFEval": 0.816,
904
- "WildBench": 0.828,
905
- "Omni-MATH": 0.548
906
- }
907
- },
908
- {
909
- "model_id": "qwen/qwen3-235b-a22b-instruct-2507-fp8",
910
- "name": "Qwen3 235B A22B Instruct 2507 FP8",
911
- "developer": "qwen",
912
- "scores": {
913
- "Mean score": 0.798,
914
- "MMLU-Pro": 0.844,
915
- "GPQA": 0.726,
916
- "IFEval": 0.835,
917
- "WildBench": 0.866,
918
- "Omni-MATH": 0.718
919
- }
920
- },
921
- {
922
- "model_id": "writer/palmyra-fin",
923
- "name": "Palmyra Fin",
924
- "developer": "writer",
925
- "scores": {
926
- "Mean score": 0.577,
927
- "MMLU-Pro": 0.591,
928
- "GPQA": 0.422,
929
- "IFEval": 0.793,
930
- "WildBench": 0.783,
931
- "Omni-MATH": 0.295
932
- }
933
- },
934
- {
935
- "model_id": "writer/palmyra-med",
936
- "name": "Palmyra Med",
937
- "developer": "writer",
938
- "scores": {
939
- "Mean score": 0.476,
940
- "MMLU-Pro": 0.411,
941
- "GPQA": 0.368,
942
- "IFEval": 0.767,
943
- "WildBench": 0.676,
944
- "Omni-MATH": 0.156
945
- }
946
- },
947
- {
948
- "model_id": "writer/palmyra-x-004",
949
- "name": "Palmyra-X-004",
950
- "developer": "writer",
951
- "scores": {
952
- "Mean score": 0.609,
953
- "MMLU-Pro": 0.657,
954
- "GPQA": 0.395,
955
- "IFEval": 0.872,
956
- "WildBench": 0.802,
957
- "Omni-MATH": 0.32
958
- }
959
- },
960
- {
961
- "model_id": "writer/palmyra-x5",
962
- "name": "Palmyra X5",
963
- "developer": "writer",
964
- "scores": {
965
- "Mean score": 0.696,
966
- "MMLU-Pro": 0.804,
967
- "GPQA": 0.661,
968
- "IFEval": 0.823,
969
- "WildBench": 0.78,
970
- "Omni-MATH": 0.414
971
- }
972
- },
973
- {
974
- "model_id": "xai/grok-3-beta",
975
- "name": "Grok 3 Beta",
976
- "developer": "xAI",
977
- "scores": {
978
- "Mean score": 0.727,
979
- "MMLU-Pro": 0.788,
980
- "GPQA": 0.65,
981
- "IFEval": 0.884,
982
- "WildBench": 0.849,
983
- "Omni-MATH": 0.464
984
- }
985
- },
986
- {
987
- "model_id": "xai/grok-3-mini-beta",
988
- "name": "Grok 3 mini Beta",
989
- "developer": "xAI",
990
- "scores": {
991
- "Mean score": 0.679,
992
- "MMLU-Pro": 0.799,
993
- "GPQA": 0.675,
994
- "IFEval": 0.951,
995
- "WildBench": 0.651,
996
- "Omni-MATH": 0.318
997
- }
998
- },
999
- {
1000
- "model_id": "xai/grok-4-0709",
1001
- "name": "grok-4-0709",
1002
- "developer": "xAI",
1003
- "scores": {
1004
- "Mean score": 0.785,
1005
- "MMLU-Pro": 0.851,
1006
- "GPQA": 0.726,
1007
- "IFEval": 0.949,
1008
- "WildBench": 0.797,
1009
- "Omni-MATH": 0.603
1010
- }
1011
- },
1012
- {
1013
- "model_id": "zai-org/glm-4.5-air-fp8",
1014
- "name": "GLM-4.5-Air-FP8",
1015
- "developer": "zai-org",
1016
- "scores": {
1017
- "Mean score": 0.67,
1018
- "MMLU-Pro": 0.762,
1019
- "GPQA": 0.594,
1020
- "IFEval": 0.812,
1021
- "WildBench": 0.789,
1022
- "Omni-MATH": 0.391
1023
- }
1024
- }
1025
- ]
1026
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/helm_classic.json DELETED
@@ -1,2030 +0,0 @@
1
- {
2
- "benchmark_cards": {
3
- "BoolQ": {
4
- "benchmark_details": {
5
- "name": "BoolQ",
6
- "overview": "BoolQ is a benchmark that measures a model's ability to answer naturally occurring yes/no questions, framed as a reading comprehension task. The questions are generated in unprompted and unconstrained settings, often querying complex, non-factoid information and requiring difficult entailment-like inference. The dataset consists of a single task: answering yes/no questions given a supporting passage.",
7
- "data_type": "text",
8
- "domains": [
9
- "natural language understanding",
10
- "reading comprehension",
11
- "natural language inference"
12
- ],
13
- "languages": [
14
- "English"
15
- ],
16
- "similar_benchmarks": [
17
- "MultiNLI",
18
- "SNLI",
19
- "QNLI",
20
- "SQuAD 2.0",
21
- "Natural Questions (NQ)",
22
- "QQP",
23
- "MS MARCO",
24
- "RACE",
25
- "bAbI stories"
26
- ],
27
- "resources": [
28
- "https://arxiv.org/abs/1905.10044",
29
- "https://huggingface.co/datasets/google/boolq",
30
- "https://goo.gl/boolq",
31
- "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json"
32
- ]
33
- },
34
- "purpose_and_intended_users": {
35
- "goal": "To test models on their ability to answer naturally occurring yes/no questions, which are challenging and require complex inferential abilities beyond surface-level reasoning.",
36
- "audience": [
37
- "Researchers in natural language understanding and reading comprehension"
38
- ],
39
- "tasks": [
40
- "Yes/no question answering",
41
- "Text-pair classification"
42
- ],
43
- "limitations": "Annotation involved some errors and ambiguous cases. The use of singly-annotated examples is a trade-off for dataset size. Potential concerns about annotation artifacts are acknowledged.",
44
- "out_of_scope_uses": "The paper does not explicitly state what the benchmark is not designed for."
45
- },
46
- "data": {
47
- "source": "The data consists of naturally occurring yes/no questions authored by people who were not prompted to write specific question types and did not know the answers. The passages are excerpts from sources like Wikipedia.",
48
- "size": "15,942 examples total, with 9,427 in the train split and 3,270 in the validation split. The dataset size category is between 10,000 and 100,000 examples.",
49
- "format": "parquet",
50
- "annotation": "Questions were answered by human annotators. A quality check on a subset showed the main annotation process achieved 90% accuracy against a gold-standard set labeled by three authors. The training, development, and test sets use singly-annotated examples."
51
- },
52
- "methodology": {
53
- "methods": [
54
- "Models are evaluated by fine-tuning on the BoolQ training set, potentially after transfer learning from other datasets or unsupervised pre-training. Zero-shot or direct use of pre-trained models without fine-tuning did not outperform the majority baseline.",
55
- "The task requires providing a yes/no (boolean) answer to a question based on a given passage."
56
- ],
57
- "metrics": [
58
- "Accuracy"
59
- ],
60
- "calculation": "The overall score is the accuracy percentage on the test set.",
61
- "interpretation": "Higher accuracy indicates better performance. Human accuracy is 90%, and the majority baseline is approximately 62%.",
62
- "baseline_results": "Paper baselines: Majority baseline: 62.17% dev, 62.31% test; Recurrent model baseline: 69.6%; Best model (BERT large pre-trained on MultiNLI then fine-tuned on BoolQ): 80.4% accuracy; Human accuracy: 90%. EEE results: Anthropic-LM v4-s3 52B: 81.5%.",
63
- "validation": "Quality assurance involved author-led gold-standard annotation on a subset, showing 90% agreement. The development set was used for model selection, such as choosing the best model from five seeds based on its performance."
64
- },
65
- "ethical_and_legal_considerations": {
66
- "privacy_and_anonymity": "Not specified",
67
- "data_licensing": "cc-by-sa-3.0",
68
- "consent_procedures": "Not specified",
69
- "compliance_with_regulations": "Not specified"
70
- },
71
- "possible_risks": [
72
- {
73
- "category": "Over- or under-reliance",
74
- "description": [
75
- "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
76
- ],
77
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
78
- },
79
- {
80
- "category": "Unrepresentative data",
81
- "description": [
82
- "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
83
- ],
84
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
85
- },
86
- {
87
- "category": "Uncertain data provenance",
88
- "description": [
89
- "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation."
90
- ],
91
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html"
92
- },
93
- {
94
- "category": "Data bias",
95
- "description": [
96
- "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
97
- ],
98
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
99
- },
100
- {
101
- "category": "Lack of data transparency",
102
- "description": [
103
- "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
104
- ],
105
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
106
- }
107
- ],
108
- "flagged_fields": {},
109
- "missing_fields": [
110
- "ethical_and_legal_considerations.privacy_and_anonymity",
111
- "ethical_and_legal_considerations.consent_procedures",
112
- "ethical_and_legal_considerations.compliance_with_regulations"
113
- ],
114
- "card_info": {
115
- "created_at": "2026-03-17T15:08:51.830946",
116
- "llm": "deepseek-ai/DeepSeek-V3.2"
117
- }
118
- },
119
- "CNN/DailyMail": {
120
- "benchmark_details": {
121
- "name": "CNN/DailyMail",
122
- "overview": "CNN/DailyMail is a benchmark for evaluating abstractive and extractive summarization models using news articles. It contains over 300,000 unique articles written by journalists from CNN and the Daily Mail. The dataset was originally created for machine reading and question answering, but later versions were restructured specifically for summarization tasks.",
123
- "data_type": "text",
124
- "domains": [
125
- "summarization",
126
- "journalism",
127
- "news media"
128
- ],
129
- "languages": [
130
- "English"
131
- ],
132
- "similar_benchmarks": "No facts provided about similar benchmarks.",
133
- "resources": [
134
- "https://huggingface.co/datasets/abisee/cnn_dailymail",
135
- "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json"
136
- ]
137
- },
138
- "purpose_and_intended_users": {
139
- "goal": "To help develop models that can summarize long paragraphs of text into one or two sentences, aiding in the efficient presentation of information from large quantities of text.",
140
- "audience": [
141
- "NLP researchers",
142
- "Summarization model developers"
143
- ],
144
- "tasks": [
145
- "Summarization"
146
- ],
147
- "limitations": "News articles often place important information in the first third, which may affect summarization. A manual study found 25% of samples in an earlier version were difficult for humans due to ambiguity and coreference errors. Also, machine-generated summaries may differ in truth values from the original articles.",
148
- "out_of_scope_uses": "No facts provided about out-of-scope uses."
149
- },
150
- "data": {
151
- "source": "The dataset consists of news articles and highlight sentences written by journalists at CNN and the Daily Mail. The CNN articles were collected from April 2007 to April 2015, and the Daily Mail articles from June 2010 to April 2015, sourced from archives on the Wayback Machine.",
152
- "size": "Over 300,000 unique articles, with 287,113 training examples, 13,368 validation examples, and 11,490 test examples.",
153
- "format": "parquet",
154
- "annotation": "The dataset does not contain additional annotations. The highlights are the original summaries written by the article authors and are used as the target for summarization."
155
- },
156
- "methodology": {
157
- "methods": [
158
- "Models generate a summary for a given news article, which is then compared to the author-written highlights."
159
- ],
160
- "metrics": [
161
- "ROUGE-2"
162
- ],
163
- "calculation": "The ROUGE-2 score measures the overlap of bigrams between the generated summary and the reference highlights.",
164
- "interpretation": "Higher scores indicate better performance, as they reflect greater overlap with the reference summaries.",
165
- "baseline_results": "Paper baseline (Zhong et al., 2020): ROUGE-1 score of 44.41 for an extractive summarization model. Evaluation suite result (Anthropic-LM v4-s3 52B): ROUGE-2 score of 0.154.",
166
- "validation": "No facts provided about validation procedures."
167
- },
168
- "ethical_and_legal_considerations": {
169
- "privacy_and_anonymity": "The dataset (version 3.0.0) is not anonymized, meaning individuals' names are present in the text.",
170
- "data_licensing": "Apache License 2.0",
171
- "consent_procedures": "Not specified",
172
- "compliance_with_regulations": "Not specified"
173
- },
174
- "possible_risks": [
175
- {
176
- "category": "Over- or under-reliance",
177
- "description": [
178
- "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
179
- ],
180
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
181
- },
182
- {
183
- "category": "Unrepresentative data",
184
- "description": [
185
- "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
186
- ],
187
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
188
- },
189
- {
190
- "category": "Data bias",
191
- "description": [
192
- "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
193
- ],
194
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
195
- },
196
- {
197
- "category": "Data contamination",
198
- "description": [
199
- "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation."
200
- ],
201
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html"
202
- },
203
- {
204
- "category": "Lack of data transparency",
205
- "description": [
206
- "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
207
- ],
208
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
209
- }
210
- ],
211
- "flagged_fields": {},
212
- "missing_fields": [
213
- "ethical_and_legal_considerations.consent_procedures",
214
- "ethical_and_legal_considerations.compliance_with_regulations"
215
- ],
216
- "card_info": {
217
- "created_at": "2026-03-17T15:15:47.316103",
218
- "llm": "deepseek-ai/DeepSeek-V3.2"
219
- }
220
- },
221
- "CivilComments": {
222
- "benchmark_details": {
223
- "name": "CivilComments",
224
- "overview": "CivilComments is a benchmark designed to measure unintended identity-based bias in toxicity classification models. It uses a large, real-world dataset of online comments from the Civil Comments platform, extended with crowd-sourced annotations for toxicity and demographic identity references. This provides a nuanced evaluation of bias beyond synthetic datasets.",
225
- "data_type": "tabular, text",
226
- "domains": [
227
- "machine learning fairness",
228
- "bias measurement",
229
- "toxic comment classification",
230
- "text classification"
231
- ],
232
- "languages": [
233
- "English"
234
- ],
235
- "similar_benchmarks": "The paper does not name other specific benchmark datasets, only referencing prior work using synthetic test sets.",
236
- "resources": [
237
- "https://arxiv.org/abs/1903.04561",
238
- "https://huggingface.co/datasets/google/civil_comments",
239
- "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json"
240
- ]
241
- },
242
- "purpose_and_intended_users": {
243
- "goal": "To evaluate unintended identity-based bias in toxicity classification models using real data and nuanced metrics.",
244
- "audience": [
245
- "Machine learning researchers and practitioners working on fairness, bias measurement, and mitigation, particularly in toxic comment classification."
246
- ],
247
- "tasks": [
248
- "Binary toxicity classification (toxic vs. non-toxic)",
249
- "Analysis of performance across identity subgroups"
250
- ],
251
- "limitations": "The labeled set of identities is not comprehensive and does not provide universal coverage, representing a balance between coverage, annotator accuracy, and example count. The real-world data is potentially noisier than synthetic alternatives.",
252
- "out_of_scope_uses": [
253
- "Developing effective strategies for choosing optimal thresholds to minimize bias"
254
- ]
255
- },
256
- "data": {
257
- "source": "The data consists of online comments sourced from the Civil Comments platform, a commenting plugin for independent English-language news sites. The comments were publicly posted between 2015 and 2017.",
258
- "size": "The dataset contains approximately 1.8 million comments for training, with separate validation and test sets of approximately 97,320 examples each. All comments were labeled for toxicity, and a subset of 450,000 comments was additionally labeled for identity references.",
259
- "format": "parquet",
260
- "annotation": "Labeling was performed by crowd raters. Toxicity labels were applied using guidelines consistent with the Perspective API. For the identity-labeled subset, raters were shown comments and selected referenced identities (e.g., genders, races, ethnicities) from a provided list. Some comments for identity labeling were pre-selected by models to increase the frequency of identity content."
261
- },
262
- "methodology": {
263
- "methods": [
264
- "Models are evaluated by applying a suite of bias metrics to their predictions on the test set. The original paper demonstrates this using publicly accessible toxicity classifiers on the dataset."
265
- ],
266
- "metrics": [
267
- "Subgroup AUC",
268
- "BPSN AUC",
269
- "BNSP AUC",
270
- "Negative Average Equality Gap (AEG)",
271
- "Positive Average Equality Gap (AEG)"
272
- ],
273
- "calculation": "The evaluation calculates five metrics for each identity subgroup to provide a multi-faceted view of bias. There is no single aggregated overall score.",
274
- "interpretation": "For the AUC metrics (Subgroup, BPSN, BNSP), higher values indicate better separability (fewer mis-orderings). For the Average Equality Gaps (Negative and Positive), lower values indicate better separability (more similar score distributions).",
275
- "baseline_results": "Paper baselines: Results for TOXICITY@1 and TOXICITY@6 from the Perspective API are reported, showing their Subgroup AUC, BPSN AUC, BNSP AUC, Negative AEG, and Positive AEG on a synthetic dataset for the lowest performing 20 subgroups. They are also compared on short comments within the human-labeled dataset for specific identities. EEE results: Anthropic-LM v4-s3 52B scored 0.6100 on the CivilComments metric.",
276
- "validation": "The evaluation assumes the human-provided labels are reliable. The identity labeling set was designed to balance coverage, crowd rater accuracy, and ensure sufficient examples per identity for meaningful results."
277
- },
278
- "ethical_and_legal_considerations": {
279
- "privacy_and_anonymity": "The paper does not discuss how personally identifiable information (PII) in the online comments was handled or if data was anonymized.",
280
- "data_licensing": "Creative Commons Zero v1.0 Universal",
281
- "consent_procedures": "The paper does not describe compensation for crowdworkers or the specific platform used for annotation.",
282
- "compliance_with_regulations": "The paper does not mention IRB approval, GDPR compliance, or any other ethical review process."
283
- },
284
- "possible_risks": [
285
- {
286
- "category": "Unrepresentative data",
287
- "description": [
288
- "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
289
- ],
290
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
291
- },
292
- {
293
- "category": "Uncertain data provenance",
294
- "description": [
295
- "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation."
296
- ],
297
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html"
298
- },
299
- {
300
- "category": "Data bias",
301
- "description": [
302
- "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
303
- ],
304
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
305
- },
306
- {
307
- "category": "Lack of data transparency",
308
- "description": [
309
- "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
310
- ],
311
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
312
- },
313
- {
314
- "category": "Output bias",
315
- "description": [
316
- "Generated content might unfairly represent certain groups or individuals."
317
- ],
318
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/output-bias.html"
319
- }
320
- ],
321
- "flagged_fields": {},
322
- "missing_fields": [],
323
- "card_info": {
324
- "created_at": "2026-03-17T12:38:43.250822",
325
- "llm": "deepseek-ai/DeepSeek-V3.2"
326
- }
327
- },
328
- "HellaSwag": {
329
- "benchmark_details": {
330
- "name": "HellaSwag",
331
- "overview": "HellaSwag is a benchmark designed to measure commonsense natural language inference by testing a model's ability to select the most plausible follow-up event from four multiple-choice options. It is adversarially constructed to be challenging for state-of-the-art models, using a method called Adversarial Filtering to create difficult wrong answers that are obvious to humans but often misclassified by models.",
332
- "data_type": "text",
333
- "domains": [
334
- "commonsense reasoning",
335
- "natural language inference"
336
- ],
337
- "languages": [
338
- "English"
339
- ],
340
- "similar_benchmarks": [
341
- "SWAG",
342
- "SNLI"
343
- ],
344
- "resources": [
345
- "https://rowanzellers.com/hellaswag",
346
- "https://arxiv.org/abs/1905.07830",
347
- "https://huggingface.co/datasets/Rowan/hellaswag",
348
- "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json"
349
- ]
350
- },
351
- "purpose_and_intended_users": {
352
- "goal": "To create a challenge dataset that reveals the difficulty of commonsense inference for state-of-the-art models, demonstrating their lack of robustness and reliance on dataset biases rather than genuine reasoning. It aims to evaluate a model's ability to select the most plausible continuation of a given event description.",
353
- "audience": [
354
- "NLP researchers"
355
- ],
356
- "tasks": [
357
- "Four-way multiple-choice selection for event continuation",
358
- "Commonsense inference"
359
- ],
360
- "limitations": "The adversarial filtering process used to create the dataset, while effective at making it difficult for models, may also select examples where the ground truth answer is not the one preferred by human annotators, necessitating manual filtering to retain the best examples.",
361
- "out_of_scope_uses": [
362
- "Not specified"
363
- ]
364
- },
365
- "data": {
366
- "source": "Contexts are sourced from WikiHow instructional articles and ActivityNet video descriptions. Incorrect answer choices are generated by machines and then adversarially filtered.",
367
- "size": "The dataset contains 70,000 examples in total, with 5,001 in-domain validation examples and 5,000 zero-shot validation examples. The training set comprises 39,905 examples.",
368
- "format": "Parquet",
369
- "annotation": "Human crowd workers on Amazon Mechanical Turk validated the endings. They were presented with a context and six endings (one true, five machine-generated) and rated their plausibility. The process involved iterative filtering and replacement of unrealistic endings. Worker quality was ensured via an autograded test and fair pay. A gold standard check by three authors on a random sample showed 90% agreement with crowd annotations."
370
- },
371
- "methodology": {
372
- "methods": [
373
- "Models are evaluated via fine-tuning on the dataset.",
374
- "The benchmark also includes zero-shot evaluation on held-out categories."
375
- ],
376
- "metrics": [
377
- "HellaSwag accuracy"
378
- ],
379
- "calculation": "The overall score is the accuracy percentage on the full validation or test sets. Performance is also broken down by subsets, such as in-domain versus zero-shot and by data source.",
380
- "interpretation": "Higher scores indicate better performance. Human performance is over 95%, which is considered strong. Model performance below 50% is reported, indicating a struggle, with a gap of over 45% from human performance on in-domain data.",
381
- "baseline_results": "Paper baselines: BERT-Large achieves 47.3% accuracy overall. ESIM + ELMo gets 33.3% accuracy. A BERT-Base model with a frozen encoder and an added LSTM performs 4.3% worse than fine-tuned BERT-Base. Evaluation suite results: Anthropic-LM v4-s3 52B achieves 0.807 (80.7%) accuracy.",
382
- "validation": "Human validation involved giving five crowd workers the same multiple-choice task and combining their answers via majority vote to establish a human performance baseline. The adversarial filtering process used iterative human ratings to ensure wrong answers were implausible."
383
- },
384
- "ethical_and_legal_considerations": {
385
- "privacy_and_anonymity": "Not specified",
386
- "data_licensing": "Not specified",
387
- "consent_procedures": "Crowdworkers on Amazon Mechanical Turk participated voluntarily and were compensated, with pay described as fair. A qualification task was used to filter workers, and those who consistently preferred generated endings over real ones were disqualified.",
388
- "compliance_with_regulations": "Not specified"
389
- },
390
- "possible_risks": [
391
- {
392
- "category": "Over- or under-reliance",
393
- "description": [
394
- "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
395
- ],
396
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
397
- },
398
- {
399
- "category": "Unrepresentative data",
400
- "description": [
401
- "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
402
- ],
403
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
404
- },
405
- {
406
- "category": "Data bias",
407
- "description": [
408
- "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
409
- ],
410
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
411
- },
412
- {
413
- "category": "Lack of data transparency",
414
- "description": [
415
- "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
416
- ],
417
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
418
- },
419
- {
420
- "category": "Improper usage",
421
- "description": [
422
- "Improper usage occurs when a model is used for a purpose that it was not originally designed for."
423
- ],
424
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html"
425
- }
426
- ],
427
- "flagged_fields": {
428
- "baseline_results": "[Possible Hallucination], no supporting evidence found in source material"
429
- },
430
- "missing_fields": [
431
- "purpose_and_intended_users.out_of_scope_uses",
432
- "ethical_and_legal_considerations.privacy_and_anonymity",
433
- "ethical_and_legal_considerations.data_licensing",
434
- "ethical_and_legal_considerations.compliance_with_regulations"
435
- ],
436
- "card_info": {
437
- "created_at": "2026-03-17T15:47:07.561060",
438
- "llm": "deepseek-ai/DeepSeek-V3.2"
439
- }
440
- },
441
- "QuAC": {
442
- "benchmark_details": {
443
- "name": "QuAC",
444
- "overview": "QuAC (Question Answering in Context) is a benchmark that measures a model's ability to answer questions within an information-seeking dialogue. It contains 14,000 dialogues comprising 100,000 question-answer pairs. The dataset is distinctive because questions are often open-ended, context-dependent, unanswerable, or only meaningful within the dialog flow, presenting challenges not found in standard machine comprehension datasets.",
445
- "data_type": "text",
446
- "domains": [
447
- "question answering",
448
- "dialogue modeling",
449
- "text generation"
450
- ],
451
- "languages": [
452
- "English"
453
- ],
454
- "similar_benchmarks": [
455
- "SQuAD"
456
- ],
457
- "resources": [
458
- "http://quac.ai",
459
- "https://arxiv.org/abs/1808.07036",
460
- "https://huggingface.co/datasets/allenai/quac",
461
- "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json"
462
- ]
463
- },
464
- "purpose_and_intended_users": {
465
- "goal": "To enable models to learn from and participate in information-seeking dialog, handling context-dependent, elliptical, and sometimes unanswerable questions.",
466
- "audience": [
467
- "Not specified"
468
- ],
469
- "tasks": [
470
- "Extractive question answering",
471
- "Text generation",
472
- "Fill mask"
473
- ],
474
- "limitations": "Some questions have lower quality annotations; the dataset filters out the noisiest ~10% of annotations where human F1 is below 40. Questions can be open-ended, unanswerable, or only meaningful within the dialog context, posing inherent challenges.",
475
- "out_of_scope_uses": [
476
- "Not specified"
477
- ]
478
- },
479
- "data": {
480
- "source": "The data is crowdsourced via an interactive dialog between two crowd workers: one acting as a student asking questions to learn about a hidden Wikipedia text, and the other acting as a teacher who answers using short excerpts from that text. The source data comes from Wikipedia.",
481
- "size": "The dataset contains 98,407 question-answer pairs from 13,594 dialogs, based on 8,854 unique sections from 3,611 unique Wikipedia articles. The training set has 83,568 questions (11,567 dialogs), the validation set has 7,354 questions (1,000 dialogs), and the test set has 7,353 questions (1,002 dialogs). The dataset size is between 10,000 and 100,000 examples.",
482
- "format": "Each dialog is a sequence of question-answer pairs centered around a Wikipedia section. The teacher's response includes a text span, a 'yes/no' indication, a 'no answer' indication, and an encouragement for follow-up questions.",
483
- "annotation": "Questions are answered by a teacher selecting short excerpts (spans) from the Wikipedia text. The training set has one reference answer per question, while the validation and test sets each have five reference answers per question to improve evaluation reliability. For evaluation, questions with a human F1 score lower than 40 are not used, as manual inspection revealed lower quality below this threshold."
484
- },
485
- "methodology": {
486
- "methods": [
487
- "Models predict a text span to answer a question about a Wikipedia section, given a dialog history of previous questions and answers.",
488
- "The evaluation uses a reading comprehension architecture extended to model dialog context."
489
- ],
490
- "metrics": [
491
- "Word-level F1"
492
- ],
493
- "calculation": "Precision and recall are computed over overlapping words after removing stopwords. For 'no answer' questions, F1 is 1 if correctly predicted and 0 otherwise. The maximum F1 among all references is computed for each question.",
494
- "interpretation": "Higher F1 scores indicate better performance. The best model underperforms humans by 20 F1, indicating significant room for improvement.",
495
- "baseline_results": "Paper baselines: The best model underperforms humans by 20 F1, but specific model names and scores are not provided. EEE results: Anthropic-LM v4-s3 52B achieves an F1 score of 0.431.",
496
- "validation": "Quality assurance includes using multiple references for development and test questions, filtering out questions with low human F1 scores, and manual inspection of low-quality annotations."
497
- },
498
- "ethical_and_legal_considerations": {
499
- "privacy_and_anonymity": "Not specified",
500
- "data_licensing": "MIT License",
501
- "consent_procedures": "Dialogs were created by two crowd workers, but the specific compensation or platform details are not provided in the paper.",
502
- "compliance_with_regulations": "Not specified"
503
- },
504
- "possible_risks": [
505
- {
506
- "category": "Over- or under-reliance",
507
- "description": [
508
- "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
509
- ],
510
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
511
- },
512
- {
513
- "category": "Unrepresentative data",
514
- "description": [
515
- "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
516
- ],
517
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
518
- },
519
- {
520
- "category": "Uncertain data provenance",
521
- "description": [
522
- "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation."
523
- ],
524
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html"
525
- },
526
- {
527
- "category": "Data bias",
528
- "description": [
529
- "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
530
- ],
531
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
532
- },
533
- {
534
- "category": "Lack of data transparency",
535
- "description": [
536
- "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
537
- ],
538
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
539
- }
540
- ],
541
- "flagged_fields": {},
542
- "missing_fields": [
543
- "purpose_and_intended_users.audience",
544
- "purpose_and_intended_users.out_of_scope_uses",
545
- "ethical_and_legal_considerations.privacy_and_anonymity",
546
- "ethical_and_legal_considerations.compliance_with_regulations"
547
- ],
548
- "card_info": {
549
- "created_at": "2026-03-17T13:45:24.009083",
550
- "llm": "deepseek-ai/DeepSeek-V3.2"
551
- }
552
- }
553
- },
554
- "models": [
555
- {
556
- "model_id": "Anthropic-LM-v4-s3-52B",
557
- "name": "Anthropic-LM v4-s3 52B",
558
- "developer": "unknown",
559
- "scores": {
560
- "Mean win rate": 0.78,
561
- "MMLU": 0.481,
562
- "BoolQ": 0.815,
563
- "NarrativeQA": 0.728,
564
- "NaturalQuestions (open-book)": 0.686,
565
- "QuAC": 0.431,
566
- "HellaSwag": 0.807,
567
- "OpenbookQA": 0.558,
568
- "TruthfulQA": 0.368,
569
- "MS MARCO (TREC)": -1,
570
- "CNN/DailyMail": 0.154,
571
- "XSUM": 0.134,
572
- "IMDB": 0.934,
573
- "CivilComments": 0.61,
574
- "RAFT": 0.699
575
- }
576
- },
577
- {
578
- "model_id": "EleutherAI/pythia-12b",
579
- "name": "Pythia 12B",
580
- "developer": "EleutherAI",
581
- "scores": {
582
- "Mean win rate": 0.257,
583
- "MMLU": 0.274,
584
- "BoolQ": 0.662,
585
- "NarrativeQA": 0.596,
586
- "NaturalQuestions (open-book)": 0.581,
587
- "QuAC": 0.313,
588
- "HellaSwag": -1,
589
- "OpenbookQA": -1,
590
- "TruthfulQA": 0.177,
591
- "MS MARCO (TREC)": -1,
592
- "CNN/DailyMail": -1,
593
- "XSUM": -1,
594
- "IMDB": 0.931,
595
- "CivilComments": 0.531,
596
- "RAFT": 0.514
597
- }
598
- },
599
- {
600
- "model_id": "EleutherAI/pythia-6.9b",
601
- "name": "Pythia 6.9B",
602
- "developer": "EleutherAI",
603
- "scores": {
604
- "Mean win rate": 0.196,
605
- "MMLU": 0.236,
606
- "BoolQ": 0.631,
607
- "NarrativeQA": 0.528,
608
- "NaturalQuestions (open-book)": 0.539,
609
- "QuAC": 0.296,
610
- "HellaSwag": -1,
611
- "OpenbookQA": -1,
612
- "TruthfulQA": 0.213,
613
- "MS MARCO (TREC)": -1,
614
- "CNN/DailyMail": -1,
615
- "XSUM": -1,
616
- "IMDB": 0.928,
617
- "CivilComments": 0.511,
618
- "RAFT": 0.502
619
- }
620
- },
621
- {
622
- "model_id": "ai21/J1-Grande-v1-17B",
623
- "name": "J1-Grande v1 17B",
624
- "developer": "ai21",
625
- "scores": {
626
- "Mean win rate": 0.433,
627
- "MMLU": 0.27,
628
- "BoolQ": 0.722,
629
- "NarrativeQA": 0.672,
630
- "NaturalQuestions (open-book)": 0.578,
631
- "QuAC": 0.362,
632
- "HellaSwag": 0.739,
633
- "OpenbookQA": 0.52,
634
- "TruthfulQA": 0.193,
635
- "MS MARCO (TREC)": 0.341,
636
- "CNN/DailyMail": 0.143,
637
- "XSUM": 0.122,
638
- "IMDB": 0.953,
639
- "CivilComments": 0.529,
640
- "RAFT": 0.658
641
- }
642
- },
643
- {
644
- "model_id": "ai21/J1-Grande-v2-beta-17B",
645
- "name": "J1-Grande v2 beta 17B",
646
- "developer": "ai21",
647
- "scores": {
648
- "Mean win rate": 0.706,
649
- "MMLU": 0.445,
650
- "BoolQ": 0.812,
651
- "NarrativeQA": 0.725,
652
- "NaturalQuestions (open-book)": 0.625,
653
- "QuAC": 0.392,
654
- "HellaSwag": 0.764,
655
- "OpenbookQA": 0.56,
656
- "TruthfulQA": 0.306,
657
- "MS MARCO (TREC)": 0.46,
658
- "CNN/DailyMail": 0.146,
659
- "XSUM": 0.152,
660
- "IMDB": 0.957,
661
- "CivilComments": 0.546,
662
- "RAFT": 0.679
663
- }
664
- },
665
- {
666
- "model_id": "ai21/J1-Jumbo-v1-178B",
667
- "name": "J1-Jumbo v1 178B",
668
- "developer": "ai21",
669
- "scores": {
670
- "Mean win rate": 0.517,
671
- "MMLU": 0.259,
672
- "BoolQ": 0.776,
673
- "NarrativeQA": 0.695,
674
- "NaturalQuestions (open-book)": 0.595,
675
- "QuAC": 0.358,
676
- "HellaSwag": 0.765,
677
- "OpenbookQA": 0.534,
678
- "TruthfulQA": 0.175,
679
- "MS MARCO (TREC)": 0.363,
680
- "CNN/DailyMail": 0.144,
681
- "XSUM": 0.129,
682
- "IMDB": 0.943,
683
- "CivilComments": 0.553,
684
- "RAFT": 0.681
685
- }
686
- },
687
- {
688
- "model_id": "ai21/J1-Large-v1-7.5B",
689
- "name": "J1-Large v1 7.5B",
690
- "developer": "ai21",
691
- "scores": {
692
- "Mean win rate": 0.285,
693
- "MMLU": 0.241,
694
- "BoolQ": 0.683,
695
- "NarrativeQA": 0.623,
696
- "NaturalQuestions (open-book)": 0.532,
697
- "QuAC": 0.328,
698
- "HellaSwag": 0.7,
699
- "OpenbookQA": 0.514,
700
- "TruthfulQA": 0.197,
701
- "MS MARCO (TREC)": 0.292,
702
- "CNN/DailyMail": 0.134,
703
- "XSUM": 0.102,
704
- "IMDB": 0.956,
705
- "CivilComments": 0.532,
706
- "RAFT": 0.545
707
- }
708
- },
709
- {
710
- "model_id": "ai21/Jurassic-2-Grande-17B",
711
- "name": "Jurassic-2 Grande 17B",
712
- "developer": "ai21",
713
- "scores": {
714
- "Mean win rate": 0.743,
715
- "MMLU": 0.475,
716
- "BoolQ": 0.826,
717
- "NarrativeQA": 0.737,
718
- "NaturalQuestions (open-book)": 0.639,
719
- "QuAC": 0.418,
720
- "HellaSwag": 0.781,
721
- "OpenbookQA": 0.542,
722
- "TruthfulQA": 0.348,
723
- "MS MARCO (TREC)": 0.514,
724
- "CNN/DailyMail": 0.144,
725
- "XSUM": 0.167,
726
- "IMDB": 0.938,
727
- "CivilComments": 0.547,
728
- "RAFT": 0.712
729
- }
730
- },
731
- {
732
- "model_id": "ai21/Jurassic-2-Jumbo-178B",
733
- "name": "Jurassic-2 Jumbo 178B",
734
- "developer": "ai21",
735
- "scores": {
736
- "Mean win rate": 0.824,
737
- "MMLU": 0.48,
738
- "BoolQ": 0.829,
739
- "NarrativeQA": 0.733,
740
- "NaturalQuestions (open-book)": 0.669,
741
- "QuAC": 0.435,
742
- "HellaSwag": 0.788,
743
- "OpenbookQA": 0.558,
744
- "TruthfulQA": 0.437,
745
- "MS MARCO (TREC)": 0.661,
746
- "CNN/DailyMail": 0.149,
747
- "XSUM": 0.182,
748
- "IMDB": 0.938,
749
- "CivilComments": 0.57,
750
- "RAFT": 0.746
751
- }
752
- },
753
- {
754
- "model_id": "ai21/Jurassic-2-Large-7.5B",
755
- "name": "Jurassic-2 Large 7.5B",
756
- "developer": "ai21",
757
- "scores": {
758
- "Mean win rate": 0.553,
759
- "MMLU": 0.339,
760
- "BoolQ": 0.742,
761
- "NarrativeQA": -1,
762
- "NaturalQuestions (open-book)": 0.589,
763
- "QuAC": -1,
764
- "HellaSwag": 0.729,
765
- "OpenbookQA": 0.53,
766
- "TruthfulQA": 0.245,
767
- "MS MARCO (TREC)": 0.464,
768
- "CNN/DailyMail": 0.136,
769
- "XSUM": 0.142,
770
- "IMDB": 0.956,
771
- "CivilComments": 0.57,
772
- "RAFT": 0.622
773
- }
774
- },
775
- {
776
- "model_id": "aleph-alpha/Luminous-Base-13B",
777
- "name": "Luminous Base 13B",
778
- "developer": "aleph-alpha",
779
- "scores": {
780
- "Mean win rate": 0.315,
781
- "MMLU": 0.27,
782
- "BoolQ": 0.719,
783
- "NarrativeQA": 0.605,
784
- "NaturalQuestions (open-book)": 0.568,
785
- "QuAC": 0.334,
786
- "HellaSwag": -1,
787
- "OpenbookQA": -1,
788
- "TruthfulQA": 0.182,
789
- "MS MARCO (TREC)": -1,
790
- "CNN/DailyMail": 0.11,
791
- "XSUM": 0.105,
792
- "IMDB": 0.939,
793
- "CivilComments": 0.544,
794
- "RAFT": 0.473
795
- }
796
- },
797
- {
798
- "model_id": "aleph-alpha/Luminous-Extended-30B",
799
- "name": "Luminous Extended 30B",
800
- "developer": "aleph-alpha",
801
- "scores": {
802
- "Mean win rate": 0.485,
803
- "MMLU": 0.321,
804
- "BoolQ": 0.767,
805
- "NarrativeQA": 0.665,
806
- "NaturalQuestions (open-book)": 0.609,
807
- "QuAC": 0.349,
808
- "HellaSwag": -1,
809
- "OpenbookQA": -1,
810
- "TruthfulQA": 0.221,
811
- "MS MARCO (TREC)": -1,
812
- "CNN/DailyMail": 0.139,
813
- "XSUM": 0.124,
814
- "IMDB": 0.947,
815
- "CivilComments": 0.524,
816
- "RAFT": 0.523
817
- }
818
- },
819
- {
820
- "model_id": "aleph-alpha/Luminous-Supreme-70B",
821
- "name": "Luminous Supreme 70B",
822
- "developer": "aleph-alpha",
823
- "scores": {
824
- "Mean win rate": 0.662,
825
- "MMLU": 0.38,
826
- "BoolQ": 0.775,
827
- "NarrativeQA": 0.711,
828
- "NaturalQuestions (open-book)": 0.649,
829
- "QuAC": 0.37,
830
- "HellaSwag": -1,
831
- "OpenbookQA": -1,
832
- "TruthfulQA": 0.222,
833
- "MS MARCO (TREC)": -1,
834
- "CNN/DailyMail": 0.15,
835
- "XSUM": 0.136,
836
- "IMDB": 0.959,
837
- "CivilComments": 0.562,
838
- "RAFT": 0.653
839
- }
840
- },
841
- {
842
- "model_id": "bigscience/BLOOM-176B",
843
- "name": "BLOOM 176B",
844
- "developer": "bigscience",
845
- "scores": {
846
- "Mean win rate": 0.446,
847
- "MMLU": 0.299,
848
- "BoolQ": 0.704,
849
- "NarrativeQA": 0.662,
850
- "NaturalQuestions (open-book)": 0.621,
851
- "QuAC": 0.361,
852
- "HellaSwag": 0.744,
853
- "OpenbookQA": 0.534,
854
- "TruthfulQA": 0.205,
855
- "MS MARCO (TREC)": 0.386,
856
- "CNN/DailyMail": 0.08,
857
- "XSUM": 0.03,
858
- "IMDB": 0.945,
859
- "CivilComments": 0.62,
860
- "RAFT": 0.592
861
- }
862
- },
863
- {
864
- "model_id": "bigscience/T0pp-11B",
865
- "name": "T0pp 11B",
866
- "developer": "bigscience",
867
- "scores": {
868
- "Mean win rate": 0.197,
869
- "MMLU": 0.407,
870
- "BoolQ": 0,
871
- "NarrativeQA": 0.151,
872
- "NaturalQuestions (open-book)": 0.19,
873
- "QuAC": 0.121,
874
- "HellaSwag": -1,
875
- "OpenbookQA": -1,
876
- "TruthfulQA": 0.377,
877
- "MS MARCO (TREC)": -1,
878
- "CNN/DailyMail": 0.122,
879
- "XSUM": 0.09,
880
- "IMDB": 0.207,
881
- "CivilComments": 0.234,
882
- "RAFT": 0.118
883
- }
884
- },
885
- {
886
- "model_id": "cohere/Cohere-Command-beta-52.4B",
887
- "name": "Cohere Command beta 52.4B",
888
- "developer": "cohere",
889
- "scores": {
890
- "Mean win rate": 0.874,
891
- "MMLU": 0.452,
892
- "BoolQ": 0.856,
893
- "NarrativeQA": 0.752,
894
- "NaturalQuestions (open-book)": 0.76,
895
- "QuAC": 0.432,
896
- "HellaSwag": 0.811,
897
- "OpenbookQA": 0.582,
898
- "TruthfulQA": 0.269,
899
- "MS MARCO (TREC)": 0.762,
900
- "CNN/DailyMail": 0.161,
901
- "XSUM": 0.152,
902
- "IMDB": 0.96,
903
- "CivilComments": 0.601,
904
- "RAFT": 0.667
905
- }
906
- },
907
- {
908
- "model_id": "cohere/Cohere-Command-beta-6.1B",
909
- "name": "Cohere Command beta 6.1B",
910
- "developer": "cohere",
911
- "scores": {
912
- "Mean win rate": 0.675,
913
- "MMLU": 0.406,
914
- "BoolQ": 0.798,
915
- "NarrativeQA": 0.709,
916
- "NaturalQuestions (open-book)": 0.717,
917
- "QuAC": 0.375,
918
- "HellaSwag": 0.752,
919
- "OpenbookQA": 0.55,
920
- "TruthfulQA": 0.203,
921
- "MS MARCO (TREC)": 0.709,
922
- "CNN/DailyMail": 0.153,
923
- "XSUM": 0.122,
924
- "IMDB": 0.961,
925
- "CivilComments": 0.54,
926
- "RAFT": 0.634
927
- }
928
- },
929
- {
930
- "model_id": "cohere/Cohere-large-v20220720-13.1B",
931
- "name": "Cohere large v20220720 13.1B",
932
- "developer": "cohere",
933
- "scores": {
934
- "Mean win rate": 0.372,
935
- "MMLU": 0.324,
936
- "BoolQ": 0.725,
937
- "NarrativeQA": 0.625,
938
- "NaturalQuestions (open-book)": 0.573,
939
- "QuAC": 0.338,
940
- "HellaSwag": 0.736,
941
- "OpenbookQA": 0.542,
942
- "TruthfulQA": 0.181,
943
- "MS MARCO (TREC)": 0.33,
944
- "CNN/DailyMail": 0.126,
945
- "XSUM": 0.108,
946
- "IMDB": 0.933,
947
- "CivilComments": 0.507,
948
- "RAFT": 0.596
949
- }
950
- },
951
- {
952
- "model_id": "cohere/Cohere-medium-v20220720-6.1B",
953
- "name": "Cohere medium v20220720 6.1B",
954
- "developer": "cohere",
955
- "scores": {
956
- "Mean win rate": 0.23,
957
- "MMLU": 0.279,
958
- "BoolQ": 0.659,
959
- "NarrativeQA": 0.559,
960
- "NaturalQuestions (open-book)": 0.504,
961
- "QuAC": 0.279,
962
- "HellaSwag": 0.706,
963
- "OpenbookQA": 0.496,
964
- "TruthfulQA": 0.19,
965
- "MS MARCO (TREC)": 0.374,
966
- "CNN/DailyMail": 0.077,
967
- "XSUM": 0.087,
968
- "IMDB": 0.935,
969
- "CivilComments": 0.504,
970
- "RAFT": 0.52
971
- }
972
- },
973
- {
974
- "model_id": "cohere/Cohere-medium-v20221108-6.1B",
975
- "name": "Cohere medium v20221108 6.1B",
976
- "developer": "cohere",
977
- "scores": {
978
- "Mean win rate": 0.312,
979
- "MMLU": 0.254,
980
- "BoolQ": 0.7,
981
- "NarrativeQA": 0.61,
982
- "NaturalQuestions (open-book)": 0.517,
983
- "QuAC": 0.314,
984
- "HellaSwag": 0.726,
985
- "OpenbookQA": 0.538,
986
- "TruthfulQA": 0.215,
987
- "MS MARCO (TREC)": 0.373,
988
- "CNN/DailyMail": 0.121,
989
- "XSUM": 0.099,
990
- "IMDB": 0.935,
991
- "CivilComments": 0.5,
992
- "RAFT": 0.591
993
- }
994
- },
995
- {
996
- "model_id": "cohere/Cohere-small-v20220720-410M",
997
- "name": "Cohere small v20220720 410M",
998
- "developer": "cohere",
999
- "scores": {
1000
- "Mean win rate": 0.109,
1001
- "MMLU": 0.264,
1002
- "BoolQ": 0.457,
1003
- "NarrativeQA": 0.294,
1004
- "NaturalQuestions (open-book)": 0.309,
1005
- "QuAC": 0.219,
1006
- "HellaSwag": 0.483,
1007
- "OpenbookQA": 0.348,
1008
- "TruthfulQA": 0.217,
1009
- "MS MARCO (TREC)": 0.304,
1010
- "CNN/DailyMail": 0.063,
1011
- "XSUM": 0.033,
1012
- "IMDB": 0.578,
1013
- "CivilComments": 0.501,
1014
- "RAFT": 0.492
1015
- }
1016
- },
1017
- {
1018
- "model_id": "cohere/Cohere-xlarge-v20220609-52.4B",
1019
- "name": "Cohere xlarge v20220609 52.4B",
1020
- "developer": "cohere",
1021
- "scores": {
1022
- "Mean win rate": 0.56,
1023
- "MMLU": 0.353,
1024
- "BoolQ": 0.718,
1025
- "NarrativeQA": 0.65,
1026
- "NaturalQuestions (open-book)": 0.595,
1027
- "QuAC": 0.361,
1028
- "HellaSwag": 0.811,
1029
- "OpenbookQA": 0.55,
1030
- "TruthfulQA": 0.198,
1031
- "MS MARCO (TREC)": 0.459,
1032
- "CNN/DailyMail": 0.144,
1033
- "XSUM": 0.129,
1034
- "IMDB": 0.956,
1035
- "CivilComments": 0.532,
1036
- "RAFT": 0.633
1037
- }
1038
- },
1039
- {
1040
- "model_id": "cohere/Cohere-xlarge-v20221108-52.4B",
1041
- "name": "Cohere xlarge v20221108 52.4B",
1042
- "developer": "cohere",
1043
- "scores": {
1044
- "Mean win rate": 0.664,
1045
- "MMLU": 0.382,
1046
- "BoolQ": 0.762,
1047
- "NarrativeQA": 0.672,
1048
- "NaturalQuestions (open-book)": 0.628,
1049
- "QuAC": 0.374,
1050
- "HellaSwag": 0.81,
1051
- "OpenbookQA": 0.588,
1052
- "TruthfulQA": 0.169,
1053
- "MS MARCO (TREC)": 0.55,
1054
- "CNN/DailyMail": 0.153,
1055
- "XSUM": 0.153,
1056
- "IMDB": 0.956,
1057
- "CivilComments": 0.524,
1058
- "RAFT": 0.624
1059
- }
1060
- },
1061
- {
1062
- "model_id": "google/Palmyra-X-43B",
1063
- "name": "Palmyra X 43B",
1064
- "developer": "Google",
1065
- "scores": {
1066
- "Mean win rate": 0.732,
1067
- "MMLU": 0.609,
1068
- "BoolQ": 0.896,
1069
- "NarrativeQA": 0.742,
1070
- "NaturalQuestions (open-book)": -1,
1071
- "QuAC": 0.473,
1072
- "HellaSwag": -1,
1073
- "OpenbookQA": -1,
1074
- "TruthfulQA": 0.616,
1075
- "MS MARCO (TREC)": -1,
1076
- "CNN/DailyMail": 0.049,
1077
- "XSUM": 0.149,
1078
- "IMDB": 0.935,
1079
- "CivilComments": 0.008,
1080
- "RAFT": 0.701
1081
- }
1082
- },
1083
- {
1084
- "model_id": "google/T5-11B",
1085
- "name": "T5 11B",
1086
- "developer": "Google",
1087
- "scores": {
1088
- "Mean win rate": 0.131,
1089
- "MMLU": 0.29,
1090
- "BoolQ": 0.761,
1091
- "NarrativeQA": 0.086,
1092
- "NaturalQuestions (open-book)": 0.477,
1093
- "QuAC": 0.116,
1094
- "HellaSwag": -1,
1095
- "OpenbookQA": -1,
1096
- "TruthfulQA": 0.133,
1097
- "MS MARCO (TREC)": -1,
1098
- "CNN/DailyMail": 0.043,
1099
- "XSUM": 0.015,
1100
- "IMDB": 0.379,
1101
- "CivilComments": 0.509,
1102
- "RAFT": 0.37
1103
- }
1104
- },
1105
- {
1106
- "model_id": "google/UL2-20B",
1107
- "name": "UL2 20B",
1108
- "developer": "Google",
1109
- "scores": {
1110
- "Mean win rate": 0.167,
1111
- "MMLU": 0.291,
1112
- "BoolQ": 0.746,
1113
- "NarrativeQA": 0.083,
1114
- "NaturalQuestions (open-book)": 0.349,
1115
- "QuAC": 0.144,
1116
- "HellaSwag": -1,
1117
- "OpenbookQA": -1,
1118
- "TruthfulQA": 0.193,
1119
- "MS MARCO (TREC)": -1,
1120
- "CNN/DailyMail": 0.03,
1121
- "XSUM": 0.058,
1122
- "IMDB": 0.337,
1123
- "CivilComments": 0.521,
1124
- "RAFT": 0.404
1125
- }
1126
- },
1127
- {
1128
- "model_id": "lmsys/Vicuna-v1.3-13B",
1129
- "name": "Vicuna v1.3 13B",
1130
- "developer": "lmsys",
1131
- "scores": {
1132
- "Mean win rate": 0.706,
1133
- "MMLU": 0.462,
1134
- "BoolQ": 0.808,
1135
- "NarrativeQA": 0.691,
1136
- "NaturalQuestions (open-book)": 0.686,
1137
- "QuAC": 0.403,
1138
- "HellaSwag": -1,
1139
- "OpenbookQA": -1,
1140
- "TruthfulQA": 0.385,
1141
- "MS MARCO (TREC)": -1,
1142
- "CNN/DailyMail": -1,
1143
- "XSUM": -1,
1144
- "IMDB": 0.762,
1145
- "CivilComments": 0.645,
1146
- "RAFT": 0.657
1147
- }
1148
- },
1149
- {
1150
- "model_id": "lmsys/Vicuna-v1.3-7B",
1151
- "name": "Vicuna v1.3 7B",
1152
- "developer": "lmsys",
1153
- "scores": {
1154
- "Mean win rate": 0.625,
1155
- "MMLU": 0.434,
1156
- "BoolQ": 0.76,
1157
- "NarrativeQA": 0.643,
1158
- "NaturalQuestions (open-book)": 0.634,
1159
- "QuAC": 0.392,
1160
- "HellaSwag": -1,
1161
- "OpenbookQA": -1,
1162
- "TruthfulQA": 0.292,
1163
- "MS MARCO (TREC)": -1,
1164
- "CNN/DailyMail": -1,
1165
- "XSUM": -1,
1166
- "IMDB": 0.916,
1167
- "CivilComments": 0.62,
1168
- "RAFT": 0.693
1169
- }
1170
- },
1171
- {
1172
- "model_id": "meta/LLaMA-13B",
1173
- "name": "LLaMA 13B",
1174
- "developer": "Meta",
1175
- "scores": {
1176
- "Mean win rate": 0.595,
1177
- "MMLU": 0.422,
1178
- "BoolQ": 0.714,
1179
- "NarrativeQA": 0.711,
1180
- "NaturalQuestions (open-book)": 0.614,
1181
- "QuAC": 0.347,
1182
- "HellaSwag": -1,
1183
- "OpenbookQA": -1,
1184
- "TruthfulQA": 0.324,
1185
- "MS MARCO (TREC)": -1,
1186
- "CNN/DailyMail": -1,
1187
- "XSUM": -1,
1188
- "IMDB": 0.928,
1189
- "CivilComments": 0.6,
1190
- "RAFT": 0.643
1191
- }
1192
- },
1193
- {
1194
- "model_id": "meta/LLaMA-30B",
1195
- "name": "LLaMA 30B",
1196
- "developer": "Meta",
1197
- "scores": {
1198
- "Mean win rate": 0.781,
1199
- "MMLU": 0.531,
1200
- "BoolQ": 0.861,
1201
- "NarrativeQA": 0.752,
1202
- "NaturalQuestions (open-book)": 0.666,
1203
- "QuAC": 0.39,
1204
- "HellaSwag": -1,
1205
- "OpenbookQA": -1,
1206
- "TruthfulQA": 0.344,
1207
- "MS MARCO (TREC)": -1,
1208
- "CNN/DailyMail": -1,
1209
- "XSUM": -1,
1210
- "IMDB": 0.927,
1211
- "CivilComments": 0.549,
1212
- "RAFT": 0.752
1213
- }
1214
- },
1215
- {
1216
- "model_id": "meta/LLaMA-65B",
1217
- "name": "LLaMA 65B",
1218
- "developer": "Meta",
1219
- "scores": {
1220
- "Mean win rate": 0.908,
1221
- "MMLU": 0.584,
1222
- "BoolQ": 0.871,
1223
- "NarrativeQA": 0.755,
1224
- "NaturalQuestions (open-book)": 0.672,
1225
- "QuAC": 0.401,
1226
- "HellaSwag": -1,
1227
- "OpenbookQA": -1,
1228
- "TruthfulQA": 0.508,
1229
- "MS MARCO (TREC)": -1,
1230
- "CNN/DailyMail": -1,
1231
- "XSUM": -1,
1232
- "IMDB": 0.962,
1233
- "CivilComments": 0.655,
1234
- "RAFT": 0.702
1235
- }
1236
- },
1237
- {
1238
- "model_id": "meta/LLaMA-7B",
1239
- "name": "LLaMA 7B",
1240
- "developer": "Meta",
1241
- "scores": {
1242
- "Mean win rate": 0.533,
1243
- "MMLU": 0.321,
1244
- "BoolQ": 0.756,
1245
- "NarrativeQA": 0.669,
1246
- "NaturalQuestions (open-book)": 0.589,
1247
- "QuAC": 0.338,
1248
- "HellaSwag": -1,
1249
- "OpenbookQA": -1,
1250
- "TruthfulQA": 0.28,
1251
- "MS MARCO (TREC)": -1,
1252
- "CNN/DailyMail": -1,
1253
- "XSUM": -1,
1254
- "IMDB": 0.947,
1255
- "CivilComments": 0.563,
1256
- "RAFT": 0.573
1257
- }
1258
- },
1259
- {
1260
- "model_id": "meta/OPT-175B",
1261
- "name": "OPT 175B",
1262
- "developer": "Meta",
1263
- "scores": {
1264
- "Mean win rate": 0.609,
1265
- "MMLU": 0.318,
1266
- "BoolQ": 0.793,
1267
- "NarrativeQA": 0.671,
1268
- "NaturalQuestions (open-book)": 0.615,
1269
- "QuAC": 0.36,
1270
- "HellaSwag": 0.791,
1271
- "OpenbookQA": 0.586,
1272
- "TruthfulQA": 0.25,
1273
- "MS MARCO (TREC)": 0.448,
1274
- "CNN/DailyMail": 0.146,
1275
- "XSUM": 0.155,
1276
- "IMDB": 0.947,
1277
- "CivilComments": 0.505,
1278
- "RAFT": 0.606
1279
- }
1280
- },
1281
- {
1282
- "model_id": "meta/OPT-66B",
1283
- "name": "OPT 66B",
1284
- "developer": "Meta",
1285
- "scores": {
1286
- "Mean win rate": 0.448,
1287
- "MMLU": 0.276,
1288
- "BoolQ": 0.76,
1289
- "NarrativeQA": 0.638,
1290
- "NaturalQuestions (open-book)": 0.596,
1291
- "QuAC": 0.357,
1292
- "HellaSwag": 0.745,
1293
- "OpenbookQA": 0.534,
1294
- "TruthfulQA": 0.201,
1295
- "MS MARCO (TREC)": 0.482,
1296
- "CNN/DailyMail": 0.136,
1297
- "XSUM": 0.126,
1298
- "IMDB": 0.917,
1299
- "CivilComments": 0.506,
1300
- "RAFT": 0.557
1301
- }
1302
- },
1303
- {
1304
- "model_id": "meta/llama-2-13b",
1305
- "name": "Llama 2 13B",
1306
- "developer": "Meta",
1307
- "scores": {
1308
- "Mean win rate": 0.823,
1309
- "MMLU": 0.507,
1310
- "BoolQ": 0.811,
1311
- "NarrativeQA": 0.744,
1312
- "NaturalQuestions (open-book)": 0.637,
1313
- "QuAC": 0.424,
1314
- "HellaSwag": -1,
1315
- "OpenbookQA": -1,
1316
- "TruthfulQA": 0.33,
1317
- "MS MARCO (TREC)": -1,
1318
- "CNN/DailyMail": -1,
1319
- "XSUM": -1,
1320
- "IMDB": 0.962,
1321
- "CivilComments": 0.588,
1322
- "RAFT": 0.707
1323
- }
1324
- },
1325
- {
1326
- "model_id": "meta/llama-2-70b",
1327
- "name": "Llama 2 70B",
1328
- "developer": "Meta",
1329
- "scores": {
1330
- "Mean win rate": 0.944,
1331
- "MMLU": 0.582,
1332
- "BoolQ": 0.886,
1333
- "NarrativeQA": 0.77,
1334
- "NaturalQuestions (open-book)": 0.674,
1335
- "QuAC": 0.484,
1336
- "HellaSwag": -1,
1337
- "OpenbookQA": -1,
1338
- "TruthfulQA": 0.554,
1339
- "MS MARCO (TREC)": -1,
1340
- "CNN/DailyMail": -1,
1341
- "XSUM": -1,
1342
- "IMDB": 0.961,
1343
- "CivilComments": 0.652,
1344
- "RAFT": 0.727
1345
- }
1346
- },
1347
- {
1348
- "model_id": "meta/llama-2-7b",
1349
- "name": "Llama 2 7B",
1350
- "developer": "Meta",
1351
- "scores": {
1352
- "Mean win rate": 0.607,
1353
- "MMLU": 0.431,
1354
- "BoolQ": 0.762,
1355
- "NarrativeQA": 0.691,
1356
- "NaturalQuestions (open-book)": 0.611,
1357
- "QuAC": 0.406,
1358
- "HellaSwag": -1,
1359
- "OpenbookQA": -1,
1360
- "TruthfulQA": 0.272,
1361
- "MS MARCO (TREC)": -1,
1362
- "CNN/DailyMail": -1,
1363
- "XSUM": -1,
1364
- "IMDB": 0.907,
1365
- "CivilComments": 0.562,
1366
- "RAFT": 0.643
1367
- }
1368
- },
1369
- {
1370
- "model_id": "microsoft/TNLG-v2-530B",
1371
- "name": "TNLG v2 530B",
1372
- "developer": "microsoft",
1373
- "scores": {
1374
- "Mean win rate": 0.787,
1375
- "MMLU": 0.469,
1376
- "BoolQ": 0.809,
1377
- "NarrativeQA": 0.722,
1378
- "NaturalQuestions (open-book)": 0.642,
1379
- "QuAC": 0.39,
1380
- "HellaSwag": 0.799,
1381
- "OpenbookQA": 0.562,
1382
- "TruthfulQA": 0.251,
1383
- "MS MARCO (TREC)": 0.643,
1384
- "CNN/DailyMail": 0.161,
1385
- "XSUM": 0.169,
1386
- "IMDB": 0.941,
1387
- "CivilComments": 0.601,
1388
- "RAFT": 0.679
1389
- }
1390
- },
1391
- {
1392
- "model_id": "microsoft/TNLG-v2-6.7B",
1393
- "name": "TNLG v2 6.7B",
1394
- "developer": "microsoft",
1395
- "scores": {
1396
- "Mean win rate": 0.309,
1397
- "MMLU": 0.242,
1398
- "BoolQ": 0.698,
1399
- "NarrativeQA": 0.631,
1400
- "NaturalQuestions (open-book)": 0.561,
1401
- "QuAC": 0.345,
1402
- "HellaSwag": 0.704,
1403
- "OpenbookQA": 0.478,
1404
- "TruthfulQA": 0.167,
1405
- "MS MARCO (TREC)": 0.332,
1406
- "CNN/DailyMail": 0.146,
1407
- "XSUM": 0.11,
1408
- "IMDB": 0.927,
1409
- "CivilComments": 0.532,
1410
- "RAFT": 0.525
1411
- }
1412
- },
1413
- {
1414
- "model_id": "mistralai/Mistral-v0.1-7B",
1415
- "name": "Mistral v0.1 7B",
1416
- "developer": "mistralai",
1417
- "scores": {
1418
- "Mean win rate": 0.884,
1419
- "MMLU": 0.572,
1420
- "BoolQ": 0.874,
1421
- "NarrativeQA": 0.716,
1422
- "NaturalQuestions (open-book)": 0.687,
1423
- "QuAC": 0.423,
1424
- "HellaSwag": -1,
1425
- "OpenbookQA": -1,
1426
- "TruthfulQA": 0.422,
1427
- "MS MARCO (TREC)": -1,
1428
- "CNN/DailyMail": -1,
1429
- "XSUM": -1,
1430
- "IMDB": 0.962,
1431
- "CivilComments": 0.624,
1432
- "RAFT": 0.707
1433
- }
1434
- },
1435
- {
1436
- "model_id": "mosaicml/MPT-30B",
1437
- "name": "MPT 30B",
1438
- "developer": "mosaicml",
1439
- "scores": {
1440
- "Mean win rate": 0.714,
1441
- "MMLU": 0.437,
1442
- "BoolQ": 0.704,
1443
- "NarrativeQA": 0.732,
1444
- "NaturalQuestions (open-book)": 0.673,
1445
- "QuAC": 0.393,
1446
- "HellaSwag": -1,
1447
- "OpenbookQA": -1,
1448
- "TruthfulQA": 0.231,
1449
- "MS MARCO (TREC)": -1,
1450
- "CNN/DailyMail": -1,
1451
- "XSUM": -1,
1452
- "IMDB": 0.959,
1453
- "CivilComments": 0.599,
1454
- "RAFT": 0.723
1455
- }
1456
- },
1457
- {
1458
- "model_id": "mosaicml/MPT-Instruct-30B",
1459
- "name": "MPT-Instruct 30B",
1460
- "developer": "mosaicml",
1461
- "scores": {
1462
- "Mean win rate": 0.716,
1463
- "MMLU": 0.444,
1464
- "BoolQ": 0.85,
1465
- "NarrativeQA": 0.733,
1466
- "NaturalQuestions (open-book)": 0.697,
1467
- "QuAC": 0.327,
1468
- "HellaSwag": -1,
1469
- "OpenbookQA": -1,
1470
- "TruthfulQA": 0.234,
1471
- "MS MARCO (TREC)": -1,
1472
- "CNN/DailyMail": -1,
1473
- "XSUM": -1,
1474
- "IMDB": 0.956,
1475
- "CivilComments": 0.573,
1476
- "RAFT": 0.68
1477
- }
1478
- },
1479
- {
1480
- "model_id": "openai/GPT-J-6B",
1481
- "name": "GPT-J 6B",
1482
- "developer": "OpenAI",
1483
- "scores": {
1484
- "Mean win rate": 0.273,
1485
- "MMLU": 0.249,
1486
- "BoolQ": 0.649,
1487
- "NarrativeQA": 0.545,
1488
- "NaturalQuestions (open-book)": 0.559,
1489
- "QuAC": 0.33,
1490
- "HellaSwag": 0.663,
1491
- "OpenbookQA": 0.514,
1492
- "TruthfulQA": 0.199,
1493
- "MS MARCO (TREC)": 0.345,
1494
- "CNN/DailyMail": 0.131,
1495
- "XSUM": 0.096,
1496
- "IMDB": 0.939,
1497
- "CivilComments": 0.52,
1498
- "RAFT": 0.619
1499
- }
1500
- },
1501
- {
1502
- "model_id": "openai/GPT-NeoX-20B",
1503
- "name": "GPT-NeoX 20B",
1504
- "developer": "OpenAI",
1505
- "scores": {
1506
- "Mean win rate": 0.351,
1507
- "MMLU": 0.276,
1508
- "BoolQ": 0.683,
1509
- "NarrativeQA": 0.599,
1510
- "NaturalQuestions (open-book)": 0.596,
1511
- "QuAC": 0.326,
1512
- "HellaSwag": 0.718,
1513
- "OpenbookQA": 0.524,
1514
- "TruthfulQA": 0.216,
1515
- "MS MARCO (TREC)": 0.398,
1516
- "CNN/DailyMail": 0.123,
1517
- "XSUM": 0.102,
1518
- "IMDB": 0.948,
1519
- "CivilComments": 0.516,
1520
- "RAFT": 0.505
1521
- }
1522
- },
1523
- {
1524
- "model_id": "openai/ada-350M",
1525
- "name": "ada 350M",
1526
- "developer": "OpenAI",
1527
- "scores": {
1528
- "Mean win rate": 0.108,
1529
- "MMLU": 0.243,
1530
- "BoolQ": 0.581,
1531
- "NarrativeQA": 0.326,
1532
- "NaturalQuestions (open-book)": 0.365,
1533
- "QuAC": 0.242,
1534
- "HellaSwag": 0.435,
1535
- "OpenbookQA": 0.38,
1536
- "TruthfulQA": 0.215,
1537
- "MS MARCO (TREC)": 0.29,
1538
- "CNN/DailyMail": 0.09,
1539
- "XSUM": 0.022,
1540
- "IMDB": 0.849,
1541
- "CivilComments": 0.517,
1542
- "RAFT": 0.423
1543
- }
1544
- },
1545
- {
1546
- "model_id": "openai/babbage-1.3B",
1547
- "name": "babbage 1.3B",
1548
- "developer": "OpenAI",
1549
- "scores": {
1550
- "Mean win rate": 0.114,
1551
- "MMLU": 0.235,
1552
- "BoolQ": 0.574,
1553
- "NarrativeQA": 0.491,
1554
- "NaturalQuestions (open-book)": 0.451,
1555
- "QuAC": 0.273,
1556
- "HellaSwag": 0.555,
1557
- "OpenbookQA": 0.438,
1558
- "TruthfulQA": 0.188,
1559
- "MS MARCO (TREC)": 0.317,
1560
- "CNN/DailyMail": 0.079,
1561
- "XSUM": 0.045,
1562
- "IMDB": 0.597,
1563
- "CivilComments": 0.519,
1564
- "RAFT": 0.455
1565
- }
1566
- },
1567
- {
1568
- "model_id": "openai/curie-6.7B",
1569
- "name": "curie 6.7B",
1570
- "developer": "OpenAI",
1571
- "scores": {
1572
- "Mean win rate": 0.247,
1573
- "MMLU": 0.243,
1574
- "BoolQ": 0.656,
1575
- "NarrativeQA": 0.604,
1576
- "NaturalQuestions (open-book)": 0.552,
1577
- "QuAC": 0.321,
1578
- "HellaSwag": 0.682,
1579
- "OpenbookQA": 0.502,
1580
- "TruthfulQA": 0.232,
1581
- "MS MARCO (TREC)": 0.3,
1582
- "CNN/DailyMail": 0.113,
1583
- "XSUM": 0.091,
1584
- "IMDB": 0.889,
1585
- "CivilComments": 0.539,
1586
- "RAFT": 0.49
1587
- }
1588
- },
1589
- {
1590
- "model_id": "openai/davinci-175B",
1591
- "name": "davinci 175B",
1592
- "developer": "OpenAI",
1593
- "scores": {
1594
- "Mean win rate": 0.538,
1595
- "MMLU": 0.422,
1596
- "BoolQ": 0.722,
1597
- "NarrativeQA": 0.687,
1598
- "NaturalQuestions (open-book)": 0.625,
1599
- "QuAC": 0.36,
1600
- "HellaSwag": 0.775,
1601
- "OpenbookQA": 0.586,
1602
- "TruthfulQA": 0.194,
1603
- "MS MARCO (TREC)": 0.378,
1604
- "CNN/DailyMail": 0.127,
1605
- "XSUM": 0.126,
1606
- "IMDB": 0.933,
1607
- "CivilComments": 0.532,
1608
- "RAFT": 0.642
1609
- }
1610
- },
1611
- {
1612
- "model_id": "openai/gpt-3.5-turbo-0301",
1613
- "name": "gpt-3.5-turbo-0301",
1614
- "developer": "OpenAI",
1615
- "scores": {
1616
- "Mean win rate": 0.76,
1617
- "MMLU": 0.59,
1618
- "BoolQ": 0.74,
1619
- "NarrativeQA": 0.663,
1620
- "NaturalQuestions (open-book)": 0.624,
1621
- "QuAC": 0.512,
1622
- "HellaSwag": -1,
1623
- "OpenbookQA": -1,
1624
- "TruthfulQA": 0.609,
1625
- "MS MARCO (TREC)": -1,
1626
- "CNN/DailyMail": -1,
1627
- "XSUM": -1,
1628
- "IMDB": 0.899,
1629
- "CivilComments": 0.674,
1630
- "RAFT": 0.768
1631
- }
1632
- },
1633
- {
1634
- "model_id": "openai/gpt-3.5-turbo-0613",
1635
- "name": "GPT-3.5 Turbo 0613",
1636
- "developer": "OpenAI",
1637
- "scores": {
1638
- "Mean win rate": 0.783,
1639
- "MMLU": 0.391,
1640
- "BoolQ": 0.87,
1641
- "NarrativeQA": 0.625,
1642
- "NaturalQuestions (open-book)": 0.675,
1643
- "QuAC": 0.485,
1644
- "HellaSwag": -1,
1645
- "OpenbookQA": -1,
1646
- "TruthfulQA": 0.339,
1647
- "MS MARCO (TREC)": -1,
1648
- "CNN/DailyMail": -1,
1649
- "XSUM": -1,
1650
- "IMDB": 0.943,
1651
- "CivilComments": 0.696,
1652
- "RAFT": 0.748
1653
- }
1654
- },
1655
- {
1656
- "model_id": "openai/text-ada-001",
1657
- "name": "text-ada-001",
1658
- "developer": "OpenAI",
1659
- "scores": {
1660
- "Mean win rate": 0.107,
1661
- "MMLU": 0.238,
1662
- "BoolQ": 0.464,
1663
- "NarrativeQA": 0.238,
1664
- "NaturalQuestions (open-book)": 0.149,
1665
- "QuAC": 0.176,
1666
- "HellaSwag": 0.429,
1667
- "OpenbookQA": 0.346,
1668
- "TruthfulQA": 0.232,
1669
- "MS MARCO (TREC)": 0.302,
1670
- "CNN/DailyMail": 0.136,
1671
- "XSUM": 0.034,
1672
- "IMDB": 0.822,
1673
- "CivilComments": 0.503,
1674
- "RAFT": 0.406
1675
- }
1676
- },
1677
- {
1678
- "model_id": "openai/text-babbage-001",
1679
- "name": "text-babbage-001",
1680
- "developer": "OpenAI",
1681
- "scores": {
1682
- "Mean win rate": 0.229,
1683
- "MMLU": 0.229,
1684
- "BoolQ": 0.451,
1685
- "NarrativeQA": 0.429,
1686
- "NaturalQuestions (open-book)": 0.33,
1687
- "QuAC": 0.284,
1688
- "HellaSwag": 0.561,
1689
- "OpenbookQA": 0.452,
1690
- "TruthfulQA": 0.233,
1691
- "MS MARCO (TREC)": 0.449,
1692
- "CNN/DailyMail": 0.151,
1693
- "XSUM": 0.046,
1694
- "IMDB": 0.913,
1695
- "CivilComments": 0.499,
1696
- "RAFT": 0.509
1697
- }
1698
- },
1699
- {
1700
- "model_id": "openai/text-curie-001",
1701
- "name": "text-curie-001",
1702
- "developer": "OpenAI",
1703
- "scores": {
1704
- "Mean win rate": 0.36,
1705
- "MMLU": 0.237,
1706
- "BoolQ": 0.62,
1707
- "NarrativeQA": 0.582,
1708
- "NaturalQuestions (open-book)": 0.571,
1709
- "QuAC": 0.358,
1710
- "HellaSwag": 0.676,
1711
- "OpenbookQA": 0.514,
1712
- "TruthfulQA": 0.257,
1713
- "MS MARCO (TREC)": 0.507,
1714
- "CNN/DailyMail": 0.152,
1715
- "XSUM": 0.076,
1716
- "IMDB": 0.923,
1717
- "CivilComments": 0.537,
1718
- "RAFT": 0.489
1719
- }
1720
- },
1721
- {
1722
- "model_id": "openai/text-davinci-002",
1723
- "name": "GPT-3.5 text-davinci-002",
1724
- "developer": "OpenAI",
1725
- "scores": {
1726
- "Mean win rate": 0.905,
1727
- "MMLU": 0.568,
1728
- "BoolQ": 0.877,
1729
- "NarrativeQA": 0.727,
1730
- "NaturalQuestions (open-book)": 0.713,
1731
- "QuAC": 0.445,
1732
- "HellaSwag": 0.815,
1733
- "OpenbookQA": 0.594,
1734
- "TruthfulQA": 0.61,
1735
- "MS MARCO (TREC)": 0.664,
1736
- "CNN/DailyMail": 0.153,
1737
- "XSUM": 0.144,
1738
- "IMDB": 0.948,
1739
- "CivilComments": 0.668,
1740
- "RAFT": 0.733
1741
- }
1742
- },
1743
- {
1744
- "model_id": "openai/text-davinci-003",
1745
- "name": "GPT-3.5 text-davinci-003",
1746
- "developer": "OpenAI",
1747
- "scores": {
1748
- "Mean win rate": 0.872,
1749
- "MMLU": 0.569,
1750
- "BoolQ": 0.881,
1751
- "NarrativeQA": 0.727,
1752
- "NaturalQuestions (open-book)": 0.77,
1753
- "QuAC": 0.525,
1754
- "HellaSwag": 0.822,
1755
- "OpenbookQA": 0.646,
1756
- "TruthfulQA": 0.593,
1757
- "MS MARCO (TREC)": 0.644,
1758
- "CNN/DailyMail": 0.156,
1759
- "XSUM": 0.124,
1760
- "IMDB": 0.848,
1761
- "CivilComments": 0.684,
1762
- "RAFT": 0.759
1763
- }
1764
- },
1765
- {
1766
- "model_id": "stanford/Alpaca-7B",
1767
- "name": "Alpaca 7B",
1768
- "developer": "stanford",
1769
- "scores": {
1770
- "Mean win rate": 0.381,
1771
- "MMLU": 0.385,
1772
- "BoolQ": 0.778,
1773
- "NarrativeQA": 0.396,
1774
- "NaturalQuestions (open-book)": 0.592,
1775
- "QuAC": 0.27,
1776
- "HellaSwag": -1,
1777
- "OpenbookQA": -1,
1778
- "TruthfulQA": 0.243,
1779
- "MS MARCO (TREC)": -1,
1780
- "CNN/DailyMail": -1,
1781
- "XSUM": -1,
1782
- "IMDB": 0.738,
1783
- "CivilComments": 0.566,
1784
- "RAFT": 0.486
1785
- }
1786
- },
1787
- {
1788
- "model_id": "tiiuae/Falcon-Instruct-40B",
1789
- "name": "Falcon-Instruct 40B",
1790
- "developer": "tiiuae",
1791
- "scores": {
1792
- "Mean win rate": 0.727,
1793
- "MMLU": 0.497,
1794
- "BoolQ": 0.829,
1795
- "NarrativeQA": 0.625,
1796
- "NaturalQuestions (open-book)": 0.666,
1797
- "QuAC": 0.371,
1798
- "HellaSwag": -1,
1799
- "OpenbookQA": -1,
1800
- "TruthfulQA": 0.384,
1801
- "MS MARCO (TREC)": -1,
1802
- "CNN/DailyMail": -1,
1803
- "XSUM": -1,
1804
- "IMDB": 0.959,
1805
- "CivilComments": 0.603,
1806
- "RAFT": 0.586
1807
- }
1808
- },
1809
- {
1810
- "model_id": "tiiuae/Falcon-Instruct-7B",
1811
- "name": "Falcon-Instruct 7B",
1812
- "developer": "tiiuae",
1813
- "scores": {
1814
- "Mean win rate": 0.244,
1815
- "MMLU": 0.275,
1816
- "BoolQ": 0.72,
1817
- "NarrativeQA": 0.476,
1818
- "NaturalQuestions (open-book)": 0.449,
1819
- "QuAC": 0.311,
1820
- "HellaSwag": -1,
1821
- "OpenbookQA": -1,
1822
- "TruthfulQA": 0.213,
1823
- "MS MARCO (TREC)": -1,
1824
- "CNN/DailyMail": -1,
1825
- "XSUM": -1,
1826
- "IMDB": 0.852,
1827
- "CivilComments": 0.511,
1828
- "RAFT": 0.523
1829
- }
1830
- },
1831
- {
1832
- "model_id": "tiiuae/falcon-40b",
1833
- "name": "Falcon 40B",
1834
- "developer": "tiiuae",
1835
- "scores": {
1836
- "Mean win rate": 0.729,
1837
- "MMLU": 0.509,
1838
- "BoolQ": 0.819,
1839
- "NarrativeQA": 0.673,
1840
- "NaturalQuestions (open-book)": 0.675,
1841
- "QuAC": 0.307,
1842
- "HellaSwag": -1,
1843
- "OpenbookQA": -1,
1844
- "TruthfulQA": 0.353,
1845
- "MS MARCO (TREC)": -1,
1846
- "CNN/DailyMail": -1,
1847
- "XSUM": -1,
1848
- "IMDB": 0.959,
1849
- "CivilComments": 0.552,
1850
- "RAFT": 0.661
1851
- }
1852
- },
1853
- {
1854
- "model_id": "tiiuae/falcon-7b",
1855
- "name": "Falcon 7B",
1856
- "developer": "tiiuae",
1857
- "scores": {
1858
- "Mean win rate": 0.378,
1859
- "MMLU": 0.286,
1860
- "BoolQ": 0.753,
1861
- "NarrativeQA": 0.621,
1862
- "NaturalQuestions (open-book)": 0.579,
1863
- "QuAC": 0.332,
1864
- "HellaSwag": -1,
1865
- "OpenbookQA": -1,
1866
- "TruthfulQA": 0.234,
1867
- "MS MARCO (TREC)": -1,
1868
- "CNN/DailyMail": -1,
1869
- "XSUM": -1,
1870
- "IMDB": 0.836,
1871
- "CivilComments": 0.514,
1872
- "RAFT": 0.602
1873
- }
1874
- },
1875
- {
1876
- "model_id": "together/RedPajama-INCITE-Base-7B",
1877
- "name": "RedPajama-INCITE-Base 7B",
1878
- "developer": "together",
1879
- "scores": {
1880
- "Mean win rate": 0.378,
1881
- "MMLU": 0.302,
1882
- "BoolQ": 0.713,
1883
- "NarrativeQA": 0.617,
1884
- "NaturalQuestions (open-book)": 0.586,
1885
- "QuAC": 0.336,
1886
- "HellaSwag": -1,
1887
- "OpenbookQA": -1,
1888
- "TruthfulQA": 0.205,
1889
- "MS MARCO (TREC)": -1,
1890
- "CNN/DailyMail": -1,
1891
- "XSUM": -1,
1892
- "IMDB": 0.752,
1893
- "CivilComments": 0.547,
1894
- "RAFT": 0.648
1895
- }
1896
- },
1897
- {
1898
- "model_id": "together/RedPajama-INCITE-Base-v1-3B",
1899
- "name": "RedPajama-INCITE-Base-v1 3B",
1900
- "developer": "together",
1901
- "scores": {
1902
- "Mean win rate": 0.311,
1903
- "MMLU": 0.263,
1904
- "BoolQ": 0.685,
1905
- "NarrativeQA": 0.555,
1906
- "NaturalQuestions (open-book)": 0.52,
1907
- "QuAC": 0.309,
1908
- "HellaSwag": -1,
1909
- "OpenbookQA": -1,
1910
- "TruthfulQA": 0.277,
1911
- "MS MARCO (TREC)": -1,
1912
- "CNN/DailyMail": -1,
1913
- "XSUM": -1,
1914
- "IMDB": 0.907,
1915
- "CivilComments": 0.549,
1916
- "RAFT": 0.502
1917
- }
1918
- },
1919
- {
1920
- "model_id": "together/RedPajama-INCITE-Instruct-7B",
1921
- "name": "RedPajama-INCITE-Instruct 7B",
1922
- "developer": "together",
1923
- "scores": {
1924
- "Mean win rate": 0.524,
1925
- "MMLU": 0.363,
1926
- "BoolQ": 0.705,
1927
- "NarrativeQA": 0.638,
1928
- "NaturalQuestions (open-book)": 0.659,
1929
- "QuAC": 0.26,
1930
- "HellaSwag": -1,
1931
- "OpenbookQA": -1,
1932
- "TruthfulQA": 0.243,
1933
- "MS MARCO (TREC)": -1,
1934
- "CNN/DailyMail": -1,
1935
- "XSUM": -1,
1936
- "IMDB": 0.927,
1937
- "CivilComments": 0.664,
1938
- "RAFT": 0.695
1939
- }
1940
- },
1941
- {
1942
- "model_id": "together/RedPajama-INCITE-Instruct-v1-3B",
1943
- "name": "RedPajama-INCITE-Instruct-v1 3B",
1944
- "developer": "together",
1945
- "scores": {
1946
- "Mean win rate": 0.366,
1947
- "MMLU": 0.257,
1948
- "BoolQ": 0.677,
1949
- "NarrativeQA": 0.638,
1950
- "NaturalQuestions (open-book)": 0.637,
1951
- "QuAC": 0.259,
1952
- "HellaSwag": -1,
1953
- "OpenbookQA": -1,
1954
- "TruthfulQA": 0.208,
1955
- "MS MARCO (TREC)": -1,
1956
- "CNN/DailyMail": -1,
1957
- "XSUM": -1,
1958
- "IMDB": 0.894,
1959
- "CivilComments": 0.549,
1960
- "RAFT": 0.661
1961
- }
1962
- },
1963
- {
1964
- "model_id": "writer/InstructPalmyra-30B",
1965
- "name": "InstructPalmyra 30B",
1966
- "developer": "writer",
1967
- "scores": {
1968
- "Mean win rate": 0.568,
1969
- "MMLU": 0.403,
1970
- "BoolQ": 0.751,
1971
- "NarrativeQA": 0.496,
1972
- "NaturalQuestions (open-book)": 0.682,
1973
- "QuAC": 0.433,
1974
- "HellaSwag": -1,
1975
- "OpenbookQA": -1,
1976
- "TruthfulQA": 0.185,
1977
- "MS MARCO (TREC)": -1,
1978
- "CNN/DailyMail": 0.152,
1979
- "XSUM": 0.104,
1980
- "IMDB": 0.94,
1981
- "CivilComments": 0.555,
1982
- "RAFT": 0.652
1983
- }
1984
- },
1985
- {
1986
- "model_id": "yandex/YaLM-100B",
1987
- "name": "YaLM 100B",
1988
- "developer": "yandex",
1989
- "scores": {
1990
- "Mean win rate": 0.075,
1991
- "MMLU": 0.243,
1992
- "BoolQ": 0.634,
1993
- "NarrativeQA": 0.252,
1994
- "NaturalQuestions (open-book)": 0.227,
1995
- "QuAC": 0.162,
1996
- "HellaSwag": -1,
1997
- "OpenbookQA": -1,
1998
- "TruthfulQA": 0.202,
1999
- "MS MARCO (TREC)": -1,
2000
- "CNN/DailyMail": 0.017,
2001
- "XSUM": 0.021,
2002
- "IMDB": 0.836,
2003
- "CivilComments": 0.49,
2004
- "RAFT": 0.395
2005
- }
2006
- },
2007
- {
2008
- "model_id": "zhipu-ai/GLM-130B",
2009
- "name": "GLM 130B",
2010
- "developer": "zhipu-ai",
2011
- "scores": {
2012
- "Mean win rate": 0.512,
2013
- "MMLU": 0.344,
2014
- "BoolQ": 0.784,
2015
- "NarrativeQA": 0.706,
2016
- "NaturalQuestions (open-book)": 0.642,
2017
- "QuAC": 0.272,
2018
- "HellaSwag": -1,
2019
- "OpenbookQA": -1,
2020
- "TruthfulQA": 0.218,
2021
- "MS MARCO (TREC)": -1,
2022
- "CNN/DailyMail": 0.154,
2023
- "XSUM": 0.132,
2024
- "IMDB": 0.955,
2025
- "CivilComments": 0.5,
2026
- "RAFT": 0.598
2027
- }
2028
- }
2029
- ]
2030
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/helm_instruct.json DELETED
@@ -1,60 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "anthropic/claude-v1.3",
5
- "name": "Anthropic Claude v1.3",
6
- "developer": "Anthropic",
7
- "scores": {
8
- "Mean win rate": 0.611,
9
- "Anthropic RLHF dataset": 4.965,
10
- "Best ChatGPT Prompts": 4.995,
11
- "Koala test dataset": 4.981,
12
- "Open Assistant": 4.975,
13
- "Self Instruct": 4.992,
14
- "Vicuna": 4.989
15
- }
16
- },
17
- {
18
- "model_id": "cohere/command-xlarge-beta",
19
- "name": "Cohere Command beta 52.4B",
20
- "developer": "cohere",
21
- "scores": {
22
- "Mean win rate": 0.089,
23
- "Anthropic RLHF dataset": 4.214,
24
- "Best ChatGPT Prompts": 4.988,
25
- "Koala test dataset": 4.969,
26
- "Open Assistant": 4.967,
27
- "Self Instruct": 4.971,
28
- "Vicuna": 4.995
29
- }
30
- },
31
- {
32
- "model_id": "openai/gpt-3.5-turbo-0613",
33
- "name": "GPT-3.5 Turbo 0613",
34
- "developer": "OpenAI",
35
- "scores": {
36
- "Mean win rate": 0.689,
37
- "Anthropic RLHF dataset": 4.964,
38
- "Best ChatGPT Prompts": 4.986,
39
- "Koala test dataset": 4.987,
40
- "Open Assistant": 4.987,
41
- "Self Instruct": 4.99,
42
- "Vicuna": 4.992
43
- }
44
- },
45
- {
46
- "model_id": "openai/gpt-4-0314",
47
- "name": "GPT-4 0314",
48
- "developer": "OpenAI",
49
- "scores": {
50
- "Mean win rate": 0.611,
51
- "Anthropic RLHF dataset": 4.934,
52
- "Best ChatGPT Prompts": 4.973,
53
- "Koala test dataset": 4.966,
54
- "Open Assistant": 4.986,
55
- "Self Instruct": 4.976,
56
- "Vicuna": 4.995
57
- }
58
- }
59
- ]
60
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/helm_lite.json DELETED
@@ -1,1893 +0,0 @@
1
- {
2
- "benchmark_cards": {
3
- "GSM8K": {
4
- "benchmark_details": {
5
- "name": "GSM8K",
6
- "overview": "GSM8K is a benchmark that measures the ability of language models to perform multi-step mathematical reasoning. It consists of 8.5K high-quality, linguistically diverse grade school math word problems. The problems are distinctive because they require 2 to 8 steps to solve using basic arithmetic, and even the largest transformer models struggle to achieve high test performance on them. Solutions are provided in natural language with step-by-step reasoning.",
7
- "data_type": "text",
8
- "domains": [
9
- "grade school mathematics",
10
- "math word problems"
11
- ],
12
- "languages": [
13
- "English"
14
- ],
15
- "similar_benchmarks": [
16
- "Not specified"
17
- ],
18
- "resources": [
19
- "https://arxiv.org/abs/2110.14168",
20
- "https://huggingface.co/datasets/openai/gsm8k",
21
- "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
22
- ]
23
- },
24
- "purpose_and_intended_users": {
25
- "goal": "To diagnose the failures of current language models in robust multi-step mathematical reasoning and to support research, particularly in methods like training verifiers to judge solution correctness. It also aims to shed light on the properties of large language models' reasoning processes.",
26
- "audience": [
27
- "Researchers working on language model capabilities and mathematical reasoning"
28
- ],
29
- "tasks": [
30
- "Solving grade school math word problems",
31
- "Text generation for question answering"
32
- ],
33
- "limitations": "Even the largest models struggle with high test performance on this dataset, and autoregressive models have no mechanism to correct their own errors during solution generation.",
34
- "out_of_scope_uses": [
35
- "Not specified"
36
- ]
37
- },
38
- "data": {
39
- "source": "The dataset was created by hiring freelance contractors via Upwork and then scaled using the NLP data labeling platform Surge AI. Problems and solutions were written by these contractors.",
40
- "size": "8.5K (8,500) problems, with a size category of 10K<n<100K. The training set contains 7,473 examples and the test set contains 1,319 examples.",
41
- "format": "parquet. The data is structured with a 'Problem:' field followed by a 'Solution:' field, where the solution includes step-by-step reasoning with intermediate calculations in special tags (e.g., `<<4*2=8>>`) and ends with a 'Final Answer:'.",
42
- "annotation": "Contractors wrote the problems and solutions. For verification, different workers re-solved all problems to check agreement with the original solutions; problematic problems were either repaired or discarded. The annotators were from Surge AI."
43
- },
44
- "methodology": {
45
- "methods": [
46
- "Models are evaluated by generating step-by-step solutions to math word problems. The dataset provides two answer formats: a standard step-by-step solution and a solution structured with Socratic sub-questions.",
47
- "The paper proposes a verification method where a separate verifier model is trained to judge the correctness of generated solutions. At test time, multiple candidate solutions are generated, and the one ranked highest by the verifier is selected."
48
- ],
49
- "metrics": [
50
- "GSM8K"
51
- ],
52
- "calculation": "The GSM8K metric is a continuous score where higher values are better. It is described as 'EM on GSM8K', indicating it measures exact match accuracy.",
53
- "interpretation": "Higher scores indicate better performance. The score is not bounded, but typical model performance ranges from low to high, with the highest reported score being 75.2.",
54
- "baseline_results": "Paper baselines: The paper notes that even the largest transformer models fail to achieve high test performance, but does not report specific scores. EEE results: Llama 3.1 8B Instruct scored 75.2, and Yi 34B scored 0.648 on GSM8K.",
55
- "validation": "The paper provides empirical evidence that the verification method scales more effectively with data than a finetuning baseline and remains effective even with a verifier much smaller than the generator."
56
- },
57
- "ethical_and_legal_considerations": {
58
- "privacy_and_anonymity": "Not specified",
59
- "data_licensing": "MIT License",
60
- "consent_procedures": "Not specified",
61
- "compliance_with_regulations": "Not specified"
62
- },
63
- "possible_risks": [
64
- {
65
- "category": "Over- or under-reliance",
66
- "description": [
67
- "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
68
- ],
69
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
70
- },
71
- {
72
- "category": "Data bias",
73
- "description": [
74
- "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
75
- ],
76
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
77
- },
78
- {
79
- "category": "Reproducibility",
80
- "description": [
81
- "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI."
82
- ],
83
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html"
84
- },
85
- {
86
- "category": "Incomplete advice",
87
- "description": [
88
- "When a model provides advice without having enough information, resulting in possible harm if the advice is followed."
89
- ],
90
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-advice.html"
91
- },
92
- {
93
- "category": "Improper usage",
94
- "description": [
95
- "Improper usage occurs when a model is used for a purpose that it was not originally designed for."
96
- ],
97
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html"
98
- }
99
- ],
100
- "flagged_fields": {},
101
- "missing_fields": [
102
- "benchmark_details.similar_benchmarks",
103
- "purpose_and_intended_users.out_of_scope_uses",
104
- "ethical_and_legal_considerations.privacy_and_anonymity",
105
- "ethical_and_legal_considerations.consent_procedures",
106
- "ethical_and_legal_considerations.compliance_with_regulations"
107
- ],
108
- "card_info": {
109
- "created_at": "2026-03-17T15:37:16.459776",
110
- "llm": "deepseek-ai/DeepSeek-V3.2"
111
- }
112
- },
113
- "LegalBench": {
114
- "benchmark_details": {
115
- "name": "LEGALBENCH",
116
- "overview": "LEGALBENCH is a benchmark designed to measure the legal reasoning capabilities of large language models. It comprises 162 tasks collaboratively constructed and hand-crafted by legal professionals, covering six distinct types of legal reasoning.",
117
- "data_type": "text",
118
- "domains": [
119
- "legal",
120
- "law",
121
- "finance"
122
- ],
123
- "languages": [
124
- "English"
125
- ],
126
- "similar_benchmarks": [
127
- "GLUE",
128
- "HELM",
129
- "BigBench",
130
- "RAFT"
131
- ],
132
- "resources": [
133
- "https://arxiv.org/abs/2308.11462",
134
- "https://huggingface.co/datasets/nguha/legalbench",
135
- "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
136
- ]
137
- },
138
- "purpose_and_intended_users": {
139
- "goal": "To enable greater study of what types of legal reasoning large language models (LLMs) can perform.",
140
- "audience": [
141
- "Practitioners (to integrate LLMs into workflows)",
142
- "Legal academics",
143
- "Computer scientists"
144
- ],
145
- "tasks": [
146
- "Text classification",
147
- "Question answering",
148
- "Text generation",
149
- "Rule-application tasks"
150
- ],
151
- "limitations": "The tasks do not generalize to all legal reasoning tasks or all types of legal documents, offering only a preliminary understanding of LLM performance.",
152
- "out_of_scope_uses": [
153
- "Predicting the legality of real-world events",
154
- "Predicting the outcome of lawsuits",
155
- "Providing legal advice"
156
- ]
157
- },
158
- "data": {
159
- "source": "Data is drawn from three categories: existing publicly available datasets and corpora (some reformatted), datasets previously created by legal professionals but not released, and tasks developed specifically for LegalBench. The tasks originate from 36 distinct corpora.",
160
- "size": "The benchmark comprises 162 tasks. The distribution of tasks by sample count is: 28 tasks have 50-100 samples, 97 have 100-500 samples, 29 have 500-2000 samples, and 8 have 2000+ samples. The overall dataset falls into the size category of 10,000 to 100,000 examples.",
161
- "format": "Examples are presented in a structured format with fields such as 'Task name', 'Question', 'Options', and 'Answer'.",
162
- "annotation": "Annotation procedures are task-dependent. For certain tasks, each data point was manually validated by a law-trained expert. Detailed annotation methodology for each task is documented in a separate section of the paper."
163
- },
164
- "methodology": {
165
- "methods": [
166
- "Models are evaluated in a few-shot setting. Train splits consist of a small random sample of between 2 to 8 instances to capture a true few-shot learning scenario.",
167
- "For rule-application tasks, a law-trained expert manually validates each model generation."
168
- ],
169
- "metrics": [
170
- "LegalBench",
171
- "Correctness",
172
- "Analysis"
173
- ],
174
- "calculation": "The primary benchmark metric is Exact Match (EM) on LegalBench. For rule-application tasks, two separate metrics are computed: 'correctness' (the proportion of generations without errors) and 'analysis'.",
175
- "interpretation": "Higher scores on the LegalBench metric indicate better performance. The metric is continuous and lower scores are not better.",
176
- "baseline_results": "The original paper evaluated 20 LLMs from 11 different families but did not provide specific scores. In a separate evaluation suite, the Yi 34B model achieved a score of 0.618.",
177
- "validation": "For rule-application tasks, a law-trained expert manually validated each model generation. For datasets reused or adapted from other sources, the original data sheets document any redactions or missing data."
178
- },
179
- "ethical_and_legal_considerations": {
180
- "privacy_and_anonymity": "For tasks that reuse or adapt existing datasets, the benchmark refers to the original data sheets for details on any data redactions or missing information.",
181
- "data_licensing": "other",
182
- "consent_procedures": "Not specified.",
183
- "compliance_with_regulations": "The benchmark includes a section for each task intended to provide information relevant to ethical review processes, but specific details are not provided in the available facts."
184
- },
185
- "possible_risks": [
186
- {
187
- "category": "Over- or under-reliance",
188
- "description": [
189
- "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
190
- ],
191
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
192
- },
193
- {
194
- "category": "Unrepresentative data",
195
- "description": [
196
- "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
197
- ],
198
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
199
- },
200
- {
201
- "category": "Data bias",
202
- "description": [
203
- "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
204
- ],
205
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
206
- },
207
- {
208
- "category": "Lack of data transparency",
209
- "description": [
210
- "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
211
- ],
212
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
213
- },
214
- {
215
- "category": "Improper usage",
216
- "description": [
217
- "Improper usage occurs when a model is used for a purpose that it was not originally designed for."
218
- ],
219
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html"
220
- }
221
- ],
222
- "flagged_fields": {},
223
- "missing_fields": [
224
- "ethical_and_legal_considerations.consent_procedures"
225
- ],
226
- "card_info": {
227
- "created_at": "2026-03-17T12:59:10.203815",
228
- "llm": "deepseek-ai/DeepSeek-V3.2"
229
- }
230
- },
231
- "MedQA": {
232
- "benchmark_details": {
233
- "name": "MEDQA",
234
- "overview": "MEDQA is a free-form multiple-choice open-domain question answering (OpenQA) benchmark designed to measure a model's ability to solve medical problems. It is distinctive as the first such dataset sourced from professional medical board exams, covering multiple languages and presenting a challenging real-world scenario.",
235
- "data_type": "text",
236
- "domains": [
237
- "medical knowledge",
238
- "professional medical exams"
239
- ],
240
- "languages": [
241
- "English"
242
- ],
243
- "similar_benchmarks": [
244
- "ARC",
245
- "OpenBookQA"
246
- ],
247
- "resources": [
248
- "https://github.com/jind11/MedQA",
249
- "https://arxiv.org/abs/2009.13081",
250
- "https://huggingface.co/datasets/GBaker/MedQA-USMLE-4-options",
251
- "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
252
- ]
253
- },
254
- "purpose_and_intended_users": {
255
- "goal": "To present a challenging open-domain question answering dataset to promote the development of stronger models capable of handling sophisticated real-world medical scenarios, specifically evaluating performance on medical knowledge as tested in professional exams.",
256
- "audience": [
257
- "The natural language processing (NLP) community"
258
- ],
259
- "tasks": [
260
- "Free-form multiple-choice question answering",
261
- "Open-domain question answering"
262
- ],
263
- "limitations": "Even the best current methods achieve relatively low accuracy (36.7% to 70.1% across languages), indicating the benchmark's difficulty and the limitations of existing models.",
264
- "out_of_scope_uses": [
265
- "Not specified"
266
- ]
267
- },
268
- "data": {
269
- "source": "The data is collected from professional medical board exams.",
270
- "size": "The dataset contains 12,723 questions in English, 34,251 in simplified Chinese, and 14,123 in traditional Chinese. The total number of examples falls within the 10K to 100K range.",
271
- "format": "JSON",
272
- "annotation": "The answer labels are the correct answers from the professional exams. No additional annotation process is described."
273
- },
274
- "methodology": {
275
- "methods": [
276
- "The benchmark uses a sequential combination of a document retriever and a machine comprehension model. It includes both rule-based and neural methods.",
277
- "The evaluation is a standard question-answering task, though the specific learning setting (e.g., zero-shot, few-shot, fine-tuning) is not explicitly defined."
278
- ],
279
- "metrics": [
280
- "Accuracy"
281
- ],
282
- "calculation": "The overall score is the accuracy on the test set.",
283
- "interpretation": "Higher accuracy indicates better performance. The best reported accuracies are 36.7% for English, 42.0% for traditional Chinese, and 70.1% for simplified Chinese questions.",
284
- "baseline_results": "Original paper baselines: The best method reported achieves 36.7% accuracy on English, 42.0% on traditional Chinese, and 70.1% on simplified Chinese questions. Model names are not specified. EEE results: Yi 34B achieves a score of 0.656 (65.6%).",
285
- "validation": "Not specified"
286
- },
287
- "ethical_and_legal_considerations": {
288
- "privacy_and_anonymity": "Not specified",
289
- "data_licensing": "Creative Commons Attribution 4.0",
290
- "consent_procedures": "Not specified",
291
- "compliance_with_regulations": "Not specified"
292
- },
293
- "possible_risks": [
294
- {
295
- "category": "Over- or under-reliance",
296
- "description": [
297
- "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
298
- ],
299
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
300
- },
301
- {
302
- "category": "Unrepresentative data",
303
- "description": [
304
- "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
305
- ],
306
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
307
- },
308
- {
309
- "category": "Uncertain data provenance",
310
- "description": [
311
- "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation."
312
- ],
313
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html"
314
- },
315
- {
316
- "category": "Data bias",
317
- "description": [
318
- "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
319
- ],
320
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
321
- },
322
- {
323
- "category": "Lack of data transparency",
324
- "description": [
325
- "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
326
- ],
327
- "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
328
- }
329
- ],
330
- "flagged_fields": {},
331
- "missing_fields": [
332
- "purpose_and_intended_users.out_of_scope_uses",
333
- "methodology.validation",
334
- "ethical_and_legal_considerations.privacy_and_anonymity",
335
- "ethical_and_legal_considerations.consent_procedures",
336
- "ethical_and_legal_considerations.compliance_with_regulations"
337
- ],
338
- "card_info": {
339
- "created_at": "2026-03-17T13:23:29.822123",
340
- "llm": "deepseek-ai/DeepSeek-V3.2"
341
- }
342
- }
343
- },
344
- "models": [
345
- {
346
- "model_id": "01-ai/yi-34b",
347
- "name": "Yi 34B",
348
- "developer": "01-ai",
349
- "scores": {
350
- "Mean win rate": 0.57,
351
- "NarrativeQA": 0.782,
352
- "NaturalQuestions (closed-book)": 0.443,
353
- "OpenbookQA": 0.92,
354
- "MMLU": 0.65,
355
- "MATH": 0.375,
356
- "GSM8K": 0.648,
357
- "LegalBench": 0.618,
358
- "MedQA": 0.656,
359
- "WMT 2014": 0.172
360
- }
361
- },
362
- {
363
- "model_id": "01-ai/yi-6b",
364
- "name": "Yi 6B",
365
- "developer": "01-ai",
366
- "scores": {
367
- "Mean win rate": 0.253,
368
- "NarrativeQA": 0.702,
369
- "NaturalQuestions (closed-book)": 0.31,
370
- "OpenbookQA": 0.8,
371
- "MMLU": 0.53,
372
- "MATH": 0.126,
373
- "GSM8K": 0.375,
374
- "LegalBench": 0.519,
375
- "MedQA": 0.497,
376
- "WMT 2014": 0.117
377
- }
378
- },
379
- {
380
- "model_id": "01-ai/yi-large-preview",
381
- "name": "Yi Large Preview",
382
- "developer": "01-ai",
383
- "scores": {
384
- "Mean win rate": 0.471,
385
- "NarrativeQA": 0.373,
386
- "NaturalQuestions (closed-book)": 0.428,
387
- "OpenbookQA": 0.946,
388
- "MMLU": 0.712,
389
- "MATH": 0.712,
390
- "GSM8K": 0.69,
391
- "LegalBench": 0.519,
392
- "MedQA": 0.66,
393
- "WMT 2014": 0.176
394
- }
395
- },
396
- {
397
- "model_id": "AlephAlpha/luminous-base",
398
- "name": "Luminous Base 13B",
399
- "developer": "AlephAlpha",
400
- "scores": {
401
- "Mean win rate": 0.041,
402
- "NarrativeQA": 0.633,
403
- "NaturalQuestions (closed-book)": 0.197,
404
- "OpenbookQA": 0.286,
405
- "MMLU": 0.243,
406
- "MATH": 0.026,
407
- "GSM8K": 0.028,
408
- "LegalBench": 0.332,
409
- "MedQA": 0.26,
410
- "WMT 2014": 0.066
411
- }
412
- },
413
- {
414
- "model_id": "AlephAlpha/luminous-extended",
415
- "name": "Luminous Extended 30B",
416
- "developer": "AlephAlpha",
417
- "scores": {
418
- "Mean win rate": 0.078,
419
- "NarrativeQA": 0.684,
420
- "NaturalQuestions (closed-book)": 0.253,
421
- "OpenbookQA": 0.272,
422
- "MMLU": 0.248,
423
- "MATH": 0.04,
424
- "GSM8K": 0.075,
425
- "LegalBench": 0.421,
426
- "MedQA": 0.276,
427
- "WMT 2014": 0.083
428
- }
429
- },
430
- {
431
- "model_id": "AlephAlpha/luminous-supreme",
432
- "name": "Luminous Supreme 70B",
433
- "developer": "AlephAlpha",
434
- "scores": {
435
- "Mean win rate": 0.145,
436
- "NarrativeQA": 0.743,
437
- "NaturalQuestions (closed-book)": 0.299,
438
- "OpenbookQA": 0.284,
439
- "MMLU": 0.316,
440
- "MATH": 0.078,
441
- "GSM8K": 0.137,
442
- "LegalBench": 0.452,
443
- "MedQA": 0.276,
444
- "WMT 2014": 0.102
445
- }
446
- },
447
- {
448
- "model_id": "ai21/j2-grande",
449
- "name": "Jurassic-2 Grande 17B",
450
- "developer": "ai21",
451
- "scores": {
452
- "Mean win rate": 0.172,
453
- "NarrativeQA": 0.744,
454
- "NaturalQuestions (closed-book)": 0.35,
455
- "OpenbookQA": 0.614,
456
- "MMLU": 0.471,
457
- "MATH": 0.064,
458
- "GSM8K": 0.159,
459
- "LegalBench": 0.468,
460
- "MedQA": 0.39,
461
- "WMT 2014": 0.102
462
- }
463
- },
464
- {
465
- "model_id": "ai21/j2-jumbo",
466
- "name": "Jurassic-2 Jumbo 178B",
467
- "developer": "ai21",
468
- "scores": {
469
- "Mean win rate": 0.215,
470
- "NarrativeQA": 0.728,
471
- "NaturalQuestions (closed-book)": 0.385,
472
- "OpenbookQA": 0.688,
473
- "MMLU": 0.483,
474
- "MATH": 0.103,
475
- "GSM8K": 0.239,
476
- "LegalBench": 0.533,
477
- "MedQA": 0.431,
478
- "WMT 2014": 0.114
479
- }
480
- },
481
- {
482
- "model_id": "ai21/jamba-1.5-large",
483
- "name": "Jamba 1.5 Large",
484
- "developer": "ai21",
485
- "scores": {
486
- "Mean win rate": 0.637,
487
- "NarrativeQA": 0.664,
488
- "NaturalQuestions (closed-book)": 0.394,
489
- "OpenbookQA": 0.948,
490
- "MMLU": 0.683,
491
- "MATH": 0.692,
492
- "GSM8K": 0.846,
493
- "LegalBench": 0.675,
494
- "MedQA": 0.698,
495
- "WMT 2014": 0.203
496
- }
497
- },
498
- {
499
- "model_id": "ai21/jamba-1.5-mini",
500
- "name": "Jamba 1.5 Mini",
501
- "developer": "ai21",
502
- "scores": {
503
- "Mean win rate": 0.414,
504
- "NarrativeQA": 0.746,
505
- "NaturalQuestions (closed-book)": 0.388,
506
- "OpenbookQA": 0.89,
507
- "MMLU": 0.582,
508
- "MATH": 0.318,
509
- "GSM8K": 0.691,
510
- "LegalBench": 0.503,
511
- "MedQA": 0.632,
512
- "WMT 2014": 0.179
513
- }
514
- },
515
- {
516
- "model_id": "ai21/jamba-instruct",
517
- "name": "Jamba Instruct",
518
- "developer": "ai21",
519
- "scores": {
520
- "Mean win rate": 0.287,
521
- "NarrativeQA": 0.658,
522
- "NaturalQuestions (closed-book)": 0.384,
523
- "OpenbookQA": 0.796,
524
- "MMLU": 0.582,
525
- "MATH": 0.38,
526
- "GSM8K": 0.67,
527
- "LegalBench": 0.54,
528
- "MedQA": 0.519,
529
- "WMT 2014": 0.164
530
- }
531
- },
532
- {
533
- "model_id": "allenai/olmo-7b",
534
- "name": "OLMo 7B",
535
- "developer": "allenai",
536
- "scores": {
537
- "Mean win rate": 0.052,
538
- "NarrativeQA": 0.597,
539
- "NaturalQuestions (closed-book)": 0.259,
540
- "OpenbookQA": 0.222,
541
- "MMLU": 0.305,
542
- "MATH": 0.029,
543
- "GSM8K": 0.044,
544
- "LegalBench": 0.341,
545
- "MedQA": 0.229,
546
- "WMT 2014": 0.097
547
- }
548
- },
549
- {
550
- "model_id": "amazon/nova-lite-v1:0",
551
- "name": "Amazon Nova Lite",
552
- "developer": "amazon",
553
- "scores": {
554
- "Mean win rate": 0.708,
555
- "NarrativeQA": 0.768,
556
- "NaturalQuestions (closed-book)": 0.352,
557
- "OpenbookQA": 0.928,
558
- "MMLU": 0.693,
559
- "MATH": 0.779,
560
- "GSM8K": 0.829,
561
- "LegalBench": 0.659,
562
- "MedQA": 0.696,
563
- "WMT 2014": 0.204
564
- }
565
- },
566
- {
567
- "model_id": "amazon/nova-micro-v1:0",
568
- "name": "Amazon Nova Micro",
569
- "developer": "amazon",
570
- "scores": {
571
- "Mean win rate": 0.524,
572
- "NarrativeQA": 0.744,
573
- "NaturalQuestions (closed-book)": 0.285,
574
- "OpenbookQA": 0.888,
575
- "MMLU": 0.64,
576
- "MATH": 0.76,
577
- "GSM8K": 0.794,
578
- "LegalBench": 0.615,
579
- "MedQA": 0.608,
580
- "WMT 2014": 0.192
581
- }
582
- },
583
- {
584
- "model_id": "amazon/nova-pro-v1:0",
585
- "name": "Amazon Nova Pro",
586
- "developer": "amazon",
587
- "scores": {
588
- "Mean win rate": 0.885,
589
- "NarrativeQA": 0.791,
590
- "NaturalQuestions (closed-book)": 0.405,
591
- "OpenbookQA": 0.96,
592
- "MMLU": 0.758,
593
- "MATH": 0.821,
594
- "GSM8K": 0.87,
595
- "LegalBench": 0.736,
596
- "MedQA": 0.811,
597
- "WMT 2014": 0.229
598
- }
599
- },
600
- {
601
- "model_id": "anthropic/claude-2.0",
602
- "name": "Claude 2.0",
603
- "developer": "Anthropic",
604
- "scores": {
605
- "Mean win rate": 0.489,
606
- "NarrativeQA": 0.718,
607
- "NaturalQuestions (closed-book)": 0.428,
608
- "OpenbookQA": 0.862,
609
- "MMLU": 0.639,
610
- "MATH": 0.603,
611
- "GSM8K": 0.583,
612
- "LegalBench": 0.643,
613
- "MedQA": 0.652,
614
- "WMT 2014": 0.219
615
- }
616
- },
617
- {
618
- "model_id": "anthropic/claude-2.1",
619
- "name": "Claude 2.1",
620
- "developer": "Anthropic",
621
- "scores": {
622
- "Mean win rate": 0.437,
623
- "NarrativeQA": 0.677,
624
- "NaturalQuestions (closed-book)": 0.375,
625
- "OpenbookQA": 0.872,
626
- "MMLU": 0.643,
627
- "MATH": 0.632,
628
- "GSM8K": 0.604,
629
- "LegalBench": 0.643,
630
- "MedQA": 0.644,
631
- "WMT 2014": 0.204
632
- }
633
- },
634
- {
635
- "model_id": "anthropic/claude-3-5-haiku-20241022",
636
- "name": "Claude 3.5 Haiku 20241022",
637
- "developer": "Anthropic",
638
- "scores": {
639
- "Mean win rate": 0.531,
640
- "NarrativeQA": 0.763,
641
- "NaturalQuestions (closed-book)": 0.344,
642
- "OpenbookQA": 0.854,
643
- "MMLU": 0.671,
644
- "MATH": 0.872,
645
- "GSM8K": 0.815,
646
- "LegalBench": 0.631,
647
- "MedQA": 0.722,
648
- "WMT 2014": 0.135
649
- }
650
- },
651
- {
652
- "model_id": "anthropic/claude-3-5-sonnet-20240620",
653
- "name": "Claude 3.5 Sonnet 20240620",
654
- "developer": "Anthropic",
655
- "scores": {
656
- "Mean win rate": 0.885,
657
- "NarrativeQA": 0.746,
658
- "NaturalQuestions (closed-book)": 0.502,
659
- "OpenbookQA": 0.972,
660
- "MMLU": 0.799,
661
- "MATH": 0.813,
662
- "GSM8K": 0.949,
663
- "LegalBench": 0.707,
664
- "MedQA": 0.825,
665
- "WMT 2014": 0.229
666
- }
667
- },
668
- {
669
- "model_id": "anthropic/claude-3-5-sonnet-20241022",
670
- "name": "Claude 3.5 Sonnet 20241022",
671
- "developer": "Anthropic",
672
- "scores": {
673
- "Mean win rate": 0.846,
674
- "NarrativeQA": 0.77,
675
- "NaturalQuestions (closed-book)": 0.467,
676
- "OpenbookQA": 0.966,
677
- "MMLU": 0.809,
678
- "MATH": 0.904,
679
- "GSM8K": 0.956,
680
- "LegalBench": 0.647,
681
- "MedQA": 0.859,
682
- "WMT 2014": 0.226
683
- }
684
- },
685
- {
686
- "model_id": "anthropic/claude-3-haiku-20240307",
687
- "name": "Claude 3 Haiku 20240307",
688
- "developer": "Anthropic",
689
- "scores": {
690
- "Mean win rate": 0.263,
691
- "NarrativeQA": 0.244,
692
- "NaturalQuestions (closed-book)": 0.144,
693
- "OpenbookQA": 0.838,
694
- "MMLU": 0.662,
695
- "MATH": 0.131,
696
- "GSM8K": 0.699,
697
- "LegalBench": 0.46,
698
- "MedQA": 0.702,
699
- "WMT 2014": 0.148
700
- }
701
- },
702
- {
703
- "model_id": "anthropic/claude-3-opus-20240229",
704
- "name": "Claude 3 Opus 20240229",
705
- "developer": "Anthropic",
706
- "scores": {
707
- "Mean win rate": 0.683,
708
- "NarrativeQA": 0.351,
709
- "NaturalQuestions (closed-book)": 0.441,
710
- "OpenbookQA": 0.956,
711
- "MMLU": 0.768,
712
- "MATH": 0.76,
713
- "GSM8K": 0.924,
714
- "LegalBench": 0.662,
715
- "MedQA": 0.775,
716
- "WMT 2014": 0.24
717
- }
718
- },
719
- {
720
- "model_id": "anthropic/claude-3-sonnet-20240229",
721
- "name": "Claude 3 Sonnet 20240229",
722
- "developer": "Anthropic",
723
- "scores": {
724
- "Mean win rate": 0.377,
725
- "NarrativeQA": 0.111,
726
- "NaturalQuestions (closed-book)": 0.028,
727
- "OpenbookQA": 0.918,
728
- "MMLU": 0.652,
729
- "MATH": 0.084,
730
- "GSM8K": 0.907,
731
- "LegalBench": 0.49,
732
- "MedQA": 0.684,
733
- "WMT 2014": 0.218
734
- }
735
- },
736
- {
737
- "model_id": "anthropic/claude-instant-1.2",
738
- "name": "Claude Instant 1.2",
739
- "developer": "Anthropic",
740
- "scores": {
741
- "Mean win rate": 0.399,
742
- "NarrativeQA": 0.616,
743
- "NaturalQuestions (closed-book)": 0.343,
744
- "OpenbookQA": 0.844,
745
- "MMLU": 0.631,
746
- "MATH": 0.499,
747
- "GSM8K": 0.721,
748
- "LegalBench": 0.586,
749
- "MedQA": 0.559,
750
- "WMT 2014": 0.194
751
- }
752
- },
753
- {
754
- "model_id": "anthropic/claude-v1.3",
755
- "name": "Anthropic Claude v1.3",
756
- "developer": "Anthropic",
757
- "scores": {
758
- "Mean win rate": 0.518,
759
- "NarrativeQA": 0.723,
760
- "NaturalQuestions (closed-book)": 0.409,
761
- "OpenbookQA": 0.908,
762
- "MMLU": 0.631,
763
- "MATH": 0.54,
764
- "GSM8K": 0.784,
765
- "LegalBench": 0.629,
766
- "MedQA": 0.618,
767
- "WMT 2014": 0.219
768
- }
769
- },
770
- {
771
- "model_id": "cohere/command",
772
- "name": "Command",
773
- "developer": "cohere",
774
- "scores": {
775
- "Mean win rate": 0.327,
776
- "NarrativeQA": 0.749,
777
- "NaturalQuestions (closed-book)": 0.391,
778
- "OpenbookQA": 0.774,
779
- "MMLU": 0.525,
780
- "MATH": 0.236,
781
- "GSM8K": 0.452,
782
- "LegalBench": 0.578,
783
- "MedQA": 0.445,
784
- "WMT 2014": 0.088
785
- }
786
- },
787
- {
788
- "model_id": "cohere/command-light",
789
- "name": "Command Light",
790
- "developer": "cohere",
791
- "scores": {
792
- "Mean win rate": 0.105,
793
- "NarrativeQA": 0.629,
794
- "NaturalQuestions (closed-book)": 0.195,
795
- "OpenbookQA": 0.398,
796
- "MMLU": 0.386,
797
- "MATH": 0.098,
798
- "GSM8K": 0.149,
799
- "LegalBench": 0.397,
800
- "MedQA": 0.312,
801
- "WMT 2014": 0.023
802
- }
803
- },
804
- {
805
- "model_id": "cohere/command-r",
806
- "name": "Command R",
807
- "developer": "cohere",
808
- "scores": {
809
- "Mean win rate": 0.299,
810
- "NarrativeQA": 0.742,
811
- "NaturalQuestions (closed-book)": 0.352,
812
- "OpenbookQA": 0.782,
813
- "MMLU": 0.567,
814
- "MATH": 0.266,
815
- "GSM8K": 0.551,
816
- "LegalBench": 0.507,
817
- "MedQA": 0.555,
818
- "WMT 2014": 0.149
819
- }
820
- },
821
- {
822
- "model_id": "cohere/command-r-plus",
823
- "name": "Command R Plus",
824
- "developer": "cohere",
825
- "scores": {
826
- "Mean win rate": 0.441,
827
- "NarrativeQA": 0.735,
828
- "NaturalQuestions (closed-book)": 0.343,
829
- "OpenbookQA": 0.828,
830
- "MMLU": 0.59,
831
- "MATH": 0.403,
832
- "GSM8K": 0.738,
833
- "LegalBench": 0.672,
834
- "MedQA": 0.567,
835
- "WMT 2014": 0.203
836
- }
837
- },
838
- {
839
- "model_id": "databricks/dbrx-instruct",
840
- "name": "DBRX Instruct",
841
- "developer": "databricks",
842
- "scores": {
843
- "Mean win rate": 0.289,
844
- "NarrativeQA": 0.488,
845
- "NaturalQuestions (closed-book)": 0.284,
846
- "OpenbookQA": 0.91,
847
- "MMLU": 0.643,
848
- "MATH": 0.358,
849
- "GSM8K": 0.671,
850
- "LegalBench": 0.426,
851
- "MedQA": 0.694,
852
- "WMT 2014": 0.131
853
- }
854
- },
855
- {
856
- "model_id": "deepseek-ai/deepseek-llm-67b-chat",
857
- "name": "DeepSeek LLM Chat 67B",
858
- "developer": "deepseek-ai",
859
- "scores": {
860
- "Mean win rate": 0.488,
861
- "NarrativeQA": 0.581,
862
- "NaturalQuestions (closed-book)": 0.412,
863
- "OpenbookQA": 0.88,
864
- "MMLU": 0.641,
865
- "MATH": 0.615,
866
- "GSM8K": 0.795,
867
- "LegalBench": 0.637,
868
- "MedQA": 0.628,
869
- "WMT 2014": 0.186
870
- }
871
- },
872
- {
873
- "model_id": "deepseek-ai/deepseek-v3",
874
- "name": "DeepSeek v3",
875
- "developer": "deepseek-ai",
876
- "scores": {
877
- "Mean win rate": 0.908,
878
- "NarrativeQA": 0.796,
879
- "NaturalQuestions (closed-book)": 0.467,
880
- "OpenbookQA": 0.954,
881
- "MMLU": 0.803,
882
- "MATH": 0.912,
883
- "GSM8K": 0.94,
884
- "LegalBench": 0.718,
885
- "MedQA": 0.809,
886
- "WMT 2014": 0.209
887
- }
888
- },
889
- {
890
- "model_id": "google/gemini-1.0-pro-002",
891
- "name": "Gemini 1.0 Pro 002",
892
- "developer": "Google",
893
- "scores": {
894
- "Mean win rate": 0.422,
895
- "NarrativeQA": 0.751,
896
- "NaturalQuestions (closed-book)": 0.391,
897
- "OpenbookQA": 0.788,
898
- "MMLU": 0.534,
899
- "MATH": 0.665,
900
- "GSM8K": 0.816,
901
- "LegalBench": 0.475,
902
- "MedQA": 0.483,
903
- "WMT 2014": 0.194
904
- }
905
- },
906
- {
907
- "model_id": "google/gemini-1.5-flash-001",
908
- "name": "Gemini 1.5 Flash 001",
909
- "developer": "Google",
910
- "scores": {
911
- "Mean win rate": 0.667,
912
- "NarrativeQA": 0.783,
913
- "NaturalQuestions (closed-book)": 0.332,
914
- "OpenbookQA": 0.928,
915
- "MMLU": 0.703,
916
- "MATH": 0.753,
917
- "GSM8K": 0.785,
918
- "LegalBench": 0.661,
919
- "MedQA": 0.68,
920
- "WMT 2014": 0.225
921
- }
922
- },
923
- {
924
- "model_id": "google/gemini-1.5-flash-002",
925
- "name": "Gemini 1.5 Flash 002",
926
- "developer": "Google",
927
- "scores": {
928
- "Mean win rate": 0.573,
929
- "NarrativeQA": 0.746,
930
- "NaturalQuestions (closed-book)": 0.323,
931
- "OpenbookQA": 0.914,
932
- "MMLU": 0.679,
933
- "MATH": 0.908,
934
- "GSM8K": 0.328,
935
- "LegalBench": 0.67,
936
- "MedQA": 0.656,
937
- "WMT 2014": 0.212
938
- }
939
- },
940
- {
941
- "model_id": "google/gemini-1.5-pro-001",
942
- "name": "Gemini 1.5 Pro 001",
943
- "developer": "Google",
944
- "scores": {
945
- "Mean win rate": 0.739,
946
- "NarrativeQA": 0.783,
947
- "NaturalQuestions (closed-book)": 0.378,
948
- "OpenbookQA": 0.902,
949
- "MMLU": 0.772,
950
- "MATH": 0.825,
951
- "GSM8K": 0.836,
952
- "LegalBench": 0.757,
953
- "MedQA": 0.692,
954
- "WMT 2014": 0.189
955
- }
956
- },
957
- {
958
- "model_id": "google/gemini-1.5-pro-002",
959
- "name": "Gemini 1.5 Pro 002",
960
- "developer": "Google",
961
- "scores": {
962
- "Mean win rate": 0.842,
963
- "NarrativeQA": 0.756,
964
- "NaturalQuestions (closed-book)": 0.455,
965
- "OpenbookQA": 0.952,
966
- "MMLU": 0.795,
967
- "MATH": 0.92,
968
- "GSM8K": 0.817,
969
- "LegalBench": 0.747,
970
- "MedQA": 0.771,
971
- "WMT 2014": 0.231
972
- }
973
- },
974
- {
975
- "model_id": "google/gemini-2.0-flash-exp",
976
- "name": "Gemini 2.0 Flash Experimental",
977
- "developer": "Google",
978
- "scores": {
979
- "Mean win rate": 0.813,
980
- "NarrativeQA": 0.783,
981
- "NaturalQuestions (closed-book)": 0.443,
982
- "OpenbookQA": 0.946,
983
- "MMLU": 0.717,
984
- "MATH": 0.901,
985
- "GSM8K": 0.946,
986
- "LegalBench": 0.674,
987
- "MedQA": 0.73,
988
- "WMT 2014": 0.212
989
- }
990
- },
991
- {
992
- "model_id": "google/gemma-2-27b-it",
993
- "name": "Gemma 2 Instruct 27B",
994
- "developer": "Google",
995
- "scores": {
996
- "Mean win rate": 0.675,
997
- "NarrativeQA": 0.79,
998
- "NaturalQuestions (closed-book)": 0.353,
999
- "OpenbookQA": 0.918,
1000
- "MMLU": 0.664,
1001
- "MATH": 0.746,
1002
- "GSM8K": 0.812,
1003
- "LegalBench": 0.7,
1004
- "MedQA": 0.684,
1005
- "WMT 2014": 0.214
1006
- }
1007
- },
1008
- {
1009
- "model_id": "google/gemma-2-9b-it",
1010
- "name": "Gemma 2 Instruct 9B",
1011
- "developer": "Google",
1012
- "scores": {
1013
- "Mean win rate": 0.562,
1014
- "NarrativeQA": 0.768,
1015
- "NaturalQuestions (closed-book)": 0.328,
1016
- "OpenbookQA": 0.91,
1017
- "MMLU": 0.645,
1018
- "MATH": 0.724,
1019
- "GSM8K": 0.762,
1020
- "LegalBench": 0.639,
1021
- "MedQA": 0.63,
1022
- "WMT 2014": 0.201
1023
- }
1024
- },
1025
- {
1026
- "model_id": "google/gemma-7b",
1027
- "name": "Gemma 7B",
1028
- "developer": "Google",
1029
- "scores": {
1030
- "Mean win rate": 0.336,
1031
- "NarrativeQA": 0.752,
1032
- "NaturalQuestions (closed-book)": 0.336,
1033
- "OpenbookQA": 0.808,
1034
- "MMLU": 0.571,
1035
- "MATH": 0.5,
1036
- "GSM8K": 0.559,
1037
- "LegalBench": 0.581,
1038
- "MedQA": 0.513,
1039
- "WMT 2014": 0.187
1040
- }
1041
- },
1042
- {
1043
- "model_id": "google/text-bison@001",
1044
- "name": "PaLM-2 Bison",
1045
- "developer": "Google",
1046
- "scores": {
1047
- "Mean win rate": 0.526,
1048
- "NarrativeQA": 0.718,
1049
- "NaturalQuestions (closed-book)": 0.39,
1050
- "OpenbookQA": 0.878,
1051
- "MMLU": 0.608,
1052
- "MATH": 0.421,
1053
- "GSM8K": 0.61,
1054
- "LegalBench": 0.645,
1055
- "MedQA": 0.547,
1056
- "WMT 2014": 0.241
1057
- }
1058
- },
1059
- {
1060
- "model_id": "google/text-unicorn@001",
1061
- "name": "PaLM-2 Unicorn",
1062
- "developer": "Google",
1063
- "scores": {
1064
- "Mean win rate": 0.644,
1065
- "NarrativeQA": 0.583,
1066
- "NaturalQuestions (closed-book)": 0.435,
1067
- "OpenbookQA": 0.938,
1068
- "MMLU": 0.702,
1069
- "MATH": 0.674,
1070
- "GSM8K": 0.831,
1071
- "LegalBench": 0.677,
1072
- "MedQA": 0.684,
1073
- "WMT 2014": 0.26
1074
- }
1075
- },
1076
- {
1077
- "model_id": "meta/LLaMA-65B",
1078
- "name": "LLaMA 65B",
1079
- "developer": "Meta",
1080
- "scores": {
1081
- "Mean win rate": 0.345,
1082
- "NarrativeQA": 0.755,
1083
- "NaturalQuestions (closed-book)": 0.433,
1084
- "OpenbookQA": 0.754,
1085
- "MMLU": 0.584,
1086
- "MATH": 0.257,
1087
- "GSM8K": 0.489,
1088
- "LegalBench": 0.48,
1089
- "MedQA": 0.507,
1090
- "WMT 2014": 0.189
1091
- }
1092
- },
1093
- {
1094
- "model_id": "meta/llama-2-13b",
1095
- "name": "Llama 2 13B",
1096
- "developer": "Meta",
1097
- "scores": {
1098
- "Mean win rate": 0.233,
1099
- "NarrativeQA": 0.741,
1100
- "NaturalQuestions (closed-book)": 0.371,
1101
- "OpenbookQA": 0.634,
1102
- "MMLU": 0.505,
1103
- "MATH": 0.102,
1104
- "GSM8K": 0.266,
1105
- "LegalBench": 0.591,
1106
- "MedQA": 0.392,
1107
- "WMT 2014": 0.167
1108
- }
1109
- },
1110
- {
1111
- "model_id": "meta/llama-2-70b",
1112
- "name": "Llama 2 70B",
1113
- "developer": "Meta",
1114
- "scores": {
1115
- "Mean win rate": 0.482,
1116
- "NarrativeQA": 0.763,
1117
- "NaturalQuestions (closed-book)": 0.46,
1118
- "OpenbookQA": 0.838,
1119
- "MMLU": 0.58,
1120
- "MATH": 0.323,
1121
- "GSM8K": 0.567,
1122
- "LegalBench": 0.673,
1123
- "MedQA": 0.618,
1124
- "WMT 2014": 0.196
1125
- }
1126
- },
1127
- {
1128
- "model_id": "meta/llama-2-7b",
1129
- "name": "Llama 2 7B",
1130
- "developer": "Meta",
1131
- "scores": {
1132
- "Mean win rate": 0.152,
1133
- "NarrativeQA": 0.686,
1134
- "NaturalQuestions (closed-book)": 0.333,
1135
- "OpenbookQA": 0.544,
1136
- "MMLU": 0.425,
1137
- "MATH": 0.097,
1138
- "GSM8K": 0.154,
1139
- "LegalBench": 0.502,
1140
- "MedQA": 0.392,
1141
- "WMT 2014": 0.144
1142
- }
1143
- },
1144
- {
1145
- "model_id": "meta/llama-3-70b",
1146
- "name": "Llama 3 70B",
1147
- "developer": "Meta",
1148
- "scores": {
1149
- "Mean win rate": 0.793,
1150
- "NarrativeQA": 0.798,
1151
- "NaturalQuestions (closed-book)": 0.475,
1152
- "OpenbookQA": 0.934,
1153
- "MMLU": 0.695,
1154
- "MATH": 0.663,
1155
- "GSM8K": 0.805,
1156
- "LegalBench": 0.733,
1157
- "MedQA": 0.777,
1158
- "WMT 2014": 0.225
1159
- }
1160
- },
1161
- {
1162
- "model_id": "meta/llama-3-8b",
1163
- "name": "Llama 3 8B",
1164
- "developer": "Meta",
1165
- "scores": {
1166
- "Mean win rate": 0.387,
1167
- "NarrativeQA": 0.754,
1168
- "NaturalQuestions (closed-book)": 0.378,
1169
- "OpenbookQA": 0.766,
1170
- "MMLU": 0.602,
1171
- "MATH": 0.391,
1172
- "GSM8K": 0.499,
1173
- "LegalBench": 0.637,
1174
- "MedQA": 0.581,
1175
- "WMT 2014": 0.183
1176
- }
1177
- },
1178
- {
1179
- "model_id": "meta/llama-3.1-405b-instruct-turbo",
1180
- "name": "Llama 3.1 Instruct Turbo 405B",
1181
- "developer": "Meta",
1182
- "scores": {
1183
- "Mean win rate": 0.854,
1184
- "NarrativeQA": 0.749,
1185
- "NaturalQuestions (closed-book)": 0.456,
1186
- "OpenbookQA": 0.94,
1187
- "MMLU": 0.759,
1188
- "MATH": 0.827,
1189
- "GSM8K": 0.949,
1190
- "LegalBench": 0.707,
1191
- "MedQA": 0.805,
1192
- "WMT 2014": 0.238
1193
- }
1194
- },
1195
- {
1196
- "model_id": "meta/llama-3.1-70b-instruct-turbo",
1197
- "name": "Llama 3.1 Instruct Turbo 70B",
1198
- "developer": "Meta",
1199
- "scores": {
1200
- "Mean win rate": 0.808,
1201
- "NarrativeQA": 0.772,
1202
- "NaturalQuestions (closed-book)": 0.452,
1203
- "OpenbookQA": 0.938,
1204
- "MMLU": 0.709,
1205
- "MATH": 0.783,
1206
- "GSM8K": 0.938,
1207
- "LegalBench": 0.687,
1208
- "MedQA": 0.769,
1209
- "WMT 2014": 0.223
1210
- }
1211
- },
1212
- {
1213
- "model_id": "meta/llama-3.1-8b-instruct-turbo",
1214
- "name": "Llama 3.1 Instruct Turbo 8B",
1215
- "developer": "Meta",
1216
- "scores": {
1217
- "Mean win rate": 0.303,
1218
- "NarrativeQA": 0.756,
1219
- "NaturalQuestions (closed-book)": 0.209,
1220
- "OpenbookQA": 0.74,
1221
- "MMLU": 0.5,
1222
- "MATH": 0.703,
1223
- "GSM8K": 0.798,
1224
- "LegalBench": 0.342,
1225
- "MedQA": 0.245,
1226
- "WMT 2014": 0.181
1227
- }
1228
- },
1229
- {
1230
- "model_id": "meta/llama-3.2-11b-vision-instruct-turbo",
1231
- "name": "Llama 3.2 Vision Instruct Turbo 11B",
1232
- "developer": "Meta",
1233
- "scores": {
1234
- "Mean win rate": 0.325,
1235
- "NarrativeQA": 0.756,
1236
- "NaturalQuestions (closed-book)": 0.234,
1237
- "OpenbookQA": 0.724,
1238
- "MMLU": 0.511,
1239
- "MATH": 0.739,
1240
- "GSM8K": 0.823,
1241
- "LegalBench": 0.435,
1242
- "MedQA": 0.27,
1243
- "WMT 2014": 0.179
1244
- }
1245
- },
1246
- {
1247
- "model_id": "meta/llama-3.2-90b-vision-instruct-turbo",
1248
- "name": "Llama 3.2 Vision Instruct Turbo 90B",
1249
- "developer": "Meta",
1250
- "scores": {
1251
- "Mean win rate": 0.819,
1252
- "NarrativeQA": 0.777,
1253
- "NaturalQuestions (closed-book)": 0.457,
1254
- "OpenbookQA": 0.942,
1255
- "MMLU": 0.703,
1256
- "MATH": 0.791,
1257
- "GSM8K": 0.936,
1258
- "LegalBench": 0.68,
1259
- "MedQA": 0.769,
1260
- "WMT 2014": 0.224
1261
- }
1262
- },
1263
- {
1264
- "model_id": "meta/llama-3.3-70b-instruct-turbo",
1265
- "name": "Llama 3.3 Instruct Turbo 70B",
1266
- "developer": "Meta",
1267
- "scores": {
1268
- "Mean win rate": 0.812,
1269
- "NarrativeQA": 0.791,
1270
- "NaturalQuestions (closed-book)": 0.431,
1271
- "OpenbookQA": 0.928,
1272
- "MMLU": 0.7,
1273
- "MATH": 0.808,
1274
- "GSM8K": 0.942,
1275
- "LegalBench": 0.725,
1276
- "MedQA": 0.761,
1277
- "WMT 2014": 0.219
1278
- }
1279
- },
1280
- {
1281
- "model_id": "microsoft/phi-2",
1282
- "name": "Phi-2",
1283
- "developer": "microsoft",
1284
- "scores": {
1285
- "Mean win rate": 0.169,
1286
- "NarrativeQA": 0.703,
1287
- "NaturalQuestions (closed-book)": 0.155,
1288
- "OpenbookQA": 0.798,
1289
- "MMLU": 0.518,
1290
- "MATH": 0.255,
1291
- "GSM8K": 0.581,
1292
- "LegalBench": 0.334,
1293
- "MedQA": 0.41,
1294
- "WMT 2014": 0.038
1295
- }
1296
- },
1297
- {
1298
- "model_id": "microsoft/phi-3-medium-4k-instruct",
1299
- "name": "Phi-3 14B",
1300
- "developer": "microsoft",
1301
- "scores": {
1302
- "Mean win rate": 0.509,
1303
- "NarrativeQA": 0.724,
1304
- "NaturalQuestions (closed-book)": 0.278,
1305
- "OpenbookQA": 0.916,
1306
- "MMLU": 0.675,
1307
- "MATH": 0.611,
1308
- "GSM8K": 0.878,
1309
- "LegalBench": 0.593,
1310
- "MedQA": 0.696,
1311
- "WMT 2014": 0.17
1312
- }
1313
- },
1314
- {
1315
- "model_id": "microsoft/phi-3-small-8k-instruct",
1316
- "name": "Phi-3 7B",
1317
- "developer": "microsoft",
1318
- "scores": {
1319
- "Mean win rate": 0.473,
1320
- "NarrativeQA": 0.754,
1321
- "NaturalQuestions (closed-book)": 0.324,
1322
- "OpenbookQA": 0.912,
1323
- "MMLU": 0.659,
1324
- "MATH": 0.703,
1325
- "GSM8K": -1,
1326
- "LegalBench": 0.584,
1327
- "MedQA": 0.672,
1328
- "WMT 2014": 0.154
1329
- }
1330
- },
1331
- {
1332
- "model_id": "mistralai/mistral-7b-instruct-v0.3",
1333
- "name": "Mistral Instruct v0.3 7B",
1334
- "developer": "mistralai",
1335
- "scores": {
1336
- "Mean win rate": 0.196,
1337
- "NarrativeQA": 0.716,
1338
- "NaturalQuestions (closed-book)": 0.253,
1339
- "OpenbookQA": 0.79,
1340
- "MMLU": 0.51,
1341
- "MATH": 0.289,
1342
- "GSM8K": 0.538,
1343
- "LegalBench": 0.331,
1344
- "MedQA": 0.517,
1345
- "WMT 2014": 0.142
1346
- }
1347
- },
1348
- {
1349
- "model_id": "mistralai/mistral-7b-v0.1",
1350
- "name": "Mistral v0.1 7B",
1351
- "developer": "mistralai",
1352
- "scores": {
1353
- "Mean win rate": 0.292,
1354
- "NarrativeQA": 0.716,
1355
- "NaturalQuestions (closed-book)": 0.367,
1356
- "OpenbookQA": 0.776,
1357
- "MMLU": 0.584,
1358
- "MATH": 0.297,
1359
- "GSM8K": 0.377,
1360
- "LegalBench": 0.58,
1361
- "MedQA": 0.525,
1362
- "WMT 2014": 0.16
1363
- }
1364
- },
1365
- {
1366
- "model_id": "mistralai/mistral-large-2402",
1367
- "name": "Mistral Large 2402",
1368
- "developer": "mistralai",
1369
- "scores": {
1370
- "Mean win rate": 0.328,
1371
- "NarrativeQA": 0.454,
1372
- "NaturalQuestions (closed-book)": 0.311,
1373
- "OpenbookQA": 0.894,
1374
- "MMLU": 0.638,
1375
- "MATH": 0.75,
1376
- "GSM8K": 0.694,
1377
- "LegalBench": 0.479,
1378
- "MedQA": 0.499,
1379
- "WMT 2014": 0.182
1380
- }
1381
- },
1382
- {
1383
- "model_id": "mistralai/mistral-large-2407",
1384
- "name": "Mistral Large 2 2407",
1385
- "developer": "mistralai",
1386
- "scores": {
1387
- "Mean win rate": 0.744,
1388
- "NarrativeQA": 0.779,
1389
- "NaturalQuestions (closed-book)": 0.453,
1390
- "OpenbookQA": 0.932,
1391
- "MMLU": 0.725,
1392
- "MATH": 0.677,
1393
- "GSM8K": 0.912,
1394
- "LegalBench": 0.646,
1395
- "MedQA": 0.775,
1396
- "WMT 2014": 0.192
1397
- }
1398
- },
1399
- {
1400
- "model_id": "mistralai/mistral-medium-2312",
1401
- "name": "Mistral Medium 2312",
1402
- "developer": "mistralai",
1403
- "scores": {
1404
- "Mean win rate": 0.268,
1405
- "NarrativeQA": 0.449,
1406
- "NaturalQuestions (closed-book)": 0.29,
1407
- "OpenbookQA": 0.83,
1408
- "MMLU": 0.618,
1409
- "MATH": 0.565,
1410
- "GSM8K": 0.706,
1411
- "LegalBench": 0.452,
1412
- "MedQA": 0.61,
1413
- "WMT 2014": 0.169
1414
- }
1415
- },
1416
- {
1417
- "model_id": "mistralai/mistral-small-2402",
1418
- "name": "Mistral Small 2402",
1419
- "developer": "mistralai",
1420
- "scores": {
1421
- "Mean win rate": 0.288,
1422
- "NarrativeQA": 0.519,
1423
- "NaturalQuestions (closed-book)": 0.304,
1424
- "OpenbookQA": 0.862,
1425
- "MMLU": 0.593,
1426
- "MATH": 0.621,
1427
- "GSM8K": 0.734,
1428
- "LegalBench": 0.389,
1429
- "MedQA": 0.616,
1430
- "WMT 2014": 0.169
1431
- }
1432
- },
1433
- {
1434
- "model_id": "mistralai/mixtral-8x22b",
1435
- "name": "Mixtral 8x22B",
1436
- "developer": "mistralai",
1437
- "scores": {
1438
- "Mean win rate": 0.705,
1439
- "NarrativeQA": 0.779,
1440
- "NaturalQuestions (closed-book)": 0.478,
1441
- "OpenbookQA": 0.882,
1442
- "MMLU": 0.701,
1443
- "MATH": 0.656,
1444
- "GSM8K": 0.8,
1445
- "LegalBench": 0.708,
1446
- "MedQA": 0.704,
1447
- "WMT 2014": 0.209
1448
- }
1449
- },
1450
- {
1451
- "model_id": "mistralai/mixtral-8x7b-32kseqlen",
1452
- "name": "Mixtral 8x7B 32K seqlen",
1453
- "developer": "mistralai",
1454
- "scores": {
1455
- "Mean win rate": 0.51,
1456
- "NarrativeQA": 0.767,
1457
- "NaturalQuestions (closed-book)": 0.427,
1458
- "OpenbookQA": 0.868,
1459
- "MMLU": 0.649,
1460
- "MATH": 0.494,
1461
- "GSM8K": 0.622,
1462
- "LegalBench": 0.63,
1463
- "MedQA": 0.652,
1464
- "WMT 2014": 0.19
1465
- }
1466
- },
1467
- {
1468
- "model_id": "mistralai/open-mistral-nemo-2407",
1469
- "name": "Mistral NeMo 2402",
1470
- "developer": "mistralai",
1471
- "scores": {
1472
- "Mean win rate": 0.333,
1473
- "NarrativeQA": 0.731,
1474
- "NaturalQuestions (closed-book)": 0.265,
1475
- "OpenbookQA": 0.822,
1476
- "MMLU": 0.604,
1477
- "MATH": 0.668,
1478
- "GSM8K": 0.782,
1479
- "LegalBench": 0.415,
1480
- "MedQA": 0.59,
1481
- "WMT 2014": 0.177
1482
- }
1483
- },
1484
- {
1485
- "model_id": "openai/gpt-3.5-turbo-0613",
1486
- "name": "GPT-3.5 Turbo 0613",
1487
- "developer": "OpenAI",
1488
- "scores": {
1489
- "Mean win rate": 0.358,
1490
- "NarrativeQA": 0.655,
1491
- "NaturalQuestions (closed-book)": 0.335,
1492
- "OpenbookQA": 0.838,
1493
- "MMLU": 0.614,
1494
- "MATH": 0.667,
1495
- "GSM8K": 0.501,
1496
- "LegalBench": 0.528,
1497
- "MedQA": 0.622,
1498
- "WMT 2014": 0.187
1499
- }
1500
- },
1501
- {
1502
- "model_id": "openai/gpt-4-0613",
1503
- "name": "GPT-4 0613",
1504
- "developer": "OpenAI",
1505
- "scores": {
1506
- "Mean win rate": 0.867,
1507
- "NarrativeQA": 0.768,
1508
- "NaturalQuestions (closed-book)": 0.457,
1509
- "OpenbookQA": 0.96,
1510
- "MMLU": 0.735,
1511
- "MATH": 0.802,
1512
- "GSM8K": 0.932,
1513
- "LegalBench": 0.713,
1514
- "MedQA": 0.815,
1515
- "WMT 2014": 0.211
1516
- }
1517
- },
1518
- {
1519
- "model_id": "openai/gpt-4-1106-preview",
1520
- "name": "GPT-4 Turbo 1106 preview",
1521
- "developer": "OpenAI",
1522
- "scores": {
1523
- "Mean win rate": 0.698,
1524
- "NarrativeQA": 0.727,
1525
- "NaturalQuestions (closed-book)": 0.435,
1526
- "OpenbookQA": 0.95,
1527
- "MMLU": 0.699,
1528
- "MATH": 0.857,
1529
- "GSM8K": 0.668,
1530
- "LegalBench": 0.626,
1531
- "MedQA": 0.817,
1532
- "WMT 2014": 0.205
1533
- }
1534
- },
1535
- {
1536
- "model_id": "openai/gpt-4-turbo-2024-04-09",
1537
- "name": "GPT-4 Turbo 2024-04-09",
1538
- "developer": "OpenAI",
1539
- "scores": {
1540
- "Mean win rate": 0.864,
1541
- "NarrativeQA": 0.761,
1542
- "NaturalQuestions (closed-book)": 0.482,
1543
- "OpenbookQA": 0.97,
1544
- "MMLU": 0.711,
1545
- "MATH": 0.833,
1546
- "GSM8K": 0.824,
1547
- "LegalBench": 0.727,
1548
- "MedQA": 0.783,
1549
- "WMT 2014": 0.218
1550
- }
1551
- },
1552
- {
1553
- "model_id": "openai/gpt-4o-2024-05-13",
1554
- "name": "GPT-4o 2024-05-13",
1555
- "developer": "OpenAI",
1556
- "scores": {
1557
- "Mean win rate": 0.938,
1558
- "NarrativeQA": 0.804,
1559
- "NaturalQuestions (closed-book)": 0.501,
1560
- "OpenbookQA": 0.966,
1561
- "MMLU": 0.748,
1562
- "MATH": 0.829,
1563
- "GSM8K": 0.905,
1564
- "LegalBench": 0.733,
1565
- "MedQA": 0.857,
1566
- "WMT 2014": 0.231
1567
- }
1568
- },
1569
- {
1570
- "model_id": "openai/gpt-4o-2024-08-06",
1571
- "name": "GPT-4o 2024-08-06",
1572
- "developer": "OpenAI",
1573
- "scores": {
1574
- "Mean win rate": 0.928,
1575
- "NarrativeQA": 0.795,
1576
- "NaturalQuestions (closed-book)": 0.496,
1577
- "OpenbookQA": 0.968,
1578
- "MMLU": 0.738,
1579
- "MATH": 0.853,
1580
- "GSM8K": 0.909,
1581
- "LegalBench": 0.721,
1582
- "MedQA": 0.863,
1583
- "WMT 2014": 0.225
1584
- }
1585
- },
1586
- {
1587
- "model_id": "openai/gpt-4o-mini-2024-07-18",
1588
- "name": "GPT-4o mini 2024-07-18",
1589
- "developer": "OpenAI",
1590
- "scores": {
1591
- "Mean win rate": 0.701,
1592
- "NarrativeQA": 0.768,
1593
- "NaturalQuestions (closed-book)": 0.386,
1594
- "OpenbookQA": 0.92,
1595
- "MMLU": 0.668,
1596
- "MATH": 0.802,
1597
- "GSM8K": 0.843,
1598
- "LegalBench": 0.653,
1599
- "MedQA": 0.748,
1600
- "WMT 2014": 0.206
1601
- }
1602
- },
1603
- {
1604
- "model_id": "openai/text-davinci-002",
1605
- "name": "GPT-3.5 text-davinci-002",
1606
- "developer": "OpenAI",
1607
- "scores": {
1608
- "Mean win rate": 0.336,
1609
- "NarrativeQA": 0.719,
1610
- "NaturalQuestions (closed-book)": 0.394,
1611
- "OpenbookQA": 0.796,
1612
- "MMLU": 0.568,
1613
- "MATH": 0.428,
1614
- "GSM8K": 0.479,
1615
- "LegalBench": 0.58,
1616
- "MedQA": 0.525,
1617
- "WMT 2014": 0.174
1618
- }
1619
- },
1620
- {
1621
- "model_id": "openai/text-davinci-003",
1622
- "name": "GPT-3.5 text-davinci-003",
1623
- "developer": "OpenAI",
1624
- "scores": {
1625
- "Mean win rate": 0.439,
1626
- "NarrativeQA": 0.731,
1627
- "NaturalQuestions (closed-book)": 0.413,
1628
- "OpenbookQA": 0.828,
1629
- "MMLU": 0.555,
1630
- "MATH": 0.449,
1631
- "GSM8K": 0.615,
1632
- "LegalBench": 0.622,
1633
- "MedQA": 0.531,
1634
- "WMT 2014": 0.191
1635
- }
1636
- },
1637
- {
1638
- "model_id": "qwen/qwen1.5-110b-chat",
1639
- "name": "Qwen1.5 Chat 110B",
1640
- "developer": "qwen",
1641
- "scores": {
1642
- "Mean win rate": 0.55,
1643
- "NarrativeQA": 0.721,
1644
- "NaturalQuestions (closed-book)": 0.35,
1645
- "OpenbookQA": 0.922,
1646
- "MMLU": 0.704,
1647
- "MATH": 0.568,
1648
- "GSM8K": 0.815,
1649
- "LegalBench": 0.624,
1650
- "MedQA": 0.64,
1651
- "WMT 2014": 0.192
1652
- }
1653
- },
1654
- {
1655
- "model_id": "qwen/qwen1.5-14b",
1656
- "name": "Qwen1.5 14B",
1657
- "developer": "qwen",
1658
- "scores": {
1659
- "Mean win rate": 0.425,
1660
- "NarrativeQA": 0.711,
1661
- "NaturalQuestions (closed-book)": 0.3,
1662
- "OpenbookQA": 0.862,
1663
- "MMLU": 0.626,
1664
- "MATH": 0.686,
1665
- "GSM8K": 0.693,
1666
- "LegalBench": 0.593,
1667
- "MedQA": 0.515,
1668
- "WMT 2014": 0.178
1669
- }
1670
- },
1671
- {
1672
- "model_id": "qwen/qwen1.5-32b",
1673
- "name": "Qwen1.5 32B",
1674
- "developer": "qwen",
1675
- "scores": {
1676
- "Mean win rate": 0.546,
1677
- "NarrativeQA": 0.589,
1678
- "NaturalQuestions (closed-book)": 0.353,
1679
- "OpenbookQA": 0.932,
1680
- "MMLU": 0.628,
1681
- "MATH": 0.733,
1682
- "GSM8K": 0.773,
1683
- "LegalBench": 0.636,
1684
- "MedQA": 0.656,
1685
- "WMT 2014": 0.193
1686
- }
1687
- },
1688
- {
1689
- "model_id": "qwen/qwen1.5-72b",
1690
- "name": "Qwen1.5 72B",
1691
- "developer": "qwen",
1692
- "scores": {
1693
- "Mean win rate": 0.608,
1694
- "NarrativeQA": 0.601,
1695
- "NaturalQuestions (closed-book)": 0.417,
1696
- "OpenbookQA": 0.93,
1697
- "MMLU": 0.647,
1698
- "MATH": 0.683,
1699
- "GSM8K": 0.799,
1700
- "LegalBench": 0.694,
1701
- "MedQA": 0.67,
1702
- "WMT 2014": 0.201
1703
- }
1704
- },
1705
- {
1706
- "model_id": "qwen/qwen1.5-7b",
1707
- "name": "Qwen1.5 7B",
1708
- "developer": "qwen",
1709
- "scores": {
1710
- "Mean win rate": 0.275,
1711
- "NarrativeQA": 0.448,
1712
- "NaturalQuestions (closed-book)": 0.27,
1713
- "OpenbookQA": 0.806,
1714
- "MMLU": 0.569,
1715
- "MATH": 0.561,
1716
- "GSM8K": 0.6,
1717
- "LegalBench": 0.523,
1718
- "MedQA": 0.479,
1719
- "WMT 2014": 0.153
1720
- }
1721
- },
1722
- {
1723
- "model_id": "qwen/qwen2-72b-instruct",
1724
- "name": "Qwen2 Instruct 72B",
1725
- "developer": "qwen",
1726
- "scores": {
1727
- "Mean win rate": 0.77,
1728
- "NarrativeQA": 0.727,
1729
- "NaturalQuestions (closed-book)": 0.39,
1730
- "OpenbookQA": 0.954,
1731
- "MMLU": 0.769,
1732
- "MATH": 0.79,
1733
- "GSM8K": 0.92,
1734
- "LegalBench": 0.712,
1735
- "MedQA": 0.746,
1736
- "WMT 2014": 0.207
1737
- }
1738
- },
1739
- {
1740
- "model_id": "qwen/qwen2.5-72b-instruct-turbo",
1741
- "name": "Qwen2.5 Instruct Turbo 72B",
1742
- "developer": "qwen",
1743
- "scores": {
1744
- "Mean win rate": 0.745,
1745
- "NarrativeQA": 0.745,
1746
- "NaturalQuestions (closed-book)": 0.359,
1747
- "OpenbookQA": 0.962,
1748
- "MMLU": 0.77,
1749
- "MATH": 0.884,
1750
- "GSM8K": 0.9,
1751
- "LegalBench": 0.74,
1752
- "MedQA": 0.753,
1753
- "WMT 2014": 0.207
1754
- }
1755
- },
1756
- {
1757
- "model_id": "qwen/qwen2.5-7b-instruct-turbo",
1758
- "name": "Qwen2.5 Instruct Turbo 7B",
1759
- "developer": "qwen",
1760
- "scores": {
1761
- "Mean win rate": 0.488,
1762
- "NarrativeQA": 0.742,
1763
- "NaturalQuestions (closed-book)": 0.205,
1764
- "OpenbookQA": 0.862,
1765
- "MMLU": 0.658,
1766
- "MATH": 0.835,
1767
- "GSM8K": 0.83,
1768
- "LegalBench": 0.632,
1769
- "MedQA": 0.6,
1770
- "WMT 2014": 0.155
1771
- }
1772
- },
1773
- {
1774
- "model_id": "snowflake/snowflake-arctic-instruct",
1775
- "name": "Arctic Instruct",
1776
- "developer": "snowflake",
1777
- "scores": {
1778
- "Mean win rate": 0.338,
1779
- "NarrativeQA": 0.654,
1780
- "NaturalQuestions (closed-book)": 0.39,
1781
- "OpenbookQA": 0.828,
1782
- "MMLU": 0.575,
1783
- "MATH": 0.519,
1784
- "GSM8K": 0.768,
1785
- "LegalBench": 0.588,
1786
- "MedQA": 0.581,
1787
- "WMT 2014": 0.172
1788
- }
1789
- },
1790
- {
1791
- "model_id": "tiiuae/falcon-40b",
1792
- "name": "Falcon 40B",
1793
- "developer": "tiiuae",
1794
- "scores": {
1795
- "Mean win rate": 0.217,
1796
- "NarrativeQA": 0.671,
1797
- "NaturalQuestions (closed-book)": 0.392,
1798
- "OpenbookQA": 0.662,
1799
- "MMLU": 0.507,
1800
- "MATH": 0.128,
1801
- "GSM8K": 0.267,
1802
- "LegalBench": 0.442,
1803
- "MedQA": 0.419,
1804
- "WMT 2014": 0.162
1805
- }
1806
- },
1807
- {
1808
- "model_id": "tiiuae/falcon-7b",
1809
- "name": "Falcon 7B",
1810
- "developer": "tiiuae",
1811
- "scores": {
1812
- "Mean win rate": 0.064,
1813
- "NarrativeQA": 0.621,
1814
- "NaturalQuestions (closed-book)": 0.285,
1815
- "OpenbookQA": 0.26,
1816
- "MMLU": 0.288,
1817
- "MATH": 0.044,
1818
- "GSM8K": 0.055,
1819
- "LegalBench": 0.346,
1820
- "MedQA": 0.254,
1821
- "WMT 2014": 0.094
1822
- }
1823
- },
1824
- {
1825
- "model_id": "upstage/solar-pro-241126",
1826
- "name": "Solar Pro",
1827
- "developer": "upstage",
1828
- "scores": {
1829
- "Mean win rate": 0.602,
1830
- "NarrativeQA": 0.753,
1831
- "NaturalQuestions (closed-book)": 0.297,
1832
- "OpenbookQA": 0.922,
1833
- "MMLU": 0.679,
1834
- "MATH": 0.567,
1835
- "GSM8K": 0.871,
1836
- "LegalBench": 0.67,
1837
- "MedQA": 0.698,
1838
- "WMT 2014": 0.169
1839
- }
1840
- },
1841
- {
1842
- "model_id": "writer/palmyra-x-004",
1843
- "name": "Palmyra-X-004",
1844
- "developer": "writer",
1845
- "scores": {
1846
- "Mean win rate": 0.808,
1847
- "NarrativeQA": 0.773,
1848
- "NaturalQuestions (closed-book)": 0.457,
1849
- "OpenbookQA": 0.926,
1850
- "MMLU": 0.739,
1851
- "MATH": 0.767,
1852
- "GSM8K": 0.905,
1853
- "LegalBench": 0.73,
1854
- "MedQA": 0.775,
1855
- "WMT 2014": 0.203
1856
- }
1857
- },
1858
- {
1859
- "model_id": "writer/palmyra-x-v2",
1860
- "name": "Palmyra X V2 33B",
1861
- "developer": "writer",
1862
- "scores": {
1863
- "Mean win rate": 0.589,
1864
- "NarrativeQA": 0.752,
1865
- "NaturalQuestions (closed-book)": 0.428,
1866
- "OpenbookQA": 0.878,
1867
- "MMLU": 0.621,
1868
- "MATH": 0.58,
1869
- "GSM8K": 0.735,
1870
- "LegalBench": 0.644,
1871
- "MedQA": 0.598,
1872
- "WMT 2014": 0.239
1873
- }
1874
- },
1875
- {
1876
- "model_id": "writer/palmyra-x-v3",
1877
- "name": "Palmyra X V3 72B",
1878
- "developer": "writer",
1879
- "scores": {
1880
- "Mean win rate": 0.679,
1881
- "NarrativeQA": 0.706,
1882
- "NaturalQuestions (closed-book)": 0.407,
1883
- "OpenbookQA": 0.938,
1884
- "MMLU": 0.702,
1885
- "MATH": 0.723,
1886
- "GSM8K": 0.831,
1887
- "LegalBench": 0.709,
1888
- "MedQA": 0.684,
1889
- "WMT 2014": 0.262
1890
- }
1891
- }
1892
- ]
1893
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/helm_mmlu.json DELETED
The diff for this file is too large to render. See raw diff
 
data/benchmarks/hfopenllm_v2.json DELETED
The diff for this file is too large to render. See raw diff
 
data/benchmarks/la_leaderboard.json DELETED
@@ -1,44 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "Qwen/Qwen2.5-7B",
5
- "name": "Qwen2.5-7B",
6
- "developer": "Qwen",
7
- "scores": {
8
- "la_leaderboard": 27.61
9
- }
10
- },
11
- {
12
- "model_id": "google/gemma-2-9b-it",
13
- "name": "Gemma 2 Instruct 9B",
14
- "developer": "Google",
15
- "scores": {
16
- "la_leaderboard": 33.62
17
- }
18
- },
19
- {
20
- "model_id": "meta-llama/Meta-Llama-3.1-8B",
21
- "name": "Meta Llama 3.1 8B",
22
- "developer": "unknown",
23
- "scores": {
24
- "la_leaderboard": 27.04
25
- }
26
- },
27
- {
28
- "model_id": "meta-llama/Meta-Llama-3.1-8B-Instruct",
29
- "name": "Meta Llama 3.1 8B Instruct",
30
- "developer": "unknown",
31
- "scores": {
32
- "la_leaderboard": 30.23
33
- }
34
- },
35
- {
36
- "model_id": "utter-project/EuroLLM-9B",
37
- "name": "EuroLLM 9B",
38
- "developer": "unknown",
39
- "scores": {
40
- "la_leaderboard": 25.87
41
- }
42
- }
43
- ]
44
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/livecodebenchpro.json DELETED
@@ -1,274 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "alibaba/qwen3-235b-a22b-thinking-2507",
5
- "name": "qwen3-235b-a22b-thinking-2507",
6
- "developer": "Alibaba",
7
- "scores": {
8
- "Hard Problems": 0.0,
9
- "Medium Problems": 0.1267605633802817,
10
- "Easy Problems": 0.7605633802816901
11
- }
12
- },
13
- {
14
- "model_id": "alibaba/qwen3-30b-a3b",
15
- "name": "qwen3-30b-a3b",
16
- "developer": "Alibaba",
17
- "scores": {
18
- "Hard Problems": 0.0,
19
- "Medium Problems": 0.028169014084507043,
20
- "Easy Problems": 0.5774647887323944
21
- }
22
- },
23
- {
24
- "model_id": "alibaba/qwen3-max",
25
- "name": "alibaba/qwen3-max",
26
- "developer": "Alibaba",
27
- "scores": {
28
- "Hard Problems": 0.0,
29
- "Medium Problems": 0.04225352112676056,
30
- "Easy Problems": 0.36619718309859156
31
- }
32
- },
33
- {
34
- "model_id": "alibaba/qwen3-next-80b-a3b-thinking",
35
- "name": "qwen3-next-80b-a3b-thinking",
36
- "developer": "Alibaba",
37
- "scores": {
38
- "Hard Problems": 0.0,
39
- "Medium Problems": 0.14084507042253522,
40
- "Easy Problems": 0.7464788732394366
41
- }
42
- },
43
- {
44
- "model_id": "aliyun/qwen3-next-80b-a3b-thinking",
45
- "name": "qwen3-next-80b-a3b-thinking",
46
- "developer": "aliyun",
47
- "scores": {
48
- "Hard Problems": 0.0,
49
- "Medium Problems": 0.0704,
50
- "Easy Problems": 0.6901
51
- }
52
- },
53
- {
54
- "model_id": "anthropic/claude-3-7-sonnet-20250219",
55
- "name": "claude-3-7-sonnet-20250219",
56
- "developer": "Anthropic",
57
- "scores": {
58
- "Hard Problems": 0.0,
59
- "Medium Problems": 0.0,
60
- "Easy Problems": 0.28169014084507044
61
- }
62
- },
63
- {
64
- "model_id": "anthropic/claude-3.7-sonnet",
65
- "name": "anthropic/claude-3.7-sonnet",
66
- "developer": "Anthropic",
67
- "scores": {
68
- "Hard Problems": 0.0,
69
- "Medium Problems": 0.014084507042253521,
70
- "Easy Problems": 0.15492957746478872
71
- }
72
- },
73
- {
74
- "model_id": "anthropic/claude-sonnet-4-5-20250929",
75
- "name": "claude-sonnet-4-5-20250929",
76
- "developer": "Anthropic",
77
- "scores": {
78
- "Hard Problems": 0.0,
79
- "Medium Problems": 0.0,
80
- "Easy Problems": 0.5352
81
- }
82
- },
83
- {
84
- "model_id": "ark/ep-20250603132404-cgpjm",
85
- "name": "ep-20250603132404-cgpjm",
86
- "developer": "ark",
87
- "scores": {
88
- "Hard Problems": 0.0,
89
- "Medium Problems": 0.0141,
90
- "Easy Problems": 0.507
91
- }
92
- },
93
- {
94
- "model_id": "bytedance/doubao-seed-1-6-thinking-250615",
95
- "name": "doubao-seed-1-6-thinking-250615",
96
- "developer": "ByteDance",
97
- "scores": {
98
- "Hard Problems": 0.0,
99
- "Medium Problems": 0.07042253521126761,
100
- "Easy Problems": 0.5774647887323944
101
- }
102
- },
103
- {
104
- "model_id": "deepseek/chat-v3-0324",
105
- "name": "deepseek/chat-v3-0324",
106
- "developer": "DeepSeek",
107
- "scores": {
108
- "Hard Problems": 0.0,
109
- "Medium Problems": 0.0,
110
- "Easy Problems": 0.19718309859154928
111
- }
112
- },
113
- {
114
- "model_id": "deepseek/ep-20250214004308-p7n89",
115
- "name": "ep-20250214004308-p7n89",
116
- "developer": "DeepSeek",
117
- "scores": {
118
- "Hard Problems": 0.0,
119
- "Medium Problems": 0.014084507042253521,
120
- "Easy Problems": 0.4225352112676056
121
- }
122
- },
123
- {
124
- "model_id": "deepseek/ep-20250228232227-z44x5",
125
- "name": "ep-20250228232227-z44x5",
126
- "developer": "DeepSeek",
127
- "scores": {
128
- "Hard Problems": 0.0,
129
- "Medium Problems": 0.0,
130
- "Easy Problems": 0.1267605633802817
131
- }
132
- },
133
- {
134
- "model_id": "deepseek/ep-20250603132404-cgpjm",
135
- "name": "ep-20250603132404-cgpjm",
136
- "developer": "DeepSeek",
137
- "scores": {
138
- "Hard Problems": 0.0,
139
- "Medium Problems": 0.08450704225352113,
140
- "Easy Problems": 0.5774647887323944
141
- }
142
- },
143
- {
144
- "model_id": "google/gemini-2.5-flash",
145
- "name": "Gemini 2.5 Flash",
146
- "developer": "Google",
147
- "scores": {
148
- "Hard Problems": 0.0,
149
- "Medium Problems": 0.028169014084507043,
150
- "Easy Problems": 0.38028169014084506
151
- }
152
- },
153
- {
154
- "model_id": "google/gemini-2.5-pro",
155
- "name": "Gemini 2.5 Pro",
156
- "developer": "Google",
157
- "scores": {
158
- "Hard Problems": 0.014084507042253521,
159
- "Medium Problems": 0.2112676056338028,
160
- "Easy Problems": 0.7183098591549296
161
- }
162
- },
163
- {
164
- "model_id": "kuaishou/kwaipilot-40b-0604",
165
- "name": "kwaipilot-40b-0604",
166
- "developer": "Kuaishou",
167
- "scores": {
168
- "Hard Problems": 0.0,
169
- "Medium Problems": 0.07042253521126761,
170
- "Easy Problems": 0.056338028169014086
171
- }
172
- },
173
- {
174
- "model_id": "meta/llama-4-maverick",
175
- "name": "meta/llama-4-maverick",
176
- "developer": "Meta",
177
- "scores": {
178
- "Hard Problems": 0.0,
179
- "Medium Problems": 0.0,
180
- "Easy Problems": 0.09859154929577464
181
- }
182
- },
183
- {
184
- "model_id": "openai/gpt-4.1",
185
- "name": "openai/gpt-4.1",
186
- "developer": "OpenAI",
187
- "scores": {
188
- "Hard Problems": 0.0,
189
- "Medium Problems": 0.0,
190
- "Easy Problems": 0.19718309859154928
191
- }
192
- },
193
- {
194
- "model_id": "openai/gpt-4o-2024-11-20",
195
- "name": "GPT-4o 2024-11-20",
196
- "developer": "OpenAI",
197
- "scores": {
198
- "Hard Problems": 0.0,
199
- "Medium Problems": 0.0,
200
- "Easy Problems": 0.07042253521126761
201
- }
202
- },
203
- {
204
- "model_id": "openai/gpt-5-2025-08-07",
205
- "name": "gpt-5-2025-08-07",
206
- "developer": "OpenAI",
207
- "scores": {
208
- "Hard Problems": 0.0423,
209
- "Medium Problems": 0.4085,
210
- "Easy Problems": 0.9014
211
- }
212
- },
213
- {
214
- "model_id": "openai/gpt-5.2-2025-12-11",
215
- "name": "gpt-5.2-2025-12-11",
216
- "developer": "OpenAI",
217
- "scores": {
218
- "Hard Problems": 0.1594,
219
- "Medium Problems": 0.5211,
220
- "Easy Problems": 0.9014
221
- }
222
- },
223
- {
224
- "model_id": "openai/gpt-oss-120b",
225
- "name": "GPT-OSS-120B",
226
- "developer": "OpenAI",
227
- "scores": {
228
- "Hard Problems": 0.0,
229
- "Medium Problems": 0.11267605633802817,
230
- "Easy Problems": 0.6619718309859155
231
- }
232
- },
233
- {
234
- "model_id": "openai/gpt-oss-20b",
235
- "name": "GPT-OSS-20B",
236
- "developer": "OpenAI",
237
- "scores": {
238
- "Hard Problems": 0.0,
239
- "Medium Problems": 0.056338028169014086,
240
- "Easy Problems": 0.5070422535211268
241
- }
242
- },
243
- {
244
- "model_id": "openai/o3-2025-04-16",
245
- "name": "o3-2025-04-16",
246
- "developer": "OpenAI",
247
- "scores": {
248
- "Hard Problems": 0.0,
249
- "Medium Problems": 0.22535211267605634,
250
- "Easy Problems": 0.7183098591549296
251
- }
252
- },
253
- {
254
- "model_id": "openai/o4-mini-2025-04-16",
255
- "name": "o4-mini-2025-04-16",
256
- "developer": "OpenAI",
257
- "scores": {
258
- "Hard Problems": 0.0143,
259
- "Medium Problems": 0.2923,
260
- "Easy Problems": 0.8571
261
- }
262
- },
263
- {
264
- "model_id": "z-ai/glm-4.5",
265
- "name": "z-ai/glm-4.5",
266
- "developer": "Z.ai",
267
- "scores": {
268
- "Hard Problems": 0.0,
269
- "Medium Problems": 0.028169014084507043,
270
- "Easy Problems": 0.1267605633802817
271
- }
272
- }
273
- ]
274
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/reward-bench.json DELETED
The diff for this file is too large to render. See raw diff
 
data/benchmarks/swe-bench.json DELETED
@@ -1,28 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "anthropic/claude-opus-4-5",
5
- "name": "claude-opus-4-5",
6
- "developer": "Anthropic",
7
- "scores": {
8
- "swe-bench": 0.65
9
- }
10
- },
11
- {
12
- "model_id": "google/gemini-3-pro-preview",
13
- "name": "gemini-3-pro-preview",
14
- "developer": "Google",
15
- "scores": {
16
- "swe-bench": 0.71
17
- }
18
- },
19
- {
20
- "model_id": "openai/gpt-5.2-2025-12-11",
21
- "name": "gpt-5.2-2025-12-11",
22
- "developer": "OpenAI",
23
- "scores": {
24
- "swe-bench": 0.57
25
- }
26
- }
27
- ]
28
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/tau-bench-2_airline.json DELETED
@@ -1,28 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "anthropic/claude-opus-4-5",
5
- "name": "claude-opus-4-5",
6
- "developer": "Anthropic",
7
- "scores": {
8
- "tau-bench-2/airline": 0.66
9
- }
10
- },
11
- {
12
- "model_id": "google/gemini-3-pro-preview",
13
- "name": "gemini-3-pro-preview",
14
- "developer": "Google",
15
- "scores": {
16
- "tau-bench-2/airline": 0.62
17
- }
18
- },
19
- {
20
- "model_id": "openai/gpt-5.2-2025-12-11",
21
- "name": "gpt-5.2-2025-12-11",
22
- "developer": "OpenAI",
23
- "scores": {
24
- "tau-bench-2/airline": 0.54
25
- }
26
- }
27
- ]
28
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/tau-bench-2_retail.json DELETED
@@ -1,28 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "anthropic/claude-opus-4-5",
5
- "name": "claude-opus-4-5",
6
- "developer": "Anthropic",
7
- "scores": {
8
- "tau-bench-2/retail": 0.78
9
- }
10
- },
11
- {
12
- "model_id": "google/gemini-3-pro-preview",
13
- "name": "gemini-3-pro-preview",
14
- "developer": "Google",
15
- "scores": {
16
- "tau-bench-2/retail": 0.7576
17
- }
18
- },
19
- {
20
- "model_id": "openai/gpt-5.2-2025-12-11",
21
- "name": "gpt-5.2-2025-12-11",
22
- "developer": "OpenAI",
23
- "scores": {
24
- "tau-bench-2/retail": 0.68
25
- }
26
- }
27
- ]
28
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/tau-bench-2_telecom.json DELETED
@@ -1,28 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "anthropic/claude-opus-4-5",
5
- "name": "claude-opus-4-5",
6
- "developer": "Anthropic",
7
- "scores": {
8
- "tau-bench-2/telecom": 0.84
9
- }
10
- },
11
- {
12
- "model_id": "google/gemini-3-pro-preview",
13
- "name": "gemini-3-pro-preview",
14
- "developer": "Google",
15
- "scores": {
16
- "tau-bench-2/telecom": 0.73
17
- }
18
- },
19
- {
20
- "model_id": "openai/gpt-5.2-2025-12-11",
21
- "name": "gpt-5.2-2025-12-11",
22
- "developer": "OpenAI",
23
- "scores": {
24
- "tau-bench-2/telecom": 0.5354
25
- }
26
- }
27
- ]
28
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/terminal-bench-2.0.json DELETED
@@ -1,300 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "alibaba/qwen-3-coder-480b",
5
- "name": "Qwen 3 Coder 480B",
6
- "developer": "Alibaba",
7
- "scores": {
8
- "terminal-bench-2.0": 23.9
9
- }
10
- },
11
- {
12
- "model_id": "anthropic/claude-haiku-4.5",
13
- "name": "Claude Haiku 4.5",
14
- "developer": "Anthropic",
15
- "scores": {
16
- "terminal-bench-2.0": 35.5
17
- }
18
- },
19
- {
20
- "model_id": "anthropic/claude-opus-4.1",
21
- "name": "Claude Opus 4.1",
22
- "developer": "Anthropic",
23
- "scores": {
24
- "terminal-bench-2.0": 35.1
25
- }
26
- },
27
- {
28
- "model_id": "anthropic/claude-opus-4.5",
29
- "name": "Claude Opus 4.5",
30
- "developer": "Anthropic",
31
- "scores": {
32
- "terminal-bench-2.0": 63.1
33
- }
34
- },
35
- {
36
- "model_id": "anthropic/claude-opus-4.6",
37
- "name": "Claude Opus 4.6",
38
- "developer": "Anthropic",
39
- "scores": {
40
- "terminal-bench-2.0": 74.7
41
- }
42
- },
43
- {
44
- "model_id": "anthropic/claude-sonnet-4.5",
45
- "name": "Claude Sonnet 4.5",
46
- "developer": "Anthropic",
47
- "scores": {
48
- "terminal-bench-2.0": 42.5
49
- }
50
- },
51
- {
52
- "model_id": "deepseek/deepseek-v3.2",
53
- "name": "DeepSeek-V3.2",
54
- "developer": "DeepSeek",
55
- "scores": {
56
- "terminal-bench-2.0": 39.6
57
- }
58
- },
59
- {
60
- "model_id": "google/gemini-2.5-flash",
61
- "name": "Gemini 2.5 Flash",
62
- "developer": "Google",
63
- "scores": {
64
- "terminal-bench-2.0": 16.9
65
- }
66
- },
67
- {
68
- "model_id": "google/gemini-2.5-pro",
69
- "name": "Gemini 2.5 Pro",
70
- "developer": "Google",
71
- "scores": {
72
- "terminal-bench-2.0": 19.6
73
- }
74
- },
75
- {
76
- "model_id": "google/gemini-3-flash",
77
- "name": "Gemini 3 Flash",
78
- "developer": "Google",
79
- "scores": {
80
- "terminal-bench-2.0": 47.4
81
- }
82
- },
83
- {
84
- "model_id": "google/gemini-3-pro",
85
- "name": "Gemini 3 Pro",
86
- "developer": "Google",
87
- "scores": {
88
- "terminal-bench-2.0": 56.9
89
- }
90
- },
91
- {
92
- "model_id": "google/gemini-3.1-pro",
93
- "name": "Gemini 3.1 Pro",
94
- "developer": "Google",
95
- "scores": {
96
- "terminal-bench-2.0": 74.8
97
- }
98
- },
99
- {
100
- "model_id": "minimax/minimax-m2",
101
- "name": "MiniMax M2",
102
- "developer": "MiniMax",
103
- "scores": {
104
- "terminal-bench-2.0": 30.0
105
- }
106
- },
107
- {
108
- "model_id": "minimax/minimax-m2.1",
109
- "name": "MiniMax M2.1",
110
- "developer": "MiniMax",
111
- "scores": {
112
- "terminal-bench-2.0": 29.2
113
- }
114
- },
115
- {
116
- "model_id": "minimax/minimax-m2.5",
117
- "name": "Minimax m2.5",
118
- "developer": "Minimax",
119
- "scores": {
120
- "terminal-bench-2.0": 42.2
121
- }
122
- },
123
- {
124
- "model_id": "moonshot-ai/kimi-k2-instruct",
125
- "name": "Kimi K2 Instruct",
126
- "developer": "Moonshot AI",
127
- "scores": {
128
- "terminal-bench-2.0": 27.8
129
- }
130
- },
131
- {
132
- "model_id": "moonshot-ai/kimi-k2-thinking",
133
- "name": "Kimi K2 Thinking",
134
- "developer": "Moonshot AI",
135
- "scores": {
136
- "terminal-bench-2.0": 35.7
137
- }
138
- },
139
- {
140
- "model_id": "moonshot-ai/kimi-k2.5",
141
- "name": "Kimi K2.5",
142
- "developer": "Kimi",
143
- "scores": {
144
- "terminal-bench-2.0": 43.2
145
- }
146
- },
147
- {
148
- "model_id": "multiple/multiple",
149
- "name": "Multiple",
150
- "developer": "Multiple",
151
- "scores": {
152
- "terminal-bench-2.0": 59.1
153
- }
154
- },
155
- {
156
- "model_id": "openai/gpt-5",
157
- "name": "GPT-5",
158
- "developer": "OpenAI",
159
- "scores": {
160
- "terminal-bench-2.0": 49.6
161
- }
162
- },
163
- {
164
- "model_id": "openai/gpt-5-codex",
165
- "name": "GPT-5-Codex",
166
- "developer": "OpenAI",
167
- "scores": {
168
- "terminal-bench-2.0": 44.3
169
- }
170
- },
171
- {
172
- "model_id": "openai/gpt-5-mini",
173
- "name": "GPT-5-Mini",
174
- "developer": "OpenAI",
175
- "scores": {
176
- "terminal-bench-2.0": 31.9
177
- }
178
- },
179
- {
180
- "model_id": "openai/gpt-5-nano",
181
- "name": "GPT-5-Nano",
182
- "developer": "OpenAI",
183
- "scores": {
184
- "terminal-bench-2.0": 7.0
185
- }
186
- },
187
- {
188
- "model_id": "openai/gpt-5.1",
189
- "name": "GPT-5.1",
190
- "developer": "OpenAI",
191
- "scores": {
192
- "terminal-bench-2.0": 47.6
193
- }
194
- },
195
- {
196
- "model_id": "openai/gpt-5.1-codex",
197
- "name": "GPT-5.1-Codex",
198
- "developer": "OpenAI",
199
- "scores": {
200
- "terminal-bench-2.0": 53.5
201
- }
202
- },
203
- {
204
- "model_id": "openai/gpt-5.1-codex-max",
205
- "name": "GPT-5.1-Codex-Max",
206
- "developer": "OpenAI",
207
- "scores": {
208
- "terminal-bench-2.0": 60.4
209
- }
210
- },
211
- {
212
- "model_id": "openai/gpt-5.1-codex-mini",
213
- "name": "GPT-5.1-Codex-Mini",
214
- "developer": "OpenAI",
215
- "scores": {
216
- "terminal-bench-2.0": 43.1
217
- }
218
- },
219
- {
220
- "model_id": "openai/gpt-5.2",
221
- "name": "GPT-5.2",
222
- "developer": "OpenAI",
223
- "scores": {
224
- "terminal-bench-2.0": 64.9
225
- }
226
- },
227
- {
228
- "model_id": "openai/gpt-5.2-codex",
229
- "name": "GPT-5.2-Codex",
230
- "developer": "OpenAI",
231
- "scores": {
232
- "terminal-bench-2.0": 66.5
233
- }
234
- },
235
- {
236
- "model_id": "openai/gpt-5.3-codex",
237
- "name": "GPT-5.3-Codex",
238
- "developer": "OpenAI",
239
- "scores": {
240
- "terminal-bench-2.0": 74.6
241
- }
242
- },
243
- {
244
- "model_id": "openai/gpt-oss-120b",
245
- "name": "GPT-OSS-120B",
246
- "developer": "OpenAI",
247
- "scores": {
248
- "terminal-bench-2.0": 18.7
249
- }
250
- },
251
- {
252
- "model_id": "openai/gpt-oss-20b",
253
- "name": "GPT-OSS-20B",
254
- "developer": "OpenAI",
255
- "scores": {
256
- "terminal-bench-2.0": 3.4
257
- }
258
- },
259
- {
260
- "model_id": "xai/grok-4",
261
- "name": "Grok 4",
262
- "developer": "xAI",
263
- "scores": {
264
- "terminal-bench-2.0": 23.1
265
- }
266
- },
267
- {
268
- "model_id": "xai/grok-code-fast-1",
269
- "name": "Grok Code Fast 1",
270
- "developer": "xAI",
271
- "scores": {
272
- "terminal-bench-2.0": 14.2
273
- }
274
- },
275
- {
276
- "model_id": "zhipu-ai/glm-4.6",
277
- "name": "GLM 4.6",
278
- "developer": "Z.ai",
279
- "scores": {
280
- "terminal-bench-2.0": 24.5
281
- }
282
- },
283
- {
284
- "model_id": "zhipu-ai/glm-4.7",
285
- "name": "GLM 4.7",
286
- "developer": "Z-AI",
287
- "scores": {
288
- "terminal-bench-2.0": 33.3
289
- }
290
- },
291
- {
292
- "model_id": "zhipu-ai/glm-5",
293
- "name": "GLM 5",
294
- "developer": "Z-AI",
295
- "scores": {
296
- "terminal-bench-2.0": 52.4
297
- }
298
- }
299
- ]
300
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
data/benchmarks/theory_of_mind.json DELETED
@@ -1,12 +0,0 @@
1
- {
2
- "models": [
3
- {
4
- "model_id": "Qwen/Qwen2.5-3B-Instruct",
5
- "name": "Qwen2.5-3B-Instruct",
6
- "developer": "Qwen",
7
- "scores": {
8
- "accuracy on theory_of_mind for scorer model_graded_fact": 0.78
9
- }
10
- }
11
- ]
12
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
lib/benchmark-metadata.ts CHANGED
@@ -1,98 +1,66 @@
1
  import "server-only"
2
 
3
- import { promises as fs, type Dirent } from "fs"
4
- import path from "path"
5
-
6
  import type { BenchmarkCard } from "@/lib/benchmark-schema"
7
- import { normalizeBenchmarkKey, candidateBenchmarkKeys as candidateKeys } from "@/lib/benchmark-metadata-utils"
 
8
 
9
  export { normalizeBenchmarkKey }
10
 
11
- interface IndexedBenchmarkDetailFile {
12
- benchmark_cards?: Record<string, BenchmarkCard>
13
- }
14
-
15
- function getBenchmarkDataDirectory() {
16
- return path.join(process.cwd(), "data", "benchmarks")
17
- }
18
 
19
- async function readEmbeddedBenchmarkCards(): Promise<Map<string, BenchmarkCard>> {
20
- const dir = getBenchmarkDataDirectory()
21
  const map = new Map<string, BenchmarkCard>()
22
 
23
- let entries: Dirent[]
24
- try {
25
- entries = await fs.readdir(dir, { withFileTypes: true })
26
- } catch {
27
- return map
28
- }
29
 
30
- const jsonFiles = entries.filter((entry) => entry.isFile() && entry.name.endsWith(".json"))
31
-
32
- await Promise.all(
33
- jsonFiles.map(async (entry) => {
34
- try {
35
- const raw = await fs.readFile(path.join(dir, entry.name), "utf8")
36
- const parsed = JSON.parse(raw) as IndexedBenchmarkDetailFile
37
- const embeddedCards = parsed.benchmark_cards
38
-
39
- if (!embeddedCards || typeof embeddedCards !== "object") {
40
- return
41
- }
42
-
43
- for (const [metricName, card] of Object.entries(embeddedCards)) {
44
- if (!card?.benchmark_details?.name) {
45
- continue
46
- }
47
-
48
- for (const key of candidateKeys(metricName)) {
49
- if (!map.has(key)) map.set(key, card)
50
- }
51
-
52
- for (const key of candidateKeys(card.benchmark_details.name)) {
53
- if (!map.has(key)) map.set(key, card)
54
- }
55
- }
56
- } catch (err) {
57
- console.warn(`benchmark-metadata: failed to load embedded cards from ${entry.name}:`, err)
58
  }
59
- })
60
- )
61
 
62
  return map
63
  }
64
 
65
- let cachedMapPromise: Promise<Map<string, BenchmarkCard>> | null = null
66
-
67
  function getMap(): Promise<Map<string, BenchmarkCard>> {
68
- if (process.env.NODE_ENV === "production") {
69
- if (!cachedMapPromise) cachedMapPromise = readEmbeddedBenchmarkCards()
70
- return cachedMapPromise
71
  }
72
- return readEmbeddedBenchmarkCards()
 
73
  }
74
 
75
- /** Look up a BenchmarkCard by any commonly-used benchmark name. Returns null if not found. */
76
  export async function getBenchmarkCard(benchmarkName: string): Promise<BenchmarkCard | null> {
77
  const map = await getMap()
 
78
  for (const key of candidateKeys(benchmarkName)) {
79
  const card = map.get(key)
80
- if (card) return card
 
 
81
  }
 
82
  return null
83
  }
84
 
85
- /** Returns all loaded BenchmarkCards keyed by their normalised canonical name. */
86
  export async function getAllBenchmarkCards(): Promise<Record<string, BenchmarkCard>> {
87
  const map = await getMap()
88
- // Deduplicate: only emit one entry per card (by canonical name)
89
  const seen = new Set<BenchmarkCard>()
90
  const result: Record<string, BenchmarkCard> = {}
91
- for (const [key, card] of map) {
92
- if (!seen.has(card)) {
93
- seen.add(card)
94
- result[normalizeBenchmarkKey(card.benchmark_details.name)] = card
95
  }
 
 
 
96
  }
 
97
  return result
98
  }
 
1
  import "server-only"
2
 
 
 
 
3
  import type { BenchmarkCard } from "@/lib/benchmark-schema"
4
+ import { candidateBenchmarkKeys as candidateKeys, normalizeBenchmarkKey } from "@/lib/benchmark-metadata-utils"
5
+ import { fetchBenchmarkMetadataMap } from "@/lib/hf-data"
6
 
7
  export { normalizeBenchmarkKey }
8
 
9
+ let cachedMapPromise: Promise<Map<string, BenchmarkCard>> | null = null
 
 
 
 
 
 
10
 
11
+ async function readPipelineBenchmarkCards(): Promise<Map<string, BenchmarkCard>> {
12
+ const cards = await fetchBenchmarkMetadataMap()
13
  const map = new Map<string, BenchmarkCard>()
14
 
15
+ for (const card of Object.values(cards)) {
16
+ if (!card?.benchmark_details?.name) {
17
+ continue
18
+ }
 
 
19
 
20
+ for (const key of candidateKeys(card.benchmark_details.name)) {
21
+ if (!map.has(key)) {
22
+ map.set(key, card)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
23
  }
24
+ }
25
+ }
26
 
27
  return map
28
  }
29
 
 
 
30
  function getMap(): Promise<Map<string, BenchmarkCard>> {
31
+ if (!cachedMapPromise) {
32
+ cachedMapPromise = readPipelineBenchmarkCards()
 
33
  }
34
+
35
+ return cachedMapPromise
36
  }
37
 
 
38
  export async function getBenchmarkCard(benchmarkName: string): Promise<BenchmarkCard | null> {
39
  const map = await getMap()
40
+
41
  for (const key of candidateKeys(benchmarkName)) {
42
  const card = map.get(key)
43
+ if (card) {
44
+ return card
45
+ }
46
  }
47
+
48
  return null
49
  }
50
 
 
51
  export async function getAllBenchmarkCards(): Promise<Record<string, BenchmarkCard>> {
52
  const map = await getMap()
 
53
  const seen = new Set<BenchmarkCard>()
54
  const result: Record<string, BenchmarkCard> = {}
55
+
56
+ for (const card of map.values()) {
57
+ if (seen.has(card)) {
58
+ continue
59
  }
60
+
61
+ seen.add(card)
62
+ result[normalizeBenchmarkKey(card.benchmark_details.name)] = card
63
  }
64
+
65
  return result
66
  }
lib/model-data.ts CHANGED
@@ -1039,6 +1039,179 @@ function aggregateBenchmarkSummaries(
1039
  }
1040
  }
1041
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1042
  // ---------------------------------------------------------------------------
1043
  // Public API
1044
  // ---------------------------------------------------------------------------
@@ -1361,6 +1534,34 @@ export async function getEvalSummaryById(evalId: string) {
1361
  return aggregateBenchmarkSummaries(validSummaries, aggregateKey)
1362
  }
1363
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1364
  // Direct eval lookup
1365
  const detail = await fetchHFEvalDetail(evalId)
1366
  if (detail) {
 
1039
  }
1040
  }
1041
 
1042
+ const SYNTHETIC_MATRIX_EVAL_PREFIX = "matrix__"
1043
+
1044
+ function buildSingleMetricSuiteMatrixSummary(
1045
+ details: HFEvalDetail[],
1046
+ suiteKey: string
1047
+ ): BenchmarkEvalSummary | null {
1048
+ if (details.length < 2) {
1049
+ return null
1050
+ }
1051
+
1052
+ const suiteDisplayName = getBenchmarkDisplayName(suiteKey)
1053
+ const validDetails = [...details]
1054
+ .filter((detail) => (detail.metrics?.length ?? 0) === 1 && extractDetailSubtasks(detail).length === 0)
1055
+ .sort((left, right) =>
1056
+ (left.benchmark_leaf_name || left.eval_summary_id).localeCompare(right.benchmark_leaf_name || right.eval_summary_id)
1057
+ )
1058
+
1059
+ if (validDetails.length < 2) {
1060
+ return null
1061
+ }
1062
+
1063
+ const leaderboardMetrics: NonNullable<BenchmarkEvalSummary["leaderboard_metrics"]> = []
1064
+ const rowStates = new Map<
1065
+ string,
1066
+ NonNullable<BenchmarkEvalSummary["leaderboard_rows"]>[number] & { _timestampValue: number }
1067
+ >()
1068
+
1069
+ let metricConfig: BenchmarkEvalSummary["metric_config"] | null = null
1070
+ let benchmarkCard: BenchmarkCard | undefined
1071
+ const metricNames = new Set<string>()
1072
+
1073
+ for (const detail of validDetails) {
1074
+ const metric = detail.metrics?.[0]
1075
+ if (!metric) {
1076
+ continue
1077
+ }
1078
+
1079
+ if (!metricConfig) {
1080
+ metricConfig = toSummaryMetricConfig(metric)
1081
+ }
1082
+
1083
+ if (!benchmarkCard && detail.benchmark_card) {
1084
+ benchmarkCard = detail.benchmark_card
1085
+ }
1086
+
1087
+ const summaryMetric = toBenchmarkSummaryMetric(metric)
1088
+ metricNames.add(summaryMetric.metric_name)
1089
+ const subtaskKey = detail.benchmark_leaf_key || slugifyEvalId(detail.eval_summary_id)
1090
+ const subtaskName = detail.benchmark_leaf_name || detail.canonical_display_name || detail.eval_summary_id || subtaskKey
1091
+ const metricToken =
1092
+ summaryMetric.metric_summary_id ||
1093
+ summaryMetric.metric_key ||
1094
+ slugifyEvalId(summaryMetric.display_name)
1095
+ const columnKey = ["subtask", subtaskKey, metricToken].join(":")
1096
+
1097
+ leaderboardMetrics.push({
1098
+ column_key: columnKey,
1099
+ metric_summary_id: summaryMetric.metric_summary_id,
1100
+ metric_name: summaryMetric.metric_name,
1101
+ display_name: summaryMetric.display_name,
1102
+ canonical_display_name: summaryMetric.canonical_display_name,
1103
+ lower_is_better: summaryMetric.lower_is_better,
1104
+ unit: summaryMetric.unit,
1105
+ scope: "subtask",
1106
+ subtask_key: subtaskKey,
1107
+ subtask_name: subtaskName,
1108
+ })
1109
+
1110
+ const benchmarkKey = detail.benchmark ?? suiteKey
1111
+ const sourceName = detail.source_data?.dataset_name || benchmarkKey
1112
+ const sourceOrganization = detail.source_data?.hf_repo || sourceName
1113
+ const sourceMetadata: SourceMetadata = {
1114
+ source_type: "documentation",
1115
+ source_name: sourceName,
1116
+ source_organization_name: sourceOrganization,
1117
+ evaluator_relationship: "other",
1118
+ }
1119
+ const sourceData = detail.source_data ?? { dataset_name: benchmarkKey }
1120
+
1121
+ for (const modelResult of metric.model_results ?? []) {
1122
+ const modelId = modelResult.model_id || modelResult.model_name
1123
+ if (!modelId) {
1124
+ continue
1125
+ }
1126
+
1127
+ const nextTimestamp = normalizeEvalTimestamp(modelResult.retrieved_timestamp ?? "")
1128
+ const existing = rowStates.get(modelId)
1129
+
1130
+ if (!existing) {
1131
+ rowStates.set(modelId, {
1132
+ model_info: {
1133
+ name: modelResult.model_name ?? "",
1134
+ id: modelId,
1135
+ developer: modelResult.developer ?? "",
1136
+ },
1137
+ model_route_id: modelResult.model_route_id,
1138
+ evaluation_timestamp: modelResult.retrieved_timestamp ?? "",
1139
+ source_metadata: sourceMetadata,
1140
+ source_data: sourceData,
1141
+ values: { [columnKey]: modelResult.score ?? null },
1142
+ metrics_present: 0,
1143
+ _timestampValue: nextTimestamp,
1144
+ })
1145
+ continue
1146
+ }
1147
+
1148
+ existing.values[columnKey] = modelResult.score ?? null
1149
+ if (!existing.model_route_id && modelResult.model_route_id) {
1150
+ existing.model_route_id = modelResult.model_route_id
1151
+ }
1152
+ if (nextTimestamp >= existing._timestampValue) {
1153
+ existing.evaluation_timestamp = modelResult.retrieved_timestamp ?? existing.evaluation_timestamp
1154
+ existing.source_metadata = sourceMetadata
1155
+ existing.source_data = sourceData
1156
+ existing._timestampValue = nextTimestamp
1157
+ }
1158
+ }
1159
+ }
1160
+
1161
+ if (leaderboardMetrics.length < 2) {
1162
+ return null
1163
+ }
1164
+
1165
+ const sharedMetricName = metricNames.size === 1 ? Array.from(metricNames)[0] : undefined
1166
+ const suiteMetricConfig = metricConfig
1167
+ ? {
1168
+ ...metricConfig,
1169
+ evaluation_description: sharedMetricName ?? metricConfig.evaluation_description,
1170
+ }
1171
+ : {
1172
+ evaluation_description: sharedMetricName ?? "",
1173
+ lower_is_better: false,
1174
+ score_type: "continuous" as const,
1175
+ min_score: 0,
1176
+ max_score: 1,
1177
+ }
1178
+
1179
+ const leaderboardRows = Array.from(rowStates.values()).map(({ _timestampValue, ...row }) => ({
1180
+ ...row,
1181
+ metrics_present: leaderboardMetrics.reduce(
1182
+ (count, metric) => count + (typeof row.values[metric.column_key] === "number" ? 1 : 0),
1183
+ 0
1184
+ ),
1185
+ }))
1186
+
1187
+ return {
1188
+ evaluation_name: suiteDisplayName,
1189
+ evaluation_id: `${SYNTHETIC_MATRIX_EVAL_PREFIX}${suiteKey}`,
1190
+ canonical_display_name: suiteDisplayName,
1191
+ composite_benchmark_key: suiteKey,
1192
+ composite_benchmark_name: suiteDisplayName,
1193
+ category: inferCategoryFromBenchmark(suiteDisplayName),
1194
+ metric_config: suiteMetricConfig,
1195
+ model_results: [],
1196
+ models_count: leaderboardRows.length,
1197
+ evaluator_names: [],
1198
+ source_types: [],
1199
+ latest_source_name: suiteDisplayName,
1200
+ third_party_ratio: 0,
1201
+ missing_generation_config_count: 0,
1202
+ best_model: null,
1203
+ worst_model: null,
1204
+ avg_score: 0,
1205
+ avg_score_norm: 0,
1206
+ benchmark_card: benchmarkCard,
1207
+ metrics_count: leaderboardMetrics.length,
1208
+ metric_names: leaderboardMetrics.map((metric) => `${metric.subtask_name} / ${metric.metric_name}`),
1209
+ source_data: { dataset_name: suiteDisplayName },
1210
+ leaderboard_metrics: leaderboardMetrics,
1211
+ leaderboard_rows: leaderboardRows,
1212
+ }
1213
+ }
1214
+
1215
  // ---------------------------------------------------------------------------
1216
  // Public API
1217
  // ---------------------------------------------------------------------------
 
1534
  return aggregateBenchmarkSummaries(validSummaries, aggregateKey)
1535
  }
1536
 
1537
+ if (evalId.startsWith(SYNTHETIC_MATRIX_EVAL_PREFIX)) {
1538
+ const suiteKey = evalId.replace(new RegExp(`^${SYNTHETIC_MATRIX_EVAL_PREFIX}`), "")
1539
+ const normalizedSuiteKey = normalizeBenchmarkKeyForLookup(suiteKey)
1540
+ const { evals } = await fetchHFEvalListLite()
1541
+ const matchingEvals = evals.filter((entry) => {
1542
+ if (entry.is_summary_score) {
1543
+ return false
1544
+ }
1545
+
1546
+ const parentKey = normalizeBenchmarkKeyForLookup(
1547
+ entry.benchmark_parent_key || entry.benchmark_family_key || entry.benchmark
1548
+ )
1549
+ return parentKey === normalizedSuiteKey
1550
+ })
1551
+
1552
+ if (matchingEvals.length < 2) {
1553
+ return null
1554
+ }
1555
+
1556
+ const details = await Promise.all(
1557
+ matchingEvals.map(async (entry) => fetchHFEvalDetail(entry.eval_summary_id))
1558
+ )
1559
+
1560
+ const validDetails = details.filter((detail): detail is HFEvalDetail => detail !== null)
1561
+ const syntheticSummary = buildSingleMetricSuiteMatrixSummary(validDetails, suiteKey)
1562
+ return syntheticSummary ? attachBenchmarkCardToSummary(syntheticSummary) : null
1563
+ }
1564
+
1565
  // Direct eval lookup
1566
  const detail = await fetchHFEvalDetail(evalId)
1567
  if (detail) {