Spaces:
Running
Running
Differentiate audience modes and tighten eval navigation
Browse files- app/evals/page.tsx +346 -123
- app/page.tsx +22 -22
- components/eval-detail.tsx +239 -202
- components/page-loading-state.tsx +120 -0
- data/benchmarks/ace.json +0 -120
- data/benchmarks/apex-agents.json +0 -218
- data/benchmarks/apex-v1.json +0 -93
- data/benchmarks/appworld_test_normal.json +0 -28
- data/benchmarks/bfcl.json +0 -0
- data/benchmarks/browsecompplus.json +0 -28
- data/benchmarks/global-mmlu-lite.json +0 -706
- data/benchmarks/helm_capabilities.json +0 -1026
- data/benchmarks/helm_classic.json +0 -2030
- data/benchmarks/helm_instruct.json +0 -60
- data/benchmarks/helm_lite.json +0 -1893
- data/benchmarks/helm_mmlu.json +0 -0
- data/benchmarks/hfopenllm_v2.json +0 -0
- data/benchmarks/la_leaderboard.json +0 -44
- data/benchmarks/livecodebenchpro.json +0 -274
- data/benchmarks/reward-bench.json +0 -0
- data/benchmarks/swe-bench.json +0 -28
- data/benchmarks/tau-bench-2_airline.json +0 -28
- data/benchmarks/tau-bench-2_retail.json +0 -28
- data/benchmarks/tau-bench-2_telecom.json +0 -28
- data/benchmarks/terminal-bench-2.0.json +0 -300
- data/benchmarks/theory_of_mind.json +0 -12
- lib/benchmark-metadata.ts +31 -63
- lib/model-data.ts +201 -0
app/evals/page.tsx
CHANGED
|
@@ -2,13 +2,14 @@
|
|
| 2 |
|
| 3 |
import { useCallback, useEffect, useMemo, useRef, useState } from "react"
|
| 4 |
import { useRouter } from "next/navigation"
|
| 5 |
-
import { ArrowLeft, Search, X } from "lucide-react"
|
| 6 |
|
| 7 |
import { useAudienceMode } from "@/components/audience-mode-provider"
|
| 8 |
import { ListPagination } from "@/components/list-pagination"
|
| 9 |
import { Navigation } from "@/components/navigation"
|
| 10 |
import { PageHeader } from "@/components/page-header"
|
| 11 |
import { Button } from "@/components/ui/button"
|
|
|
|
| 12 |
import { Input } from "@/components/ui/input"
|
| 13 |
import type { EvalHierarchy } from "@/lib/backend-artifacts"
|
| 14 |
import type { BenchmarkCard, CategoryType } from "@/lib/benchmark-schema"
|
|
@@ -262,6 +263,7 @@ interface EvalBrowserNode {
|
|
| 262 |
domains: string[]
|
| 263 |
dataType?: string
|
| 264 |
license?: string
|
|
|
|
| 265 |
modelsCount: number
|
| 266 |
metricCount: number
|
| 267 |
topScore?: number
|
|
@@ -353,6 +355,71 @@ function formatCompactScore(value: number | undefined) {
|
|
| 353 |
return value.toFixed(value >= 100 ? 0 : 2)
|
| 354 |
}
|
| 355 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 356 |
function getNodeCard(
|
| 357 |
benchmarkCards: Record<string, BenchmarkCard>,
|
| 358 |
...candidates: Array<string | undefined>
|
|
@@ -452,6 +519,7 @@ function mapHierarchyCategory(value: string | undefined | null): CategoryType {
|
|
| 452 |
export default function EvalsPage() {
|
| 453 |
const { mode } = useAudienceMode()
|
| 454 |
const router = useRouter()
|
|
|
|
| 455 |
|
| 456 |
const [summaries, setSummaries] = useState<BenchmarkEvalListItem[]>([])
|
| 457 |
const [benchmarkCards, setBenchmarkCards] = useState<Record<string, BenchmarkCard>>({})
|
|
@@ -464,6 +532,7 @@ export default function EvalsPage() {
|
|
| 464 |
const [selectedNodeKind, setSelectedNodeKind] = useState<EvalBrowserNodeKind | null>(null)
|
| 465 |
const [currentNodeId, setCurrentNodeId] = useState<string | null>(null)
|
| 466 |
const [page, setPage] = useState(1)
|
|
|
|
| 467 |
const pendingHistoryActionRef = useRef<"push" | "replace">("replace")
|
| 468 |
|
| 469 |
useEffect(() => {
|
|
@@ -635,6 +704,7 @@ export default function EvalsPage() {
|
|
| 635 |
domains: Array.from(new Set(domains.flatMap((domain) => normalizeDomainList(domain)))),
|
| 636 |
dataType: card?.benchmark_details?.data_type,
|
| 637 |
license: card?.ethical_and_legal_considerations?.data_licensing,
|
|
|
|
| 638 |
modelsCount: stats.modelsCount,
|
| 639 |
metricCount: stats.metricCount,
|
| 640 |
topScore: stats.topScore,
|
|
@@ -655,9 +725,9 @@ export default function EvalsPage() {
|
|
| 655 |
metrics?: Array<{ key: string; display_name: string }>
|
| 656 |
}>,
|
| 657 |
scopeKeys: string[]
|
| 658 |
-
): EvalBrowserNode["matrixPreview"] |
|
| 659 |
if (benchmarks.length < 2) {
|
| 660 |
-
return
|
| 661 |
}
|
| 662 |
|
| 663 |
const metricLabels = new Set<string>()
|
|
@@ -665,7 +735,7 @@ export default function EvalsPage() {
|
|
| 665 |
|
| 666 |
for (const benchmark of benchmarks) {
|
| 667 |
if ((benchmark.slices?.length ?? 0) > 0 || (benchmark.metrics?.length ?? 0) !== 1) {
|
| 668 |
-
return
|
| 669 |
}
|
| 670 |
|
| 671 |
const metric = benchmark.metrics?.[0]
|
|
@@ -679,7 +749,7 @@ export default function EvalsPage() {
|
|
| 679 |
}
|
| 680 |
|
| 681 |
if (metricLabels.size !== 1) {
|
| 682 |
-
return
|
| 683 |
}
|
| 684 |
|
| 685 |
return {
|
|
@@ -747,6 +817,10 @@ export default function EvalsPage() {
|
|
| 747 |
const card = summary?.benchmark_card ?? getNodeCard(benchmarkCards, ...cardCandidates)
|
| 748 |
const childSlices = slices.filter((slice) => !isSameHierarchyKey(slice.key, benchmarkKey))
|
| 749 |
const drilldownSlices = childSlices.filter((slice) => (slice.metrics?.length ?? 0) > 1)
|
|
|
|
|
|
|
|
|
|
|
|
|
| 750 |
const isParentRollupBenchmark =
|
| 751 |
Boolean(parentId) && scopeKeys.some((scopeKey) => isSameHierarchyKey(scopeKey, benchmarkKey))
|
| 752 |
|
|
@@ -775,9 +849,14 @@ export default function EvalsPage() {
|
|
| 775 |
domains,
|
| 776 |
summaries: summary ? [summary] : [],
|
| 777 |
card,
|
| 778 |
-
href:
|
| 779 |
-
|
| 780 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 781 |
scopeKeys,
|
| 782 |
descriptionFallback: `Browse the {label} benchmark and its lower-level breakdowns.`,
|
| 783 |
})
|
|
@@ -874,6 +953,8 @@ export default function EvalsPage() {
|
|
| 874 |
const suiteBenchmarks = (composite.benchmarks ?? []).filter((benchmark) => !isSameHierarchyKey(benchmark.key, composite.key))
|
| 875 |
const suiteMatrixPreview = buildSingleMetricMatrixPreview(suiteBenchmarks, suiteScopeKeys)
|
| 876 |
const rollupSummary = pickSummaryForKey(summariesWithCards, composite.key, suiteScopeKeys)
|
|
|
|
|
|
|
| 877 |
|
| 878 |
buildNode({
|
| 879 |
id: suiteId,
|
|
@@ -886,7 +967,13 @@ export default function EvalsPage() {
|
|
| 886 |
summaries: suiteSummaries,
|
| 887 |
card: suiteCard,
|
| 888 |
sourceLabel: suiteLabel,
|
| 889 |
-
href: suiteMatrixPreview
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 890 |
scopeKeys: suiteScopeKeys,
|
| 891 |
matrixPreview: suiteMatrixPreview,
|
| 892 |
descriptionFallback: `Browse the {label} suite and then open its benchmark children.`,
|
|
@@ -928,8 +1015,12 @@ export default function EvalsPage() {
|
|
| 928 |
}
|
| 929 |
|
| 930 |
const suiteNode = nodes.get(suiteId)
|
| 931 |
-
if (suiteNode && suiteNode.childIds.length === 0
|
| 932 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 933 |
}
|
| 934 |
}
|
| 935 |
|
|
@@ -979,6 +1070,8 @@ export default function EvalsPage() {
|
|
| 979 |
const visibleBenchmarks = suiteBenchmarks.filter((benchmark) => !isSameHierarchyKey(benchmark.key, suiteKey))
|
| 980 |
const suiteMatrixPreview = buildSingleMetricMatrixPreview(visibleBenchmarks, suiteScopeKeys)
|
| 981 |
const rollupSummary = pickSummaryForKey(summariesWithCards, suiteKey, suiteScopeKeys)
|
|
|
|
|
|
|
| 982 |
const suiteSummaries = summariesWithCards.filter((summary) => {
|
| 983 |
const familyScope = getSummaryScopeKey(summary.benchmark_family_key ?? summary.composite_benchmark_key)
|
| 984 |
return familyScope === getSummaryScopeKey(suiteKey)
|
|
@@ -1002,7 +1095,13 @@ export default function EvalsPage() {
|
|
| 1002 |
family.key
|
| 1003 |
),
|
| 1004 |
sourceLabel: suiteLabel,
|
| 1005 |
-
href: suiteMatrixPreview
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1006 |
scopeKeys: suiteScopeKeys,
|
| 1007 |
matrixPreview: suiteMatrixPreview,
|
| 1008 |
descriptionFallback: `Browse the {label} suite and then open its benchmark children.`,
|
|
@@ -1044,8 +1143,12 @@ export default function EvalsPage() {
|
|
| 1044 |
}
|
| 1045 |
|
| 1046 |
const suiteNode = nodes.get(suiteId)
|
| 1047 |
-
if (suiteNode && suiteNode.childIds.length === 0
|
| 1048 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1049 |
}
|
| 1050 |
|
| 1051 |
continue
|
|
@@ -1250,6 +1353,12 @@ export default function EvalsPage() {
|
|
| 1250 |
}
|
| 1251 |
}, [currentLevelKinds, selectedNodeKind])
|
| 1252 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1253 |
const handleNodeOpen = useCallback(
|
| 1254 |
(node: EvalBrowserNode) => {
|
| 1255 |
if (node.childIds.length > 0) {
|
|
@@ -1455,116 +1564,150 @@ export default function EvalsPage() {
|
|
| 1455 |
</div>
|
| 1456 |
</div>
|
| 1457 |
|
| 1458 |
-
|
| 1459 |
-
|
| 1460 |
-
|
| 1461 |
-
|
| 1462 |
-
|
| 1463 |
-
|
| 1464 |
-
|
| 1465 |
-
|
| 1466 |
-
|
| 1467 |
-
|
| 1468 |
-
|
| 1469 |
-
|
| 1470 |
-
|
| 1471 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1472 |
)}
|
| 1473 |
-
|
| 1474 |
-
|
| 1475 |
-
|
| 1476 |
-
|
| 1477 |
-
|
| 1478 |
-
|
| 1479 |
-
|
| 1480 |
-
|
| 1481 |
-
|
| 1482 |
-
|
| 1483 |
-
|
| 1484 |
-
|
| 1485 |
-
|
| 1486 |
-
|
| 1487 |
-
|
| 1488 |
-
|
| 1489 |
-
|
| 1490 |
-
|
| 1491 |
-
|
| 1492 |
-
|
| 1493 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1494 |
|
| 1495 |
-
|
| 1496 |
-
|
| 1497 |
-
|
| 1498 |
-
|
| 1499 |
-
|
| 1500 |
-
|
| 1501 |
-
|
| 1502 |
-
|
| 1503 |
-
|
| 1504 |
-
|
| 1505 |
-
|
| 1506 |
-
|
| 1507 |
-
|
| 1508 |
-
|
| 1509 |
-
|
| 1510 |
-
|
| 1511 |
-
|
| 1512 |
-
|
| 1513 |
-
|
| 1514 |
-
|
| 1515 |
-
|
| 1516 |
-
|
| 1517 |
-
|
| 1518 |
-
|
| 1519 |
-
|
| 1520 |
-
|
| 1521 |
-
|
| 1522 |
-
|
| 1523 |
-
|
| 1524 |
-
|
| 1525 |
-
|
| 1526 |
-
|
| 1527 |
-
|
| 1528 |
-
|
| 1529 |
-
|
| 1530 |
-
|
| 1531 |
|
| 1532 |
-
|
| 1533 |
-
|
| 1534 |
-
|
| 1535 |
-
|
| 1536 |
-
|
| 1537 |
-
|
| 1538 |
-
|
| 1539 |
-
|
| 1540 |
-
|
| 1541 |
-
|
| 1542 |
-
|
| 1543 |
-
|
| 1544 |
-
|
| 1545 |
-
|
| 1546 |
-
|
| 1547 |
-
|
| 1548 |
-
|
| 1549 |
-
|
| 1550 |
-
|
| 1551 |
-
|
| 1552 |
-
|
| 1553 |
-
|
| 1554 |
-
|
| 1555 |
-
|
| 1556 |
-
|
| 1557 |
-
|
| 1558 |
-
|
| 1559 |
-
|
| 1560 |
-
|
| 1561 |
-
|
| 1562 |
-
|
| 1563 |
-
|
| 1564 |
-
|
|
|
|
|
|
|
|
|
|
| 1565 |
</div>
|
| 1566 |
-
</
|
| 1567 |
-
|
| 1568 |
</section>
|
| 1569 |
|
| 1570 |
{filtered.length === 0 ? (
|
|
@@ -1580,6 +1723,8 @@ export default function EvalsPage() {
|
|
| 1580 |
const isNavigable = node.childIds.length > 0 || Boolean(node.href)
|
| 1581 |
const actionLabel = node.childIds.length > 0 ? "Open level" : node.matrixPreview ? "View rollup" : "View benchmark"
|
| 1582 |
const kindLabel = getBrowserNodeKindLabel(node.kind)
|
|
|
|
|
|
|
| 1583 |
|
| 1584 |
return (
|
| 1585 |
<button
|
|
@@ -1655,6 +1800,84 @@ export default function EvalsPage() {
|
|
| 1655 |
</p>
|
| 1656 |
)}
|
| 1657 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1658 |
{node.matrixPreview && (
|
| 1659 |
<div className="mb-4 overflow-hidden rounded-[1.2rem] border border-stone-200/80 bg-stone-50/85 dark:border-stone-800/80 dark:bg-stone-900/85">
|
| 1660 |
<div className="space-y-2 bg-white/92 px-3 py-3 dark:bg-stone-950/92">
|
|
@@ -1679,18 +1902,18 @@ export default function EvalsPage() {
|
|
| 1679 |
<div className="mb-4 grid gap-px overflow-hidden rounded-[1.2rem] border border-stone-200/80 bg-stone-200/80 dark:border-stone-800/80 dark:bg-stone-800/80 sm:grid-cols-2 xl:grid-cols-3">
|
| 1680 |
{topScoreLabel !== "—" && (
|
| 1681 |
<div className="bg-white/90 px-3 py-3 dark:bg-stone-950/90">
|
| 1682 |
-
<div className="text-[10px] font-semibold uppercase tracking-[0.16em] text-stone-500 dark:text-stone-400">
|
| 1683 |
<div className="mt-1 text-sm font-semibold text-stone-900 dark:text-stone-100">{topScoreLabel}</div>
|
| 1684 |
</div>
|
| 1685 |
)}
|
| 1686 |
<div className="bg-white/90 px-3 py-3 dark:bg-stone-950/90">
|
| 1687 |
-
<div className="text-[10px] font-semibold uppercase tracking-[0.16em] text-stone-500 dark:text-stone-400">
|
| 1688 |
<div className="mt-1 truncate text-sm font-semibold text-stone-900 dark:text-stone-100">
|
| 1689 |
{node.sourceLabel}
|
| 1690 |
</div>
|
| 1691 |
</div>
|
| 1692 |
<div className="bg-white/90 px-3 py-3 dark:bg-stone-950/90">
|
| 1693 |
-
<div className="text-[10px] font-semibold uppercase tracking-[0.16em] text-stone-500 dark:text-stone-400">
|
| 1694 |
<div className="mt-1 text-sm font-semibold text-stone-900 dark:text-stone-100">
|
| 1695 |
{node.instanceDataLabel}
|
| 1696 |
</div>
|
|
|
|
| 2 |
|
| 3 |
import { useCallback, useEffect, useMemo, useRef, useState } from "react"
|
| 4 |
import { useRouter } from "next/navigation"
|
| 5 |
+
import { ArrowLeft, ChevronDown, Search, SlidersHorizontal, X } from "lucide-react"
|
| 6 |
|
| 7 |
import { useAudienceMode } from "@/components/audience-mode-provider"
|
| 8 |
import { ListPagination } from "@/components/list-pagination"
|
| 9 |
import { Navigation } from "@/components/navigation"
|
| 10 |
import { PageHeader } from "@/components/page-header"
|
| 11 |
import { Button } from "@/components/ui/button"
|
| 12 |
+
import { Collapsible, CollapsibleContent, CollapsibleTrigger } from "@/components/ui/collapsible"
|
| 13 |
import { Input } from "@/components/ui/input"
|
| 14 |
import type { EvalHierarchy } from "@/lib/backend-artifacts"
|
| 15 |
import type { BenchmarkCard, CategoryType } from "@/lib/benchmark-schema"
|
|
|
|
| 263 |
domains: string[]
|
| 264 |
dataType?: string
|
| 265 |
license?: string
|
| 266 |
+
card?: BenchmarkCard
|
| 267 |
modelsCount: number
|
| 268 |
metricCount: number
|
| 269 |
topScore?: number
|
|
|
|
| 355 |
return value.toFixed(value >= 100 ? 0 : 2)
|
| 356 |
}
|
| 357 |
|
| 358 |
+
function hasConcreteText(value: string | undefined | null) {
|
| 359 |
+
if (!value) return false
|
| 360 |
+
const normalized = value.trim().toLowerCase()
|
| 361 |
+
return Boolean(normalized) && normalized !== "not specified" && normalized !== "unknown"
|
| 362 |
+
}
|
| 363 |
+
|
| 364 |
+
function firstConcreteListValue(value: string[] | string | undefined | null) {
|
| 365 |
+
if (Array.isArray(value)) {
|
| 366 |
+
return value.find((entry) => hasConcreteText(entry))
|
| 367 |
+
}
|
| 368 |
+
|
| 369 |
+
return hasConcreteText(value) ? value : undefined
|
| 370 |
+
}
|
| 371 |
+
|
| 372 |
+
function countConcreteListValues(value: string[] | undefined | null) {
|
| 373 |
+
return (value ?? []).filter((entry) => hasConcreteText(entry)).length
|
| 374 |
+
}
|
| 375 |
+
|
| 376 |
+
function getNodePolicySummary(node: EvalBrowserNode) {
|
| 377 |
+
const card = node.card
|
| 378 |
+
if (!card) return null
|
| 379 |
+
|
| 380 |
+
const riskCount = card.possible_risks?.length ?? 0
|
| 381 |
+
const reportingGapCount =
|
| 382 |
+
card.missing_fields?.filter(
|
| 383 |
+
(field) => field.startsWith("methodology") || field.startsWith("purpose_and_intended_users")
|
| 384 |
+
).length ?? 0
|
| 385 |
+
|
| 386 |
+
return {
|
| 387 |
+
goal: hasConcreteText(card.purpose_and_intended_users?.goal)
|
| 388 |
+
? card.purpose_and_intended_users.goal
|
| 389 |
+
: undefined,
|
| 390 |
+
limitations: hasConcreteText(card.purpose_and_intended_users?.limitations)
|
| 391 |
+
? card.purpose_and_intended_users.limitations
|
| 392 |
+
: undefined,
|
| 393 |
+
audience: firstConcreteListValue(card.purpose_and_intended_users?.audience),
|
| 394 |
+
compliance: hasConcreteText(card.ethical_and_legal_considerations?.compliance_with_regulations)
|
| 395 |
+
? card.ethical_and_legal_considerations.compliance_with_regulations
|
| 396 |
+
: undefined,
|
| 397 |
+
riskCount,
|
| 398 |
+
reportingGapCount,
|
| 399 |
+
}
|
| 400 |
+
}
|
| 401 |
+
|
| 402 |
+
function getNodeResearchSummary(node: EvalBrowserNode) {
|
| 403 |
+
const card = node.card
|
| 404 |
+
if (!card) return null
|
| 405 |
+
|
| 406 |
+
const similarBenchmarks = Array.isArray(card.benchmark_details?.similar_benchmarks)
|
| 407 |
+
? card.benchmark_details.similar_benchmarks
|
| 408 |
+
: card.benchmark_details?.similar_benchmarks
|
| 409 |
+
? [card.benchmark_details.similar_benchmarks]
|
| 410 |
+
: []
|
| 411 |
+
|
| 412 |
+
return {
|
| 413 |
+
methodsCount: countConcreteListValues(card.methodology?.methods),
|
| 414 |
+
metricsCount: countConcreteListValues(card.methodology?.metrics),
|
| 415 |
+
similarCount: countConcreteListValues(similarBenchmarks),
|
| 416 |
+
interpretation: hasConcreteText(card.methodology?.interpretation)
|
| 417 |
+
? card.methodology.interpretation
|
| 418 |
+
: undefined,
|
| 419 |
+
missingMethodCount: card.missing_fields?.filter((field) => field.startsWith("methodology")).length ?? 0,
|
| 420 |
+
}
|
| 421 |
+
}
|
| 422 |
+
|
| 423 |
function getNodeCard(
|
| 424 |
benchmarkCards: Record<string, BenchmarkCard>,
|
| 425 |
...candidates: Array<string | undefined>
|
|
|
|
| 519 |
export default function EvalsPage() {
|
| 520 |
const { mode } = useAudienceMode()
|
| 521 |
const router = useRouter()
|
| 522 |
+
const isResearchView = mode === "research"
|
| 523 |
|
| 524 |
const [summaries, setSummaries] = useState<BenchmarkEvalListItem[]>([])
|
| 525 |
const [benchmarkCards, setBenchmarkCards] = useState<Record<string, BenchmarkCard>>({})
|
|
|
|
| 532 |
const [selectedNodeKind, setSelectedNodeKind] = useState<EvalBrowserNodeKind | null>(null)
|
| 533 |
const [currentNodeId, setCurrentNodeId] = useState<string | null>(null)
|
| 534 |
const [page, setPage] = useState(1)
|
| 535 |
+
const [filtersOpen, setFiltersOpen] = useState(false)
|
| 536 |
const pendingHistoryActionRef = useRef<"push" | "replace">("replace")
|
| 537 |
|
| 538 |
useEffect(() => {
|
|
|
|
| 704 |
domains: Array.from(new Set(domains.flatMap((domain) => normalizeDomainList(domain)))),
|
| 705 |
dataType: card?.benchmark_details?.data_type,
|
| 706 |
license: card?.ethical_and_legal_considerations?.data_licensing,
|
| 707 |
+
card,
|
| 708 |
modelsCount: stats.modelsCount,
|
| 709 |
metricCount: stats.metricCount,
|
| 710 |
topScore: stats.topScore,
|
|
|
|
| 725 |
metrics?: Array<{ key: string; display_name: string }>
|
| 726 |
}>,
|
| 727 |
scopeKeys: string[]
|
| 728 |
+
): EvalBrowserNode["matrixPreview"] | undefined => {
|
| 729 |
if (benchmarks.length < 2) {
|
| 730 |
+
return undefined
|
| 731 |
}
|
| 732 |
|
| 733 |
const metricLabels = new Set<string>()
|
|
|
|
| 735 |
|
| 736 |
for (const benchmark of benchmarks) {
|
| 737 |
if ((benchmark.slices?.length ?? 0) > 0 || (benchmark.metrics?.length ?? 0) !== 1) {
|
| 738 |
+
return undefined
|
| 739 |
}
|
| 740 |
|
| 741 |
const metric = benchmark.metrics?.[0]
|
|
|
|
| 749 |
}
|
| 750 |
|
| 751 |
if (metricLabels.size !== 1) {
|
| 752 |
+
return undefined
|
| 753 |
}
|
| 754 |
|
| 755 |
return {
|
|
|
|
| 817 |
const card = summary?.benchmark_card ?? getNodeCard(benchmarkCards, ...cardCandidates)
|
| 818 |
const childSlices = slices.filter((slice) => !isSameHierarchyKey(slice.key, benchmarkKey))
|
| 819 |
const drilldownSlices = childSlices.filter((slice) => (slice.metrics?.length ?? 0) > 1)
|
| 820 |
+
const fallbackSummary =
|
| 821 |
+
!summary && metrics.length > 0
|
| 822 |
+
? scopeKeys.map((scopeKey) => pickSummaryForKey(summariesWithCards, scopeKey, scopeKeys)).find(Boolean)
|
| 823 |
+
: undefined
|
| 824 |
const isParentRollupBenchmark =
|
| 825 |
Boolean(parentId) && scopeKeys.some((scopeKey) => isSameHierarchyKey(scopeKey, benchmarkKey))
|
| 826 |
|
|
|
|
| 849 |
domains,
|
| 850 |
summaries: summary ? [summary] : [],
|
| 851 |
card,
|
| 852 |
+
href:
|
| 853 |
+
drilldownSlices.length === 0
|
| 854 |
+
? summary
|
| 855 |
+
? `/evals/${summary.evaluation_id}`
|
| 856 |
+
: fallbackSummary
|
| 857 |
+
? `/evals/${fallbackSummary.evaluation_id}`
|
| 858 |
+
: undefined
|
| 859 |
+
: undefined,
|
| 860 |
scopeKeys,
|
| 861 |
descriptionFallback: `Browse the {label} benchmark and its lower-level breakdowns.`,
|
| 862 |
})
|
|
|
|
| 953 |
const suiteBenchmarks = (composite.benchmarks ?? []).filter((benchmark) => !isSameHierarchyKey(benchmark.key, composite.key))
|
| 954 |
const suiteMatrixPreview = buildSingleMetricMatrixPreview(suiteBenchmarks, suiteScopeKeys)
|
| 955 |
const rollupSummary = pickSummaryForKey(summariesWithCards, composite.key, suiteScopeKeys)
|
| 956 |
+
const hasSuiteRollup = Boolean(rollupBenchmark && rollupSummary)
|
| 957 |
+
const syntheticMatrixEvalId = suiteMatrixPreview && !hasSuiteRollup ? `matrix__${composite.key}` : undefined
|
| 958 |
|
| 959 |
buildNode({
|
| 960 |
id: suiteId,
|
|
|
|
| 967 |
summaries: suiteSummaries,
|
| 968 |
card: suiteCard,
|
| 969 |
sourceLabel: suiteLabel,
|
| 970 |
+
href: suiteMatrixPreview
|
| 971 |
+
? hasSuiteRollup
|
| 972 |
+
? `/evals/${rollupSummary.evaluation_id}`
|
| 973 |
+
: syntheticMatrixEvalId
|
| 974 |
+
? `/evals/${syntheticMatrixEvalId}`
|
| 975 |
+
: undefined
|
| 976 |
+
: undefined,
|
| 977 |
scopeKeys: suiteScopeKeys,
|
| 978 |
matrixPreview: suiteMatrixPreview,
|
| 979 |
descriptionFallback: `Browse the {label} suite and then open its benchmark children.`,
|
|
|
|
| 1015 |
}
|
| 1016 |
|
| 1017 |
const suiteNode = nodes.get(suiteId)
|
| 1018 |
+
if (suiteNode && suiteNode.childIds.length === 0) {
|
| 1019 |
+
if (hasSuiteRollup) {
|
| 1020 |
+
suiteNode.href = `/evals/${rollupSummary.evaluation_id}`
|
| 1021 |
+
} else if (syntheticMatrixEvalId) {
|
| 1022 |
+
suiteNode.href = `/evals/${syntheticMatrixEvalId}`
|
| 1023 |
+
}
|
| 1024 |
}
|
| 1025 |
}
|
| 1026 |
|
|
|
|
| 1070 |
const visibleBenchmarks = suiteBenchmarks.filter((benchmark) => !isSameHierarchyKey(benchmark.key, suiteKey))
|
| 1071 |
const suiteMatrixPreview = buildSingleMetricMatrixPreview(visibleBenchmarks, suiteScopeKeys)
|
| 1072 |
const rollupSummary = pickSummaryForKey(summariesWithCards, suiteKey, suiteScopeKeys)
|
| 1073 |
+
const hasSuiteRollup = Boolean(rollupBenchmark && rollupSummary)
|
| 1074 |
+
const syntheticMatrixEvalId = suiteMatrixPreview && !hasSuiteRollup ? `matrix__${suiteKey}` : undefined
|
| 1075 |
const suiteSummaries = summariesWithCards.filter((summary) => {
|
| 1076 |
const familyScope = getSummaryScopeKey(summary.benchmark_family_key ?? summary.composite_benchmark_key)
|
| 1077 |
return familyScope === getSummaryScopeKey(suiteKey)
|
|
|
|
| 1095 |
family.key
|
| 1096 |
),
|
| 1097 |
sourceLabel: suiteLabel,
|
| 1098 |
+
href: suiteMatrixPreview
|
| 1099 |
+
? hasSuiteRollup
|
| 1100 |
+
? `/evals/${rollupSummary.evaluation_id}`
|
| 1101 |
+
: syntheticMatrixEvalId
|
| 1102 |
+
? `/evals/${syntheticMatrixEvalId}`
|
| 1103 |
+
: undefined
|
| 1104 |
+
: undefined,
|
| 1105 |
scopeKeys: suiteScopeKeys,
|
| 1106 |
matrixPreview: suiteMatrixPreview,
|
| 1107 |
descriptionFallback: `Browse the {label} suite and then open its benchmark children.`,
|
|
|
|
| 1143 |
}
|
| 1144 |
|
| 1145 |
const suiteNode = nodes.get(suiteId)
|
| 1146 |
+
if (suiteNode && suiteNode.childIds.length === 0) {
|
| 1147 |
+
if (hasSuiteRollup) {
|
| 1148 |
+
suiteNode.href = `/evals/${rollupSummary.evaluation_id}`
|
| 1149 |
+
} else if (syntheticMatrixEvalId) {
|
| 1150 |
+
suiteNode.href = `/evals/${syntheticMatrixEvalId}`
|
| 1151 |
+
}
|
| 1152 |
}
|
| 1153 |
|
| 1154 |
continue
|
|
|
|
| 1353 |
}
|
| 1354 |
}, [currentLevelKinds, selectedNodeKind])
|
| 1355 |
|
| 1356 |
+
useEffect(() => {
|
| 1357 |
+
if (activeFilterCount > 0) {
|
| 1358 |
+
setFiltersOpen(true)
|
| 1359 |
+
}
|
| 1360 |
+
}, [activeFilterCount])
|
| 1361 |
+
|
| 1362 |
const handleNodeOpen = useCallback(
|
| 1363 |
(node: EvalBrowserNode) => {
|
| 1364 |
if (node.childIds.length > 0) {
|
|
|
|
| 1564 |
</div>
|
| 1565 |
</div>
|
| 1566 |
|
| 1567 |
+
<Collapsible
|
| 1568 |
+
open={filtersOpen}
|
| 1569 |
+
onOpenChange={setFiltersOpen}
|
| 1570 |
+
className="mt-4 rounded-[1.35rem] border border-stone-200/80 bg-white/70 dark:border-stone-800/80 dark:bg-stone-950/60"
|
| 1571 |
+
>
|
| 1572 |
+
<CollapsibleTrigger asChild>
|
| 1573 |
+
<button type="button" className="flex w-full items-center justify-between gap-3 px-4 py-3 text-left">
|
| 1574 |
+
<div className="min-w-0">
|
| 1575 |
+
<div className="flex items-center gap-2 text-[11px] font-semibold uppercase tracking-[0.2em] text-stone-500 dark:text-stone-400">
|
| 1576 |
+
<SlidersHorizontal className="h-3.5 w-3.5" />
|
| 1577 |
+
Refine this list
|
| 1578 |
+
</div>
|
| 1579 |
+
<p className="mt-1 text-sm text-stone-600 dark:text-stone-300">
|
| 1580 |
+
{activeFilterCount > 0
|
| 1581 |
+
? `${activeFilterCount} filter${activeFilterCount === 1 ? "" : "s"} active`
|
| 1582 |
+
: "Open filters only when you need to narrow by node type, domain tags, or category."}
|
| 1583 |
+
</p>
|
| 1584 |
+
</div>
|
| 1585 |
+
<div className="flex items-center gap-2">
|
| 1586 |
+
{activeFilterCount > 0 && (
|
| 1587 |
+
<span className="rounded-full border border-stone-200/80 bg-stone-100/80 px-2.5 py-1 text-[11px] font-medium text-stone-700 dark:border-stone-700/80 dark:bg-stone-900/80 dark:text-stone-200">
|
| 1588 |
+
{activeFilterCount} active
|
| 1589 |
+
</span>
|
| 1590 |
)}
|
| 1591 |
+
<ChevronDown className={cn("h-4 w-4 text-stone-500 transition-transform dark:text-stone-400", filtersOpen && "rotate-180")} />
|
| 1592 |
+
</div>
|
| 1593 |
+
</button>
|
| 1594 |
+
</CollapsibleTrigger>
|
| 1595 |
+
|
| 1596 |
+
<CollapsibleContent>
|
| 1597 |
+
<div className="border-t border-stone-200/80 px-4 pb-4 pt-4 dark:border-stone-800/80">
|
| 1598 |
+
{currentLevelKinds.length > 1 && (
|
| 1599 |
+
<div className="space-y-1.5">
|
| 1600 |
+
<div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-stone-500 dark:text-stone-400">
|
| 1601 |
+
Node type
|
| 1602 |
+
</div>
|
| 1603 |
+
<div className="flex flex-wrap items-center gap-1.5">
|
| 1604 |
+
<button
|
| 1605 |
+
type="button"
|
| 1606 |
+
onClick={() => setSelectedNodeKind(null)}
|
| 1607 |
+
className={cn(
|
| 1608 |
+
"shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
|
| 1609 |
+
selectedNodeKind === null
|
| 1610 |
+
? "border-stone-950 bg-stone-950 text-stone-50 dark:border-stone-100 dark:bg-stone-100 dark:text-stone-950"
|
| 1611 |
+
: "border-stone-200/80 bg-stone-50/80 text-stone-600 hover:bg-stone-100 dark:border-stone-700/80 dark:bg-stone-900/70 dark:text-stone-300 dark:hover:bg-stone-800"
|
| 1612 |
+
)}
|
| 1613 |
+
>
|
| 1614 |
+
All
|
| 1615 |
+
</button>
|
| 1616 |
+
{currentLevelKinds.map((kind) => (
|
| 1617 |
+
<button
|
| 1618 |
+
key={kind}
|
| 1619 |
+
type="button"
|
| 1620 |
+
onClick={() => setSelectedNodeKind(selectedNodeKind === kind ? null : kind)}
|
| 1621 |
+
className={cn(
|
| 1622 |
+
"shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
|
| 1623 |
+
selectedNodeKind === kind
|
| 1624 |
+
? "border-sky-300 bg-sky-50 text-sky-800 dark:border-sky-800 dark:bg-sky-950/50 dark:text-sky-200"
|
| 1625 |
+
: "border-stone-200/80 bg-white text-stone-600 hover:bg-stone-50 dark:border-stone-700/80 dark:bg-stone-900 dark:text-stone-300 dark:hover:bg-stone-800"
|
| 1626 |
+
)}
|
| 1627 |
+
>
|
| 1628 |
+
{getBrowserNodeKindLabel(kind)}
|
| 1629 |
+
</button>
|
| 1630 |
+
))}
|
| 1631 |
+
</div>
|
| 1632 |
+
</div>
|
| 1633 |
+
)}
|
| 1634 |
|
| 1635 |
+
{allDomains.length > 0 && (
|
| 1636 |
+
<div className="mt-4 space-y-1.5">
|
| 1637 |
+
<div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-stone-500 dark:text-stone-400">
|
| 1638 |
+
Domain tags
|
| 1639 |
+
</div>
|
| 1640 |
+
<div className="flex max-h-40 flex-wrap items-center gap-1.5 overflow-y-auto pr-1">
|
| 1641 |
+
<button
|
| 1642 |
+
type="button"
|
| 1643 |
+
onClick={() => setSelectedDomain(null)}
|
| 1644 |
+
className={cn(
|
| 1645 |
+
"shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
|
| 1646 |
+
selectedDomain === null
|
| 1647 |
+
? "border-stone-950 bg-stone-950 text-stone-50 dark:border-stone-100 dark:bg-stone-100 dark:text-stone-950"
|
| 1648 |
+
: "border-stone-200/80 bg-stone-50/80 text-stone-600 hover:bg-stone-100 dark:border-stone-700/80 dark:bg-stone-900/70 dark:text-stone-300 dark:hover:bg-stone-800"
|
| 1649 |
+
)}
|
| 1650 |
+
>
|
| 1651 |
+
All
|
| 1652 |
+
</button>
|
| 1653 |
+
{allDomains.map((domain) => (
|
| 1654 |
+
<button
|
| 1655 |
+
key={domain}
|
| 1656 |
+
type="button"
|
| 1657 |
+
onClick={() => setSelectedDomain(selectedDomain === domain ? null : domain)}
|
| 1658 |
+
className={cn(
|
| 1659 |
+
"shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
|
| 1660 |
+
selectedDomain === domain
|
| 1661 |
+
? "border-sky-300 bg-sky-50 text-sky-800 dark:border-sky-800 dark:bg-sky-950/50 dark:text-sky-200"
|
| 1662 |
+
: "border-stone-200/80 bg-white text-stone-600 hover:bg-stone-50 dark:border-stone-700/80 dark:bg-stone-900 dark:text-stone-300 dark:hover:bg-stone-800"
|
| 1663 |
+
)}
|
| 1664 |
+
>
|
| 1665 |
+
{domain}
|
| 1666 |
+
</button>
|
| 1667 |
+
))}
|
| 1668 |
+
</div>
|
| 1669 |
+
</div>
|
| 1670 |
+
)}
|
| 1671 |
|
| 1672 |
+
{allCategories.length > 0 && (
|
| 1673 |
+
<div className="mt-4 space-y-1.5">
|
| 1674 |
+
<div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-stone-500 dark:text-stone-400">
|
| 1675 |
+
Category
|
| 1676 |
+
</div>
|
| 1677 |
+
<div className="flex flex-wrap items-center gap-1.5">
|
| 1678 |
+
<button
|
| 1679 |
+
type="button"
|
| 1680 |
+
onClick={() => setSelectedCategory(null)}
|
| 1681 |
+
className={cn(
|
| 1682 |
+
"shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
|
| 1683 |
+
selectedCategory === null
|
| 1684 |
+
? "border-stone-950 bg-stone-950 text-stone-50 dark:border-stone-100 dark:bg-stone-100 dark:text-stone-950"
|
| 1685 |
+
: "border-stone-200/80 bg-stone-50/80 text-stone-600 hover:bg-stone-100 dark:border-stone-700/80 dark:bg-stone-900/70 dark:text-stone-300 dark:hover:bg-stone-800"
|
| 1686 |
+
)}
|
| 1687 |
+
>
|
| 1688 |
+
All
|
| 1689 |
+
</button>
|
| 1690 |
+
{allCategories.map((category) => (
|
| 1691 |
+
<button
|
| 1692 |
+
key={category}
|
| 1693 |
+
type="button"
|
| 1694 |
+
onClick={() => setSelectedCategory(selectedCategory === category ? null : category)}
|
| 1695 |
+
className={cn(
|
| 1696 |
+
"shrink-0 rounded-full border px-3 py-1.5 text-xs font-medium transition-colors",
|
| 1697 |
+
selectedCategory === category
|
| 1698 |
+
? `${getCategoryColor(category as CategoryType)} border`
|
| 1699 |
+
: "border-stone-200/80 bg-white text-stone-600 hover:bg-stone-50 dark:border-stone-700/80 dark:bg-stone-900 dark:text-stone-300 dark:hover:bg-stone-800"
|
| 1700 |
+
)}
|
| 1701 |
+
>
|
| 1702 |
+
{category}
|
| 1703 |
+
</button>
|
| 1704 |
+
))}
|
| 1705 |
+
</div>
|
| 1706 |
+
</div>
|
| 1707 |
+
)}
|
| 1708 |
</div>
|
| 1709 |
+
</CollapsibleContent>
|
| 1710 |
+
</Collapsible>
|
| 1711 |
</section>
|
| 1712 |
|
| 1713 |
{filtered.length === 0 ? (
|
|
|
|
| 1723 |
const isNavigable = node.childIds.length > 0 || Boolean(node.href)
|
| 1724 |
const actionLabel = node.childIds.length > 0 ? "Open level" : node.matrixPreview ? "View rollup" : "View benchmark"
|
| 1725 |
const kindLabel = getBrowserNodeKindLabel(node.kind)
|
| 1726 |
+
const policySummary = getNodePolicySummary(node)
|
| 1727 |
+
const researchSummary = getNodeResearchSummary(node)
|
| 1728 |
|
| 1729 |
return (
|
| 1730 |
<button
|
|
|
|
| 1800 |
</p>
|
| 1801 |
)}
|
| 1802 |
|
| 1803 |
+
{!isResearchView && policySummary && (
|
| 1804 |
+
<div className="mb-4 space-y-2 rounded-[1.1rem] border border-amber-200/80 bg-amber-50/70 p-3 dark:border-amber-900/50 dark:bg-amber-950/20">
|
| 1805 |
+
<div className="text-[10px] font-semibold uppercase tracking-[0.18em] text-amber-800 dark:text-amber-200">
|
| 1806 |
+
Policy notes
|
| 1807 |
+
</div>
|
| 1808 |
+
{policySummary.goal && (
|
| 1809 |
+
<p className="text-sm leading-6 text-stone-700 dark:text-stone-200">
|
| 1810 |
+
<span className="font-semibold">What it measures: </span>
|
| 1811 |
+
{policySummary.goal}
|
| 1812 |
+
</p>
|
| 1813 |
+
)}
|
| 1814 |
+
{policySummary.limitations && (
|
| 1815 |
+
<p className="text-sm leading-6 text-stone-700 dark:text-stone-200">
|
| 1816 |
+
<span className="font-semibold">Main caveat: </span>
|
| 1817 |
+
{policySummary.limitations}
|
| 1818 |
+
</p>
|
| 1819 |
+
)}
|
| 1820 |
+
<div className="flex flex-wrap gap-2 text-[11px] text-stone-700 dark:text-stone-200">
|
| 1821 |
+
{policySummary.audience && (
|
| 1822 |
+
<span className="rounded-full border border-amber-200/80 bg-white/80 px-2.5 py-1 dark:border-amber-900/50 dark:bg-stone-950/70">
|
| 1823 |
+
Intended for {policySummary.audience}
|
| 1824 |
+
</span>
|
| 1825 |
+
)}
|
| 1826 |
+
{policySummary.compliance && (
|
| 1827 |
+
<span className="rounded-full border border-amber-200/80 bg-white/80 px-2.5 py-1 dark:border-amber-900/50 dark:bg-stone-950/70">
|
| 1828 |
+
Regulation note documented
|
| 1829 |
+
</span>
|
| 1830 |
+
)}
|
| 1831 |
+
{policySummary.riskCount > 0 && (
|
| 1832 |
+
<span className="rounded-full border border-amber-200/80 bg-white/80 px-2.5 py-1 dark:border-amber-900/50 dark:bg-stone-950/70">
|
| 1833 |
+
{policySummary.riskCount} risk note{policySummary.riskCount === 1 ? "" : "s"}
|
| 1834 |
+
</span>
|
| 1835 |
+
)}
|
| 1836 |
+
{policySummary.reportingGapCount > 0 && (
|
| 1837 |
+
<span className="rounded-full border border-rose-200/80 bg-white/80 px-2.5 py-1 text-rose-700 dark:border-rose-900/50 dark:bg-stone-950/70 dark:text-rose-200">
|
| 1838 |
+
{policySummary.reportingGapCount} missing reporting field{policySummary.reportingGapCount === 1 ? "" : "s"}
|
| 1839 |
+
</span>
|
| 1840 |
+
)}
|
| 1841 |
+
</div>
|
| 1842 |
+
</div>
|
| 1843 |
+
)}
|
| 1844 |
+
|
| 1845 |
+
{isResearchView && researchSummary && (
|
| 1846 |
+
<div className="mb-4 space-y-2 rounded-[1.1rem] border border-sky-200/80 bg-sky-50/70 p-3 dark:border-sky-900/50 dark:bg-sky-950/20">
|
| 1847 |
+
<div className="text-[10px] font-semibold uppercase tracking-[0.18em] text-sky-800 dark:text-sky-200">
|
| 1848 |
+
Research notes
|
| 1849 |
+
</div>
|
| 1850 |
+
{researchSummary.interpretation && (
|
| 1851 |
+
<p className="text-sm leading-6 text-stone-700 dark:text-stone-200">
|
| 1852 |
+
<span className="font-semibold">Score interpretation: </span>
|
| 1853 |
+
{researchSummary.interpretation}
|
| 1854 |
+
</p>
|
| 1855 |
+
)}
|
| 1856 |
+
<div className="flex flex-wrap gap-2 text-[11px] text-stone-700 dark:text-stone-200">
|
| 1857 |
+
{researchSummary.methodsCount > 0 && (
|
| 1858 |
+
<span className="rounded-full border border-sky-200/80 bg-white/80 px-2.5 py-1 dark:border-sky-900/50 dark:bg-stone-950/70">
|
| 1859 |
+
{researchSummary.methodsCount} method note{researchSummary.methodsCount === 1 ? "" : "s"}
|
| 1860 |
+
</span>
|
| 1861 |
+
)}
|
| 1862 |
+
{researchSummary.metricsCount > 0 && (
|
| 1863 |
+
<span className="rounded-full border border-sky-200/80 bg-white/80 px-2.5 py-1 dark:border-sky-900/50 dark:bg-stone-950/70">
|
| 1864 |
+
{researchSummary.metricsCount} documented metric{researchSummary.metricsCount === 1 ? "" : "s"}
|
| 1865 |
+
</span>
|
| 1866 |
+
)}
|
| 1867 |
+
{researchSummary.similarCount > 0 && (
|
| 1868 |
+
<span className="rounded-full border border-sky-200/80 bg-white/80 px-2.5 py-1 dark:border-sky-900/50 dark:bg-stone-950/70">
|
| 1869 |
+
{researchSummary.similarCount} related benchmark{researchSummary.similarCount === 1 ? "" : "s"}
|
| 1870 |
+
</span>
|
| 1871 |
+
)}
|
| 1872 |
+
{researchSummary.missingMethodCount > 0 && (
|
| 1873 |
+
<span className="rounded-full border border-rose-200/80 bg-white/80 px-2.5 py-1 text-rose-700 dark:border-rose-900/50 dark:bg-stone-950/70 dark:text-rose-200">
|
| 1874 |
+
{researchSummary.missingMethodCount} missing method field{researchSummary.missingMethodCount === 1 ? "" : "s"}
|
| 1875 |
+
</span>
|
| 1876 |
+
)}
|
| 1877 |
+
</div>
|
| 1878 |
+
</div>
|
| 1879 |
+
)}
|
| 1880 |
+
|
| 1881 |
{node.matrixPreview && (
|
| 1882 |
<div className="mb-4 overflow-hidden rounded-[1.2rem] border border-stone-200/80 bg-stone-50/85 dark:border-stone-800/80 dark:bg-stone-900/85">
|
| 1883 |
<div className="space-y-2 bg-white/92 px-3 py-3 dark:bg-stone-950/92">
|
|
|
|
| 1902 |
<div className="mb-4 grid gap-px overflow-hidden rounded-[1.2rem] border border-stone-200/80 bg-stone-200/80 dark:border-stone-800/80 dark:bg-stone-800/80 sm:grid-cols-2 xl:grid-cols-3">
|
| 1903 |
{topScoreLabel !== "—" && (
|
| 1904 |
<div className="bg-white/90 px-3 py-3 dark:bg-stone-950/90">
|
| 1905 |
+
<div className="text-[10px] font-semibold uppercase tracking-[0.16em] text-stone-500 dark:text-stone-400">Reported score</div>
|
| 1906 |
<div className="mt-1 text-sm font-semibold text-stone-900 dark:text-stone-100">{topScoreLabel}</div>
|
| 1907 |
</div>
|
| 1908 |
)}
|
| 1909 |
<div className="bg-white/90 px-3 py-3 dark:bg-stone-950/90">
|
| 1910 |
+
<div className="text-[10px] font-semibold uppercase tracking-[0.16em] text-stone-500 dark:text-stone-400">Dataset or record</div>
|
| 1911 |
<div className="mt-1 truncate text-sm font-semibold text-stone-900 dark:text-stone-100">
|
| 1912 |
{node.sourceLabel}
|
| 1913 |
</div>
|
| 1914 |
</div>
|
| 1915 |
<div className="bg-white/90 px-3 py-3 dark:bg-stone-950/90">
|
| 1916 |
+
<div className="text-[10px] font-semibold uppercase tracking-[0.16em] text-stone-500 dark:text-stone-400">Linked instances</div>
|
| 1917 |
<div className="mt-1 text-sm font-semibold text-stone-900 dark:text-stone-100">
|
| 1918 |
{node.instanceDataLabel}
|
| 1919 |
</div>
|
app/page.tsx
CHANGED
|
@@ -57,7 +57,7 @@ export default async function HomePage() {
|
|
| 57 |
Public reporting for AI evaluations.
|
| 58 |
</h1>
|
| 59 |
<p className="max-w-3xl text-lg leading-8 text-muted-foreground sm:text-xl">
|
| 60 |
-
Start from the question you have: which models
|
| 61 |
</p>
|
| 62 |
</div>
|
| 63 |
|
|
@@ -86,20 +86,20 @@ export default async function HomePage() {
|
|
| 86 |
<div className="grid md:grid-cols-3 md:divide-x md:divide-border/60">
|
| 87 |
<SignalCard
|
| 88 |
icon={<Database className="h-4 w-4" />}
|
| 89 |
-
title="
|
| 90 |
-
body="
|
| 91 |
tone="sky"
|
| 92 |
/>
|
| 93 |
<SignalCard
|
| 94 |
icon={<Scale className="h-4 w-4" />}
|
| 95 |
-
title="
|
| 96 |
-
body="Configuration gaps and evaluator relationships
|
| 97 |
tone="amber"
|
| 98 |
/>
|
| 99 |
<SignalCard
|
| 100 |
icon={<BookOpenText className="h-4 w-4" />}
|
| 101 |
-
title="
|
| 102 |
-
body="Research and policy views
|
| 103 |
tone="emerald"
|
| 104 |
/>
|
| 105 |
</div>
|
|
@@ -113,7 +113,7 @@ export default async function HomePage() {
|
|
| 113 |
Start from a reader question
|
| 114 |
</div>
|
| 115 |
<div className="mt-1 text-sm leading-6 text-muted-foreground">
|
| 116 |
-
The site
|
| 117 |
</div>
|
| 118 |
</div>
|
| 119 |
<Database className="h-4 w-4 text-muted-foreground" />
|
|
@@ -121,27 +121,27 @@ export default async function HomePage() {
|
|
| 121 |
|
| 122 |
<div className="mt-5 space-y-3">
|
| 123 |
<InquiryRow
|
| 124 |
-
title="Which models have
|
| 125 |
-
body="Use the models view to scan
|
| 126 |
href="/models"
|
| 127 |
/>
|
| 128 |
<InquiryRow
|
| 129 |
title="How is one benchmark being reported?"
|
| 130 |
-
body="Use the evaluations view to inspect
|
| 131 |
href="/evals"
|
| 132 |
/>
|
| 133 |
<InquiryRow
|
| 134 |
-
title="
|
| 135 |
-
body="
|
| 136 |
href="/developers"
|
| 137 |
/>
|
| 138 |
</div>
|
| 139 |
|
| 140 |
<div className="mt-5 grid gap-3 border-t border-border/60 pt-5 sm:grid-cols-2">
|
| 141 |
<QuietStat label="Models" value={models.length.toString()} detail="Tracked in the current corpus" tone="amber" />
|
| 142 |
-
<QuietStat label="Evaluations" value={evalSummaries.length.toString()} detail="Benchmark
|
| 143 |
<QuietStat label="Developers" value={developerCount.toString()} detail="Organizations represented" tone="emerald" />
|
| 144 |
-
<QuietStat label="Reported results" value={totalReportedResults.toLocaleString()} detail={`Avg ${avgBenchmarksPerModel.toFixed(1)}
|
| 145 |
</div>
|
| 146 |
</aside>
|
| 147 |
</div>
|
|
@@ -151,20 +151,20 @@ export default async function HomePage() {
|
|
| 151 |
<RoutePanel
|
| 152 |
href="/models"
|
| 153 |
icon={<Database className="h-4 w-4" />}
|
| 154 |
-
title="Model
|
| 155 |
-
body="See
|
| 156 |
/>
|
| 157 |
<RoutePanel
|
| 158 |
href="/evals"
|
| 159 |
icon={<BookOpenText className="h-4 w-4" />}
|
| 160 |
-
title="Benchmark
|
| 161 |
-
body="Inspect how
|
| 162 |
/>
|
| 163 |
<RoutePanel
|
| 164 |
href="/about"
|
| 165 |
icon={<Scale className="h-4 w-4" />}
|
| 166 |
-
title="Project
|
| 167 |
-
body="Read why this reporting format exists and how
|
| 168 |
/>
|
| 169 |
</div>
|
| 170 |
|
|
@@ -200,7 +200,7 @@ function InquiryRow({
|
|
| 200 |
href: string
|
| 201 |
}) {
|
| 202 |
return (
|
| 203 |
-
<Link href={href} className="group rounded-[1.3rem] border border-border/70 bg-muted/10 p-4 transition-colors hover:bg-muted/20">
|
| 204 |
<div className="flex items-start justify-between gap-3">
|
| 205 |
<div>
|
| 206 |
<div className="text-sm font-semibold text-foreground">{title}</div>
|
|
|
|
| 57 |
Public reporting for AI evaluations.
|
| 58 |
</h1>
|
| 59 |
<p className="max-w-3xl text-lg leading-8 text-muted-foreground sm:text-xl">
|
| 60 |
+
Start from the question you have: which models have more reported benchmarks, which benchmarks are sparsely reported, and where the record is still incomplete.
|
| 61 |
</p>
|
| 62 |
</div>
|
| 63 |
|
|
|
|
| 86 |
<div className="grid md:grid-cols-3 md:divide-x md:divide-border/60">
|
| 87 |
<SignalCard
|
| 88 |
icon={<Database className="h-4 w-4" />}
|
| 89 |
+
title="Reported benchmarks"
|
| 90 |
+
body="See which benchmarks, settings, and sources are actually documented."
|
| 91 |
tone="sky"
|
| 92 |
/>
|
| 93 |
<SignalCard
|
| 94 |
icon={<Scale className="h-4 w-4" />}
|
| 95 |
+
title="Comparison context"
|
| 96 |
+
body="Configuration gaps and evaluator relationships stay attached to each record."
|
| 97 |
tone="amber"
|
| 98 |
/>
|
| 99 |
<SignalCard
|
| 100 |
icon={<BookOpenText className="h-4 w-4" />}
|
| 101 |
+
title="Reader modes"
|
| 102 |
+
body="Research and policy views prioritize different fields without hiding the same record."
|
| 103 |
tone="emerald"
|
| 104 |
/>
|
| 105 |
</div>
|
|
|
|
| 113 |
Start from a reader question
|
| 114 |
</div>
|
| 115 |
<div className="mt-1 text-sm leading-6 text-muted-foreground">
|
| 116 |
+
The site works best when you inspect reported records, not just the ranking order.
|
| 117 |
</div>
|
| 118 |
</div>
|
| 119 |
<Database className="h-4 w-4 text-muted-foreground" />
|
|
|
|
| 121 |
|
| 122 |
<div className="mt-5 space-y-3">
|
| 123 |
<InquiryRow
|
| 124 |
+
title="Which models have the most reported benchmarks?"
|
| 125 |
+
body="Use the models view to scan reported benchmarks, result counts, and where reporting is sparse."
|
| 126 |
href="/models"
|
| 127 |
/>
|
| 128 |
<InquiryRow
|
| 129 |
title="How is one benchmark being reported?"
|
| 130 |
+
body="Use the evaluations view to inspect benchmark notes, score ranges, and missing setup details."
|
| 131 |
href="/evals"
|
| 132 |
/>
|
| 133 |
<InquiryRow
|
| 134 |
+
title="Which organizations are publishing results?"
|
| 135 |
+
body="Open the developer pages to see which organizations appear most often in the reported records."
|
| 136 |
href="/developers"
|
| 137 |
/>
|
| 138 |
</div>
|
| 139 |
|
| 140 |
<div className="mt-5 grid gap-3 border-t border-border/60 pt-5 sm:grid-cols-2">
|
| 141 |
<QuietStat label="Models" value={models.length.toString()} detail="Tracked in the current corpus" tone="amber" />
|
| 142 |
+
<QuietStat label="Evaluations" value={evalSummaries.length.toString()} detail="Benchmark records with linked details" tone="sky" />
|
| 143 |
<QuietStat label="Developers" value={developerCount.toString()} detail="Organizations represented" tone="emerald" />
|
| 144 |
+
<QuietStat label="Reported results" value={totalReportedResults.toLocaleString()} detail={`Avg ${avgBenchmarksPerModel.toFixed(1)} reported benchmarks per model`} tone="slate" />
|
| 145 |
</div>
|
| 146 |
</aside>
|
| 147 |
</div>
|
|
|
|
| 151 |
<RoutePanel
|
| 152 |
href="/models"
|
| 153 |
icon={<Database className="h-4 w-4" />}
|
| 154 |
+
title="Model records"
|
| 155 |
+
body="See which benchmarks are reported for a model and where reporting is missing or thin."
|
| 156 |
/>
|
| 157 |
<RoutePanel
|
| 158 |
href="/evals"
|
| 159 |
icon={<BookOpenText className="h-4 w-4" />}
|
| 160 |
+
title="Benchmark records"
|
| 161 |
+
body="Inspect how one benchmark is reported across models, including slices, setup notes, and sources."
|
| 162 |
/>
|
| 163 |
<RoutePanel
|
| 164 |
href="/about"
|
| 165 |
icon={<Scale className="h-4 w-4" />}
|
| 166 |
+
title="Project notes"
|
| 167 |
+
body="Read why this reporting format exists and how the same records support research and policy use."
|
| 168 |
/>
|
| 169 |
</div>
|
| 170 |
|
|
|
|
| 200 |
href: string
|
| 201 |
}) {
|
| 202 |
return (
|
| 203 |
+
<Link href={href} className="group block rounded-[1.3rem] border border-border/70 bg-muted/10 p-4 transition-colors hover:bg-muted/20">
|
| 204 |
<div className="flex items-start justify-between gap-3">
|
| 205 |
<div>
|
| 206 |
<div className="text-sm font-semibold text-foreground">{title}</div>
|
components/eval-detail.tsx
CHANGED
|
@@ -300,6 +300,7 @@ export function EvalDetail({ summary }: EvalDetailProps) {
|
|
| 300 |
const hasMultiMetricLeaderboard =
|
| 301 |
(summary.leaderboard_metrics?.length ?? 0) > 1 &&
|
| 302 |
(summary.leaderboard_rows?.length ?? 0) > 0
|
|
|
|
| 303 |
const [expandedRows, setExpandedRows] = useState<Record<string, boolean>>({})
|
| 304 |
const [leaderboardPage, setLeaderboardPage] = useState(1)
|
| 305 |
const [minParamStep, setMinParamStep] = useState(0)
|
|
@@ -409,203 +410,198 @@ export function EvalDetail({ summary }: EvalDetailProps) {
|
|
| 409 |
return (
|
| 410 |
<div className="space-y-6">
|
| 411 |
<Card className="overflow-hidden">
|
| 412 |
-
<
|
| 413 |
-
<
|
| 414 |
-
<
|
| 415 |
-
|
| 416 |
-
|
| 417 |
-
|
| 418 |
-
|
| 419 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 420 |
<Badge variant="secondary" className="font-normal">
|
| 421 |
-
{summary.
|
| 422 |
</Badge>
|
| 423 |
-
) : (
|
| 424 |
<Badge variant="secondary" className="font-normal">
|
| 425 |
-
|
|
|
|
|
|
|
| 426 |
</Badge>
|
| 427 |
-
|
| 428 |
-
<Badge variant="secondary" className="font-normal capitalize">
|
| 429 |
-
{summary.metric_config.score_type}
|
| 430 |
-
</Badge>
|
| 431 |
-
<Badge variant="secondary" className="font-normal">
|
| 432 |
-
{summary.metric_config.lower_is_better ? "Lower is better" : "Higher is better"}
|
| 433 |
-
</Badge>
|
| 434 |
-
{summary.tags?.languages && summary.tags.languages.length > 0 && (
|
| 435 |
-
<Badge variant="secondary" className="font-normal">
|
| 436 |
-
{summary.tags.languages.join(", ")}
|
| 437 |
-
</Badge>
|
| 438 |
-
)}
|
| 439 |
-
</div>
|
| 440 |
-
|
| 441 |
-
<div className="space-y-1">
|
| 442 |
-
<div className="text-2xl font-semibold tracking-tight sm:text-[1.9rem]">{summary.evaluation_name}</div>
|
| 443 |
-
<p className="max-w-3xl text-sm leading-6 text-muted-foreground">
|
| 444 |
-
{summary.metric_config.evaluation_description}
|
| 445 |
-
</p>
|
| 446 |
</div>
|
| 447 |
-
|
| 448 |
-
|
| 449 |
-
|
| 450 |
-
|
| 451 |
-
</p>
|
| 452 |
)}
|
| 453 |
-
</
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 454 |
|
| 455 |
-
|
| 456 |
-
|
| 457 |
-
|
| 458 |
-
|
| 459 |
-
|
| 460 |
-
|
| 461 |
-
|
| 462 |
-
|
| 463 |
-
|
| 464 |
-
<div className="mt-1 text-2xl font-semibold">
|
| 465 |
-
{hasMultiMetricLeaderboard ? summary.metrics_count ?? summary.leaderboard_metrics?.length ?? 1 : isResearchView ? avgScoreLabel : summary.metrics_count ?? 1}
|
| 466 |
-
</div>
|
| 467 |
-
</div>
|
| 468 |
-
<div className="rounded-2xl border border-emerald-200/80 bg-emerald-50/80 px-4 py-3 dark:border-emerald-900/40 dark:bg-emerald-950/20">
|
| 469 |
-
<div className="text-[11px] font-semibold uppercase tracking-[0.18em] text-emerald-700 dark:text-emerald-200">
|
| 470 |
-
{hasMultiMetricLeaderboard || !isResearchView ? "Source dataset" : "Top model"}
|
| 471 |
-
</div>
|
| 472 |
-
<div className="mt-1 text-sm font-semibold text-emerald-950 dark:text-emerald-50">
|
| 473 |
-
{hasMultiMetricLeaderboard || !isResearchView
|
| 474 |
-
? sourceDatasetLabel
|
| 475 |
-
: summary.best_model?.name ?? "Unknown"}
|
| 476 |
</div>
|
| 477 |
-
|
| 478 |
-
|
| 479 |
-
|
|
|
|
|
|
|
| 480 |
</div>
|
| 481 |
-
|
| 482 |
-
|
| 483 |
-
|
| 484 |
-
|
| 485 |
-
|
| 486 |
-
|
| 487 |
-
|
| 488 |
-
{hasMultiMetricLeaderboard || !isResearchView
|
| 489 |
-
? instanceDataLabel
|
| 490 |
-
: summary.worst_model?.name ?? "Unknown"}
|
| 491 |
-
</div>
|
| 492 |
-
{!hasMultiMetricLeaderboard && isResearchView && summary.worst_model && (
|
| 493 |
-
<div className="mt-1 text-xs text-amber-700/80 dark:text-amber-200/80">
|
| 494 |
-
{formatRawScore(summary.worst_model.score, summary.metric_config.unit)}
|
| 495 |
</div>
|
| 496 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 497 |
</div>
|
| 498 |
-
</div>
|
| 499 |
-
</div>
|
| 500 |
|
| 501 |
-
|
| 502 |
-
|
| 503 |
-
|
| 504 |
-
</div>
|
| 505 |
-
<dl className="mt-3 grid gap-x-6 gap-y-3 text-sm sm:grid-cols-2 xl:grid-cols-5">
|
| 506 |
-
<div>
|
| 507 |
-
<dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
|
| 508 |
-
Composite benchmark
|
| 509 |
-
</dt>
|
| 510 |
-
<dd className="mt-1 break-words font-medium">
|
| 511 |
-
{summary.is_aggregated
|
| 512 |
-
? summary.aggregate_sources?.map((source) => source.composite_benchmark_name).join(", ") || "Multiple composite benchmarks"
|
| 513 |
-
: summary.composite_benchmark_name}
|
| 514 |
-
</dd>
|
| 515 |
-
</div>
|
| 516 |
-
<div>
|
| 517 |
-
<dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
|
| 518 |
-
{isResearchView ? "Single benchmark ID" : "What this covers"}
|
| 519 |
-
</dt>
|
| 520 |
-
<dd className="mt-1 break-words font-medium">
|
| 521 |
-
{isResearchView
|
| 522 |
-
? summary.evaluation_id
|
| 523 |
-
: summary.is_aggregated
|
| 524 |
-
? summary.metric_config.evaluation_description
|
| 525 |
-
: summary.metric_config.evaluation_description}
|
| 526 |
-
</dd>
|
| 527 |
-
</div>
|
| 528 |
-
<div>
|
| 529 |
-
<dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
|
| 530 |
-
{isResearchView ? "Score scale" : "How to read scores"}
|
| 531 |
-
</dt>
|
| 532 |
-
<dd className="mt-1 font-medium">
|
| 533 |
-
{isResearchView
|
| 534 |
-
? `${summary.metric_config.min_score ?? 0} - ${summary.metric_config.max_score ?? 1}`
|
| 535 |
-
: scoreDirectionLabel}
|
| 536 |
-
</dd>
|
| 537 |
-
</div>
|
| 538 |
-
{summary.tags?.domains && summary.tags.domains.length > 0 && (
|
| 539 |
-
<div>
|
| 540 |
-
<dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">Domain coverage</dt>
|
| 541 |
-
<dd className="mt-1 font-medium capitalize">
|
| 542 |
-
{summary.tags.domains.slice(0, 2).join(", ")}
|
| 543 |
-
{summary.tags.domains.length > 2 ? ` +${summary.tags.domains.length - 2} more` : ""}
|
| 544 |
-
</dd>
|
| 545 |
</div>
|
| 546 |
-
|
| 547 |
-
|
| 548 |
-
|
| 549 |
-
|
| 550 |
-
|
| 551 |
-
|
| 552 |
-
|
| 553 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 554 |
</div>
|
| 555 |
-
</dl>
|
| 556 |
-
</div>
|
| 557 |
-
</CardContent>
|
| 558 |
-
</Card>
|
| 559 |
|
| 560 |
-
|
| 561 |
-
|
| 562 |
-
|
| 563 |
-
|
| 564 |
-
|
| 565 |
-
|
| 566 |
-
|
| 567 |
-
</CardHeader>
|
| 568 |
-
<CardContent className="space-y-6 p-5 sm:p-6">
|
| 569 |
-
{summary.root_metrics && summary.root_metrics.length > 0 && (
|
| 570 |
-
<section className="space-y-3">
|
| 571 |
-
<div>
|
| 572 |
-
<div className="text-sm font-semibold">Benchmark-level metrics</div>
|
| 573 |
-
<div className="text-xs text-muted-foreground">
|
| 574 |
-
Benchmark summary metrics used in this evaluation view.
|
| 575 |
</div>
|
| 576 |
-
</div>
|
| 577 |
-
<div className="flex flex-wrap gap-2">
|
| 578 |
-
{summary.root_metrics.map((metric) => (
|
| 579 |
-
<span
|
| 580 |
-
key={metric.metric_summary_id}
|
| 581 |
-
className="rounded-full border border-border/70 bg-background px-3 py-1.5 text-xs font-medium"
|
| 582 |
-
title={metric.canonical_display_name || metric.display_name}
|
| 583 |
-
>
|
| 584 |
-
{getCompactMetricLabel(metric.display_name)}
|
| 585 |
-
{typeof metric.top_score === "number" ? ` · ${formatRawScore(metric.top_score, metric.unit)}` : ""}
|
| 586 |
-
</span>
|
| 587 |
-
))}
|
| 588 |
-
</div>
|
| 589 |
-
</section>
|
| 590 |
-
)}
|
| 591 |
|
| 592 |
-
|
| 593 |
-
|
| 594 |
-
|
| 595 |
-
|
| 596 |
-
|
| 597 |
-
|
| 598 |
-
|
| 599 |
-
<div key={subtask.subtask_key} className="rounded-2xl border bg-background p-4">
|
| 600 |
-
<div className="font-semibold">{subtask.display_name || subtask.subtask_name}</div>
|
| 601 |
-
<div className="mt-1 text-xs text-muted-foreground" title={subtask.canonical_display_name || subtask.display_name}>
|
| 602 |
-
{subtask.canonical_display_name || subtask.display_name}
|
| 603 |
</div>
|
| 604 |
-
<div className="
|
| 605 |
-
{
|
| 606 |
<span
|
| 607 |
key={metric.metric_summary_id}
|
| 608 |
-
className="rounded-full border border-border/70 bg-
|
| 609 |
title={metric.canonical_display_name || metric.display_name}
|
| 610 |
>
|
| 611 |
{getCompactMetricLabel(metric.display_name)}
|
|
@@ -614,18 +610,50 @@ export function EvalDetail({ summary }: EvalDetailProps) {
|
|
| 614 |
))}
|
| 615 |
</div>
|
| 616 |
</div>
|
| 617 |
-
)
|
| 618 |
-
</div>
|
| 619 |
-
</section>
|
| 620 |
-
)}
|
| 621 |
-
</CardContent>
|
| 622 |
-
</Card>
|
| 623 |
-
) : null}
|
| 624 |
|
| 625 |
-
|
| 626 |
-
|
| 627 |
-
|
| 628 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 629 |
|
| 630 |
{hasMultiMetricLeaderboard ? (
|
| 631 |
<MultiMetricLeaderboard summary={summary} isResearchView={isResearchView} />
|
|
@@ -1135,11 +1163,6 @@ export function EvalDetail({ summary }: EvalDetailProps) {
|
|
| 1135 |
</CardContent>
|
| 1136 |
</Card>
|
| 1137 |
)}
|
| 1138 |
-
|
| 1139 |
-
{/* Research: benchmark card details AFTER the leaderboard, collapsed by default */}
|
| 1140 |
-
{isResearchView && summary.benchmark_card && (
|
| 1141 |
-
<ResearchBenchmarkCardCollapsible card={summary.benchmark_card} />
|
| 1142 |
-
)}
|
| 1143 |
</div>
|
| 1144 |
)
|
| 1145 |
}
|
|
@@ -1433,8 +1456,8 @@ function MultiMetricLeaderboard({
|
|
| 1433 |
</div>
|
| 1434 |
<CardDescription>
|
| 1435 |
{isResearchView
|
| 1436 |
-
? "Each column is a reported benchmark
|
| 1437 |
-
: "Each column is a separately reported
|
| 1438 |
</CardDescription>
|
| 1439 |
</div>
|
| 1440 |
|
|
@@ -1446,8 +1469,8 @@ function MultiMetricLeaderboard({
|
|
| 1446 |
</Badge>
|
| 1447 |
<Badge variant="outline">
|
| 1448 |
{visibleMetrics.length === leaderboardMetrics.length
|
| 1449 |
-
? `${leaderboardMetrics.length}
|
| 1450 |
-
: `${visibleMetrics.length} of ${leaderboardMetrics.length}
|
| 1451 |
</Badge>
|
| 1452 |
{hasParameterData && (numericMinParams != null || numericMaxParams != null) && (
|
| 1453 |
<Badge variant="outline">
|
|
@@ -1462,7 +1485,7 @@ function MultiMetricLeaderboard({
|
|
| 1462 |
</Button>
|
| 1463 |
</DropdownMenuTrigger>
|
| 1464 |
<DropdownMenuContent align="end" className="w-80">
|
| 1465 |
-
<DropdownMenuLabel>Visible
|
| 1466 |
<DropdownMenuItem onSelect={() => setVisibleMetricKeys(allMetricKeys)}>
|
| 1467 |
Show all
|
| 1468 |
</DropdownMenuItem>
|
|
@@ -1638,7 +1661,7 @@ function MultiMetricLeaderboard({
|
|
| 1638 |
onClick={() => handleSort("coverage")}
|
| 1639 |
className="w-full text-right font-semibold transition-colors hover:text-primary"
|
| 1640 |
>
|
| 1641 |
-
|
| 1642 |
</button>
|
| 1643 |
</TableHead>
|
| 1644 |
{visibleMetrics.map((metric) => (
|
|
@@ -1773,8 +1796,18 @@ function MultiMetricLeaderboard({
|
|
| 1773 |
)
|
| 1774 |
}
|
| 1775 |
|
| 1776 |
-
function
|
| 1777 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1778 |
return (
|
| 1779 |
<Collapsible open={open} onOpenChange={setOpen}>
|
| 1780 |
<CollapsibleTrigger asChild>
|
|
@@ -1797,7 +1830,11 @@ function ResearchBenchmarkCardCollapsible({ card }: { card: BenchmarkCard }) {
|
|
| 1797 |
</button>
|
| 1798 |
</CollapsibleTrigger>
|
| 1799 |
<CollapsibleContent className="mt-2">
|
| 1800 |
-
<BenchmarkCardPanel
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1801 |
</CollapsibleContent>
|
| 1802 |
</Collapsible>
|
| 1803 |
)
|
|
|
|
| 300 |
const hasMultiMetricLeaderboard =
|
| 301 |
(summary.leaderboard_metrics?.length ?? 0) > 1 &&
|
| 302 |
(summary.leaderboard_rows?.length ?? 0) > 0
|
| 303 |
+
const [overviewOpen, setOverviewOpen] = useState(true)
|
| 304 |
const [expandedRows, setExpandedRows] = useState<Record<string, boolean>>({})
|
| 305 |
const [leaderboardPage, setLeaderboardPage] = useState(1)
|
| 306 |
const [minParamStep, setMinParamStep] = useState(0)
|
|
|
|
| 410 |
return (
|
| 411 |
<div className="space-y-6">
|
| 412 |
<Card className="overflow-hidden">
|
| 413 |
+
<Collapsible open={overviewOpen} onOpenChange={setOverviewOpen}>
|
| 414 |
+
<CollapsibleTrigger asChild>
|
| 415 |
+
<button
|
| 416 |
+
type="button"
|
| 417 |
+
className="flex w-full items-center justify-between gap-4 border-b bg-muted/10 px-4 py-3 text-left transition-colors hover:bg-muted/15 sm:px-5"
|
| 418 |
+
>
|
| 419 |
+
<div className="min-w-0 space-y-1">
|
| 420 |
+
<div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-muted-foreground">
|
| 421 |
+
{isResearchView ? "Benchmark overview" : "Reading overview"}
|
| 422 |
+
</div>
|
| 423 |
+
<div className="flex flex-wrap items-center gap-2">
|
| 424 |
+
<span className="text-base font-semibold tracking-tight sm:text-lg">{summary.evaluation_name}</span>
|
| 425 |
<Badge variant="secondary" className="font-normal">
|
| 426 |
+
{summary.models_count} models
|
| 427 |
</Badge>
|
|
|
|
| 428 |
<Badge variant="secondary" className="font-normal">
|
| 429 |
+
{hasMultiMetricLeaderboard
|
| 430 |
+
? `${summary.metrics_count ?? summary.leaderboard_metrics?.length ?? 1} measures`
|
| 431 |
+
: `${summary.metrics_count ?? 1} ${(summary.metrics_count ?? 1) === 1 ? "measure" : "measures"}`}
|
| 432 |
</Badge>
|
| 433 |
+
</div>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 434 |
</div>
|
| 435 |
+
{overviewOpen ? (
|
| 436 |
+
<ChevronUp className="h-4 w-4 shrink-0 text-muted-foreground" />
|
| 437 |
+
) : (
|
| 438 |
+
<ChevronDown className="h-4 w-4 shrink-0 text-muted-foreground" />
|
|
|
|
| 439 |
)}
|
| 440 |
+
</button>
|
| 441 |
+
</CollapsibleTrigger>
|
| 442 |
+
|
| 443 |
+
<CollapsibleContent>
|
| 444 |
+
<CardContent className="space-y-4 p-4 sm:p-5">
|
| 445 |
+
<div className="flex flex-col gap-4 xl:flex-row xl:items-start xl:justify-between">
|
| 446 |
+
<div className="space-y-2.5">
|
| 447 |
+
<div className="flex flex-wrap items-center gap-2">
|
| 448 |
+
<Badge variant="outline" className="border-border/60 bg-background/80 text-[11px] uppercase tracking-[0.18em]">
|
| 449 |
+
{summary.is_aggregated ? "Merged Benchmark" : "Single Benchmark"}
|
| 450 |
+
</Badge>
|
| 451 |
+
{summary.is_aggregated ? (
|
| 452 |
+
<Badge variant="secondary" className="font-normal">
|
| 453 |
+
{summary.aggregate_sources?.length ?? 0} composite benchmarks
|
| 454 |
+
</Badge>
|
| 455 |
+
) : (
|
| 456 |
+
<Badge variant="secondary" className="font-normal">
|
| 457 |
+
Composite: {summary.composite_benchmark_name}
|
| 458 |
+
</Badge>
|
| 459 |
+
)}
|
| 460 |
+
<Badge variant="secondary" className="font-normal capitalize">
|
| 461 |
+
{summary.metric_config.score_type}
|
| 462 |
+
</Badge>
|
| 463 |
+
<Badge variant="secondary" className="font-normal">
|
| 464 |
+
{summary.metric_config.lower_is_better ? "Lower is better" : "Higher is better"}
|
| 465 |
+
</Badge>
|
| 466 |
+
{summary.tags?.languages && summary.tags.languages.length > 0 && (
|
| 467 |
+
<Badge variant="secondary" className="font-normal">
|
| 468 |
+
{summary.tags.languages.join(", ")}
|
| 469 |
+
</Badge>
|
| 470 |
+
)}
|
| 471 |
+
</div>
|
| 472 |
|
| 473 |
+
<p className="max-w-3xl text-sm leading-6 text-muted-foreground">
|
| 474 |
+
{summary.metric_config.evaluation_description}
|
| 475 |
+
</p>
|
| 476 |
+
|
| 477 |
+
{!isResearchView && (
|
| 478 |
+
<p className="max-w-3xl text-sm leading-6 text-muted-foreground">
|
| 479 |
+
{`${summary.benchmark_card?.purpose_and_intended_users?.goal ?? "This benchmark reports a capability result."} Scores should be read alongside benchmark scope, metric definitions, and the source dataset context.`}
|
| 480 |
+
</p>
|
| 481 |
+
)}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 482 |
</div>
|
| 483 |
+
|
| 484 |
+
<div className="grid w-full grid-cols-2 gap-2 xl:grid-cols-4">
|
| 485 |
+
<div className="rounded-xl border border-sky-200/80 bg-sky-50/80 px-3 py-2.5 dark:border-sky-900/40 dark:bg-sky-950/20">
|
| 486 |
+
<div className="text-[11px] font-semibold uppercase tracking-[0.18em] text-sky-700 dark:text-sky-200">Models</div>
|
| 487 |
+
<div className="mt-1 text-xl font-semibold text-sky-950 dark:text-sky-50">{summary.models_count}</div>
|
| 488 |
</div>
|
| 489 |
+
<div className="rounded-xl border border-border/70 bg-muted/20 px-3 py-2.5">
|
| 490 |
+
<div className="text-[11px] font-semibold uppercase tracking-[0.18em] text-muted-foreground">
|
| 491 |
+
{hasMultiMetricLeaderboard ? "Measures" : isResearchView ? "Avg score" : "Measures"}
|
| 492 |
+
</div>
|
| 493 |
+
<div className="mt-1 text-xl font-semibold">
|
| 494 |
+
{hasMultiMetricLeaderboard ? summary.metrics_count ?? summary.leaderboard_metrics?.length ?? 1 : isResearchView ? avgScoreLabel : summary.metrics_count ?? 1}
|
| 495 |
+
</div>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 496 |
</div>
|
| 497 |
+
<div className="rounded-xl border border-emerald-200/80 bg-emerald-50/80 px-3 py-2.5 dark:border-emerald-900/40 dark:bg-emerald-950/20">
|
| 498 |
+
<div className="text-[11px] font-semibold uppercase tracking-[0.18em] text-emerald-700 dark:text-emerald-200">
|
| 499 |
+
{hasMultiMetricLeaderboard || !isResearchView ? "Source dataset" : "Top model"}
|
| 500 |
+
</div>
|
| 501 |
+
<div className="mt-1 text-sm font-semibold text-emerald-950 dark:text-emerald-50">
|
| 502 |
+
{hasMultiMetricLeaderboard || !isResearchView
|
| 503 |
+
? sourceDatasetLabel
|
| 504 |
+
: summary.best_model?.name ?? "Unknown"}
|
| 505 |
+
</div>
|
| 506 |
+
{!hasMultiMetricLeaderboard && isResearchView && summary.best_model && (
|
| 507 |
+
<div className="mt-1 text-xs text-emerald-700/80 dark:text-emerald-200/80">
|
| 508 |
+
{formatRawScore(summary.best_model.score, summary.metric_config.unit)}
|
| 509 |
+
</div>
|
| 510 |
+
)}
|
| 511 |
+
</div>
|
| 512 |
+
<div className="rounded-xl border border-amber-200/80 bg-amber-50/80 px-3 py-2.5 dark:border-amber-900/40 dark:bg-amber-950/20">
|
| 513 |
+
<div className="text-[11px] font-semibold uppercase tracking-[0.18em] text-amber-700 dark:text-amber-200">
|
| 514 |
+
{hasMultiMetricLeaderboard || !isResearchView ? "Instance data" : "Bottom model"}
|
| 515 |
+
</div>
|
| 516 |
+
<div className="mt-1 text-sm font-semibold text-amber-950 dark:text-amber-50">
|
| 517 |
+
{hasMultiMetricLeaderboard || !isResearchView
|
| 518 |
+
? instanceDataLabel
|
| 519 |
+
: summary.worst_model?.name ?? "Unknown"}
|
| 520 |
+
</div>
|
| 521 |
+
{!hasMultiMetricLeaderboard && isResearchView && summary.worst_model && (
|
| 522 |
+
<div className="mt-1 text-xs text-amber-700/80 dark:text-amber-200/80">
|
| 523 |
+
{formatRawScore(summary.worst_model.score, summary.metric_config.unit)}
|
| 524 |
+
</div>
|
| 525 |
+
)}
|
| 526 |
+
</div>
|
| 527 |
+
</div>
|
| 528 |
</div>
|
|
|
|
|
|
|
| 529 |
|
| 530 |
+
<div className="rounded-2xl border bg-muted/10 p-3.5">
|
| 531 |
+
<div className="text-[11px] font-semibold uppercase tracking-[0.2em] text-muted-foreground">
|
| 532 |
+
{isResearchView ? "Metric specification" : "Reading context"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 533 |
</div>
|
| 534 |
+
<dl className="mt-3 grid gap-x-5 gap-y-3 text-sm sm:grid-cols-2 xl:grid-cols-5">
|
| 535 |
+
<div>
|
| 536 |
+
<dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
|
| 537 |
+
Composite benchmark
|
| 538 |
+
</dt>
|
| 539 |
+
<dd className="mt-1 break-words font-medium">
|
| 540 |
+
{summary.is_aggregated
|
| 541 |
+
? summary.aggregate_sources?.map((source) => source.composite_benchmark_name).join(", ") || "Multiple composite benchmarks"
|
| 542 |
+
: summary.composite_benchmark_name}
|
| 543 |
+
</dd>
|
| 544 |
+
</div>
|
| 545 |
+
<div>
|
| 546 |
+
<dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
|
| 547 |
+
{isResearchView ? "Single benchmark ID" : "What this covers"}
|
| 548 |
+
</dt>
|
| 549 |
+
<dd className="mt-1 break-words font-medium">
|
| 550 |
+
{isResearchView ? summary.evaluation_id : summary.metric_config.evaluation_description}
|
| 551 |
+
</dd>
|
| 552 |
+
</div>
|
| 553 |
+
<div>
|
| 554 |
+
<dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
|
| 555 |
+
{isResearchView ? "Score scale" : "How to read scores"}
|
| 556 |
+
</dt>
|
| 557 |
+
<dd className="mt-1 font-medium">
|
| 558 |
+
{isResearchView
|
| 559 |
+
? `${summary.metric_config.min_score ?? 0} - ${summary.metric_config.max_score ?? 1}`
|
| 560 |
+
: scoreDirectionLabel}
|
| 561 |
+
</dd>
|
| 562 |
+
</div>
|
| 563 |
+
{summary.tags?.domains && summary.tags.domains.length > 0 && (
|
| 564 |
+
<div>
|
| 565 |
+
<dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">Domain tags</dt>
|
| 566 |
+
<dd className="mt-1 font-medium capitalize">
|
| 567 |
+
{summary.tags.domains.slice(0, 2).join(", ")}
|
| 568 |
+
{summary.tags.domains.length > 2 ? ` +${summary.tags.domains.length - 2} more` : ""}
|
| 569 |
+
</dd>
|
| 570 |
+
</div>
|
| 571 |
+
)}
|
| 572 |
+
<div>
|
| 573 |
+
<dt className="text-[11px] font-semibold uppercase tracking-[0.16em] text-muted-foreground">
|
| 574 |
+
{isResearchView ? "Source dataset" : "Instance data"}
|
| 575 |
+
</dt>
|
| 576 |
+
<dd className="mt-1 font-medium">
|
| 577 |
+
{isResearchView ? sourceDatasetLabel : instanceDataLabel}
|
| 578 |
+
</dd>
|
| 579 |
+
</div>
|
| 580 |
+
</dl>
|
| 581 |
</div>
|
|
|
|
|
|
|
|
|
|
|
|
|
| 582 |
|
| 583 |
+
{!hasMultiMetricLeaderboard && (summary.root_metrics?.length || summary.subtasks?.length) ? (
|
| 584 |
+
<section className="rounded-2xl border bg-muted/5 p-3.5">
|
| 585 |
+
<div className="space-y-1">
|
| 586 |
+
<div className="text-sm font-semibold">Benchmark structure</div>
|
| 587 |
+
<div className="text-xs text-muted-foreground">
|
| 588 |
+
Benchmark-level summary metrics and subtask slices grouped in one compact section.
|
| 589 |
+
</div>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 590 |
</div>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 591 |
|
| 592 |
+
{summary.root_metrics && summary.root_metrics.length > 0 && (
|
| 593 |
+
<div className="mt-4 space-y-2.5">
|
| 594 |
+
<div>
|
| 595 |
+
<div className="text-sm font-semibold">Benchmark-level metrics</div>
|
| 596 |
+
<div className="text-xs text-muted-foreground">
|
| 597 |
+
Benchmark summary metrics used in this evaluation view.
|
| 598 |
+
</div>
|
|
|
|
|
|
|
|
|
|
|
|
|
| 599 |
</div>
|
| 600 |
+
<div className="flex flex-wrap gap-2">
|
| 601 |
+
{summary.root_metrics.map((metric) => (
|
| 602 |
<span
|
| 603 |
key={metric.metric_summary_id}
|
| 604 |
+
className="rounded-full border border-border/70 bg-background px-3 py-1.5 text-xs font-medium"
|
| 605 |
title={metric.canonical_display_name || metric.display_name}
|
| 606 |
>
|
| 607 |
{getCompactMetricLabel(metric.display_name)}
|
|
|
|
| 610 |
))}
|
| 611 |
</div>
|
| 612 |
</div>
|
| 613 |
+
)}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 614 |
|
| 615 |
+
{summary.subtasks && summary.subtasks.length > 0 && (
|
| 616 |
+
<div className="mt-4 space-y-2.5">
|
| 617 |
+
<div className="text-sm font-semibold">Subtask breakdown</div>
|
| 618 |
+
<div className="grid gap-3 lg:grid-cols-2">
|
| 619 |
+
{summary.subtasks.map((subtask) => (
|
| 620 |
+
<div key={subtask.subtask_key} className="rounded-xl border bg-background p-3.5">
|
| 621 |
+
<div className="font-semibold">{subtask.display_name || subtask.subtask_name}</div>
|
| 622 |
+
<div className="mt-1 text-xs text-muted-foreground" title={subtask.canonical_display_name || subtask.display_name}>
|
| 623 |
+
{subtask.canonical_display_name || subtask.display_name}
|
| 624 |
+
</div>
|
| 625 |
+
<div className="mt-3 flex flex-wrap gap-2">
|
| 626 |
+
{subtask.metrics.map((metric) => (
|
| 627 |
+
<span
|
| 628 |
+
key={metric.metric_summary_id}
|
| 629 |
+
className="rounded-full border border-border/70 bg-muted/20 px-2.5 py-1 text-[11px] font-medium"
|
| 630 |
+
title={metric.canonical_display_name || metric.display_name}
|
| 631 |
+
>
|
| 632 |
+
{getCompactMetricLabel(metric.display_name)}
|
| 633 |
+
{typeof metric.top_score === "number" ? ` · ${formatRawScore(metric.top_score, metric.unit)}` : ""}
|
| 634 |
+
</span>
|
| 635 |
+
))}
|
| 636 |
+
</div>
|
| 637 |
+
</div>
|
| 638 |
+
))}
|
| 639 |
+
</div>
|
| 640 |
+
</div>
|
| 641 |
+
)}
|
| 642 |
+
</section>
|
| 643 |
+
) : null}
|
| 644 |
+
|
| 645 |
+
{summary.benchmark_card && (
|
| 646 |
+
<BenchmarkCardCollapsible
|
| 647 |
+
card={summary.benchmark_card}
|
| 648 |
+
isResearchView={isResearchView}
|
| 649 |
+
defaultOpen
|
| 650 |
+
defaultRisksOpen={!isResearchView}
|
| 651 |
+
/>
|
| 652 |
+
)}
|
| 653 |
+
</CardContent>
|
| 654 |
+
</CollapsibleContent>
|
| 655 |
+
</Collapsible>
|
| 656 |
+
</Card>
|
| 657 |
|
| 658 |
{hasMultiMetricLeaderboard ? (
|
| 659 |
<MultiMetricLeaderboard summary={summary} isResearchView={isResearchView} />
|
|
|
|
| 1163 |
</CardContent>
|
| 1164 |
</Card>
|
| 1165 |
)}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1166 |
</div>
|
| 1167 |
)
|
| 1168 |
}
|
|
|
|
| 1456 |
</div>
|
| 1457 |
<CardDescription>
|
| 1458 |
{isResearchView
|
| 1459 |
+
? "Each column is a reported benchmark measure. Distinct measures stay separate instead of collapsing into a single raw score."
|
| 1460 |
+
: "Each column is a separately reported measure so the benchmark can be read without flattening different results into one number."}
|
| 1461 |
</CardDescription>
|
| 1462 |
</div>
|
| 1463 |
|
|
|
|
| 1469 |
</Badge>
|
| 1470 |
<Badge variant="outline">
|
| 1471 |
{visibleMetrics.length === leaderboardMetrics.length
|
| 1472 |
+
? `${leaderboardMetrics.length} measures`
|
| 1473 |
+
: `${visibleMetrics.length} of ${leaderboardMetrics.length} measures`}
|
| 1474 |
</Badge>
|
| 1475 |
{hasParameterData && (numericMinParams != null || numericMaxParams != null) && (
|
| 1476 |
<Badge variant="outline">
|
|
|
|
| 1485 |
</Button>
|
| 1486 |
</DropdownMenuTrigger>
|
| 1487 |
<DropdownMenuContent align="end" className="w-80">
|
| 1488 |
+
<DropdownMenuLabel>Visible measure columns</DropdownMenuLabel>
|
| 1489 |
<DropdownMenuItem onSelect={() => setVisibleMetricKeys(allMetricKeys)}>
|
| 1490 |
Show all
|
| 1491 |
</DropdownMenuItem>
|
|
|
|
| 1661 |
onClick={() => handleSort("coverage")}
|
| 1662 |
className="w-full text-right font-semibold transition-colors hover:text-primary"
|
| 1663 |
>
|
| 1664 |
+
Measures present{getSortIndicator("coverage")}
|
| 1665 |
</button>
|
| 1666 |
</TableHead>
|
| 1667 |
{visibleMetrics.map((metric) => (
|
|
|
|
| 1796 |
)
|
| 1797 |
}
|
| 1798 |
|
| 1799 |
+
function BenchmarkCardCollapsible({
|
| 1800 |
+
card,
|
| 1801 |
+
isResearchView,
|
| 1802 |
+
defaultOpen = true,
|
| 1803 |
+
defaultRisksOpen = false,
|
| 1804 |
+
}: {
|
| 1805 |
+
card: BenchmarkCard
|
| 1806 |
+
isResearchView: boolean
|
| 1807 |
+
defaultOpen?: boolean
|
| 1808 |
+
defaultRisksOpen?: boolean
|
| 1809 |
+
}) {
|
| 1810 |
+
const [open, setOpen] = useState(defaultOpen)
|
| 1811 |
return (
|
| 1812 |
<Collapsible open={open} onOpenChange={setOpen}>
|
| 1813 |
<CollapsibleTrigger asChild>
|
|
|
|
| 1830 |
</button>
|
| 1831 |
</CollapsibleTrigger>
|
| 1832 |
<CollapsibleContent className="mt-2">
|
| 1833 |
+
<BenchmarkCardPanel
|
| 1834 |
+
card={card}
|
| 1835 |
+
isResearchView={isResearchView}
|
| 1836 |
+
defaultRisksOpen={defaultRisksOpen}
|
| 1837 |
+
/>
|
| 1838 |
</CollapsibleContent>
|
| 1839 |
</Collapsible>
|
| 1840 |
)
|
components/page-loading-state.tsx
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"use client"
|
| 2 |
+
|
| 3 |
+
import { useEffect, useMemo, useState } from "react"
|
| 4 |
+
|
| 5 |
+
import { cn } from "@/lib/utils"
|
| 6 |
+
|
| 7 |
+
export interface PageLoadingStage {
|
| 8 |
+
label: string
|
| 9 |
+
done: boolean
|
| 10 |
+
}
|
| 11 |
+
|
| 12 |
+
interface PageLoadingStateProps {
|
| 13 |
+
title: string
|
| 14 |
+
description?: string
|
| 15 |
+
stages: PageLoadingStage[]
|
| 16 |
+
className?: string
|
| 17 |
+
}
|
| 18 |
+
|
| 19 |
+
function clamp(value: number, min: number, max: number) {
|
| 20 |
+
return Math.min(max, Math.max(min, value))
|
| 21 |
+
}
|
| 22 |
+
|
| 23 |
+
function getTargetProgress(completedStages: number, totalStages: number) {
|
| 24 |
+
const start = 12
|
| 25 |
+
const ceiling = 94
|
| 26 |
+
|
| 27 |
+
if (totalStages <= 0) {
|
| 28 |
+
return start
|
| 29 |
+
}
|
| 30 |
+
|
| 31 |
+
const stageWidth = (ceiling - start) / totalStages
|
| 32 |
+
const hardTarget = start + completedStages * stageWidth
|
| 33 |
+
|
| 34 |
+
if (completedStages >= totalStages) {
|
| 35 |
+
return 100
|
| 36 |
+
}
|
| 37 |
+
|
| 38 |
+
return Math.min(ceiling, hardTarget + Math.min(stageWidth * 0.35, 12))
|
| 39 |
+
}
|
| 40 |
+
|
| 41 |
+
export function PageLoadingState({ title, description, stages, className }: PageLoadingStateProps) {
|
| 42 |
+
const safeStages = stages.length > 0 ? stages : [{ label: "Loading", done: false }]
|
| 43 |
+
|
| 44 |
+
const { completedStages, currentStageLabel, targetProgress } = useMemo(() => {
|
| 45 |
+
const completed = safeStages.filter((stage) => stage.done).length
|
| 46 |
+
const currentStage = safeStages.find((stage) => !stage.done)?.label ?? "Finalizing"
|
| 47 |
+
|
| 48 |
+
return {
|
| 49 |
+
completedStages: completed,
|
| 50 |
+
currentStageLabel: currentStage,
|
| 51 |
+
targetProgress: getTargetProgress(completed, safeStages.length),
|
| 52 |
+
}
|
| 53 |
+
}, [safeStages])
|
| 54 |
+
|
| 55 |
+
const [displayProgress, setDisplayProgress] = useState(() => clamp(targetProgress, 0, 100))
|
| 56 |
+
|
| 57 |
+
useEffect(() => {
|
| 58 |
+
setDisplayProgress((current) => {
|
| 59 |
+
if (targetProgress < current) {
|
| 60 |
+
return clamp(targetProgress, 0, 100)
|
| 61 |
+
}
|
| 62 |
+
|
| 63 |
+
return current
|
| 64 |
+
})
|
| 65 |
+
}, [targetProgress])
|
| 66 |
+
|
| 67 |
+
useEffect(() => {
|
| 68 |
+
if (Math.abs(displayProgress - targetProgress) < 0.5) {
|
| 69 |
+
if (displayProgress !== targetProgress) {
|
| 70 |
+
setDisplayProgress(targetProgress)
|
| 71 |
+
}
|
| 72 |
+
return
|
| 73 |
+
}
|
| 74 |
+
|
| 75 |
+
const interval = window.setInterval(() => {
|
| 76 |
+
setDisplayProgress((current) => {
|
| 77 |
+
const difference = targetProgress - current
|
| 78 |
+
if (Math.abs(difference) < 0.5) {
|
| 79 |
+
return targetProgress
|
| 80 |
+
}
|
| 81 |
+
|
| 82 |
+
const increment =
|
| 83 |
+
difference > 18 ? 6 : difference > 10 ? 4 : difference > 4 ? 2 : 1
|
| 84 |
+
|
| 85 |
+
return clamp(current + increment, 0, targetProgress)
|
| 86 |
+
})
|
| 87 |
+
}, 110)
|
| 88 |
+
|
| 89 |
+
return () => {
|
| 90 |
+
window.clearInterval(interval)
|
| 91 |
+
}
|
| 92 |
+
}, [displayProgress, targetProgress])
|
| 93 |
+
|
| 94 |
+
const progressStyle = {
|
| 95 |
+
background: `conic-gradient(from 180deg, hsl(var(--primary)) 0deg ${displayProgress * 3.6}deg, hsl(var(--border)) ${displayProgress * 3.6}deg 360deg)`,
|
| 96 |
+
}
|
| 97 |
+
|
| 98 |
+
return (
|
| 99 |
+
<div className={cn("flex min-h-[20rem] items-center justify-center px-4", className)}>
|
| 100 |
+
<section className="flex w-full max-w-md flex-col items-center gap-5 rounded-[1.75rem] border border-border/60 bg-background/95 px-6 py-8 text-center shadow-[0_24px_70px_-52px_rgba(15,23,42,0.5)] backdrop-blur">
|
| 101 |
+
<div className="relative flex h-28 w-28 items-center justify-center rounded-full" style={progressStyle}>
|
| 102 |
+
<div className="flex h-[5.4rem] w-[5.4rem] items-center justify-center rounded-full border border-border/60 bg-background text-2xl font-semibold tracking-tight text-foreground tabular-nums">
|
| 103 |
+
{Math.round(displayProgress)}%
|
| 104 |
+
</div>
|
| 105 |
+
</div>
|
| 106 |
+
|
| 107 |
+
<div className="space-y-1.5">
|
| 108 |
+
<h2 className="text-xl font-semibold tracking-tight text-foreground sm:text-2xl">{title}</h2>
|
| 109 |
+
{description ? (
|
| 110 |
+
<p className="text-sm leading-6 text-muted-foreground">{description}</p>
|
| 111 |
+
) : null}
|
| 112 |
+
</div>
|
| 113 |
+
|
| 114 |
+
<p className="text-[11px] font-medium uppercase tracking-[0.22em] text-muted-foreground">
|
| 115 |
+
{currentStageLabel} · {completedStages}/{safeStages.length} ready
|
| 116 |
+
</p>
|
| 117 |
+
</section>
|
| 118 |
+
</div>
|
| 119 |
+
)
|
| 120 |
+
}
|
data/benchmarks/ace.json
DELETED
|
@@ -1,120 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "anthropic/Opus 4.1",
|
| 5 |
-
"name": "Opus 4.1",
|
| 6 |
-
"developer": "Anthropic",
|
| 7 |
-
"scores": {
|
| 8 |
-
"Overall Score": 0.4,
|
| 9 |
-
"Gaming Score": 0.318
|
| 10 |
-
}
|
| 11 |
-
},
|
| 12 |
-
{
|
| 13 |
-
"model_id": "anthropic/Opus 4.5",
|
| 14 |
-
"name": "Opus 4.5",
|
| 15 |
-
"developer": "Anthropic",
|
| 16 |
-
"scores": {
|
| 17 |
-
"Overall Score": 0.478,
|
| 18 |
-
"Gaming Score": 0.391
|
| 19 |
-
}
|
| 20 |
-
},
|
| 21 |
-
{
|
| 22 |
-
"model_id": "anthropic/Sonnet 4.5",
|
| 23 |
-
"name": "Sonnet 4.5",
|
| 24 |
-
"developer": "Anthropic",
|
| 25 |
-
"scores": {
|
| 26 |
-
"Overall Score": 0.44,
|
| 27 |
-
"Gaming Score": 0.373
|
| 28 |
-
}
|
| 29 |
-
},
|
| 30 |
-
{
|
| 31 |
-
"model_id": "google/Gemini 2.5 Flash",
|
| 32 |
-
"name": "Gemini 2.5 Flash",
|
| 33 |
-
"developer": "Google",
|
| 34 |
-
"scores": {
|
| 35 |
-
"Overall Score": 0.38,
|
| 36 |
-
"Gaming Score": 0.284
|
| 37 |
-
}
|
| 38 |
-
},
|
| 39 |
-
{
|
| 40 |
-
"model_id": "google/Gemini 2.5 Pro",
|
| 41 |
-
"name": "Gemini 2.5 Pro",
|
| 42 |
-
"developer": "Google",
|
| 43 |
-
"scores": {
|
| 44 |
-
"Overall Score": 0.4,
|
| 45 |
-
"Gaming Score": 0.285
|
| 46 |
-
}
|
| 47 |
-
},
|
| 48 |
-
{
|
| 49 |
-
"model_id": "google/Gemini 3 Flash",
|
| 50 |
-
"name": "Gemini 3 Flash",
|
| 51 |
-
"developer": "Google",
|
| 52 |
-
"scores": {
|
| 53 |
-
"Gaming Score": 0.415
|
| 54 |
-
}
|
| 55 |
-
},
|
| 56 |
-
{
|
| 57 |
-
"model_id": "google/Gemini 3 Pro",
|
| 58 |
-
"name": "Gemini 3 Pro",
|
| 59 |
-
"developer": "Google",
|
| 60 |
-
"scores": {
|
| 61 |
-
"Overall Score": 0.47,
|
| 62 |
-
"Gaming Score": 0.509
|
| 63 |
-
}
|
| 64 |
-
},
|
| 65 |
-
{
|
| 66 |
-
"model_id": "openai/GPT 5",
|
| 67 |
-
"name": "GPT 5",
|
| 68 |
-
"developer": "OpenAI",
|
| 69 |
-
"scores": {
|
| 70 |
-
"Overall Score": 0.561,
|
| 71 |
-
"DIY Score": 0.55,
|
| 72 |
-
"Food Score": 0.7,
|
| 73 |
-
"Gaming Score": 0.575
|
| 74 |
-
}
|
| 75 |
-
},
|
| 76 |
-
{
|
| 77 |
-
"model_id": "openai/GPT 5.1",
|
| 78 |
-
"name": "GPT 5.1",
|
| 79 |
-
"developer": "OpenAI",
|
| 80 |
-
"scores": {
|
| 81 |
-
"Overall Score": 0.551,
|
| 82 |
-
"DIY Score": 0.56,
|
| 83 |
-
"Gaming Score": 0.61,
|
| 84 |
-
"Shopping Score": 0.45
|
| 85 |
-
}
|
| 86 |
-
},
|
| 87 |
-
{
|
| 88 |
-
"model_id": "openai/GPT 5.2",
|
| 89 |
-
"name": "GPT 5.2",
|
| 90 |
-
"developer": "OpenAI",
|
| 91 |
-
"scores": {
|
| 92 |
-
"Overall Score": 0.515,
|
| 93 |
-
"Food Score": 0.65,
|
| 94 |
-
"Gaming Score": 0.578
|
| 95 |
-
}
|
| 96 |
-
},
|
| 97 |
-
{
|
| 98 |
-
"model_id": "openai/o3",
|
| 99 |
-
"name": "o3",
|
| 100 |
-
"developer": "OpenAI",
|
| 101 |
-
"scores": {
|
| 102 |
-
"Overall Score": 0.529,
|
| 103 |
-
"Gaming Score": 0.585,
|
| 104 |
-
"Shopping Score": 0.45
|
| 105 |
-
}
|
| 106 |
-
},
|
| 107 |
-
{
|
| 108 |
-
"model_id": "openai/o3 Pro",
|
| 109 |
-
"name": "o3 Pro",
|
| 110 |
-
"developer": "OpenAI",
|
| 111 |
-
"scores": {
|
| 112 |
-
"Overall Score": 0.552,
|
| 113 |
-
"DIY Score": 0.54,
|
| 114 |
-
"Food Score": 0.6,
|
| 115 |
-
"Gaming Score": 0.613,
|
| 116 |
-
"Shopping Score": 0.45
|
| 117 |
-
}
|
| 118 |
-
}
|
| 119 |
-
]
|
| 120 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/apex-agents.json
DELETED
|
@@ -1,218 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "anthropic/Opus 4.5",
|
| 5 |
-
"name": "Opus 4.5",
|
| 6 |
-
"developer": "Anthropic",
|
| 7 |
-
"scores": {
|
| 8 |
-
"Overall Pass@1": 0.184,
|
| 9 |
-
"Overall Pass@8": 0.34,
|
| 10 |
-
"Overall Mean Score": 0.348,
|
| 11 |
-
"Investment Banking Pass@1": 0.216,
|
| 12 |
-
"Management Consulting Pass@1": 0.132,
|
| 13 |
-
"Corporate Law Pass@1": 0.202,
|
| 14 |
-
"Corporate Lawyer Mean Score": 0.471
|
| 15 |
-
}
|
| 16 |
-
},
|
| 17 |
-
{
|
| 18 |
-
"model_id": "anthropic/Opus 4.6",
|
| 19 |
-
"name": "Opus 4.6",
|
| 20 |
-
"developer": "Anthropic",
|
| 21 |
-
"scores": {
|
| 22 |
-
"Overall Pass@1": 0.298,
|
| 23 |
-
"Corporate Lawyer Mean Score": 0.502
|
| 24 |
-
}
|
| 25 |
-
},
|
| 26 |
-
{
|
| 27 |
-
"model_id": "applied-compute/Applied Compute: Small",
|
| 28 |
-
"name": "Applied Compute: Small",
|
| 29 |
-
"developer": "applied-compute",
|
| 30 |
-
"scores": {
|
| 31 |
-
"Overall Pass@1": 0.23,
|
| 32 |
-
"Overall Mean Score": 0.401,
|
| 33 |
-
"Corporate Law Pass@1": 0.266,
|
| 34 |
-
"Corporate Lawyer Mean Score": 0.548
|
| 35 |
-
}
|
| 36 |
-
},
|
| 37 |
-
{
|
| 38 |
-
"model_id": "google/Gemini 3 Flash",
|
| 39 |
-
"name": "Gemini 3 Flash",
|
| 40 |
-
"developer": "Google",
|
| 41 |
-
"scores": {
|
| 42 |
-
"Overall Pass@1": 0.24,
|
| 43 |
-
"Overall Pass@8": 0.367,
|
| 44 |
-
"Overall Mean Score": 0.395,
|
| 45 |
-
"Investment Banking Pass@1": 0.267,
|
| 46 |
-
"Management Consulting Pass@1": 0.193,
|
| 47 |
-
"Corporate Law Pass@1": 0.259,
|
| 48 |
-
"Corporate Lawyer Mean Score": 0.524
|
| 49 |
-
}
|
| 50 |
-
},
|
| 51 |
-
{
|
| 52 |
-
"model_id": "google/Gemini 3 Pro",
|
| 53 |
-
"name": "Gemini 3 Pro",
|
| 54 |
-
"developer": "Google",
|
| 55 |
-
"scores": {
|
| 56 |
-
"Overall Pass@1": 0.184,
|
| 57 |
-
"Overall Pass@8": 0.373,
|
| 58 |
-
"Overall Mean Score": 0.341,
|
| 59 |
-
"Investment Banking Pass@1": 0.188,
|
| 60 |
-
"Management Consulting Pass@1": 0.124,
|
| 61 |
-
"Corporate Law Pass@1": 0.239,
|
| 62 |
-
"Corporate Lawyer Mean Score": 0.487
|
| 63 |
-
}
|
| 64 |
-
},
|
| 65 |
-
{
|
| 66 |
-
"model_id": "google/Gemini 3.1 Pro",
|
| 67 |
-
"name": "Gemini 3.1 Pro",
|
| 68 |
-
"developer": "Google",
|
| 69 |
-
"scores": {
|
| 70 |
-
"Overall Pass@1": 0.335,
|
| 71 |
-
"Corporate Lawyer Mean Score": 0.494
|
| 72 |
-
}
|
| 73 |
-
},
|
| 74 |
-
{
|
| 75 |
-
"model_id": "minimax/Minimax-2.5",
|
| 76 |
-
"name": "Minimax-2.5",
|
| 77 |
-
"developer": "minimax",
|
| 78 |
-
"scores": {
|
| 79 |
-
"Corporate Lawyer Mean Score": 0.339
|
| 80 |
-
}
|
| 81 |
-
},
|
| 82 |
-
{
|
| 83 |
-
"model_id": "moonshot/Kimi K2 Thinking",
|
| 84 |
-
"name": "Kimi K2 Thinking",
|
| 85 |
-
"developer": "moonshot",
|
| 86 |
-
"scores": {
|
| 87 |
-
"Overall Pass@1": 0.04,
|
| 88 |
-
"Overall Pass@8": 0.144,
|
| 89 |
-
"Overall Mean Score": 0.115,
|
| 90 |
-
"Investment Banking Pass@1": 0.012,
|
| 91 |
-
"Management Consulting Pass@1": 0.029,
|
| 92 |
-
"Corporate Law Pass@1": 0.08,
|
| 93 |
-
"Corporate Lawyer Mean Score": 0.223
|
| 94 |
-
}
|
| 95 |
-
},
|
| 96 |
-
{
|
| 97 |
-
"model_id": "moonshot/Kimi K2.5",
|
| 98 |
-
"name": "Kimi K2.5",
|
| 99 |
-
"developer": "moonshot",
|
| 100 |
-
"scores": {
|
| 101 |
-
"Corporate Lawyer Mean Score": 0.402
|
| 102 |
-
}
|
| 103 |
-
},
|
| 104 |
-
{
|
| 105 |
-
"model_id": "openai/GPT 5",
|
| 106 |
-
"name": "GPT 5",
|
| 107 |
-
"developer": "OpenAI",
|
| 108 |
-
"scores": {
|
| 109 |
-
"Overall Pass@1": 0.183,
|
| 110 |
-
"Overall Pass@8": 0.31,
|
| 111 |
-
"Overall Mean Score": 0.329,
|
| 112 |
-
"Investment Banking Pass@1": 0.273,
|
| 113 |
-
"Management Consulting Pass@1": 0.123,
|
| 114 |
-
"Corporate Law Pass@1": 0.153,
|
| 115 |
-
"Corporate Lawyer Mean Score": 0.382
|
| 116 |
-
}
|
| 117 |
-
},
|
| 118 |
-
{
|
| 119 |
-
"model_id": "openai/GPT 5 Codex",
|
| 120 |
-
"name": "GPT 5 Codex",
|
| 121 |
-
"developer": "OpenAI",
|
| 122 |
-
"scores": {
|
| 123 |
-
"Corporate Lawyer Mean Score": 0.362
|
| 124 |
-
}
|
| 125 |
-
},
|
| 126 |
-
{
|
| 127 |
-
"model_id": "openai/GPT 5.1",
|
| 128 |
-
"name": "GPT 5.1",
|
| 129 |
-
"developer": "OpenAI",
|
| 130 |
-
"scores": {
|
| 131 |
-
"Corporate Lawyer Mean Score": 0.376
|
| 132 |
-
}
|
| 133 |
-
},
|
| 134 |
-
{
|
| 135 |
-
"model_id": "openai/GPT 5.1 Codex",
|
| 136 |
-
"name": "GPT 5.1 Codex",
|
| 137 |
-
"developer": "OpenAI",
|
| 138 |
-
"scores": {
|
| 139 |
-
"Corporate Lawyer Mean Score": 0.366
|
| 140 |
-
}
|
| 141 |
-
},
|
| 142 |
-
{
|
| 143 |
-
"model_id": "openai/GPT 5.2",
|
| 144 |
-
"name": "GPT 5.2",
|
| 145 |
-
"developer": "OpenAI",
|
| 146 |
-
"scores": {
|
| 147 |
-
"Overall Pass@1": 0.23,
|
| 148 |
-
"Overall Pass@8": 0.4,
|
| 149 |
-
"Overall Mean Score": 0.387,
|
| 150 |
-
"Investment Banking Pass@1": 0.273,
|
| 151 |
-
"Management Consulting Pass@1": 0.227,
|
| 152 |
-
"Corporate Law Pass@1": 0.189,
|
| 153 |
-
"Corporate Lawyer Mean Score": 0.443
|
| 154 |
-
}
|
| 155 |
-
},
|
| 156 |
-
{
|
| 157 |
-
"model_id": "openai/GPT 5.2 Codex",
|
| 158 |
-
"name": "GPT 5.2 Codex",
|
| 159 |
-
"developer": "OpenAI",
|
| 160 |
-
"scores": {
|
| 161 |
-
"Overall Pass@1": 0.276,
|
| 162 |
-
"Corporate Lawyer Mean Score": 0.394
|
| 163 |
-
}
|
| 164 |
-
},
|
| 165 |
-
{
|
| 166 |
-
"model_id": "openai/GPT 5.3 Codex",
|
| 167 |
-
"name": "GPT 5.3 Codex",
|
| 168 |
-
"developer": "OpenAI",
|
| 169 |
-
"scores": {
|
| 170 |
-
"Overall Pass@1": 0.317
|
| 171 |
-
}
|
| 172 |
-
},
|
| 173 |
-
{
|
| 174 |
-
"model_id": "openai/GPT OSS 120B",
|
| 175 |
-
"name": "GPT OSS 120B",
|
| 176 |
-
"developer": "OpenAI",
|
| 177 |
-
"scores": {
|
| 178 |
-
"Overall Pass@1": 0.047,
|
| 179 |
-
"Overall Pass@8": 0.115,
|
| 180 |
-
"Overall Mean Score": 0.145,
|
| 181 |
-
"Investment Banking Pass@1": 0.027,
|
| 182 |
-
"Management Consulting Pass@1": 0.035,
|
| 183 |
-
"Corporate Law Pass@1": 0.078,
|
| 184 |
-
"Corporate Lawyer Mean Score": 0.269
|
| 185 |
-
}
|
| 186 |
-
},
|
| 187 |
-
{
|
| 188 |
-
"model_id": "xai/Grok 4",
|
| 189 |
-
"name": "Grok 4",
|
| 190 |
-
"developer": "xAI",
|
| 191 |
-
"scores": {
|
| 192 |
-
"Overall Pass@1": 0.152,
|
| 193 |
-
"Overall Pass@8": 0.329,
|
| 194 |
-
"Overall Mean Score": 0.303,
|
| 195 |
-
"Investment Banking Pass@1": 0.17,
|
| 196 |
-
"Management Consulting Pass@1": 0.12,
|
| 197 |
-
"Corporate Law Pass@1": 0.165,
|
| 198 |
-
"Corporate Lawyer Mean Score": 0.41
|
| 199 |
-
}
|
| 200 |
-
},
|
| 201 |
-
{
|
| 202 |
-
"model_id": "zhipu/GLM 4.6",
|
| 203 |
-
"name": "GLM 4.6",
|
| 204 |
-
"developer": "zhipu",
|
| 205 |
-
"scores": {
|
| 206 |
-
"Corporate Lawyer Mean Score": 0.196
|
| 207 |
-
}
|
| 208 |
-
},
|
| 209 |
-
{
|
| 210 |
-
"model_id": "zhipu/GLM 4.7",
|
| 211 |
-
"name": "GLM 4.7",
|
| 212 |
-
"developer": "zhipu",
|
| 213 |
-
"scores": {
|
| 214 |
-
"Corporate Lawyer Mean Score": 0.147
|
| 215 |
-
}
|
| 216 |
-
}
|
| 217 |
-
]
|
| 218 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/apex-v1.json
DELETED
|
@@ -1,93 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "anthropic/Opus 4.5",
|
| 5 |
-
"name": "Opus 4.5",
|
| 6 |
-
"developer": "Anthropic",
|
| 7 |
-
"scores": {
|
| 8 |
-
"Medicine (MD) Score": 0.65
|
| 9 |
-
}
|
| 10 |
-
},
|
| 11 |
-
{
|
| 12 |
-
"model_id": "google/Gemini 2.5 Flash",
|
| 13 |
-
"name": "Gemini 2.5 Flash",
|
| 14 |
-
"developer": "Google",
|
| 15 |
-
"scores": {
|
| 16 |
-
"Overall Score": 0.604
|
| 17 |
-
}
|
| 18 |
-
},
|
| 19 |
-
{
|
| 20 |
-
"model_id": "google/Gemini 3 Flash",
|
| 21 |
-
"name": "Gemini 3 Flash",
|
| 22 |
-
"developer": "Google",
|
| 23 |
-
"scores": {
|
| 24 |
-
"Overall Score": 0.64,
|
| 25 |
-
"Consulting Score": 0.64
|
| 26 |
-
}
|
| 27 |
-
},
|
| 28 |
-
{
|
| 29 |
-
"model_id": "google/Gemini 3 Pro",
|
| 30 |
-
"name": "Gemini 3 Pro",
|
| 31 |
-
"developer": "Google",
|
| 32 |
-
"scores": {
|
| 33 |
-
"Overall Score": 0.643,
|
| 34 |
-
"Consulting Score": 0.64,
|
| 35 |
-
"Investment Banking Score": 0.63
|
| 36 |
-
}
|
| 37 |
-
},
|
| 38 |
-
{
|
| 39 |
-
"model_id": "openai/GPT 4o",
|
| 40 |
-
"name": "GPT 4o",
|
| 41 |
-
"developer": "OpenAI",
|
| 42 |
-
"scores": {
|
| 43 |
-
"Overall Score": 0.359
|
| 44 |
-
}
|
| 45 |
-
},
|
| 46 |
-
{
|
| 47 |
-
"model_id": "openai/GPT 5",
|
| 48 |
-
"name": "GPT 5",
|
| 49 |
-
"developer": "OpenAI",
|
| 50 |
-
"scores": {
|
| 51 |
-
"Overall Score": 0.67,
|
| 52 |
-
"Big Law Score": 0.78,
|
| 53 |
-
"Medicine (MD) Score": 0.66,
|
| 54 |
-
"Investment Banking Score": 0.61
|
| 55 |
-
}
|
| 56 |
-
},
|
| 57 |
-
{
|
| 58 |
-
"model_id": "openai/GPT 5.1",
|
| 59 |
-
"name": "GPT 5.1",
|
| 60 |
-
"developer": "OpenAI",
|
| 61 |
-
"scores": {
|
| 62 |
-
"Big Law Score": 0.77
|
| 63 |
-
}
|
| 64 |
-
},
|
| 65 |
-
{
|
| 66 |
-
"model_id": "openai/GPT 5.2 Pro",
|
| 67 |
-
"name": "GPT 5.2 Pro",
|
| 68 |
-
"developer": "OpenAI",
|
| 69 |
-
"scores": {
|
| 70 |
-
"Overall Score": 0.668,
|
| 71 |
-
"Consulting Score": 0.64,
|
| 72 |
-
"Medicine (MD) Score": 0.65,
|
| 73 |
-
"Investment Banking Score": 0.64
|
| 74 |
-
}
|
| 75 |
-
},
|
| 76 |
-
{
|
| 77 |
-
"model_id": "openai/o3",
|
| 78 |
-
"name": "o3",
|
| 79 |
-
"developer": "OpenAI",
|
| 80 |
-
"scores": {
|
| 81 |
-
"Big Law Score": 0.76
|
| 82 |
-
}
|
| 83 |
-
},
|
| 84 |
-
{
|
| 85 |
-
"model_id": "xai/Grok 4",
|
| 86 |
-
"name": "Grok 4",
|
| 87 |
-
"developer": "xAI",
|
| 88 |
-
"scores": {
|
| 89 |
-
"Overall Score": 0.635
|
| 90 |
-
}
|
| 91 |
-
}
|
| 92 |
-
]
|
| 93 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/appworld_test_normal.json
DELETED
|
@@ -1,28 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "anthropic/claude-opus-4-5",
|
| 5 |
-
"name": "claude-opus-4-5",
|
| 6 |
-
"developer": "Anthropic",
|
| 7 |
-
"scores": {
|
| 8 |
-
"appworld/test_normal": 0.68
|
| 9 |
-
}
|
| 10 |
-
},
|
| 11 |
-
{
|
| 12 |
-
"model_id": "google/gemini-3-pro-preview",
|
| 13 |
-
"name": "gemini-3-pro-preview",
|
| 14 |
-
"developer": "Google",
|
| 15 |
-
"scores": {
|
| 16 |
-
"appworld/test_normal": 0.505
|
| 17 |
-
}
|
| 18 |
-
},
|
| 19 |
-
{
|
| 20 |
-
"model_id": "openai/gpt-5.2-2025-12-11",
|
| 21 |
-
"name": "gpt-5.2-2025-12-11",
|
| 22 |
-
"developer": "OpenAI",
|
| 23 |
-
"scores": {
|
| 24 |
-
"appworld/test_normal": 0.0
|
| 25 |
-
}
|
| 26 |
-
}
|
| 27 |
-
]
|
| 28 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/bfcl.json
DELETED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/benchmarks/browsecompplus.json
DELETED
|
@@ -1,28 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "anthropic/claude-opus-4-5",
|
| 5 |
-
"name": "claude-opus-4-5",
|
| 6 |
-
"developer": "Anthropic",
|
| 7 |
-
"scores": {
|
| 8 |
-
"browsecompplus": 0.61
|
| 9 |
-
}
|
| 10 |
-
},
|
| 11 |
-
{
|
| 12 |
-
"model_id": "google/gemini-3-pro-preview",
|
| 13 |
-
"name": "gemini-3-pro-preview",
|
| 14 |
-
"developer": "Google",
|
| 15 |
-
"scores": {
|
| 16 |
-
"browsecompplus": 0.48
|
| 17 |
-
}
|
| 18 |
-
},
|
| 19 |
-
{
|
| 20 |
-
"model_id": "openai/gpt-5.2-2025-12-11",
|
| 21 |
-
"name": "gpt-5.2-2025-12-11",
|
| 22 |
-
"developer": "OpenAI",
|
| 23 |
-
"scores": {
|
| 24 |
-
"browsecompplus": 0.26
|
| 25 |
-
}
|
| 26 |
-
}
|
| 27 |
-
]
|
| 28 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/global-mmlu-lite.json
DELETED
|
@@ -1,706 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "alibaba/qwen3-235b-a22b-instruct-2507",
|
| 5 |
-
"name": "qwen3-235b-a22b-instruct-2507",
|
| 6 |
-
"developer": "alibaba",
|
| 7 |
-
"scores": {
|
| 8 |
-
"Global MMLU Lite": 0.8798,
|
| 9 |
-
"Culturally Sensitive": 0.8522,
|
| 10 |
-
"Culturally Agnostic": 0.9075,
|
| 11 |
-
"Arabic": 0.88,
|
| 12 |
-
"English": 0.89,
|
| 13 |
-
"Bengali": 0.8875,
|
| 14 |
-
"German": 0.885,
|
| 15 |
-
"French": 0.88,
|
| 16 |
-
"Hindi": 0.8775,
|
| 17 |
-
"Indonesian": 0.88,
|
| 18 |
-
"Italian": 0.88,
|
| 19 |
-
"Japanese": 0.88,
|
| 20 |
-
"Korean": 0.875,
|
| 21 |
-
"Portuguese": 0.8875,
|
| 22 |
-
"Spanish": 0.875,
|
| 23 |
-
"Swahili": 0.87,
|
| 24 |
-
"Yoruba": 0.8725,
|
| 25 |
-
"Chinese": 0.8775,
|
| 26 |
-
"Burmese": 0.88
|
| 27 |
-
}
|
| 28 |
-
},
|
| 29 |
-
{
|
| 30 |
-
"model_id": "anthropic/claude-3-5-haiku-20241022",
|
| 31 |
-
"name": "Claude 3.5 Haiku 20241022",
|
| 32 |
-
"developer": "Anthropic",
|
| 33 |
-
"scores": {
|
| 34 |
-
"Global MMLU Lite": 0.6114,
|
| 35 |
-
"Culturally Sensitive": 0.5834,
|
| 36 |
-
"Culturally Agnostic": 0.6394,
|
| 37 |
-
"Arabic": 0.695,
|
| 38 |
-
"English": 0.485,
|
| 39 |
-
"Bengali": 0.675,
|
| 40 |
-
"German": 0.565,
|
| 41 |
-
"French": 0.61,
|
| 42 |
-
"Hindi": 0.6575,
|
| 43 |
-
"Indonesian": 0.5475,
|
| 44 |
-
"Italian": 0.48,
|
| 45 |
-
"Japanese": 0.655,
|
| 46 |
-
"Korean": 0.6575,
|
| 47 |
-
"Portuguese": 0.5225,
|
| 48 |
-
"Spanish": 0.485,
|
| 49 |
-
"Swahili": 0.69,
|
| 50 |
-
"Yoruba": 0.6675,
|
| 51 |
-
"Chinese": 0.69,
|
| 52 |
-
"Burmese": 0.7
|
| 53 |
-
}
|
| 54 |
-
},
|
| 55 |
-
{
|
| 56 |
-
"model_id": "anthropic/claude-3-7-sonnet-20250219",
|
| 57 |
-
"name": "claude-3-7-sonnet-20250219",
|
| 58 |
-
"developer": "Anthropic",
|
| 59 |
-
"scores": {
|
| 60 |
-
"Global MMLU Lite": 0.8078,
|
| 61 |
-
"Culturally Sensitive": 0.7794,
|
| 62 |
-
"Culturally Agnostic": 0.8362,
|
| 63 |
-
"Arabic": 0.7925,
|
| 64 |
-
"English": 0.7625,
|
| 65 |
-
"Bengali": 0.825,
|
| 66 |
-
"German": 0.8125,
|
| 67 |
-
"French": 0.7675,
|
| 68 |
-
"Hindi": 0.805,
|
| 69 |
-
"Indonesian": 0.8175,
|
| 70 |
-
"Italian": 0.8225,
|
| 71 |
-
"Japanese": 0.8425,
|
| 72 |
-
"Korean": 0.83,
|
| 73 |
-
"Portuguese": 0.77,
|
| 74 |
-
"Spanish": 0.8075,
|
| 75 |
-
"Swahili": 0.8125,
|
| 76 |
-
"Yoruba": 0.81,
|
| 77 |
-
"Chinese": 0.835,
|
| 78 |
-
"Burmese": 0.8125
|
| 79 |
-
}
|
| 80 |
-
},
|
| 81 |
-
{
|
| 82 |
-
"model_id": "anthropic/claude-opus-4-1-20250805",
|
| 83 |
-
"name": "claude-opus-4-1-20250805",
|
| 84 |
-
"developer": "Anthropic",
|
| 85 |
-
"scores": {
|
| 86 |
-
"Global MMLU Lite": 0.943,
|
| 87 |
-
"Culturally Sensitive": 0.9331,
|
| 88 |
-
"Culturally Agnostic": 0.9528,
|
| 89 |
-
"Arabic": 0.945,
|
| 90 |
-
"English": 0.9475,
|
| 91 |
-
"Bengali": 0.9425,
|
| 92 |
-
"German": 0.94,
|
| 93 |
-
"French": 0.945,
|
| 94 |
-
"Hindi": 0.9475,
|
| 95 |
-
"Indonesian": 0.9425,
|
| 96 |
-
"Italian": 0.94,
|
| 97 |
-
"Japanese": 0.94,
|
| 98 |
-
"Korean": 0.95,
|
| 99 |
-
"Portuguese": 0.945,
|
| 100 |
-
"Spanish": 0.945,
|
| 101 |
-
"Swahili": 0.93,
|
| 102 |
-
"Yoruba": 0.9375,
|
| 103 |
-
"Chinese": 0.945,
|
| 104 |
-
"Burmese": 0.945
|
| 105 |
-
}
|
| 106 |
-
},
|
| 107 |
-
{
|
| 108 |
-
"model_id": "anthropic/claude-sonnet-4-20250514",
|
| 109 |
-
"name": "claude-sonnet-4-20250514",
|
| 110 |
-
"developer": "Anthropic",
|
| 111 |
-
"scores": {
|
| 112 |
-
"Global MMLU Lite": 0.9058,
|
| 113 |
-
"Culturally Sensitive": 0.8913,
|
| 114 |
-
"Culturally Agnostic": 0.9203,
|
| 115 |
-
"Arabic": 0.9125,
|
| 116 |
-
"English": 0.905,
|
| 117 |
-
"Bengali": 0.9075,
|
| 118 |
-
"German": 0.9125,
|
| 119 |
-
"French": 0.91,
|
| 120 |
-
"Hindi": 0.9,
|
| 121 |
-
"Indonesian": 0.9025,
|
| 122 |
-
"Italian": 0.9075,
|
| 123 |
-
"Japanese": 0.9,
|
| 124 |
-
"Korean": 0.9125,
|
| 125 |
-
"Portuguese": 0.91,
|
| 126 |
-
"Spanish": 0.9075,
|
| 127 |
-
"Swahili": 0.8975,
|
| 128 |
-
"Yoruba": 0.8975,
|
| 129 |
-
"Chinese": 0.9175,
|
| 130 |
-
"Burmese": 0.8925
|
| 131 |
-
}
|
| 132 |
-
},
|
| 133 |
-
{
|
| 134 |
-
"model_id": "cohere/aya-expanse-32b",
|
| 135 |
-
"name": "aya-expanse-32b",
|
| 136 |
-
"developer": "cohere",
|
| 137 |
-
"scores": {
|
| 138 |
-
"Global MMLU Lite": 0.7353,
|
| 139 |
-
"Culturally Sensitive": 0.6891,
|
| 140 |
-
"Culturally Agnostic": 0.7815,
|
| 141 |
-
"Arabic": 0.7425,
|
| 142 |
-
"English": 0.7544,
|
| 143 |
-
"Bengali": 0.7343,
|
| 144 |
-
"German": 0.7425,
|
| 145 |
-
"French": 0.7325,
|
| 146 |
-
"Hindi": 0.7375,
|
| 147 |
-
"Indonesian": 0.7594,
|
| 148 |
-
"Italian": 0.7305,
|
| 149 |
-
"Japanese": 0.7419,
|
| 150 |
-
"Korean": 0.7525,
|
| 151 |
-
"Portuguese": 0.7544,
|
| 152 |
-
"Spanish": 0.7362,
|
| 153 |
-
"Swahili": 0.7071,
|
| 154 |
-
"Yoruba": 0.6942,
|
| 155 |
-
"Chinese": 0.743,
|
| 156 |
-
"Burmese": 0.7025
|
| 157 |
-
}
|
| 158 |
-
},
|
| 159 |
-
{
|
| 160 |
-
"model_id": "cohere/command-a-03-2025",
|
| 161 |
-
"name": "command-a-03-2025",
|
| 162 |
-
"developer": "cohere",
|
| 163 |
-
"scores": {
|
| 164 |
-
"Global MMLU Lite": 0.8385,
|
| 165 |
-
"Culturally Sensitive": 0.7993,
|
| 166 |
-
"Culturally Agnostic": 0.8778,
|
| 167 |
-
"Arabic": 0.8425,
|
| 168 |
-
"English": 0.855,
|
| 169 |
-
"Bengali": 0.8225,
|
| 170 |
-
"German": 0.8425,
|
| 171 |
-
"French": 0.8375,
|
| 172 |
-
"Hindi": 0.8421,
|
| 173 |
-
"Indonesian": 0.8546,
|
| 174 |
-
"Italian": 0.8375,
|
| 175 |
-
"Japanese": 0.845,
|
| 176 |
-
"Korean": 0.85,
|
| 177 |
-
"Portuguese": 0.84,
|
| 178 |
-
"Spanish": 0.8525,
|
| 179 |
-
"Swahili": 0.8275,
|
| 180 |
-
"Yoruba": 0.815,
|
| 181 |
-
"Chinese": 0.835,
|
| 182 |
-
"Burmese": 0.8175
|
| 183 |
-
}
|
| 184 |
-
},
|
| 185 |
-
{
|
| 186 |
-
"model_id": "deepseek/deepseek-r1-0528",
|
| 187 |
-
"name": "deepseek-r1-0528",
|
| 188 |
-
"developer": "deepseek",
|
| 189 |
-
"scores": {
|
| 190 |
-
"Global MMLU Lite": 0.6744,
|
| 191 |
-
"Culturally Sensitive": 0.6672,
|
| 192 |
-
"Culturally Agnostic": 0.6816,
|
| 193 |
-
"Arabic": 0.6825,
|
| 194 |
-
"English": 0.715,
|
| 195 |
-
"Bengali": 0.655,
|
| 196 |
-
"German": 0.6375,
|
| 197 |
-
"French": 0.6925,
|
| 198 |
-
"Hindi": 0.6475,
|
| 199 |
-
"Indonesian": 0.655,
|
| 200 |
-
"Italian": 0.6775,
|
| 201 |
-
"Japanese": 0.7725,
|
| 202 |
-
"Korean": 0.6575,
|
| 203 |
-
"Portuguese": 0.635,
|
| 204 |
-
"Spanish": 0.7175,
|
| 205 |
-
"Swahili": 0.6775,
|
| 206 |
-
"Yoruba": 0.77,
|
| 207 |
-
"Chinese": 0.5075,
|
| 208 |
-
"Burmese": 0.69
|
| 209 |
-
}
|
| 210 |
-
},
|
| 211 |
-
{
|
| 212 |
-
"model_id": "deepseek/deepseek-v3.1",
|
| 213 |
-
"name": "deepseek-v3.1",
|
| 214 |
-
"developer": "deepseek",
|
| 215 |
-
"scores": {
|
| 216 |
-
"Global MMLU Lite": 0.8044,
|
| 217 |
-
"Culturally Sensitive": 0.7793,
|
| 218 |
-
"Culturally Agnostic": 0.8295,
|
| 219 |
-
"Arabic": 0.805,
|
| 220 |
-
"English": 0.825,
|
| 221 |
-
"Bengali": 0.8157,
|
| 222 |
-
"German": 0.7925,
|
| 223 |
-
"French": 0.8175,
|
| 224 |
-
"Hindi": 0.7569,
|
| 225 |
-
"Indonesian": 0.7764,
|
| 226 |
-
"Italian": 0.8075,
|
| 227 |
-
"Japanese": 0.8312,
|
| 228 |
-
"Korean": 0.8125,
|
| 229 |
-
"Portuguese": 0.8246,
|
| 230 |
-
"Spanish": 0.8125,
|
| 231 |
-
"Swahili": 0.801,
|
| 232 |
-
"Yoruba": 0.7831,
|
| 233 |
-
"Chinese": 0.8161,
|
| 234 |
-
"Burmese": 0.7925
|
| 235 |
-
}
|
| 236 |
-
},
|
| 237 |
-
{
|
| 238 |
-
"model_id": "google/gemini-2.5-flash",
|
| 239 |
-
"name": "Gemini 2.5 Flash",
|
| 240 |
-
"developer": "Google",
|
| 241 |
-
"scores": {
|
| 242 |
-
"Global MMLU Lite": 0.9145,
|
| 243 |
-
"Culturally Sensitive": 0.9,
|
| 244 |
-
"Culturally Agnostic": 0.9291,
|
| 245 |
-
"Arabic": 0.9125,
|
| 246 |
-
"English": 0.9325,
|
| 247 |
-
"Bengali": 0.91,
|
| 248 |
-
"German": 0.9025,
|
| 249 |
-
"French": 0.91,
|
| 250 |
-
"Hindi": 0.925,
|
| 251 |
-
"Indonesian": 0.9075,
|
| 252 |
-
"Italian": 0.9225,
|
| 253 |
-
"Japanese": 0.9125,
|
| 254 |
-
"Korean": 0.915,
|
| 255 |
-
"Portuguese": 0.9125,
|
| 256 |
-
"Spanish": 0.9175,
|
| 257 |
-
"Swahili": 0.915,
|
| 258 |
-
"Yoruba": 0.9075,
|
| 259 |
-
"Chinese": 0.915,
|
| 260 |
-
"Burmese": 0.915
|
| 261 |
-
}
|
| 262 |
-
},
|
| 263 |
-
{
|
| 264 |
-
"model_id": "google/gemini-2.5-flash-preview-05-20",
|
| 265 |
-
"name": "gemini-2.5-flash-preview-05-20",
|
| 266 |
-
"developer": "Google",
|
| 267 |
-
"scores": {
|
| 268 |
-
"Global MMLU Lite": 0.9092,
|
| 269 |
-
"Culturally Sensitive": 0.8925,
|
| 270 |
-
"Culturally Agnostic": 0.9259,
|
| 271 |
-
"Arabic": 0.905,
|
| 272 |
-
"English": 0.9225,
|
| 273 |
-
"Bengali": 0.91,
|
| 274 |
-
"German": 0.905,
|
| 275 |
-
"French": 0.925,
|
| 276 |
-
"Hindi": 0.9125,
|
| 277 |
-
"Indonesian": 0.9075,
|
| 278 |
-
"Italian": 0.89,
|
| 279 |
-
"Japanese": 0.9125,
|
| 280 |
-
"Korean": 0.9075,
|
| 281 |
-
"Portuguese": 0.915,
|
| 282 |
-
"Spanish": 0.915,
|
| 283 |
-
"Swahili": 0.905,
|
| 284 |
-
"Yoruba": 0.8825,
|
| 285 |
-
"Chinese": 0.93,
|
| 286 |
-
"Burmese": 0.9025
|
| 287 |
-
}
|
| 288 |
-
},
|
| 289 |
-
{
|
| 290 |
-
"model_id": "google/gemini-2.5-pro",
|
| 291 |
-
"name": "Gemini 2.5 Pro",
|
| 292 |
-
"developer": "Google",
|
| 293 |
-
"scores": {
|
| 294 |
-
"Global MMLU Lite": 0.9323,
|
| 295 |
-
"Culturally Sensitive": 0.9241,
|
| 296 |
-
"Culturally Agnostic": 0.9406,
|
| 297 |
-
"Arabic": 0.9475,
|
| 298 |
-
"English": 0.9275,
|
| 299 |
-
"Bengali": 0.9275,
|
| 300 |
-
"German": 0.93,
|
| 301 |
-
"French": 0.9425,
|
| 302 |
-
"Hindi": 0.9275,
|
| 303 |
-
"Indonesian": 0.925,
|
| 304 |
-
"Italian": 0.935,
|
| 305 |
-
"Japanese": 0.9375,
|
| 306 |
-
"Korean": 0.9275,
|
| 307 |
-
"Portuguese": 0.93,
|
| 308 |
-
"Spanish": 0.94,
|
| 309 |
-
"Swahili": 0.9375,
|
| 310 |
-
"Yoruba": 0.925,
|
| 311 |
-
"Chinese": 0.9275,
|
| 312 |
-
"Burmese": 0.93
|
| 313 |
-
}
|
| 314 |
-
},
|
| 315 |
-
{
|
| 316 |
-
"model_id": "google/gemini-3-pro-preview",
|
| 317 |
-
"name": "gemini-3-pro-preview",
|
| 318 |
-
"developer": "Google",
|
| 319 |
-
"scores": {
|
| 320 |
-
"Global MMLU Lite": 0.9453,
|
| 321 |
-
"Culturally Sensitive": 0.9397,
|
| 322 |
-
"Culturally Agnostic": 0.9509,
|
| 323 |
-
"Arabic": 0.9475,
|
| 324 |
-
"English": 0.9425,
|
| 325 |
-
"Bengali": 0.9425,
|
| 326 |
-
"German": 0.94,
|
| 327 |
-
"French": 0.9575,
|
| 328 |
-
"Hindi": 0.9425,
|
| 329 |
-
"Indonesian": 0.955,
|
| 330 |
-
"Italian": 0.955,
|
| 331 |
-
"Japanese": 0.94,
|
| 332 |
-
"Korean": 0.94,
|
| 333 |
-
"Portuguese": 0.9425,
|
| 334 |
-
"Spanish": 0.9475,
|
| 335 |
-
"Swahili": 0.94,
|
| 336 |
-
"Yoruba": 0.9425,
|
| 337 |
-
"Chinese": 0.9475,
|
| 338 |
-
"Burmese": 0.9425
|
| 339 |
-
}
|
| 340 |
-
},
|
| 341 |
-
{
|
| 342 |
-
"model_id": "google/gemma-3-27b-it",
|
| 343 |
-
"name": "gemma-3-27b-it",
|
| 344 |
-
"developer": "Google",
|
| 345 |
-
"scores": {
|
| 346 |
-
"Global MMLU Lite": 0.763,
|
| 347 |
-
"Culturally Sensitive": 0.7528,
|
| 348 |
-
"Culturally Agnostic": 0.7733,
|
| 349 |
-
"Arabic": 0.78,
|
| 350 |
-
"English": 0.7337,
|
| 351 |
-
"Bengali": 0.75,
|
| 352 |
-
"German": 0.775,
|
| 353 |
-
"French": 0.7481,
|
| 354 |
-
"Hindi": 0.7335,
|
| 355 |
-
"Indonesian": 0.7563,
|
| 356 |
-
"Italian": 0.75,
|
| 357 |
-
"Japanese": 0.7925,
|
| 358 |
-
"Korean": 0.798,
|
| 359 |
-
"Portuguese": 0.7481,
|
| 360 |
-
"Spanish": 0.7494,
|
| 361 |
-
"Swahili": 0.785,
|
| 362 |
-
"Yoruba": 0.7444,
|
| 363 |
-
"Chinese": 0.7925,
|
| 364 |
-
"Burmese": 0.7719
|
| 365 |
-
}
|
| 366 |
-
},
|
| 367 |
-
{
|
| 368 |
-
"model_id": "google/gemma-3-4b-it",
|
| 369 |
-
"name": "gemma-3-4b-it",
|
| 370 |
-
"developer": "Google",
|
| 371 |
-
"scores": {
|
| 372 |
-
"Global MMLU Lite": 0.6511,
|
| 373 |
-
"Culturally Sensitive": 0.6116,
|
| 374 |
-
"Culturally Agnostic": 0.6906,
|
| 375 |
-
"Arabic": 0.6525,
|
| 376 |
-
"English": 0.67,
|
| 377 |
-
"Bengali": 0.68,
|
| 378 |
-
"German": 0.6525,
|
| 379 |
-
"French": 0.6575,
|
| 380 |
-
"Hindi": 0.6475,
|
| 381 |
-
"Indonesian": 0.6775,
|
| 382 |
-
"Italian": 0.6675,
|
| 383 |
-
"Japanese": 0.6325,
|
| 384 |
-
"Korean": 0.66,
|
| 385 |
-
"Portuguese": 0.68,
|
| 386 |
-
"Spanish": 0.6725,
|
| 387 |
-
"Swahili": 0.6075,
|
| 388 |
-
"Yoruba": 0.5825,
|
| 389 |
-
"Chinese": 0.6475,
|
| 390 |
-
"Burmese": 0.63
|
| 391 |
-
}
|
| 392 |
-
},
|
| 393 |
-
{
|
| 394 |
-
"model_id": "ibm/granite-4.0-h-small",
|
| 395 |
-
"name": "granite-4.0-h-small",
|
| 396 |
-
"developer": "ibm",
|
| 397 |
-
"scores": {
|
| 398 |
-
"Global MMLU Lite": 0.7503,
|
| 399 |
-
"Culturally Sensitive": 0.7182,
|
| 400 |
-
"Culturally Agnostic": 0.7826,
|
| 401 |
-
"Arabic": 0.7613,
|
| 402 |
-
"English": 0.77,
|
| 403 |
-
"Bengali": 0.7613,
|
| 404 |
-
"German": 0.755,
|
| 405 |
-
"French": 0.7594,
|
| 406 |
-
"Hindi": 0.7575,
|
| 407 |
-
"Indonesian": 0.7614,
|
| 408 |
-
"Italian": 0.7525,
|
| 409 |
-
"Japanese": 0.7406,
|
| 410 |
-
"Korean": 0.7525,
|
| 411 |
-
"Portuguese": 0.757,
|
| 412 |
-
"Spanish": 0.7638,
|
| 413 |
-
"Swahili": 0.7318,
|
| 414 |
-
"Yoruba": 0.6921,
|
| 415 |
-
"Chinese": 0.7475,
|
| 416 |
-
"Burmese": 0.7419
|
| 417 |
-
}
|
| 418 |
-
},
|
| 419 |
-
{
|
| 420 |
-
"model_id": "mistralai/mistral-medium-3",
|
| 421 |
-
"name": "mistral-medium-3",
|
| 422 |
-
"developer": "mistralai",
|
| 423 |
-
"scores": {
|
| 424 |
-
"Global MMLU Lite": 0.5511,
|
| 425 |
-
"Culturally Sensitive": 0.5391,
|
| 426 |
-
"Culturally Agnostic": 0.5631,
|
| 427 |
-
"Arabic": 0.455,
|
| 428 |
-
"English": 0.38,
|
| 429 |
-
"Bengali": 0.5175,
|
| 430 |
-
"German": 0.4775,
|
| 431 |
-
"French": 0.41,
|
| 432 |
-
"Hindi": 0.555,
|
| 433 |
-
"Indonesian": 0.515,
|
| 434 |
-
"Italian": 0.535,
|
| 435 |
-
"Japanese": 0.58,
|
| 436 |
-
"Korean": 0.595,
|
| 437 |
-
"Portuguese": 0.5175,
|
| 438 |
-
"Spanish": 0.5375,
|
| 439 |
-
"Swahili": 0.7075,
|
| 440 |
-
"Yoruba": 0.7675,
|
| 441 |
-
"Chinese": 0.535,
|
| 442 |
-
"Burmese": 0.7325
|
| 443 |
-
}
|
| 444 |
-
},
|
| 445 |
-
{
|
| 446 |
-
"model_id": "mistralai/mistral-small-2503",
|
| 447 |
-
"name": "mistral-small-2503",
|
| 448 |
-
"developer": "mistralai",
|
| 449 |
-
"scores": {
|
| 450 |
-
"Global MMLU Lite": 0.7852,
|
| 451 |
-
"Culturally Sensitive": 0.7537,
|
| 452 |
-
"Culturally Agnostic": 0.8166,
|
| 453 |
-
"Arabic": 0.7875,
|
| 454 |
-
"English": 0.8,
|
| 455 |
-
"Bengali": 0.7725,
|
| 456 |
-
"German": 0.7975,
|
| 457 |
-
"French": 0.8,
|
| 458 |
-
"Hindi": 0.795,
|
| 459 |
-
"Indonesian": 0.785,
|
| 460 |
-
"Italian": 0.805,
|
| 461 |
-
"Japanese": 0.77,
|
| 462 |
-
"Korean": 0.79,
|
| 463 |
-
"Portuguese": 0.7925,
|
| 464 |
-
"Spanish": 0.7825,
|
| 465 |
-
"Swahili": 0.775,
|
| 466 |
-
"Yoruba": 0.735,
|
| 467 |
-
"Chinese": 0.7925,
|
| 468 |
-
"Burmese": 0.7825
|
| 469 |
-
}
|
| 470 |
-
},
|
| 471 |
-
{
|
| 472 |
-
"model_id": "openai/gpt-4.1-2025-04-14",
|
| 473 |
-
"name": "gpt-4.1-2025-04-14",
|
| 474 |
-
"developer": "OpenAI",
|
| 475 |
-
"scores": {
|
| 476 |
-
"Global MMLU Lite": 0.8755,
|
| 477 |
-
"Culturally Sensitive": 0.8541,
|
| 478 |
-
"Culturally Agnostic": 0.8969,
|
| 479 |
-
"Arabic": 0.88,
|
| 480 |
-
"English": 0.8825,
|
| 481 |
-
"Bengali": 0.8625,
|
| 482 |
-
"German": 0.875,
|
| 483 |
-
"French": 0.8875,
|
| 484 |
-
"Hindi": 0.8775,
|
| 485 |
-
"Indonesian": 0.885,
|
| 486 |
-
"Italian": 0.88,
|
| 487 |
-
"Japanese": 0.8725,
|
| 488 |
-
"Korean": 0.87,
|
| 489 |
-
"Portuguese": 0.875,
|
| 490 |
-
"Spanish": 0.885,
|
| 491 |
-
"Swahili": 0.8725,
|
| 492 |
-
"Yoruba": 0.875,
|
| 493 |
-
"Chinese": 0.87,
|
| 494 |
-
"Burmese": 0.8575
|
| 495 |
-
}
|
| 496 |
-
},
|
| 497 |
-
{
|
| 498 |
-
"model_id": "openai/gpt-5-2025-08-07",
|
| 499 |
-
"name": "gpt-5-2025-08-07",
|
| 500 |
-
"developer": "OpenAI",
|
| 501 |
-
"scores": {
|
| 502 |
-
"Global MMLU Lite": 0.8895,
|
| 503 |
-
"Culturally Sensitive": 0.8913,
|
| 504 |
-
"Culturally Agnostic": 0.8878,
|
| 505 |
-
"Arabic": 0.8925,
|
| 506 |
-
"English": 0.8725,
|
| 507 |
-
"Bengali": 0.9,
|
| 508 |
-
"German": 0.91,
|
| 509 |
-
"French": 0.9075,
|
| 510 |
-
"Hindi": 0.865,
|
| 511 |
-
"Indonesian": 0.795,
|
| 512 |
-
"Italian": 0.9075,
|
| 513 |
-
"Japanese": 0.8875,
|
| 514 |
-
"Korean": 0.915,
|
| 515 |
-
"Portuguese": 0.8875,
|
| 516 |
-
"Spanish": 0.905,
|
| 517 |
-
"Swahili": 0.865,
|
| 518 |
-
"Yoruba": 0.9125,
|
| 519 |
-
"Chinese": 0.895,
|
| 520 |
-
"Burmese": 0.915
|
| 521 |
-
}
|
| 522 |
-
},
|
| 523 |
-
{
|
| 524 |
-
"model_id": "openai/o3-mini-2025-01-31",
|
| 525 |
-
"name": "o3-mini-2025-01-31",
|
| 526 |
-
"developer": "OpenAI",
|
| 527 |
-
"scores": {
|
| 528 |
-
"Global MMLU Lite": 0.78,
|
| 529 |
-
"Culturally Sensitive": 0.765,
|
| 530 |
-
"Culturally Agnostic": 0.795,
|
| 531 |
-
"Arabic": 0.7725,
|
| 532 |
-
"English": 0.8025,
|
| 533 |
-
"Bengali": 0.77,
|
| 534 |
-
"German": 0.7525,
|
| 535 |
-
"French": 0.74,
|
| 536 |
-
"Hindi": 0.7525,
|
| 537 |
-
"Indonesian": 0.7425,
|
| 538 |
-
"Italian": 0.8,
|
| 539 |
-
"Japanese": 0.81,
|
| 540 |
-
"Korean": 0.8075,
|
| 541 |
-
"Portuguese": 0.7975,
|
| 542 |
-
"Spanish": 0.775,
|
| 543 |
-
"Swahili": 0.765,
|
| 544 |
-
"Yoruba": 0.7725,
|
| 545 |
-
"Chinese": 0.8125,
|
| 546 |
-
"Burmese": 0.8075
|
| 547 |
-
}
|
| 548 |
-
},
|
| 549 |
-
{
|
| 550 |
-
"model_id": "openai/o4-mini-2025-04-16",
|
| 551 |
-
"name": "o4-mini-2025-04-16",
|
| 552 |
-
"developer": "OpenAI",
|
| 553 |
-
"scores": {
|
| 554 |
-
"Global MMLU Lite": 0.8705,
|
| 555 |
-
"Culturally Sensitive": 0.8503,
|
| 556 |
-
"Culturally Agnostic": 0.8906,
|
| 557 |
-
"Arabic": 0.865,
|
| 558 |
-
"English": 0.8675,
|
| 559 |
-
"Bengali": 0.8875,
|
| 560 |
-
"German": 0.8775,
|
| 561 |
-
"French": 0.87,
|
| 562 |
-
"Hindi": 0.87,
|
| 563 |
-
"Indonesian": 0.8675,
|
| 564 |
-
"Italian": 0.855,
|
| 565 |
-
"Japanese": 0.885,
|
| 566 |
-
"Korean": 0.88,
|
| 567 |
-
"Portuguese": 0.88,
|
| 568 |
-
"Spanish": 0.855,
|
| 569 |
-
"Swahili": 0.8525,
|
| 570 |
-
"Yoruba": 0.8525,
|
| 571 |
-
"Chinese": 0.89,
|
| 572 |
-
"Burmese": 0.8725
|
| 573 |
-
}
|
| 574 |
-
},
|
| 575 |
-
{
|
| 576 |
-
"model_id": "unknown/aya-expanse-32b",
|
| 577 |
-
"name": "aya-expanse-32b",
|
| 578 |
-
"developer": "unknown",
|
| 579 |
-
"scores": {
|
| 580 |
-
"Global MMLU Lite": 0.7353,
|
| 581 |
-
"Culturally Sensitive": 0.6891,
|
| 582 |
-
"Culturally Agnostic": 0.7815,
|
| 583 |
-
"Arabic": 0.7425,
|
| 584 |
-
"English": 0.7544,
|
| 585 |
-
"Bengali": 0.7343,
|
| 586 |
-
"German": 0.7425,
|
| 587 |
-
"French": 0.7325,
|
| 588 |
-
"Hindi": 0.7375,
|
| 589 |
-
"Indonesian": 0.7594,
|
| 590 |
-
"Italian": 0.7305,
|
| 591 |
-
"Japanese": 0.7419,
|
| 592 |
-
"Korean": 0.7525,
|
| 593 |
-
"Portuguese": 0.7544,
|
| 594 |
-
"Spanish": 0.7362,
|
| 595 |
-
"Swahili": 0.7071,
|
| 596 |
-
"Yoruba": 0.6942,
|
| 597 |
-
"Chinese": 0.743,
|
| 598 |
-
"Burmese": 0.7025
|
| 599 |
-
}
|
| 600 |
-
},
|
| 601 |
-
{
|
| 602 |
-
"model_id": "unknown/granite-4.0-h-small",
|
| 603 |
-
"name": "granite-4.0-h-small",
|
| 604 |
-
"developer": "unknown",
|
| 605 |
-
"scores": {
|
| 606 |
-
"Global MMLU Lite": 0.7503,
|
| 607 |
-
"Culturally Sensitive": 0.7182,
|
| 608 |
-
"Culturally Agnostic": 0.7826,
|
| 609 |
-
"Arabic": 0.7613,
|
| 610 |
-
"English": 0.77,
|
| 611 |
-
"Bengali": 0.7613,
|
| 612 |
-
"German": 0.755,
|
| 613 |
-
"French": 0.7594,
|
| 614 |
-
"Hindi": 0.7575,
|
| 615 |
-
"Indonesian": 0.7614,
|
| 616 |
-
"Italian": 0.7525,
|
| 617 |
-
"Japanese": 0.7406,
|
| 618 |
-
"Korean": 0.7525,
|
| 619 |
-
"Portuguese": 0.757,
|
| 620 |
-
"Spanish": 0.7638,
|
| 621 |
-
"Swahili": 0.7318,
|
| 622 |
-
"Yoruba": 0.6921,
|
| 623 |
-
"Chinese": 0.7475,
|
| 624 |
-
"Burmese": 0.7419
|
| 625 |
-
}
|
| 626 |
-
},
|
| 627 |
-
{
|
| 628 |
-
"model_id": "unknown/o4-mini-2025-04-16",
|
| 629 |
-
"name": "o4-mini-2025-04-16",
|
| 630 |
-
"developer": "unknown",
|
| 631 |
-
"scores": {
|
| 632 |
-
"Global MMLU Lite": 0.8705,
|
| 633 |
-
"Culturally Sensitive": 0.8503,
|
| 634 |
-
"Culturally Agnostic": 0.8906,
|
| 635 |
-
"Arabic": 0.865,
|
| 636 |
-
"English": 0.8675,
|
| 637 |
-
"Bengali": 0.8875,
|
| 638 |
-
"German": 0.8775,
|
| 639 |
-
"French": 0.87,
|
| 640 |
-
"Hindi": 0.87,
|
| 641 |
-
"Indonesian": 0.8675,
|
| 642 |
-
"Italian": 0.855,
|
| 643 |
-
"Japanese": 0.885,
|
| 644 |
-
"Korean": 0.88,
|
| 645 |
-
"Portuguese": 0.88,
|
| 646 |
-
"Spanish": 0.855,
|
| 647 |
-
"Swahili": 0.8525,
|
| 648 |
-
"Yoruba": 0.8525,
|
| 649 |
-
"Chinese": 0.89,
|
| 650 |
-
"Burmese": 0.8725
|
| 651 |
-
}
|
| 652 |
-
},
|
| 653 |
-
{
|
| 654 |
-
"model_id": "xai/grok-3-mini",
|
| 655 |
-
"name": "grok-3-mini",
|
| 656 |
-
"developer": "xAI",
|
| 657 |
-
"scores": {
|
| 658 |
-
"Global MMLU Lite": 0.673,
|
| 659 |
-
"Culturally Sensitive": 0.6717,
|
| 660 |
-
"Culturally Agnostic": 0.6743,
|
| 661 |
-
"Arabic": 0.755,
|
| 662 |
-
"English": 0.5075,
|
| 663 |
-
"Bengali": 0.7355,
|
| 664 |
-
"German": 0.6591,
|
| 665 |
-
"French": 0.485,
|
| 666 |
-
"Hindi": 0.56,
|
| 667 |
-
"Indonesian": 0.725,
|
| 668 |
-
"Italian": 0.696,
|
| 669 |
-
"Japanese": 0.6575,
|
| 670 |
-
"Korean": 0.7325,
|
| 671 |
-
"Portuguese": 0.6275,
|
| 672 |
-
"Spanish": 0.61,
|
| 673 |
-
"Swahili": 0.7625,
|
| 674 |
-
"Yoruba": 0.8296,
|
| 675 |
-
"Chinese": 0.5564,
|
| 676 |
-
"Burmese": 0.8693
|
| 677 |
-
}
|
| 678 |
-
},
|
| 679 |
-
{
|
| 680 |
-
"model_id": "xai/grok-4-0709",
|
| 681 |
-
"name": "grok-4-0709",
|
| 682 |
-
"developer": "xAI",
|
| 683 |
-
"scores": {
|
| 684 |
-
"Global MMLU Lite": 0.8881,
|
| 685 |
-
"Culturally Sensitive": 0.8862,
|
| 686 |
-
"Culturally Agnostic": 0.89,
|
| 687 |
-
"Arabic": 0.885,
|
| 688 |
-
"English": 0.905,
|
| 689 |
-
"Bengali": 0.8925,
|
| 690 |
-
"German": 0.8725,
|
| 691 |
-
"French": 0.875,
|
| 692 |
-
"Hindi": 0.8675,
|
| 693 |
-
"Indonesian": 0.89,
|
| 694 |
-
"Italian": 0.9025,
|
| 695 |
-
"Japanese": 0.87,
|
| 696 |
-
"Korean": 0.895,
|
| 697 |
-
"Portuguese": 0.8725,
|
| 698 |
-
"Spanish": 0.9075,
|
| 699 |
-
"Swahili": 0.91,
|
| 700 |
-
"Yoruba": 0.905,
|
| 701 |
-
"Chinese": 0.8525,
|
| 702 |
-
"Burmese": 0.9075
|
| 703 |
-
}
|
| 704 |
-
}
|
| 705 |
-
]
|
| 706 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/helm_capabilities.json
DELETED
|
@@ -1,1026 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"benchmark_cards": {
|
| 3 |
-
"Omni-MATH": {
|
| 4 |
-
"benchmark_details": {
|
| 5 |
-
"name": "Omni-MATH",
|
| 6 |
-
"overview": "Omni-MATH is a benchmark designed to measure the mathematical reasoning capabilities of large language models at the Olympiad competition level. It comprises 4,428 challenging problems, categorized into over 33 sub-domains and more than 10 distinct difficulty levels. It was created because existing mathematical benchmarks are nearing saturation and are inadequate for assessing models on truly difficult tasks.",
|
| 7 |
-
"data_type": "text",
|
| 8 |
-
"domains": [
|
| 9 |
-
"math",
|
| 10 |
-
"olympiads"
|
| 11 |
-
],
|
| 12 |
-
"languages": [
|
| 13 |
-
"English"
|
| 14 |
-
],
|
| 15 |
-
"similar_benchmarks": [
|
| 16 |
-
"GSM8K",
|
| 17 |
-
"MATH"
|
| 18 |
-
],
|
| 19 |
-
"resources": [
|
| 20 |
-
"https://arxiv.org/abs/2410.07985",
|
| 21 |
-
"https://huggingface.co/datasets/KbsdJames/Omni-MATH",
|
| 22 |
-
"https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
|
| 23 |
-
]
|
| 24 |
-
},
|
| 25 |
-
"purpose_and_intended_users": {
|
| 26 |
-
"goal": "To provide a challenging benchmark for assessing and advancing the mathematical reasoning capabilities of large language models at the Olympiad and competition level, as existing benchmarks have become less challenging. It enables a nuanced analysis of performance across various mathematical disciplines and complexity levels.",
|
| 27 |
-
"audience": [
|
| 28 |
-
"Researchers evaluating large language models"
|
| 29 |
-
],
|
| 30 |
-
"tasks": [
|
| 31 |
-
"Solving Olympiad-level mathematical problems",
|
| 32 |
-
"Solving competition-level mathematical problems",
|
| 33 |
-
"Rule-based evaluation on a filtered subset of problems (Omni-MATH-Rule)"
|
| 34 |
-
],
|
| 35 |
-
"limitations": "Some problems are not suitable for reliable rule-based evaluation, leading to a filtered 'testable' set and an 'untestable' set.",
|
| 36 |
-
"out_of_scope_uses": [
|
| 37 |
-
"Not specified"
|
| 38 |
-
]
|
| 39 |
-
},
|
| 40 |
-
"data": {
|
| 41 |
-
"source": "The data is sourced from contest pages, the AoPS Wiki, and the AoPS Forum. Initial raw counts from these sources were filtered to create the final dataset.",
|
| 42 |
-
"size": "4,428 problems. A subset of 2,821 problems (Omni-MATH-Rule) is suitable for rule-based evaluation, and a separate subset of 100 samples was used for meta-evaluation.",
|
| 43 |
-
"format": "JSON",
|
| 44 |
-
"annotation": "Problems underwent rigorous human annotation by a team including PhD and Master's students. Each problem was reviewed by two annotators for cross-validation, achieving high agreement rates."
|
| 45 |
-
},
|
| 46 |
-
"methodology": {
|
| 47 |
-
"methods": [
|
| 48 |
-
"Models are evaluated by generating solutions to the mathematical problems.",
|
| 49 |
-
"Evaluation is conducted using both model-based judgment (e.g., GPT-4o or the open-source Omni-Judge) and rule-based evaluation for a specific subset of problems (Omni-MATH-Rule)."
|
| 50 |
-
],
|
| 51 |
-
"metrics": [
|
| 52 |
-
"Accuracy (Acc)"
|
| 53 |
-
],
|
| 54 |
-
"calculation": "The overall score is the accuracy percentage, calculated as the number of correct solutions divided by the total number of problems.",
|
| 55 |
-
"interpretation": "Higher accuracy indicates stronger performance. The benchmark is considered challenging, as even advanced models achieve only moderate accuracy.",
|
| 56 |
-
"baseline_results": "Paper baselines (accuracy on full Omni-MATH / Omni-MATH-Rule subset): o1-mini (60.54% / 62.2%), o1-preview (52.55% / 51.7%), qwen2.5-MATH-72b-Instruct (36.20% / 35.7%), qwen2.5-MATH-7b-Instruct (33.22% / 32.3%), GPT-4o (30.49% / 29.2%), NuminaMATH-72b-cot (28.45% / 27.1%), DeepseekMATH-7b-RL (16.12% / 14.9%). EEE results: OLMo 2 32B Instruct March 2025 scored 0.161 (16.1%).",
|
| 57 |
-
"validation": "For the Omni-MATH-Rule subset, two PhD students annotated problems to determine suitability for rule-based evaluation, achieving 98% cross-validation accuracy. A meta-evaluation compared model-generated judgments against human gold-standard annotations on a 100-sample subset to assess reliability."
|
| 58 |
-
},
|
| 59 |
-
"ethical_and_legal_considerations": {
|
| 60 |
-
"privacy_and_anonymity": "Not specified",
|
| 61 |
-
"data_licensing": "Apache License 2.0",
|
| 62 |
-
"consent_procedures": "Not specified",
|
| 63 |
-
"compliance_with_regulations": "Not specified"
|
| 64 |
-
},
|
| 65 |
-
"possible_risks": [
|
| 66 |
-
{
|
| 67 |
-
"category": "Over- or under-reliance",
|
| 68 |
-
"description": [
|
| 69 |
-
"In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
|
| 70 |
-
],
|
| 71 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
|
| 72 |
-
},
|
| 73 |
-
{
|
| 74 |
-
"category": "Unrepresentative data",
|
| 75 |
-
"description": [
|
| 76 |
-
"Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
|
| 77 |
-
],
|
| 78 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
|
| 79 |
-
},
|
| 80 |
-
{
|
| 81 |
-
"category": "Data bias",
|
| 82 |
-
"description": [
|
| 83 |
-
"Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
|
| 84 |
-
],
|
| 85 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
|
| 86 |
-
},
|
| 87 |
-
{
|
| 88 |
-
"category": "Lack of data transparency",
|
| 89 |
-
"description": [
|
| 90 |
-
"Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
|
| 91 |
-
],
|
| 92 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
|
| 93 |
-
},
|
| 94 |
-
{
|
| 95 |
-
"category": "Improper usage",
|
| 96 |
-
"description": [
|
| 97 |
-
"Improper usage occurs when a model is used for a purpose that it was not originally designed for."
|
| 98 |
-
],
|
| 99 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html"
|
| 100 |
-
}
|
| 101 |
-
],
|
| 102 |
-
"flagged_fields": {},
|
| 103 |
-
"missing_fields": [
|
| 104 |
-
"purpose_and_intended_users.out_of_scope_uses",
|
| 105 |
-
"ethical_and_legal_considerations.privacy_and_anonymity",
|
| 106 |
-
"ethical_and_legal_considerations.consent_procedures",
|
| 107 |
-
"ethical_and_legal_considerations.compliance_with_regulations"
|
| 108 |
-
],
|
| 109 |
-
"card_info": {
|
| 110 |
-
"created_at": "2026-03-17T13:34:44.331592",
|
| 111 |
-
"llm": "deepseek-ai/DeepSeek-V3.2"
|
| 112 |
-
}
|
| 113 |
-
},
|
| 114 |
-
"WildBench": {
|
| 115 |
-
"benchmark_details": {
|
| 116 |
-
"name": "WildBench",
|
| 117 |
-
"overview": "WildBench is a benchmark that measures the performance of large language models on challenging, open-ended tasks derived from real-world user queries. It consists of 1,024 tasks selected from human-chatbot conversation logs. Its distinctive features include using a natural distribution of real user queries and employing automated evaluation with two novel metrics, WB-Reward and WB-Score, which utilize task-specific checklists and structured explanations.",
|
| 118 |
-
"data_type": "tabular, text",
|
| 119 |
-
"domains": [
|
| 120 |
-
"Info Seeking",
|
| 121 |
-
"Math & Data",
|
| 122 |
-
"Reasoning & Planning",
|
| 123 |
-
"Creative Tasks"
|
| 124 |
-
],
|
| 125 |
-
"languages": [
|
| 126 |
-
"English"
|
| 127 |
-
],
|
| 128 |
-
"similar_benchmarks": [
|
| 129 |
-
"AlpacaEval",
|
| 130 |
-
"ArenaHard",
|
| 131 |
-
"MT-bench",
|
| 132 |
-
"Chatbot Arena"
|
| 133 |
-
],
|
| 134 |
-
"resources": [
|
| 135 |
-
"https://arxiv.org/abs/2406.04770",
|
| 136 |
-
"https://huggingface.co/datasets/allenai/WildBench",
|
| 137 |
-
"https://huggingface.co/spaces/allenai/WildBench",
|
| 138 |
-
"https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json"
|
| 139 |
-
]
|
| 140 |
-
},
|
| 141 |
-
"purpose_and_intended_users": {
|
| 142 |
-
"goal": "To provide a realistic, automated, and cost-effective evaluation framework for large language models that reflects their capabilities on challenging, real-world user tasks. It aims to be contamination-resilient and dynamically updated.",
|
| 143 |
-
"audience": [
|
| 144 |
-
"Researchers and practitioners evaluating large language models"
|
| 145 |
-
],
|
| 146 |
-
"tasks": [
|
| 147 |
-
"Open-ended text generation in response to diverse user queries"
|
| 148 |
-
],
|
| 149 |
-
"limitations": "The benchmark acknowledges a common issue in LLM-as-a-judge evaluations: bias towards longer outputs. It introduces a length-penalty method to mitigate this.",
|
| 150 |
-
"out_of_scope_uses": [
|
| 151 |
-
"Not specified"
|
| 152 |
-
]
|
| 153 |
-
},
|
| 154 |
-
"data": {
|
| 155 |
-
"source": "The data is derived from real user-chatbot conversation logs collected by the AI2 WildChat project. Over one million logs were filtered to create the benchmark.",
|
| 156 |
-
"size": "The benchmark contains 1,024 tasks. A separate 'v2-hard' configuration contains 256 examples. The dataset falls into the size category of 1K to 10K examples.",
|
| 157 |
-
"format": "The data is stored in Parquet format.",
|
| 158 |
-
"annotation": "Human annotation was used for quality control. GPT-4-Turbo assisted by summarizing query intents to help human reviewers filter out nonsensical tasks. Human reviewers manually assessed tasks to ensure they were challenging, diverse, and that associated checklist questions were clear. The final set of tasks was retained after this process. Each example includes fine-grained annotations such as task types and checklists for evaluating response quality."
|
| 159 |
-
},
|
| 160 |
-
"methodology": {
|
| 161 |
-
"methods": [
|
| 162 |
-
"Models are evaluated automatically using LLM-as-a-judge, with GPT-4-turbo as the evaluator. The evaluation is not fine-tuning-based and appears to be zero-shot or prompted.",
|
| 163 |
-
"The evaluation uses length-penalized pairwise comparisons and individual scoring to prevent bias towards longer outputs."
|
| 164 |
-
],
|
| 165 |
-
"metrics": [
|
| 166 |
-
"WB-Reward (for pairwise comparisons)",
|
| 167 |
-
"WB-Score (for individual scoring)",
|
| 168 |
-
"WB-Elo (for leaderboard ranking, merging WB-Reward and WB-Score)"
|
| 169 |
-
],
|
| 170 |
-
"calculation": "WB-Reward compares a test model against three baseline models of varying performance, producing outcomes of 'much better', 'slightly better', 'tie', 'slightly worse', or 'much worse'. A length penalty converts 'slightly better/worse' to 'tie' if the winner's response is excessively longer. WB-Score evaluates individual model outputs. WB-Elo merges the pairwise comparisons from WB-Reward and WB-Score to perform Elo rating updates.",
|
| 171 |
-
"interpretation": "Higher scores on WB-Reward and WB-Score indicate better performance. Strong performance is validated by a high correlation with human-voted Elo ratings from Chatbot Arena, with WB-Reward achieving a Pearson correlation of 0.98 and WB-Score reaching 0.95.",
|
| 172 |
-
"baseline_results": "Paper baselines: The paper reports results for 40 LLMs but does not list specific scores. It notes that using Haiku as a baseline yields the best correlation with human ratings. EEE results: OLMo 2 32B Instruct March 2025 scored 0.7340 on WildBench.",
|
| 173 |
-
"validation": "The evaluation metrics are validated by their strong correlation with human-voted Elo ratings from Chatbot Arena on hard tasks. Checklist questions used in the evaluation were manually verified for clarity and relevance."
|
| 174 |
-
},
|
| 175 |
-
"ethical_and_legal_considerations": {
|
| 176 |
-
"privacy_and_anonymity": "Not specified",
|
| 177 |
-
"data_licensing": "The dataset is made available under a CC BY license, intended for research and educational use in accordance with AI2's Responsible Use Guidelines.",
|
| 178 |
-
"consent_procedures": "Not specified",
|
| 179 |
-
"compliance_with_regulations": "Not specified"
|
| 180 |
-
},
|
| 181 |
-
"possible_risks": [
|
| 182 |
-
{
|
| 183 |
-
"category": "Over- or under-reliance",
|
| 184 |
-
"description": [
|
| 185 |
-
"In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
|
| 186 |
-
],
|
| 187 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
|
| 188 |
-
},
|
| 189 |
-
{
|
| 190 |
-
"category": "Unrepresentative data",
|
| 191 |
-
"description": [
|
| 192 |
-
"Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
|
| 193 |
-
],
|
| 194 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
|
| 195 |
-
},
|
| 196 |
-
{
|
| 197 |
-
"category": "Data bias",
|
| 198 |
-
"description": [
|
| 199 |
-
"Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
|
| 200 |
-
],
|
| 201 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
|
| 202 |
-
},
|
| 203 |
-
{
|
| 204 |
-
"category": "Data contamination",
|
| 205 |
-
"description": [
|
| 206 |
-
"Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation."
|
| 207 |
-
],
|
| 208 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html"
|
| 209 |
-
},
|
| 210 |
-
{
|
| 211 |
-
"category": "Lack of data transparency",
|
| 212 |
-
"description": [
|
| 213 |
-
"Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
|
| 214 |
-
],
|
| 215 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
|
| 216 |
-
}
|
| 217 |
-
],
|
| 218 |
-
"flagged_fields": {},
|
| 219 |
-
"missing_fields": [
|
| 220 |
-
"purpose_and_intended_users.out_of_scope_uses",
|
| 221 |
-
"ethical_and_legal_considerations.privacy_and_anonymity",
|
| 222 |
-
"ethical_and_legal_considerations.consent_procedures",
|
| 223 |
-
"ethical_and_legal_considerations.compliance_with_regulations"
|
| 224 |
-
],
|
| 225 |
-
"card_info": {
|
| 226 |
-
"created_at": "2026-03-17T13:56:24.159440",
|
| 227 |
-
"llm": "deepseek-ai/DeepSeek-V3.2"
|
| 228 |
-
}
|
| 229 |
-
}
|
| 230 |
-
},
|
| 231 |
-
"models": [
|
| 232 |
-
{
|
| 233 |
-
"model_id": "allenai/OLMo-2-1124-7B-Instruct",
|
| 234 |
-
"name": "OLMo 2 7B Instruct November 2024",
|
| 235 |
-
"developer": "allenai",
|
| 236 |
-
"scores": {
|
| 237 |
-
"Mean score": 0.405,
|
| 238 |
-
"MMLU-Pro": 0.292,
|
| 239 |
-
"GPQA": 0.296,
|
| 240 |
-
"IFEval": 0.693,
|
| 241 |
-
"WildBench": 0.628,
|
| 242 |
-
"Omni-MATH": 0.116
|
| 243 |
-
}
|
| 244 |
-
},
|
| 245 |
-
{
|
| 246 |
-
"model_id": "allenai/OLMoE-1B-7B-0125-Instruct",
|
| 247 |
-
"name": "OLMoE 1B-7B Instruct January 2025",
|
| 248 |
-
"developer": "allenai",
|
| 249 |
-
"scores": {
|
| 250 |
-
"Mean score": 0.332,
|
| 251 |
-
"MMLU-Pro": 0.169,
|
| 252 |
-
"GPQA": 0.22,
|
| 253 |
-
"IFEval": 0.628,
|
| 254 |
-
"WildBench": 0.551,
|
| 255 |
-
"Omni-MATH": 0.093
|
| 256 |
-
}
|
| 257 |
-
},
|
| 258 |
-
{
|
| 259 |
-
"model_id": "allenai/olmo-2-0325-32b-instruct",
|
| 260 |
-
"name": "OLMo 2 32B Instruct March 2025",
|
| 261 |
-
"developer": "allenai",
|
| 262 |
-
"scores": {
|
| 263 |
-
"Mean score": 0.475,
|
| 264 |
-
"MMLU-Pro": 0.414,
|
| 265 |
-
"GPQA": 0.287,
|
| 266 |
-
"IFEval": 0.78,
|
| 267 |
-
"WildBench": 0.734,
|
| 268 |
-
"Omni-MATH": 0.161
|
| 269 |
-
}
|
| 270 |
-
},
|
| 271 |
-
{
|
| 272 |
-
"model_id": "allenai/olmo-2-1124-13b-instruct",
|
| 273 |
-
"name": "OLMo 2 13B Instruct November 2024",
|
| 274 |
-
"developer": "allenai",
|
| 275 |
-
"scores": {
|
| 276 |
-
"Mean score": 0.44,
|
| 277 |
-
"MMLU-Pro": 0.31,
|
| 278 |
-
"GPQA": 0.316,
|
| 279 |
-
"IFEval": 0.73,
|
| 280 |
-
"WildBench": 0.689,
|
| 281 |
-
"Omni-MATH": 0.156
|
| 282 |
-
}
|
| 283 |
-
},
|
| 284 |
-
{
|
| 285 |
-
"model_id": "amazon/nova-lite-v1:0",
|
| 286 |
-
"name": "Amazon Nova Lite",
|
| 287 |
-
"developer": "amazon",
|
| 288 |
-
"scores": {
|
| 289 |
-
"Mean score": 0.551,
|
| 290 |
-
"MMLU-Pro": 0.6,
|
| 291 |
-
"GPQA": 0.397,
|
| 292 |
-
"IFEval": 0.776,
|
| 293 |
-
"WildBench": 0.75,
|
| 294 |
-
"Omni-MATH": 0.233
|
| 295 |
-
}
|
| 296 |
-
},
|
| 297 |
-
{
|
| 298 |
-
"model_id": "amazon/nova-micro-v1:0",
|
| 299 |
-
"name": "Amazon Nova Micro",
|
| 300 |
-
"developer": "amazon",
|
| 301 |
-
"scores": {
|
| 302 |
-
"Mean score": 0.522,
|
| 303 |
-
"MMLU-Pro": 0.511,
|
| 304 |
-
"GPQA": 0.383,
|
| 305 |
-
"IFEval": 0.76,
|
| 306 |
-
"WildBench": 0.743,
|
| 307 |
-
"Omni-MATH": 0.214
|
| 308 |
-
}
|
| 309 |
-
},
|
| 310 |
-
{
|
| 311 |
-
"model_id": "amazon/nova-premier-v1:0",
|
| 312 |
-
"name": "Amazon Nova Premier",
|
| 313 |
-
"developer": "amazon",
|
| 314 |
-
"scores": {
|
| 315 |
-
"Mean score": 0.637,
|
| 316 |
-
"MMLU-Pro": 0.726,
|
| 317 |
-
"GPQA": 0.518,
|
| 318 |
-
"IFEval": 0.803,
|
| 319 |
-
"WildBench": 0.788,
|
| 320 |
-
"Omni-MATH": 0.35
|
| 321 |
-
}
|
| 322 |
-
},
|
| 323 |
-
{
|
| 324 |
-
"model_id": "amazon/nova-pro-v1:0",
|
| 325 |
-
"name": "Amazon Nova Pro",
|
| 326 |
-
"developer": "amazon",
|
| 327 |
-
"scores": {
|
| 328 |
-
"Mean score": 0.591,
|
| 329 |
-
"MMLU-Pro": 0.673,
|
| 330 |
-
"GPQA": 0.446,
|
| 331 |
-
"IFEval": 0.815,
|
| 332 |
-
"WildBench": 0.777,
|
| 333 |
-
"Omni-MATH": 0.242
|
| 334 |
-
}
|
| 335 |
-
},
|
| 336 |
-
{
|
| 337 |
-
"model_id": "anthropic/claude-3-5-haiku-20241022",
|
| 338 |
-
"name": "Claude 3.5 Haiku 20241022",
|
| 339 |
-
"developer": "Anthropic",
|
| 340 |
-
"scores": {
|
| 341 |
-
"Mean score": 0.549,
|
| 342 |
-
"MMLU-Pro": 0.605,
|
| 343 |
-
"GPQA": 0.363,
|
| 344 |
-
"IFEval": 0.792,
|
| 345 |
-
"WildBench": 0.76,
|
| 346 |
-
"Omni-MATH": 0.224
|
| 347 |
-
}
|
| 348 |
-
},
|
| 349 |
-
{
|
| 350 |
-
"model_id": "anthropic/claude-3-5-sonnet-20241022",
|
| 351 |
-
"name": "Claude 3.5 Sonnet 20241022",
|
| 352 |
-
"developer": "Anthropic",
|
| 353 |
-
"scores": {
|
| 354 |
-
"Mean score": 0.653,
|
| 355 |
-
"MMLU-Pro": 0.777,
|
| 356 |
-
"GPQA": 0.565,
|
| 357 |
-
"IFEval": 0.856,
|
| 358 |
-
"WildBench": 0.792,
|
| 359 |
-
"Omni-MATH": 0.276
|
| 360 |
-
}
|
| 361 |
-
},
|
| 362 |
-
{
|
| 363 |
-
"model_id": "anthropic/claude-3-7-sonnet-20250219",
|
| 364 |
-
"name": "claude-3-7-sonnet-20250219",
|
| 365 |
-
"developer": "Anthropic",
|
| 366 |
-
"scores": {
|
| 367 |
-
"Mean score": 0.674,
|
| 368 |
-
"MMLU-Pro": 0.784,
|
| 369 |
-
"GPQA": 0.608,
|
| 370 |
-
"IFEval": 0.834,
|
| 371 |
-
"WildBench": 0.814,
|
| 372 |
-
"Omni-MATH": 0.33
|
| 373 |
-
}
|
| 374 |
-
},
|
| 375 |
-
{
|
| 376 |
-
"model_id": "anthropic/claude-opus-4-20250514",
|
| 377 |
-
"name": "Claude 4 Opus 20250514",
|
| 378 |
-
"developer": "Anthropic",
|
| 379 |
-
"scores": {
|
| 380 |
-
"Mean score": 0.757,
|
| 381 |
-
"MMLU-Pro": 0.859,
|
| 382 |
-
"GPQA": 0.666,
|
| 383 |
-
"IFEval": 0.918,
|
| 384 |
-
"WildBench": 0.833,
|
| 385 |
-
"Omni-MATH": 0.511
|
| 386 |
-
}
|
| 387 |
-
},
|
| 388 |
-
{
|
| 389 |
-
"model_id": "anthropic/claude-opus-4-20250514-thinking-10k",
|
| 390 |
-
"name": "Claude 4 Opus 20250514, extended thinking",
|
| 391 |
-
"developer": "Anthropic",
|
| 392 |
-
"scores": {
|
| 393 |
-
"Mean score": 0.78,
|
| 394 |
-
"MMLU-Pro": 0.875,
|
| 395 |
-
"GPQA": 0.709,
|
| 396 |
-
"IFEval": 0.849,
|
| 397 |
-
"WildBench": 0.852,
|
| 398 |
-
"Omni-MATH": 0.616
|
| 399 |
-
}
|
| 400 |
-
},
|
| 401 |
-
{
|
| 402 |
-
"model_id": "anthropic/claude-sonnet-4-20250514",
|
| 403 |
-
"name": "claude-sonnet-4-20250514",
|
| 404 |
-
"developer": "Anthropic",
|
| 405 |
-
"scores": {
|
| 406 |
-
"Mean score": 0.733,
|
| 407 |
-
"MMLU-Pro": 0.843,
|
| 408 |
-
"GPQA": 0.643,
|
| 409 |
-
"IFEval": 0.839,
|
| 410 |
-
"WildBench": 0.825,
|
| 411 |
-
"Omni-MATH": 0.512
|
| 412 |
-
}
|
| 413 |
-
},
|
| 414 |
-
{
|
| 415 |
-
"model_id": "anthropic/claude-sonnet-4-20250514-thinking-10k",
|
| 416 |
-
"name": "Claude 4 Sonnet 20250514, extended thinking",
|
| 417 |
-
"developer": "Anthropic",
|
| 418 |
-
"scores": {
|
| 419 |
-
"Mean score": 0.766,
|
| 420 |
-
"MMLU-Pro": 0.843,
|
| 421 |
-
"GPQA": 0.706,
|
| 422 |
-
"IFEval": 0.84,
|
| 423 |
-
"WildBench": 0.838,
|
| 424 |
-
"Omni-MATH": 0.602
|
| 425 |
-
}
|
| 426 |
-
},
|
| 427 |
-
{
|
| 428 |
-
"model_id": "deepseek-ai/deepseek-r1-0528",
|
| 429 |
-
"name": "DeepSeek-R1-0528",
|
| 430 |
-
"developer": "deepseek-ai",
|
| 431 |
-
"scores": {
|
| 432 |
-
"Mean score": 0.699,
|
| 433 |
-
"MMLU-Pro": 0.793,
|
| 434 |
-
"GPQA": 0.666,
|
| 435 |
-
"IFEval": 0.784,
|
| 436 |
-
"WildBench": 0.828,
|
| 437 |
-
"Omni-MATH": 0.424
|
| 438 |
-
}
|
| 439 |
-
},
|
| 440 |
-
{
|
| 441 |
-
"model_id": "deepseek-ai/deepseek-v3",
|
| 442 |
-
"name": "DeepSeek v3",
|
| 443 |
-
"developer": "deepseek-ai",
|
| 444 |
-
"scores": {
|
| 445 |
-
"Mean score": 0.665,
|
| 446 |
-
"MMLU-Pro": 0.723,
|
| 447 |
-
"GPQA": 0.538,
|
| 448 |
-
"IFEval": 0.832,
|
| 449 |
-
"WildBench": 0.831,
|
| 450 |
-
"Omni-MATH": 0.403
|
| 451 |
-
}
|
| 452 |
-
},
|
| 453 |
-
{
|
| 454 |
-
"model_id": "google/gemini-1.5-flash-002",
|
| 455 |
-
"name": "Gemini 1.5 Flash 002",
|
| 456 |
-
"developer": "Google",
|
| 457 |
-
"scores": {
|
| 458 |
-
"Mean score": 0.609,
|
| 459 |
-
"MMLU-Pro": 0.678,
|
| 460 |
-
"GPQA": 0.437,
|
| 461 |
-
"IFEval": 0.831,
|
| 462 |
-
"WildBench": 0.792,
|
| 463 |
-
"Omni-MATH": 0.305
|
| 464 |
-
}
|
| 465 |
-
},
|
| 466 |
-
{
|
| 467 |
-
"model_id": "google/gemini-1.5-pro-002",
|
| 468 |
-
"name": "Gemini 1.5 Pro 002",
|
| 469 |
-
"developer": "Google",
|
| 470 |
-
"scores": {
|
| 471 |
-
"Mean score": 0.657,
|
| 472 |
-
"MMLU-Pro": 0.737,
|
| 473 |
-
"GPQA": 0.534,
|
| 474 |
-
"IFEval": 0.837,
|
| 475 |
-
"WildBench": 0.813,
|
| 476 |
-
"Omni-MATH": 0.364
|
| 477 |
-
}
|
| 478 |
-
},
|
| 479 |
-
{
|
| 480 |
-
"model_id": "google/gemini-2.0-flash-001",
|
| 481 |
-
"name": "Gemini 2.0 Flash",
|
| 482 |
-
"developer": "Google",
|
| 483 |
-
"scores": {
|
| 484 |
-
"Mean score": 0.679,
|
| 485 |
-
"MMLU-Pro": 0.737,
|
| 486 |
-
"GPQA": 0.556,
|
| 487 |
-
"IFEval": 0.841,
|
| 488 |
-
"WildBench": 0.8,
|
| 489 |
-
"Omni-MATH": 0.459
|
| 490 |
-
}
|
| 491 |
-
},
|
| 492 |
-
{
|
| 493 |
-
"model_id": "google/gemini-2.0-flash-lite-preview-02-05",
|
| 494 |
-
"name": "Gemini 2.0 Flash Lite 02-05 preview",
|
| 495 |
-
"developer": "Google",
|
| 496 |
-
"scores": {
|
| 497 |
-
"Mean score": 0.642,
|
| 498 |
-
"MMLU-Pro": 0.72,
|
| 499 |
-
"GPQA": 0.5,
|
| 500 |
-
"IFEval": 0.824,
|
| 501 |
-
"WildBench": 0.79,
|
| 502 |
-
"Omni-MATH": 0.374
|
| 503 |
-
}
|
| 504 |
-
},
|
| 505 |
-
{
|
| 506 |
-
"model_id": "google/gemini-2.5-flash-lite",
|
| 507 |
-
"name": "Gemini 2.5 Flash-Lite",
|
| 508 |
-
"developer": "Google",
|
| 509 |
-
"scores": {
|
| 510 |
-
"Mean score": 0.591,
|
| 511 |
-
"MMLU-Pro": 0.537,
|
| 512 |
-
"GPQA": 0.309,
|
| 513 |
-
"IFEval": 0.81,
|
| 514 |
-
"WildBench": 0.818,
|
| 515 |
-
"Omni-MATH": 0.48
|
| 516 |
-
}
|
| 517 |
-
},
|
| 518 |
-
{
|
| 519 |
-
"model_id": "google/gemini-2.5-flash-preview-04-17",
|
| 520 |
-
"name": "Gemini 2.5 Flash 04-17 preview",
|
| 521 |
-
"developer": "Google",
|
| 522 |
-
"scores": {
|
| 523 |
-
"Mean score": 0.626,
|
| 524 |
-
"MMLU-Pro": 0.639,
|
| 525 |
-
"GPQA": 0.39,
|
| 526 |
-
"IFEval": 0.898,
|
| 527 |
-
"WildBench": 0.817,
|
| 528 |
-
"Omni-MATH": 0.384
|
| 529 |
-
}
|
| 530 |
-
},
|
| 531 |
-
{
|
| 532 |
-
"model_id": "google/gemini-2.5-pro-preview-03-25",
|
| 533 |
-
"name": "Gemini 2.5 Pro 03-25 preview",
|
| 534 |
-
"developer": "Google",
|
| 535 |
-
"scores": {
|
| 536 |
-
"Mean score": 0.745,
|
| 537 |
-
"MMLU-Pro": 0.863,
|
| 538 |
-
"GPQA": 0.749,
|
| 539 |
-
"IFEval": 0.84,
|
| 540 |
-
"WildBench": 0.857,
|
| 541 |
-
"Omni-MATH": 0.416
|
| 542 |
-
}
|
| 543 |
-
},
|
| 544 |
-
{
|
| 545 |
-
"model_id": "ibm/granite-3.3-8b-instruct",
|
| 546 |
-
"name": "IBM Granite 3.3 8B Instruct",
|
| 547 |
-
"developer": "ibm",
|
| 548 |
-
"scores": {
|
| 549 |
-
"Mean score": 0.463,
|
| 550 |
-
"MMLU-Pro": 0.343,
|
| 551 |
-
"GPQA": 0.325,
|
| 552 |
-
"IFEval": 0.729,
|
| 553 |
-
"WildBench": 0.741,
|
| 554 |
-
"Omni-MATH": 0.176
|
| 555 |
-
}
|
| 556 |
-
},
|
| 557 |
-
{
|
| 558 |
-
"model_id": "marin-community/marin-8b-instruct",
|
| 559 |
-
"name": "Marin 8B Instruct",
|
| 560 |
-
"developer": "marin-community",
|
| 561 |
-
"scores": {
|
| 562 |
-
"Mean score": 0.325,
|
| 563 |
-
"MMLU-Pro": 0.188,
|
| 564 |
-
"GPQA": 0.168,
|
| 565 |
-
"IFEval": 0.632,
|
| 566 |
-
"WildBench": 0.477,
|
| 567 |
-
"Omni-MATH": 0.16
|
| 568 |
-
}
|
| 569 |
-
},
|
| 570 |
-
{
|
| 571 |
-
"model_id": "meta/llama-3.1-405b-instruct-turbo",
|
| 572 |
-
"name": "Llama 3.1 Instruct Turbo 405B",
|
| 573 |
-
"developer": "Meta",
|
| 574 |
-
"scores": {
|
| 575 |
-
"Mean score": 0.618,
|
| 576 |
-
"MMLU-Pro": 0.723,
|
| 577 |
-
"GPQA": 0.522,
|
| 578 |
-
"IFEval": 0.811,
|
| 579 |
-
"WildBench": 0.783,
|
| 580 |
-
"Omni-MATH": 0.249
|
| 581 |
-
}
|
| 582 |
-
},
|
| 583 |
-
{
|
| 584 |
-
"model_id": "meta/llama-3.1-70b-instruct-turbo",
|
| 585 |
-
"name": "Llama 3.1 Instruct Turbo 70B",
|
| 586 |
-
"developer": "Meta",
|
| 587 |
-
"scores": {
|
| 588 |
-
"Mean score": 0.574,
|
| 589 |
-
"MMLU-Pro": 0.653,
|
| 590 |
-
"GPQA": 0.426,
|
| 591 |
-
"IFEval": 0.821,
|
| 592 |
-
"WildBench": 0.758,
|
| 593 |
-
"Omni-MATH": 0.21
|
| 594 |
-
}
|
| 595 |
-
},
|
| 596 |
-
{
|
| 597 |
-
"model_id": "meta/llama-3.1-8b-instruct-turbo",
|
| 598 |
-
"name": "Llama 3.1 Instruct Turbo 8B",
|
| 599 |
-
"developer": "Meta",
|
| 600 |
-
"scores": {
|
| 601 |
-
"Mean score": 0.444,
|
| 602 |
-
"MMLU-Pro": 0.406,
|
| 603 |
-
"GPQA": 0.247,
|
| 604 |
-
"IFEval": 0.743,
|
| 605 |
-
"WildBench": 0.686,
|
| 606 |
-
"Omni-MATH": 0.137
|
| 607 |
-
}
|
| 608 |
-
},
|
| 609 |
-
{
|
| 610 |
-
"model_id": "meta/llama-4-maverick-17b-128e-instruct-fp8",
|
| 611 |
-
"name": "Llama 4 Maverick 17Bx128E Instruct FP8",
|
| 612 |
-
"developer": "Meta",
|
| 613 |
-
"scores": {
|
| 614 |
-
"Mean score": 0.718,
|
| 615 |
-
"MMLU-Pro": 0.81,
|
| 616 |
-
"GPQA": 0.65,
|
| 617 |
-
"IFEval": 0.908,
|
| 618 |
-
"WildBench": 0.8,
|
| 619 |
-
"Omni-MATH": 0.422
|
| 620 |
-
}
|
| 621 |
-
},
|
| 622 |
-
{
|
| 623 |
-
"model_id": "meta/llama-4-scout-17b-16e-instruct",
|
| 624 |
-
"name": "Llama 4 Scout 17Bx16E Instruct",
|
| 625 |
-
"developer": "Meta",
|
| 626 |
-
"scores": {
|
| 627 |
-
"Mean score": 0.644,
|
| 628 |
-
"MMLU-Pro": 0.742,
|
| 629 |
-
"GPQA": 0.507,
|
| 630 |
-
"IFEval": 0.818,
|
| 631 |
-
"WildBench": 0.779,
|
| 632 |
-
"Omni-MATH": 0.373
|
| 633 |
-
}
|
| 634 |
-
},
|
| 635 |
-
{
|
| 636 |
-
"model_id": "mistralai/Mixtral-8x22B-Instruct-v0.1",
|
| 637 |
-
"name": "Mixtral-8x22B-Instruct-v0.1",
|
| 638 |
-
"developer": "mistralai",
|
| 639 |
-
"scores": {
|
| 640 |
-
"Mean score": 0.478,
|
| 641 |
-
"MMLU-Pro": 0.46,
|
| 642 |
-
"GPQA": 0.334,
|
| 643 |
-
"IFEval": 0.724,
|
| 644 |
-
"WildBench": 0.711,
|
| 645 |
-
"Omni-MATH": 0.163
|
| 646 |
-
}
|
| 647 |
-
},
|
| 648 |
-
{
|
| 649 |
-
"model_id": "mistralai/Mixtral-8x7B-Instruct-v0.1",
|
| 650 |
-
"name": "Mixtral-8x7B-Instruct-v0.1",
|
| 651 |
-
"developer": "mistralai",
|
| 652 |
-
"scores": {
|
| 653 |
-
"Mean score": 0.397,
|
| 654 |
-
"MMLU-Pro": 0.335,
|
| 655 |
-
"GPQA": 0.296,
|
| 656 |
-
"IFEval": 0.575,
|
| 657 |
-
"WildBench": 0.673,
|
| 658 |
-
"Omni-MATH": 0.105
|
| 659 |
-
}
|
| 660 |
-
},
|
| 661 |
-
{
|
| 662 |
-
"model_id": "mistralai/mistral-7b-instruct-v0.3",
|
| 663 |
-
"name": "Mistral Instruct v0.3 7B",
|
| 664 |
-
"developer": "mistralai",
|
| 665 |
-
"scores": {
|
| 666 |
-
"Mean score": 0.376,
|
| 667 |
-
"MMLU-Pro": 0.277,
|
| 668 |
-
"GPQA": 0.303,
|
| 669 |
-
"IFEval": 0.567,
|
| 670 |
-
"WildBench": 0.66,
|
| 671 |
-
"Omni-MATH": 0.072
|
| 672 |
-
}
|
| 673 |
-
},
|
| 674 |
-
{
|
| 675 |
-
"model_id": "mistralai/mistral-large-2411",
|
| 676 |
-
"name": "Mistral Large 2411",
|
| 677 |
-
"developer": "mistralai",
|
| 678 |
-
"scores": {
|
| 679 |
-
"Mean score": 0.598,
|
| 680 |
-
"MMLU-Pro": 0.599,
|
| 681 |
-
"GPQA": 0.435,
|
| 682 |
-
"IFEval": 0.876,
|
| 683 |
-
"WildBench": 0.801,
|
| 684 |
-
"Omni-MATH": 0.281
|
| 685 |
-
}
|
| 686 |
-
},
|
| 687 |
-
{
|
| 688 |
-
"model_id": "mistralai/mistral-small-2503",
|
| 689 |
-
"name": "mistral-small-2503",
|
| 690 |
-
"developer": "mistralai",
|
| 691 |
-
"scores": {
|
| 692 |
-
"Mean score": 0.558,
|
| 693 |
-
"MMLU-Pro": 0.61,
|
| 694 |
-
"GPQA": 0.392,
|
| 695 |
-
"IFEval": 0.75,
|
| 696 |
-
"WildBench": 0.788,
|
| 697 |
-
"Omni-MATH": 0.248
|
| 698 |
-
}
|
| 699 |
-
},
|
| 700 |
-
{
|
| 701 |
-
"model_id": "moonshotai/kimi-k2-instruct",
|
| 702 |
-
"name": "Kimi K2 Instruct",
|
| 703 |
-
"developer": "moonshotai",
|
| 704 |
-
"scores": {
|
| 705 |
-
"Mean score": 0.768,
|
| 706 |
-
"MMLU-Pro": 0.819,
|
| 707 |
-
"GPQA": 0.652,
|
| 708 |
-
"IFEval": 0.85,
|
| 709 |
-
"WildBench": 0.862,
|
| 710 |
-
"Omni-MATH": 0.654
|
| 711 |
-
}
|
| 712 |
-
},
|
| 713 |
-
{
|
| 714 |
-
"model_id": "openai/gpt-4.1-2025-04-14",
|
| 715 |
-
"name": "gpt-4.1-2025-04-14",
|
| 716 |
-
"developer": "OpenAI",
|
| 717 |
-
"scores": {
|
| 718 |
-
"Mean score": 0.727,
|
| 719 |
-
"MMLU-Pro": 0.811,
|
| 720 |
-
"GPQA": 0.659,
|
| 721 |
-
"IFEval": 0.838,
|
| 722 |
-
"WildBench": 0.854,
|
| 723 |
-
"Omni-MATH": 0.471
|
| 724 |
-
}
|
| 725 |
-
},
|
| 726 |
-
{
|
| 727 |
-
"model_id": "openai/gpt-4.1-mini-2025-04-14",
|
| 728 |
-
"name": "GPT-4.1 mini 2025-04-14",
|
| 729 |
-
"developer": "OpenAI",
|
| 730 |
-
"scores": {
|
| 731 |
-
"Mean score": 0.726,
|
| 732 |
-
"MMLU-Pro": 0.783,
|
| 733 |
-
"GPQA": 0.614,
|
| 734 |
-
"IFEval": 0.904,
|
| 735 |
-
"WildBench": 0.838,
|
| 736 |
-
"Omni-MATH": 0.491
|
| 737 |
-
}
|
| 738 |
-
},
|
| 739 |
-
{
|
| 740 |
-
"model_id": "openai/gpt-4.1-nano-2025-04-14",
|
| 741 |
-
"name": "GPT-4.1 nano 2025-04-14",
|
| 742 |
-
"developer": "OpenAI",
|
| 743 |
-
"scores": {
|
| 744 |
-
"Mean score": 0.616,
|
| 745 |
-
"MMLU-Pro": 0.55,
|
| 746 |
-
"GPQA": 0.507,
|
| 747 |
-
"IFEval": 0.843,
|
| 748 |
-
"WildBench": 0.811,
|
| 749 |
-
"Omni-MATH": 0.367
|
| 750 |
-
}
|
| 751 |
-
},
|
| 752 |
-
{
|
| 753 |
-
"model_id": "openai/gpt-4o-2024-11-20",
|
| 754 |
-
"name": "GPT-4o 2024-11-20",
|
| 755 |
-
"developer": "OpenAI",
|
| 756 |
-
"scores": {
|
| 757 |
-
"Mean score": 0.634,
|
| 758 |
-
"MMLU-Pro": 0.713,
|
| 759 |
-
"GPQA": 0.52,
|
| 760 |
-
"IFEval": 0.817,
|
| 761 |
-
"WildBench": 0.828,
|
| 762 |
-
"Omni-MATH": 0.293
|
| 763 |
-
}
|
| 764 |
-
},
|
| 765 |
-
{
|
| 766 |
-
"model_id": "openai/gpt-4o-mini-2024-07-18",
|
| 767 |
-
"name": "GPT-4o mini 2024-07-18",
|
| 768 |
-
"developer": "OpenAI",
|
| 769 |
-
"scores": {
|
| 770 |
-
"Mean score": 0.565,
|
| 771 |
-
"MMLU-Pro": 0.603,
|
| 772 |
-
"GPQA": 0.368,
|
| 773 |
-
"IFEval": 0.782,
|
| 774 |
-
"WildBench": 0.791,
|
| 775 |
-
"Omni-MATH": 0.28
|
| 776 |
-
}
|
| 777 |
-
},
|
| 778 |
-
{
|
| 779 |
-
"model_id": "openai/gpt-5-2025-08-07",
|
| 780 |
-
"name": "gpt-5-2025-08-07",
|
| 781 |
-
"developer": "OpenAI",
|
| 782 |
-
"scores": {
|
| 783 |
-
"Mean score": 0.807,
|
| 784 |
-
"MMLU-Pro": 0.863,
|
| 785 |
-
"GPQA": 0.791,
|
| 786 |
-
"IFEval": 0.875,
|
| 787 |
-
"WildBench": 0.857,
|
| 788 |
-
"Omni-MATH": 0.647
|
| 789 |
-
}
|
| 790 |
-
},
|
| 791 |
-
{
|
| 792 |
-
"model_id": "openai/gpt-5-mini-2025-08-07",
|
| 793 |
-
"name": "GPT-5 mini 2025-08-07",
|
| 794 |
-
"developer": "OpenAI",
|
| 795 |
-
"scores": {
|
| 796 |
-
"Mean score": 0.819,
|
| 797 |
-
"MMLU-Pro": 0.835,
|
| 798 |
-
"GPQA": 0.756,
|
| 799 |
-
"IFEval": 0.927,
|
| 800 |
-
"WildBench": 0.855,
|
| 801 |
-
"Omni-MATH": 0.722
|
| 802 |
-
}
|
| 803 |
-
},
|
| 804 |
-
{
|
| 805 |
-
"model_id": "openai/gpt-5-nano-2025-08-07",
|
| 806 |
-
"name": "GPT-5 nano 2025-08-07",
|
| 807 |
-
"developer": "OpenAI",
|
| 808 |
-
"scores": {
|
| 809 |
-
"Mean score": 0.748,
|
| 810 |
-
"MMLU-Pro": 0.778,
|
| 811 |
-
"GPQA": 0.679,
|
| 812 |
-
"IFEval": 0.932,
|
| 813 |
-
"WildBench": 0.806,
|
| 814 |
-
"Omni-MATH": 0.547
|
| 815 |
-
}
|
| 816 |
-
},
|
| 817 |
-
{
|
| 818 |
-
"model_id": "openai/gpt-oss-120b",
|
| 819 |
-
"name": "GPT-OSS-120B",
|
| 820 |
-
"developer": "OpenAI",
|
| 821 |
-
"scores": {
|
| 822 |
-
"Mean score": 0.77,
|
| 823 |
-
"MMLU-Pro": 0.795,
|
| 824 |
-
"GPQA": 0.684,
|
| 825 |
-
"IFEval": 0.836,
|
| 826 |
-
"WildBench": 0.845,
|
| 827 |
-
"Omni-MATH": 0.688
|
| 828 |
-
}
|
| 829 |
-
},
|
| 830 |
-
{
|
| 831 |
-
"model_id": "openai/gpt-oss-20b",
|
| 832 |
-
"name": "GPT-OSS-20B",
|
| 833 |
-
"developer": "OpenAI",
|
| 834 |
-
"scores": {
|
| 835 |
-
"Mean score": 0.674,
|
| 836 |
-
"MMLU-Pro": 0.74,
|
| 837 |
-
"GPQA": 0.594,
|
| 838 |
-
"IFEval": 0.732,
|
| 839 |
-
"WildBench": 0.737,
|
| 840 |
-
"Omni-MATH": 0.565
|
| 841 |
-
}
|
| 842 |
-
},
|
| 843 |
-
{
|
| 844 |
-
"model_id": "openai/o3-2025-04-16",
|
| 845 |
-
"name": "o3-2025-04-16",
|
| 846 |
-
"developer": "OpenAI",
|
| 847 |
-
"scores": {
|
| 848 |
-
"Mean score": 0.811,
|
| 849 |
-
"MMLU-Pro": 0.859,
|
| 850 |
-
"GPQA": 0.753,
|
| 851 |
-
"IFEval": 0.869,
|
| 852 |
-
"WildBench": 0.861,
|
| 853 |
-
"Omni-MATH": 0.714
|
| 854 |
-
}
|
| 855 |
-
},
|
| 856 |
-
{
|
| 857 |
-
"model_id": "openai/o4-mini-2025-04-16",
|
| 858 |
-
"name": "o4-mini-2025-04-16",
|
| 859 |
-
"developer": "OpenAI",
|
| 860 |
-
"scores": {
|
| 861 |
-
"Mean score": 0.812,
|
| 862 |
-
"MMLU-Pro": 0.82,
|
| 863 |
-
"GPQA": 0.735,
|
| 864 |
-
"IFEval": 0.929,
|
| 865 |
-
"WildBench": 0.854,
|
| 866 |
-
"Omni-MATH": 0.72
|
| 867 |
-
}
|
| 868 |
-
},
|
| 869 |
-
{
|
| 870 |
-
"model_id": "qwen/qwen2.5-72b-instruct-turbo",
|
| 871 |
-
"name": "Qwen2.5 Instruct Turbo 72B",
|
| 872 |
-
"developer": "qwen",
|
| 873 |
-
"scores": {
|
| 874 |
-
"Mean score": 0.599,
|
| 875 |
-
"MMLU-Pro": 0.631,
|
| 876 |
-
"GPQA": 0.426,
|
| 877 |
-
"IFEval": 0.806,
|
| 878 |
-
"WildBench": 0.802,
|
| 879 |
-
"Omni-MATH": 0.33
|
| 880 |
-
}
|
| 881 |
-
},
|
| 882 |
-
{
|
| 883 |
-
"model_id": "qwen/qwen2.5-7b-instruct-turbo",
|
| 884 |
-
"name": "Qwen2.5 Instruct Turbo 7B",
|
| 885 |
-
"developer": "qwen",
|
| 886 |
-
"scores": {
|
| 887 |
-
"Mean score": 0.529,
|
| 888 |
-
"MMLU-Pro": 0.539,
|
| 889 |
-
"GPQA": 0.341,
|
| 890 |
-
"IFEval": 0.741,
|
| 891 |
-
"WildBench": 0.731,
|
| 892 |
-
"Omni-MATH": 0.294
|
| 893 |
-
}
|
| 894 |
-
},
|
| 895 |
-
{
|
| 896 |
-
"model_id": "qwen/qwen3-235b-a22b-fp8-tput",
|
| 897 |
-
"name": "Qwen3 235B A22B FP8 Throughput",
|
| 898 |
-
"developer": "qwen",
|
| 899 |
-
"scores": {
|
| 900 |
-
"Mean score": 0.726,
|
| 901 |
-
"MMLU-Pro": 0.817,
|
| 902 |
-
"GPQA": 0.623,
|
| 903 |
-
"IFEval": 0.816,
|
| 904 |
-
"WildBench": 0.828,
|
| 905 |
-
"Omni-MATH": 0.548
|
| 906 |
-
}
|
| 907 |
-
},
|
| 908 |
-
{
|
| 909 |
-
"model_id": "qwen/qwen3-235b-a22b-instruct-2507-fp8",
|
| 910 |
-
"name": "Qwen3 235B A22B Instruct 2507 FP8",
|
| 911 |
-
"developer": "qwen",
|
| 912 |
-
"scores": {
|
| 913 |
-
"Mean score": 0.798,
|
| 914 |
-
"MMLU-Pro": 0.844,
|
| 915 |
-
"GPQA": 0.726,
|
| 916 |
-
"IFEval": 0.835,
|
| 917 |
-
"WildBench": 0.866,
|
| 918 |
-
"Omni-MATH": 0.718
|
| 919 |
-
}
|
| 920 |
-
},
|
| 921 |
-
{
|
| 922 |
-
"model_id": "writer/palmyra-fin",
|
| 923 |
-
"name": "Palmyra Fin",
|
| 924 |
-
"developer": "writer",
|
| 925 |
-
"scores": {
|
| 926 |
-
"Mean score": 0.577,
|
| 927 |
-
"MMLU-Pro": 0.591,
|
| 928 |
-
"GPQA": 0.422,
|
| 929 |
-
"IFEval": 0.793,
|
| 930 |
-
"WildBench": 0.783,
|
| 931 |
-
"Omni-MATH": 0.295
|
| 932 |
-
}
|
| 933 |
-
},
|
| 934 |
-
{
|
| 935 |
-
"model_id": "writer/palmyra-med",
|
| 936 |
-
"name": "Palmyra Med",
|
| 937 |
-
"developer": "writer",
|
| 938 |
-
"scores": {
|
| 939 |
-
"Mean score": 0.476,
|
| 940 |
-
"MMLU-Pro": 0.411,
|
| 941 |
-
"GPQA": 0.368,
|
| 942 |
-
"IFEval": 0.767,
|
| 943 |
-
"WildBench": 0.676,
|
| 944 |
-
"Omni-MATH": 0.156
|
| 945 |
-
}
|
| 946 |
-
},
|
| 947 |
-
{
|
| 948 |
-
"model_id": "writer/palmyra-x-004",
|
| 949 |
-
"name": "Palmyra-X-004",
|
| 950 |
-
"developer": "writer",
|
| 951 |
-
"scores": {
|
| 952 |
-
"Mean score": 0.609,
|
| 953 |
-
"MMLU-Pro": 0.657,
|
| 954 |
-
"GPQA": 0.395,
|
| 955 |
-
"IFEval": 0.872,
|
| 956 |
-
"WildBench": 0.802,
|
| 957 |
-
"Omni-MATH": 0.32
|
| 958 |
-
}
|
| 959 |
-
},
|
| 960 |
-
{
|
| 961 |
-
"model_id": "writer/palmyra-x5",
|
| 962 |
-
"name": "Palmyra X5",
|
| 963 |
-
"developer": "writer",
|
| 964 |
-
"scores": {
|
| 965 |
-
"Mean score": 0.696,
|
| 966 |
-
"MMLU-Pro": 0.804,
|
| 967 |
-
"GPQA": 0.661,
|
| 968 |
-
"IFEval": 0.823,
|
| 969 |
-
"WildBench": 0.78,
|
| 970 |
-
"Omni-MATH": 0.414
|
| 971 |
-
}
|
| 972 |
-
},
|
| 973 |
-
{
|
| 974 |
-
"model_id": "xai/grok-3-beta",
|
| 975 |
-
"name": "Grok 3 Beta",
|
| 976 |
-
"developer": "xAI",
|
| 977 |
-
"scores": {
|
| 978 |
-
"Mean score": 0.727,
|
| 979 |
-
"MMLU-Pro": 0.788,
|
| 980 |
-
"GPQA": 0.65,
|
| 981 |
-
"IFEval": 0.884,
|
| 982 |
-
"WildBench": 0.849,
|
| 983 |
-
"Omni-MATH": 0.464
|
| 984 |
-
}
|
| 985 |
-
},
|
| 986 |
-
{
|
| 987 |
-
"model_id": "xai/grok-3-mini-beta",
|
| 988 |
-
"name": "Grok 3 mini Beta",
|
| 989 |
-
"developer": "xAI",
|
| 990 |
-
"scores": {
|
| 991 |
-
"Mean score": 0.679,
|
| 992 |
-
"MMLU-Pro": 0.799,
|
| 993 |
-
"GPQA": 0.675,
|
| 994 |
-
"IFEval": 0.951,
|
| 995 |
-
"WildBench": 0.651,
|
| 996 |
-
"Omni-MATH": 0.318
|
| 997 |
-
}
|
| 998 |
-
},
|
| 999 |
-
{
|
| 1000 |
-
"model_id": "xai/grok-4-0709",
|
| 1001 |
-
"name": "grok-4-0709",
|
| 1002 |
-
"developer": "xAI",
|
| 1003 |
-
"scores": {
|
| 1004 |
-
"Mean score": 0.785,
|
| 1005 |
-
"MMLU-Pro": 0.851,
|
| 1006 |
-
"GPQA": 0.726,
|
| 1007 |
-
"IFEval": 0.949,
|
| 1008 |
-
"WildBench": 0.797,
|
| 1009 |
-
"Omni-MATH": 0.603
|
| 1010 |
-
}
|
| 1011 |
-
},
|
| 1012 |
-
{
|
| 1013 |
-
"model_id": "zai-org/glm-4.5-air-fp8",
|
| 1014 |
-
"name": "GLM-4.5-Air-FP8",
|
| 1015 |
-
"developer": "zai-org",
|
| 1016 |
-
"scores": {
|
| 1017 |
-
"Mean score": 0.67,
|
| 1018 |
-
"MMLU-Pro": 0.762,
|
| 1019 |
-
"GPQA": 0.594,
|
| 1020 |
-
"IFEval": 0.812,
|
| 1021 |
-
"WildBench": 0.789,
|
| 1022 |
-
"Omni-MATH": 0.391
|
| 1023 |
-
}
|
| 1024 |
-
}
|
| 1025 |
-
]
|
| 1026 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/helm_classic.json
DELETED
|
@@ -1,2030 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"benchmark_cards": {
|
| 3 |
-
"BoolQ": {
|
| 4 |
-
"benchmark_details": {
|
| 5 |
-
"name": "BoolQ",
|
| 6 |
-
"overview": "BoolQ is a benchmark that measures a model's ability to answer naturally occurring yes/no questions, framed as a reading comprehension task. The questions are generated in unprompted and unconstrained settings, often querying complex, non-factoid information and requiring difficult entailment-like inference. The dataset consists of a single task: answering yes/no questions given a supporting passage.",
|
| 7 |
-
"data_type": "text",
|
| 8 |
-
"domains": [
|
| 9 |
-
"natural language understanding",
|
| 10 |
-
"reading comprehension",
|
| 11 |
-
"natural language inference"
|
| 12 |
-
],
|
| 13 |
-
"languages": [
|
| 14 |
-
"English"
|
| 15 |
-
],
|
| 16 |
-
"similar_benchmarks": [
|
| 17 |
-
"MultiNLI",
|
| 18 |
-
"SNLI",
|
| 19 |
-
"QNLI",
|
| 20 |
-
"SQuAD 2.0",
|
| 21 |
-
"Natural Questions (NQ)",
|
| 22 |
-
"QQP",
|
| 23 |
-
"MS MARCO",
|
| 24 |
-
"RACE",
|
| 25 |
-
"bAbI stories"
|
| 26 |
-
],
|
| 27 |
-
"resources": [
|
| 28 |
-
"https://arxiv.org/abs/1905.10044",
|
| 29 |
-
"https://huggingface.co/datasets/google/boolq",
|
| 30 |
-
"https://goo.gl/boolq",
|
| 31 |
-
"https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json"
|
| 32 |
-
]
|
| 33 |
-
},
|
| 34 |
-
"purpose_and_intended_users": {
|
| 35 |
-
"goal": "To test models on their ability to answer naturally occurring yes/no questions, which are challenging and require complex inferential abilities beyond surface-level reasoning.",
|
| 36 |
-
"audience": [
|
| 37 |
-
"Researchers in natural language understanding and reading comprehension"
|
| 38 |
-
],
|
| 39 |
-
"tasks": [
|
| 40 |
-
"Yes/no question answering",
|
| 41 |
-
"Text-pair classification"
|
| 42 |
-
],
|
| 43 |
-
"limitations": "Annotation involved some errors and ambiguous cases. The use of singly-annotated examples is a trade-off for dataset size. Potential concerns about annotation artifacts are acknowledged.",
|
| 44 |
-
"out_of_scope_uses": "The paper does not explicitly state what the benchmark is not designed for."
|
| 45 |
-
},
|
| 46 |
-
"data": {
|
| 47 |
-
"source": "The data consists of naturally occurring yes/no questions authored by people who were not prompted to write specific question types and did not know the answers. The passages are excerpts from sources like Wikipedia.",
|
| 48 |
-
"size": "15,942 examples total, with 9,427 in the train split and 3,270 in the validation split. The dataset size category is between 10,000 and 100,000 examples.",
|
| 49 |
-
"format": "parquet",
|
| 50 |
-
"annotation": "Questions were answered by human annotators. A quality check on a subset showed the main annotation process achieved 90% accuracy against a gold-standard set labeled by three authors. The training, development, and test sets use singly-annotated examples."
|
| 51 |
-
},
|
| 52 |
-
"methodology": {
|
| 53 |
-
"methods": [
|
| 54 |
-
"Models are evaluated by fine-tuning on the BoolQ training set, potentially after transfer learning from other datasets or unsupervised pre-training. Zero-shot or direct use of pre-trained models without fine-tuning did not outperform the majority baseline.",
|
| 55 |
-
"The task requires providing a yes/no (boolean) answer to a question based on a given passage."
|
| 56 |
-
],
|
| 57 |
-
"metrics": [
|
| 58 |
-
"Accuracy"
|
| 59 |
-
],
|
| 60 |
-
"calculation": "The overall score is the accuracy percentage on the test set.",
|
| 61 |
-
"interpretation": "Higher accuracy indicates better performance. Human accuracy is 90%, and the majority baseline is approximately 62%.",
|
| 62 |
-
"baseline_results": "Paper baselines: Majority baseline: 62.17% dev, 62.31% test; Recurrent model baseline: 69.6%; Best model (BERT large pre-trained on MultiNLI then fine-tuned on BoolQ): 80.4% accuracy; Human accuracy: 90%. EEE results: Anthropic-LM v4-s3 52B: 81.5%.",
|
| 63 |
-
"validation": "Quality assurance involved author-led gold-standard annotation on a subset, showing 90% agreement. The development set was used for model selection, such as choosing the best model from five seeds based on its performance."
|
| 64 |
-
},
|
| 65 |
-
"ethical_and_legal_considerations": {
|
| 66 |
-
"privacy_and_anonymity": "Not specified",
|
| 67 |
-
"data_licensing": "cc-by-sa-3.0",
|
| 68 |
-
"consent_procedures": "Not specified",
|
| 69 |
-
"compliance_with_regulations": "Not specified"
|
| 70 |
-
},
|
| 71 |
-
"possible_risks": [
|
| 72 |
-
{
|
| 73 |
-
"category": "Over- or under-reliance",
|
| 74 |
-
"description": [
|
| 75 |
-
"In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
|
| 76 |
-
],
|
| 77 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
|
| 78 |
-
},
|
| 79 |
-
{
|
| 80 |
-
"category": "Unrepresentative data",
|
| 81 |
-
"description": [
|
| 82 |
-
"Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
|
| 83 |
-
],
|
| 84 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
|
| 85 |
-
},
|
| 86 |
-
{
|
| 87 |
-
"category": "Uncertain data provenance",
|
| 88 |
-
"description": [
|
| 89 |
-
"Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation."
|
| 90 |
-
],
|
| 91 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html"
|
| 92 |
-
},
|
| 93 |
-
{
|
| 94 |
-
"category": "Data bias",
|
| 95 |
-
"description": [
|
| 96 |
-
"Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
|
| 97 |
-
],
|
| 98 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
|
| 99 |
-
},
|
| 100 |
-
{
|
| 101 |
-
"category": "Lack of data transparency",
|
| 102 |
-
"description": [
|
| 103 |
-
"Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
|
| 104 |
-
],
|
| 105 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
|
| 106 |
-
}
|
| 107 |
-
],
|
| 108 |
-
"flagged_fields": {},
|
| 109 |
-
"missing_fields": [
|
| 110 |
-
"ethical_and_legal_considerations.privacy_and_anonymity",
|
| 111 |
-
"ethical_and_legal_considerations.consent_procedures",
|
| 112 |
-
"ethical_and_legal_considerations.compliance_with_regulations"
|
| 113 |
-
],
|
| 114 |
-
"card_info": {
|
| 115 |
-
"created_at": "2026-03-17T15:08:51.830946",
|
| 116 |
-
"llm": "deepseek-ai/DeepSeek-V3.2"
|
| 117 |
-
}
|
| 118 |
-
},
|
| 119 |
-
"CNN/DailyMail": {
|
| 120 |
-
"benchmark_details": {
|
| 121 |
-
"name": "CNN/DailyMail",
|
| 122 |
-
"overview": "CNN/DailyMail is a benchmark for evaluating abstractive and extractive summarization models using news articles. It contains over 300,000 unique articles written by journalists from CNN and the Daily Mail. The dataset was originally created for machine reading and question answering, but later versions were restructured specifically for summarization tasks.",
|
| 123 |
-
"data_type": "text",
|
| 124 |
-
"domains": [
|
| 125 |
-
"summarization",
|
| 126 |
-
"journalism",
|
| 127 |
-
"news media"
|
| 128 |
-
],
|
| 129 |
-
"languages": [
|
| 130 |
-
"English"
|
| 131 |
-
],
|
| 132 |
-
"similar_benchmarks": "No facts provided about similar benchmarks.",
|
| 133 |
-
"resources": [
|
| 134 |
-
"https://huggingface.co/datasets/abisee/cnn_dailymail",
|
| 135 |
-
"https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json"
|
| 136 |
-
]
|
| 137 |
-
},
|
| 138 |
-
"purpose_and_intended_users": {
|
| 139 |
-
"goal": "To help develop models that can summarize long paragraphs of text into one or two sentences, aiding in the efficient presentation of information from large quantities of text.",
|
| 140 |
-
"audience": [
|
| 141 |
-
"NLP researchers",
|
| 142 |
-
"Summarization model developers"
|
| 143 |
-
],
|
| 144 |
-
"tasks": [
|
| 145 |
-
"Summarization"
|
| 146 |
-
],
|
| 147 |
-
"limitations": "News articles often place important information in the first third, which may affect summarization. A manual study found 25% of samples in an earlier version were difficult for humans due to ambiguity and coreference errors. Also, machine-generated summaries may differ in truth values from the original articles.",
|
| 148 |
-
"out_of_scope_uses": "No facts provided about out-of-scope uses."
|
| 149 |
-
},
|
| 150 |
-
"data": {
|
| 151 |
-
"source": "The dataset consists of news articles and highlight sentences written by journalists at CNN and the Daily Mail. The CNN articles were collected from April 2007 to April 2015, and the Daily Mail articles from June 2010 to April 2015, sourced from archives on the Wayback Machine.",
|
| 152 |
-
"size": "Over 300,000 unique articles, with 287,113 training examples, 13,368 validation examples, and 11,490 test examples.",
|
| 153 |
-
"format": "parquet",
|
| 154 |
-
"annotation": "The dataset does not contain additional annotations. The highlights are the original summaries written by the article authors and are used as the target for summarization."
|
| 155 |
-
},
|
| 156 |
-
"methodology": {
|
| 157 |
-
"methods": [
|
| 158 |
-
"Models generate a summary for a given news article, which is then compared to the author-written highlights."
|
| 159 |
-
],
|
| 160 |
-
"metrics": [
|
| 161 |
-
"ROUGE-2"
|
| 162 |
-
],
|
| 163 |
-
"calculation": "The ROUGE-2 score measures the overlap of bigrams between the generated summary and the reference highlights.",
|
| 164 |
-
"interpretation": "Higher scores indicate better performance, as they reflect greater overlap with the reference summaries.",
|
| 165 |
-
"baseline_results": "Paper baseline (Zhong et al., 2020): ROUGE-1 score of 44.41 for an extractive summarization model. Evaluation suite result (Anthropic-LM v4-s3 52B): ROUGE-2 score of 0.154.",
|
| 166 |
-
"validation": "No facts provided about validation procedures."
|
| 167 |
-
},
|
| 168 |
-
"ethical_and_legal_considerations": {
|
| 169 |
-
"privacy_and_anonymity": "The dataset (version 3.0.0) is not anonymized, meaning individuals' names are present in the text.",
|
| 170 |
-
"data_licensing": "Apache License 2.0",
|
| 171 |
-
"consent_procedures": "Not specified",
|
| 172 |
-
"compliance_with_regulations": "Not specified"
|
| 173 |
-
},
|
| 174 |
-
"possible_risks": [
|
| 175 |
-
{
|
| 176 |
-
"category": "Over- or under-reliance",
|
| 177 |
-
"description": [
|
| 178 |
-
"In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
|
| 179 |
-
],
|
| 180 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
|
| 181 |
-
},
|
| 182 |
-
{
|
| 183 |
-
"category": "Unrepresentative data",
|
| 184 |
-
"description": [
|
| 185 |
-
"Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
|
| 186 |
-
],
|
| 187 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
|
| 188 |
-
},
|
| 189 |
-
{
|
| 190 |
-
"category": "Data bias",
|
| 191 |
-
"description": [
|
| 192 |
-
"Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
|
| 193 |
-
],
|
| 194 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
|
| 195 |
-
},
|
| 196 |
-
{
|
| 197 |
-
"category": "Data contamination",
|
| 198 |
-
"description": [
|
| 199 |
-
"Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation."
|
| 200 |
-
],
|
| 201 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html"
|
| 202 |
-
},
|
| 203 |
-
{
|
| 204 |
-
"category": "Lack of data transparency",
|
| 205 |
-
"description": [
|
| 206 |
-
"Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
|
| 207 |
-
],
|
| 208 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
|
| 209 |
-
}
|
| 210 |
-
],
|
| 211 |
-
"flagged_fields": {},
|
| 212 |
-
"missing_fields": [
|
| 213 |
-
"ethical_and_legal_considerations.consent_procedures",
|
| 214 |
-
"ethical_and_legal_considerations.compliance_with_regulations"
|
| 215 |
-
],
|
| 216 |
-
"card_info": {
|
| 217 |
-
"created_at": "2026-03-17T15:15:47.316103",
|
| 218 |
-
"llm": "deepseek-ai/DeepSeek-V3.2"
|
| 219 |
-
}
|
| 220 |
-
},
|
| 221 |
-
"CivilComments": {
|
| 222 |
-
"benchmark_details": {
|
| 223 |
-
"name": "CivilComments",
|
| 224 |
-
"overview": "CivilComments is a benchmark designed to measure unintended identity-based bias in toxicity classification models. It uses a large, real-world dataset of online comments from the Civil Comments platform, extended with crowd-sourced annotations for toxicity and demographic identity references. This provides a nuanced evaluation of bias beyond synthetic datasets.",
|
| 225 |
-
"data_type": "tabular, text",
|
| 226 |
-
"domains": [
|
| 227 |
-
"machine learning fairness",
|
| 228 |
-
"bias measurement",
|
| 229 |
-
"toxic comment classification",
|
| 230 |
-
"text classification"
|
| 231 |
-
],
|
| 232 |
-
"languages": [
|
| 233 |
-
"English"
|
| 234 |
-
],
|
| 235 |
-
"similar_benchmarks": "The paper does not name other specific benchmark datasets, only referencing prior work using synthetic test sets.",
|
| 236 |
-
"resources": [
|
| 237 |
-
"https://arxiv.org/abs/1903.04561",
|
| 238 |
-
"https://huggingface.co/datasets/google/civil_comments",
|
| 239 |
-
"https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json"
|
| 240 |
-
]
|
| 241 |
-
},
|
| 242 |
-
"purpose_and_intended_users": {
|
| 243 |
-
"goal": "To evaluate unintended identity-based bias in toxicity classification models using real data and nuanced metrics.",
|
| 244 |
-
"audience": [
|
| 245 |
-
"Machine learning researchers and practitioners working on fairness, bias measurement, and mitigation, particularly in toxic comment classification."
|
| 246 |
-
],
|
| 247 |
-
"tasks": [
|
| 248 |
-
"Binary toxicity classification (toxic vs. non-toxic)",
|
| 249 |
-
"Analysis of performance across identity subgroups"
|
| 250 |
-
],
|
| 251 |
-
"limitations": "The labeled set of identities is not comprehensive and does not provide universal coverage, representing a balance between coverage, annotator accuracy, and example count. The real-world data is potentially noisier than synthetic alternatives.",
|
| 252 |
-
"out_of_scope_uses": [
|
| 253 |
-
"Developing effective strategies for choosing optimal thresholds to minimize bias"
|
| 254 |
-
]
|
| 255 |
-
},
|
| 256 |
-
"data": {
|
| 257 |
-
"source": "The data consists of online comments sourced from the Civil Comments platform, a commenting plugin for independent English-language news sites. The comments were publicly posted between 2015 and 2017.",
|
| 258 |
-
"size": "The dataset contains approximately 1.8 million comments for training, with separate validation and test sets of approximately 97,320 examples each. All comments were labeled for toxicity, and a subset of 450,000 comments was additionally labeled for identity references.",
|
| 259 |
-
"format": "parquet",
|
| 260 |
-
"annotation": "Labeling was performed by crowd raters. Toxicity labels were applied using guidelines consistent with the Perspective API. For the identity-labeled subset, raters were shown comments and selected referenced identities (e.g., genders, races, ethnicities) from a provided list. Some comments for identity labeling were pre-selected by models to increase the frequency of identity content."
|
| 261 |
-
},
|
| 262 |
-
"methodology": {
|
| 263 |
-
"methods": [
|
| 264 |
-
"Models are evaluated by applying a suite of bias metrics to their predictions on the test set. The original paper demonstrates this using publicly accessible toxicity classifiers on the dataset."
|
| 265 |
-
],
|
| 266 |
-
"metrics": [
|
| 267 |
-
"Subgroup AUC",
|
| 268 |
-
"BPSN AUC",
|
| 269 |
-
"BNSP AUC",
|
| 270 |
-
"Negative Average Equality Gap (AEG)",
|
| 271 |
-
"Positive Average Equality Gap (AEG)"
|
| 272 |
-
],
|
| 273 |
-
"calculation": "The evaluation calculates five metrics for each identity subgroup to provide a multi-faceted view of bias. There is no single aggregated overall score.",
|
| 274 |
-
"interpretation": "For the AUC metrics (Subgroup, BPSN, BNSP), higher values indicate better separability (fewer mis-orderings). For the Average Equality Gaps (Negative and Positive), lower values indicate better separability (more similar score distributions).",
|
| 275 |
-
"baseline_results": "Paper baselines: Results for TOXICITY@1 and TOXICITY@6 from the Perspective API are reported, showing their Subgroup AUC, BPSN AUC, BNSP AUC, Negative AEG, and Positive AEG on a synthetic dataset for the lowest performing 20 subgroups. They are also compared on short comments within the human-labeled dataset for specific identities. EEE results: Anthropic-LM v4-s3 52B scored 0.6100 on the CivilComments metric.",
|
| 276 |
-
"validation": "The evaluation assumes the human-provided labels are reliable. The identity labeling set was designed to balance coverage, crowd rater accuracy, and ensure sufficient examples per identity for meaningful results."
|
| 277 |
-
},
|
| 278 |
-
"ethical_and_legal_considerations": {
|
| 279 |
-
"privacy_and_anonymity": "The paper does not discuss how personally identifiable information (PII) in the online comments was handled or if data was anonymized.",
|
| 280 |
-
"data_licensing": "Creative Commons Zero v1.0 Universal",
|
| 281 |
-
"consent_procedures": "The paper does not describe compensation for crowdworkers or the specific platform used for annotation.",
|
| 282 |
-
"compliance_with_regulations": "The paper does not mention IRB approval, GDPR compliance, or any other ethical review process."
|
| 283 |
-
},
|
| 284 |
-
"possible_risks": [
|
| 285 |
-
{
|
| 286 |
-
"category": "Unrepresentative data",
|
| 287 |
-
"description": [
|
| 288 |
-
"Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
|
| 289 |
-
],
|
| 290 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
|
| 291 |
-
},
|
| 292 |
-
{
|
| 293 |
-
"category": "Uncertain data provenance",
|
| 294 |
-
"description": [
|
| 295 |
-
"Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation."
|
| 296 |
-
],
|
| 297 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html"
|
| 298 |
-
},
|
| 299 |
-
{
|
| 300 |
-
"category": "Data bias",
|
| 301 |
-
"description": [
|
| 302 |
-
"Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
|
| 303 |
-
],
|
| 304 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
|
| 305 |
-
},
|
| 306 |
-
{
|
| 307 |
-
"category": "Lack of data transparency",
|
| 308 |
-
"description": [
|
| 309 |
-
"Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
|
| 310 |
-
],
|
| 311 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
|
| 312 |
-
},
|
| 313 |
-
{
|
| 314 |
-
"category": "Output bias",
|
| 315 |
-
"description": [
|
| 316 |
-
"Generated content might unfairly represent certain groups or individuals."
|
| 317 |
-
],
|
| 318 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/output-bias.html"
|
| 319 |
-
}
|
| 320 |
-
],
|
| 321 |
-
"flagged_fields": {},
|
| 322 |
-
"missing_fields": [],
|
| 323 |
-
"card_info": {
|
| 324 |
-
"created_at": "2026-03-17T12:38:43.250822",
|
| 325 |
-
"llm": "deepseek-ai/DeepSeek-V3.2"
|
| 326 |
-
}
|
| 327 |
-
},
|
| 328 |
-
"HellaSwag": {
|
| 329 |
-
"benchmark_details": {
|
| 330 |
-
"name": "HellaSwag",
|
| 331 |
-
"overview": "HellaSwag is a benchmark designed to measure commonsense natural language inference by testing a model's ability to select the most plausible follow-up event from four multiple-choice options. It is adversarially constructed to be challenging for state-of-the-art models, using a method called Adversarial Filtering to create difficult wrong answers that are obvious to humans but often misclassified by models.",
|
| 332 |
-
"data_type": "text",
|
| 333 |
-
"domains": [
|
| 334 |
-
"commonsense reasoning",
|
| 335 |
-
"natural language inference"
|
| 336 |
-
],
|
| 337 |
-
"languages": [
|
| 338 |
-
"English"
|
| 339 |
-
],
|
| 340 |
-
"similar_benchmarks": [
|
| 341 |
-
"SWAG",
|
| 342 |
-
"SNLI"
|
| 343 |
-
],
|
| 344 |
-
"resources": [
|
| 345 |
-
"https://rowanzellers.com/hellaswag",
|
| 346 |
-
"https://arxiv.org/abs/1905.07830",
|
| 347 |
-
"https://huggingface.co/datasets/Rowan/hellaswag",
|
| 348 |
-
"https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json"
|
| 349 |
-
]
|
| 350 |
-
},
|
| 351 |
-
"purpose_and_intended_users": {
|
| 352 |
-
"goal": "To create a challenge dataset that reveals the difficulty of commonsense inference for state-of-the-art models, demonstrating their lack of robustness and reliance on dataset biases rather than genuine reasoning. It aims to evaluate a model's ability to select the most plausible continuation of a given event description.",
|
| 353 |
-
"audience": [
|
| 354 |
-
"NLP researchers"
|
| 355 |
-
],
|
| 356 |
-
"tasks": [
|
| 357 |
-
"Four-way multiple-choice selection for event continuation",
|
| 358 |
-
"Commonsense inference"
|
| 359 |
-
],
|
| 360 |
-
"limitations": "The adversarial filtering process used to create the dataset, while effective at making it difficult for models, may also select examples where the ground truth answer is not the one preferred by human annotators, necessitating manual filtering to retain the best examples.",
|
| 361 |
-
"out_of_scope_uses": [
|
| 362 |
-
"Not specified"
|
| 363 |
-
]
|
| 364 |
-
},
|
| 365 |
-
"data": {
|
| 366 |
-
"source": "Contexts are sourced from WikiHow instructional articles and ActivityNet video descriptions. Incorrect answer choices are generated by machines and then adversarially filtered.",
|
| 367 |
-
"size": "The dataset contains 70,000 examples in total, with 5,001 in-domain validation examples and 5,000 zero-shot validation examples. The training set comprises 39,905 examples.",
|
| 368 |
-
"format": "Parquet",
|
| 369 |
-
"annotation": "Human crowd workers on Amazon Mechanical Turk validated the endings. They were presented with a context and six endings (one true, five machine-generated) and rated their plausibility. The process involved iterative filtering and replacement of unrealistic endings. Worker quality was ensured via an autograded test and fair pay. A gold standard check by three authors on a random sample showed 90% agreement with crowd annotations."
|
| 370 |
-
},
|
| 371 |
-
"methodology": {
|
| 372 |
-
"methods": [
|
| 373 |
-
"Models are evaluated via fine-tuning on the dataset.",
|
| 374 |
-
"The benchmark also includes zero-shot evaluation on held-out categories."
|
| 375 |
-
],
|
| 376 |
-
"metrics": [
|
| 377 |
-
"HellaSwag accuracy"
|
| 378 |
-
],
|
| 379 |
-
"calculation": "The overall score is the accuracy percentage on the full validation or test sets. Performance is also broken down by subsets, such as in-domain versus zero-shot and by data source.",
|
| 380 |
-
"interpretation": "Higher scores indicate better performance. Human performance is over 95%, which is considered strong. Model performance below 50% is reported, indicating a struggle, with a gap of over 45% from human performance on in-domain data.",
|
| 381 |
-
"baseline_results": "Paper baselines: BERT-Large achieves 47.3% accuracy overall. ESIM + ELMo gets 33.3% accuracy. A BERT-Base model with a frozen encoder and an added LSTM performs 4.3% worse than fine-tuned BERT-Base. Evaluation suite results: Anthropic-LM v4-s3 52B achieves 0.807 (80.7%) accuracy.",
|
| 382 |
-
"validation": "Human validation involved giving five crowd workers the same multiple-choice task and combining their answers via majority vote to establish a human performance baseline. The adversarial filtering process used iterative human ratings to ensure wrong answers were implausible."
|
| 383 |
-
},
|
| 384 |
-
"ethical_and_legal_considerations": {
|
| 385 |
-
"privacy_and_anonymity": "Not specified",
|
| 386 |
-
"data_licensing": "Not specified",
|
| 387 |
-
"consent_procedures": "Crowdworkers on Amazon Mechanical Turk participated voluntarily and were compensated, with pay described as fair. A qualification task was used to filter workers, and those who consistently preferred generated endings over real ones were disqualified.",
|
| 388 |
-
"compliance_with_regulations": "Not specified"
|
| 389 |
-
},
|
| 390 |
-
"possible_risks": [
|
| 391 |
-
{
|
| 392 |
-
"category": "Over- or under-reliance",
|
| 393 |
-
"description": [
|
| 394 |
-
"In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
|
| 395 |
-
],
|
| 396 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
|
| 397 |
-
},
|
| 398 |
-
{
|
| 399 |
-
"category": "Unrepresentative data",
|
| 400 |
-
"description": [
|
| 401 |
-
"Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
|
| 402 |
-
],
|
| 403 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
|
| 404 |
-
},
|
| 405 |
-
{
|
| 406 |
-
"category": "Data bias",
|
| 407 |
-
"description": [
|
| 408 |
-
"Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
|
| 409 |
-
],
|
| 410 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
|
| 411 |
-
},
|
| 412 |
-
{
|
| 413 |
-
"category": "Lack of data transparency",
|
| 414 |
-
"description": [
|
| 415 |
-
"Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
|
| 416 |
-
],
|
| 417 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
|
| 418 |
-
},
|
| 419 |
-
{
|
| 420 |
-
"category": "Improper usage",
|
| 421 |
-
"description": [
|
| 422 |
-
"Improper usage occurs when a model is used for a purpose that it was not originally designed for."
|
| 423 |
-
],
|
| 424 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html"
|
| 425 |
-
}
|
| 426 |
-
],
|
| 427 |
-
"flagged_fields": {
|
| 428 |
-
"baseline_results": "[Possible Hallucination], no supporting evidence found in source material"
|
| 429 |
-
},
|
| 430 |
-
"missing_fields": [
|
| 431 |
-
"purpose_and_intended_users.out_of_scope_uses",
|
| 432 |
-
"ethical_and_legal_considerations.privacy_and_anonymity",
|
| 433 |
-
"ethical_and_legal_considerations.data_licensing",
|
| 434 |
-
"ethical_and_legal_considerations.compliance_with_regulations"
|
| 435 |
-
],
|
| 436 |
-
"card_info": {
|
| 437 |
-
"created_at": "2026-03-17T15:47:07.561060",
|
| 438 |
-
"llm": "deepseek-ai/DeepSeek-V3.2"
|
| 439 |
-
}
|
| 440 |
-
},
|
| 441 |
-
"QuAC": {
|
| 442 |
-
"benchmark_details": {
|
| 443 |
-
"name": "QuAC",
|
| 444 |
-
"overview": "QuAC (Question Answering in Context) is a benchmark that measures a model's ability to answer questions within an information-seeking dialogue. It contains 14,000 dialogues comprising 100,000 question-answer pairs. The dataset is distinctive because questions are often open-ended, context-dependent, unanswerable, or only meaningful within the dialog flow, presenting challenges not found in standard machine comprehension datasets.",
|
| 445 |
-
"data_type": "text",
|
| 446 |
-
"domains": [
|
| 447 |
-
"question answering",
|
| 448 |
-
"dialogue modeling",
|
| 449 |
-
"text generation"
|
| 450 |
-
],
|
| 451 |
-
"languages": [
|
| 452 |
-
"English"
|
| 453 |
-
],
|
| 454 |
-
"similar_benchmarks": [
|
| 455 |
-
"SQuAD"
|
| 456 |
-
],
|
| 457 |
-
"resources": [
|
| 458 |
-
"http://quac.ai",
|
| 459 |
-
"https://arxiv.org/abs/1808.07036",
|
| 460 |
-
"https://huggingface.co/datasets/allenai/quac",
|
| 461 |
-
"https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json"
|
| 462 |
-
]
|
| 463 |
-
},
|
| 464 |
-
"purpose_and_intended_users": {
|
| 465 |
-
"goal": "To enable models to learn from and participate in information-seeking dialog, handling context-dependent, elliptical, and sometimes unanswerable questions.",
|
| 466 |
-
"audience": [
|
| 467 |
-
"Not specified"
|
| 468 |
-
],
|
| 469 |
-
"tasks": [
|
| 470 |
-
"Extractive question answering",
|
| 471 |
-
"Text generation",
|
| 472 |
-
"Fill mask"
|
| 473 |
-
],
|
| 474 |
-
"limitations": "Some questions have lower quality annotations; the dataset filters out the noisiest ~10% of annotations where human F1 is below 40. Questions can be open-ended, unanswerable, or only meaningful within the dialog context, posing inherent challenges.",
|
| 475 |
-
"out_of_scope_uses": [
|
| 476 |
-
"Not specified"
|
| 477 |
-
]
|
| 478 |
-
},
|
| 479 |
-
"data": {
|
| 480 |
-
"source": "The data is crowdsourced via an interactive dialog between two crowd workers: one acting as a student asking questions to learn about a hidden Wikipedia text, and the other acting as a teacher who answers using short excerpts from that text. The source data comes from Wikipedia.",
|
| 481 |
-
"size": "The dataset contains 98,407 question-answer pairs from 13,594 dialogs, based on 8,854 unique sections from 3,611 unique Wikipedia articles. The training set has 83,568 questions (11,567 dialogs), the validation set has 7,354 questions (1,000 dialogs), and the test set has 7,353 questions (1,002 dialogs). The dataset size is between 10,000 and 100,000 examples.",
|
| 482 |
-
"format": "Each dialog is a sequence of question-answer pairs centered around a Wikipedia section. The teacher's response includes a text span, a 'yes/no' indication, a 'no answer' indication, and an encouragement for follow-up questions.",
|
| 483 |
-
"annotation": "Questions are answered by a teacher selecting short excerpts (spans) from the Wikipedia text. The training set has one reference answer per question, while the validation and test sets each have five reference answers per question to improve evaluation reliability. For evaluation, questions with a human F1 score lower than 40 are not used, as manual inspection revealed lower quality below this threshold."
|
| 484 |
-
},
|
| 485 |
-
"methodology": {
|
| 486 |
-
"methods": [
|
| 487 |
-
"Models predict a text span to answer a question about a Wikipedia section, given a dialog history of previous questions and answers.",
|
| 488 |
-
"The evaluation uses a reading comprehension architecture extended to model dialog context."
|
| 489 |
-
],
|
| 490 |
-
"metrics": [
|
| 491 |
-
"Word-level F1"
|
| 492 |
-
],
|
| 493 |
-
"calculation": "Precision and recall are computed over overlapping words after removing stopwords. For 'no answer' questions, F1 is 1 if correctly predicted and 0 otherwise. The maximum F1 among all references is computed for each question.",
|
| 494 |
-
"interpretation": "Higher F1 scores indicate better performance. The best model underperforms humans by 20 F1, indicating significant room for improvement.",
|
| 495 |
-
"baseline_results": "Paper baselines: The best model underperforms humans by 20 F1, but specific model names and scores are not provided. EEE results: Anthropic-LM v4-s3 52B achieves an F1 score of 0.431.",
|
| 496 |
-
"validation": "Quality assurance includes using multiple references for development and test questions, filtering out questions with low human F1 scores, and manual inspection of low-quality annotations."
|
| 497 |
-
},
|
| 498 |
-
"ethical_and_legal_considerations": {
|
| 499 |
-
"privacy_and_anonymity": "Not specified",
|
| 500 |
-
"data_licensing": "MIT License",
|
| 501 |
-
"consent_procedures": "Dialogs were created by two crowd workers, but the specific compensation or platform details are not provided in the paper.",
|
| 502 |
-
"compliance_with_regulations": "Not specified"
|
| 503 |
-
},
|
| 504 |
-
"possible_risks": [
|
| 505 |
-
{
|
| 506 |
-
"category": "Over- or under-reliance",
|
| 507 |
-
"description": [
|
| 508 |
-
"In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
|
| 509 |
-
],
|
| 510 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
|
| 511 |
-
},
|
| 512 |
-
{
|
| 513 |
-
"category": "Unrepresentative data",
|
| 514 |
-
"description": [
|
| 515 |
-
"Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
|
| 516 |
-
],
|
| 517 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
|
| 518 |
-
},
|
| 519 |
-
{
|
| 520 |
-
"category": "Uncertain data provenance",
|
| 521 |
-
"description": [
|
| 522 |
-
"Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation."
|
| 523 |
-
],
|
| 524 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html"
|
| 525 |
-
},
|
| 526 |
-
{
|
| 527 |
-
"category": "Data bias",
|
| 528 |
-
"description": [
|
| 529 |
-
"Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
|
| 530 |
-
],
|
| 531 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
|
| 532 |
-
},
|
| 533 |
-
{
|
| 534 |
-
"category": "Lack of data transparency",
|
| 535 |
-
"description": [
|
| 536 |
-
"Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
|
| 537 |
-
],
|
| 538 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
|
| 539 |
-
}
|
| 540 |
-
],
|
| 541 |
-
"flagged_fields": {},
|
| 542 |
-
"missing_fields": [
|
| 543 |
-
"purpose_and_intended_users.audience",
|
| 544 |
-
"purpose_and_intended_users.out_of_scope_uses",
|
| 545 |
-
"ethical_and_legal_considerations.privacy_and_anonymity",
|
| 546 |
-
"ethical_and_legal_considerations.compliance_with_regulations"
|
| 547 |
-
],
|
| 548 |
-
"card_info": {
|
| 549 |
-
"created_at": "2026-03-17T13:45:24.009083",
|
| 550 |
-
"llm": "deepseek-ai/DeepSeek-V3.2"
|
| 551 |
-
}
|
| 552 |
-
}
|
| 553 |
-
},
|
| 554 |
-
"models": [
|
| 555 |
-
{
|
| 556 |
-
"model_id": "Anthropic-LM-v4-s3-52B",
|
| 557 |
-
"name": "Anthropic-LM v4-s3 52B",
|
| 558 |
-
"developer": "unknown",
|
| 559 |
-
"scores": {
|
| 560 |
-
"Mean win rate": 0.78,
|
| 561 |
-
"MMLU": 0.481,
|
| 562 |
-
"BoolQ": 0.815,
|
| 563 |
-
"NarrativeQA": 0.728,
|
| 564 |
-
"NaturalQuestions (open-book)": 0.686,
|
| 565 |
-
"QuAC": 0.431,
|
| 566 |
-
"HellaSwag": 0.807,
|
| 567 |
-
"OpenbookQA": 0.558,
|
| 568 |
-
"TruthfulQA": 0.368,
|
| 569 |
-
"MS MARCO (TREC)": -1,
|
| 570 |
-
"CNN/DailyMail": 0.154,
|
| 571 |
-
"XSUM": 0.134,
|
| 572 |
-
"IMDB": 0.934,
|
| 573 |
-
"CivilComments": 0.61,
|
| 574 |
-
"RAFT": 0.699
|
| 575 |
-
}
|
| 576 |
-
},
|
| 577 |
-
{
|
| 578 |
-
"model_id": "EleutherAI/pythia-12b",
|
| 579 |
-
"name": "Pythia 12B",
|
| 580 |
-
"developer": "EleutherAI",
|
| 581 |
-
"scores": {
|
| 582 |
-
"Mean win rate": 0.257,
|
| 583 |
-
"MMLU": 0.274,
|
| 584 |
-
"BoolQ": 0.662,
|
| 585 |
-
"NarrativeQA": 0.596,
|
| 586 |
-
"NaturalQuestions (open-book)": 0.581,
|
| 587 |
-
"QuAC": 0.313,
|
| 588 |
-
"HellaSwag": -1,
|
| 589 |
-
"OpenbookQA": -1,
|
| 590 |
-
"TruthfulQA": 0.177,
|
| 591 |
-
"MS MARCO (TREC)": -1,
|
| 592 |
-
"CNN/DailyMail": -1,
|
| 593 |
-
"XSUM": -1,
|
| 594 |
-
"IMDB": 0.931,
|
| 595 |
-
"CivilComments": 0.531,
|
| 596 |
-
"RAFT": 0.514
|
| 597 |
-
}
|
| 598 |
-
},
|
| 599 |
-
{
|
| 600 |
-
"model_id": "EleutherAI/pythia-6.9b",
|
| 601 |
-
"name": "Pythia 6.9B",
|
| 602 |
-
"developer": "EleutherAI",
|
| 603 |
-
"scores": {
|
| 604 |
-
"Mean win rate": 0.196,
|
| 605 |
-
"MMLU": 0.236,
|
| 606 |
-
"BoolQ": 0.631,
|
| 607 |
-
"NarrativeQA": 0.528,
|
| 608 |
-
"NaturalQuestions (open-book)": 0.539,
|
| 609 |
-
"QuAC": 0.296,
|
| 610 |
-
"HellaSwag": -1,
|
| 611 |
-
"OpenbookQA": -1,
|
| 612 |
-
"TruthfulQA": 0.213,
|
| 613 |
-
"MS MARCO (TREC)": -1,
|
| 614 |
-
"CNN/DailyMail": -1,
|
| 615 |
-
"XSUM": -1,
|
| 616 |
-
"IMDB": 0.928,
|
| 617 |
-
"CivilComments": 0.511,
|
| 618 |
-
"RAFT": 0.502
|
| 619 |
-
}
|
| 620 |
-
},
|
| 621 |
-
{
|
| 622 |
-
"model_id": "ai21/J1-Grande-v1-17B",
|
| 623 |
-
"name": "J1-Grande v1 17B",
|
| 624 |
-
"developer": "ai21",
|
| 625 |
-
"scores": {
|
| 626 |
-
"Mean win rate": 0.433,
|
| 627 |
-
"MMLU": 0.27,
|
| 628 |
-
"BoolQ": 0.722,
|
| 629 |
-
"NarrativeQA": 0.672,
|
| 630 |
-
"NaturalQuestions (open-book)": 0.578,
|
| 631 |
-
"QuAC": 0.362,
|
| 632 |
-
"HellaSwag": 0.739,
|
| 633 |
-
"OpenbookQA": 0.52,
|
| 634 |
-
"TruthfulQA": 0.193,
|
| 635 |
-
"MS MARCO (TREC)": 0.341,
|
| 636 |
-
"CNN/DailyMail": 0.143,
|
| 637 |
-
"XSUM": 0.122,
|
| 638 |
-
"IMDB": 0.953,
|
| 639 |
-
"CivilComments": 0.529,
|
| 640 |
-
"RAFT": 0.658
|
| 641 |
-
}
|
| 642 |
-
},
|
| 643 |
-
{
|
| 644 |
-
"model_id": "ai21/J1-Grande-v2-beta-17B",
|
| 645 |
-
"name": "J1-Grande v2 beta 17B",
|
| 646 |
-
"developer": "ai21",
|
| 647 |
-
"scores": {
|
| 648 |
-
"Mean win rate": 0.706,
|
| 649 |
-
"MMLU": 0.445,
|
| 650 |
-
"BoolQ": 0.812,
|
| 651 |
-
"NarrativeQA": 0.725,
|
| 652 |
-
"NaturalQuestions (open-book)": 0.625,
|
| 653 |
-
"QuAC": 0.392,
|
| 654 |
-
"HellaSwag": 0.764,
|
| 655 |
-
"OpenbookQA": 0.56,
|
| 656 |
-
"TruthfulQA": 0.306,
|
| 657 |
-
"MS MARCO (TREC)": 0.46,
|
| 658 |
-
"CNN/DailyMail": 0.146,
|
| 659 |
-
"XSUM": 0.152,
|
| 660 |
-
"IMDB": 0.957,
|
| 661 |
-
"CivilComments": 0.546,
|
| 662 |
-
"RAFT": 0.679
|
| 663 |
-
}
|
| 664 |
-
},
|
| 665 |
-
{
|
| 666 |
-
"model_id": "ai21/J1-Jumbo-v1-178B",
|
| 667 |
-
"name": "J1-Jumbo v1 178B",
|
| 668 |
-
"developer": "ai21",
|
| 669 |
-
"scores": {
|
| 670 |
-
"Mean win rate": 0.517,
|
| 671 |
-
"MMLU": 0.259,
|
| 672 |
-
"BoolQ": 0.776,
|
| 673 |
-
"NarrativeQA": 0.695,
|
| 674 |
-
"NaturalQuestions (open-book)": 0.595,
|
| 675 |
-
"QuAC": 0.358,
|
| 676 |
-
"HellaSwag": 0.765,
|
| 677 |
-
"OpenbookQA": 0.534,
|
| 678 |
-
"TruthfulQA": 0.175,
|
| 679 |
-
"MS MARCO (TREC)": 0.363,
|
| 680 |
-
"CNN/DailyMail": 0.144,
|
| 681 |
-
"XSUM": 0.129,
|
| 682 |
-
"IMDB": 0.943,
|
| 683 |
-
"CivilComments": 0.553,
|
| 684 |
-
"RAFT": 0.681
|
| 685 |
-
}
|
| 686 |
-
},
|
| 687 |
-
{
|
| 688 |
-
"model_id": "ai21/J1-Large-v1-7.5B",
|
| 689 |
-
"name": "J1-Large v1 7.5B",
|
| 690 |
-
"developer": "ai21",
|
| 691 |
-
"scores": {
|
| 692 |
-
"Mean win rate": 0.285,
|
| 693 |
-
"MMLU": 0.241,
|
| 694 |
-
"BoolQ": 0.683,
|
| 695 |
-
"NarrativeQA": 0.623,
|
| 696 |
-
"NaturalQuestions (open-book)": 0.532,
|
| 697 |
-
"QuAC": 0.328,
|
| 698 |
-
"HellaSwag": 0.7,
|
| 699 |
-
"OpenbookQA": 0.514,
|
| 700 |
-
"TruthfulQA": 0.197,
|
| 701 |
-
"MS MARCO (TREC)": 0.292,
|
| 702 |
-
"CNN/DailyMail": 0.134,
|
| 703 |
-
"XSUM": 0.102,
|
| 704 |
-
"IMDB": 0.956,
|
| 705 |
-
"CivilComments": 0.532,
|
| 706 |
-
"RAFT": 0.545
|
| 707 |
-
}
|
| 708 |
-
},
|
| 709 |
-
{
|
| 710 |
-
"model_id": "ai21/Jurassic-2-Grande-17B",
|
| 711 |
-
"name": "Jurassic-2 Grande 17B",
|
| 712 |
-
"developer": "ai21",
|
| 713 |
-
"scores": {
|
| 714 |
-
"Mean win rate": 0.743,
|
| 715 |
-
"MMLU": 0.475,
|
| 716 |
-
"BoolQ": 0.826,
|
| 717 |
-
"NarrativeQA": 0.737,
|
| 718 |
-
"NaturalQuestions (open-book)": 0.639,
|
| 719 |
-
"QuAC": 0.418,
|
| 720 |
-
"HellaSwag": 0.781,
|
| 721 |
-
"OpenbookQA": 0.542,
|
| 722 |
-
"TruthfulQA": 0.348,
|
| 723 |
-
"MS MARCO (TREC)": 0.514,
|
| 724 |
-
"CNN/DailyMail": 0.144,
|
| 725 |
-
"XSUM": 0.167,
|
| 726 |
-
"IMDB": 0.938,
|
| 727 |
-
"CivilComments": 0.547,
|
| 728 |
-
"RAFT": 0.712
|
| 729 |
-
}
|
| 730 |
-
},
|
| 731 |
-
{
|
| 732 |
-
"model_id": "ai21/Jurassic-2-Jumbo-178B",
|
| 733 |
-
"name": "Jurassic-2 Jumbo 178B",
|
| 734 |
-
"developer": "ai21",
|
| 735 |
-
"scores": {
|
| 736 |
-
"Mean win rate": 0.824,
|
| 737 |
-
"MMLU": 0.48,
|
| 738 |
-
"BoolQ": 0.829,
|
| 739 |
-
"NarrativeQA": 0.733,
|
| 740 |
-
"NaturalQuestions (open-book)": 0.669,
|
| 741 |
-
"QuAC": 0.435,
|
| 742 |
-
"HellaSwag": 0.788,
|
| 743 |
-
"OpenbookQA": 0.558,
|
| 744 |
-
"TruthfulQA": 0.437,
|
| 745 |
-
"MS MARCO (TREC)": 0.661,
|
| 746 |
-
"CNN/DailyMail": 0.149,
|
| 747 |
-
"XSUM": 0.182,
|
| 748 |
-
"IMDB": 0.938,
|
| 749 |
-
"CivilComments": 0.57,
|
| 750 |
-
"RAFT": 0.746
|
| 751 |
-
}
|
| 752 |
-
},
|
| 753 |
-
{
|
| 754 |
-
"model_id": "ai21/Jurassic-2-Large-7.5B",
|
| 755 |
-
"name": "Jurassic-2 Large 7.5B",
|
| 756 |
-
"developer": "ai21",
|
| 757 |
-
"scores": {
|
| 758 |
-
"Mean win rate": 0.553,
|
| 759 |
-
"MMLU": 0.339,
|
| 760 |
-
"BoolQ": 0.742,
|
| 761 |
-
"NarrativeQA": -1,
|
| 762 |
-
"NaturalQuestions (open-book)": 0.589,
|
| 763 |
-
"QuAC": -1,
|
| 764 |
-
"HellaSwag": 0.729,
|
| 765 |
-
"OpenbookQA": 0.53,
|
| 766 |
-
"TruthfulQA": 0.245,
|
| 767 |
-
"MS MARCO (TREC)": 0.464,
|
| 768 |
-
"CNN/DailyMail": 0.136,
|
| 769 |
-
"XSUM": 0.142,
|
| 770 |
-
"IMDB": 0.956,
|
| 771 |
-
"CivilComments": 0.57,
|
| 772 |
-
"RAFT": 0.622
|
| 773 |
-
}
|
| 774 |
-
},
|
| 775 |
-
{
|
| 776 |
-
"model_id": "aleph-alpha/Luminous-Base-13B",
|
| 777 |
-
"name": "Luminous Base 13B",
|
| 778 |
-
"developer": "aleph-alpha",
|
| 779 |
-
"scores": {
|
| 780 |
-
"Mean win rate": 0.315,
|
| 781 |
-
"MMLU": 0.27,
|
| 782 |
-
"BoolQ": 0.719,
|
| 783 |
-
"NarrativeQA": 0.605,
|
| 784 |
-
"NaturalQuestions (open-book)": 0.568,
|
| 785 |
-
"QuAC": 0.334,
|
| 786 |
-
"HellaSwag": -1,
|
| 787 |
-
"OpenbookQA": -1,
|
| 788 |
-
"TruthfulQA": 0.182,
|
| 789 |
-
"MS MARCO (TREC)": -1,
|
| 790 |
-
"CNN/DailyMail": 0.11,
|
| 791 |
-
"XSUM": 0.105,
|
| 792 |
-
"IMDB": 0.939,
|
| 793 |
-
"CivilComments": 0.544,
|
| 794 |
-
"RAFT": 0.473
|
| 795 |
-
}
|
| 796 |
-
},
|
| 797 |
-
{
|
| 798 |
-
"model_id": "aleph-alpha/Luminous-Extended-30B",
|
| 799 |
-
"name": "Luminous Extended 30B",
|
| 800 |
-
"developer": "aleph-alpha",
|
| 801 |
-
"scores": {
|
| 802 |
-
"Mean win rate": 0.485,
|
| 803 |
-
"MMLU": 0.321,
|
| 804 |
-
"BoolQ": 0.767,
|
| 805 |
-
"NarrativeQA": 0.665,
|
| 806 |
-
"NaturalQuestions (open-book)": 0.609,
|
| 807 |
-
"QuAC": 0.349,
|
| 808 |
-
"HellaSwag": -1,
|
| 809 |
-
"OpenbookQA": -1,
|
| 810 |
-
"TruthfulQA": 0.221,
|
| 811 |
-
"MS MARCO (TREC)": -1,
|
| 812 |
-
"CNN/DailyMail": 0.139,
|
| 813 |
-
"XSUM": 0.124,
|
| 814 |
-
"IMDB": 0.947,
|
| 815 |
-
"CivilComments": 0.524,
|
| 816 |
-
"RAFT": 0.523
|
| 817 |
-
}
|
| 818 |
-
},
|
| 819 |
-
{
|
| 820 |
-
"model_id": "aleph-alpha/Luminous-Supreme-70B",
|
| 821 |
-
"name": "Luminous Supreme 70B",
|
| 822 |
-
"developer": "aleph-alpha",
|
| 823 |
-
"scores": {
|
| 824 |
-
"Mean win rate": 0.662,
|
| 825 |
-
"MMLU": 0.38,
|
| 826 |
-
"BoolQ": 0.775,
|
| 827 |
-
"NarrativeQA": 0.711,
|
| 828 |
-
"NaturalQuestions (open-book)": 0.649,
|
| 829 |
-
"QuAC": 0.37,
|
| 830 |
-
"HellaSwag": -1,
|
| 831 |
-
"OpenbookQA": -1,
|
| 832 |
-
"TruthfulQA": 0.222,
|
| 833 |
-
"MS MARCO (TREC)": -1,
|
| 834 |
-
"CNN/DailyMail": 0.15,
|
| 835 |
-
"XSUM": 0.136,
|
| 836 |
-
"IMDB": 0.959,
|
| 837 |
-
"CivilComments": 0.562,
|
| 838 |
-
"RAFT": 0.653
|
| 839 |
-
}
|
| 840 |
-
},
|
| 841 |
-
{
|
| 842 |
-
"model_id": "bigscience/BLOOM-176B",
|
| 843 |
-
"name": "BLOOM 176B",
|
| 844 |
-
"developer": "bigscience",
|
| 845 |
-
"scores": {
|
| 846 |
-
"Mean win rate": 0.446,
|
| 847 |
-
"MMLU": 0.299,
|
| 848 |
-
"BoolQ": 0.704,
|
| 849 |
-
"NarrativeQA": 0.662,
|
| 850 |
-
"NaturalQuestions (open-book)": 0.621,
|
| 851 |
-
"QuAC": 0.361,
|
| 852 |
-
"HellaSwag": 0.744,
|
| 853 |
-
"OpenbookQA": 0.534,
|
| 854 |
-
"TruthfulQA": 0.205,
|
| 855 |
-
"MS MARCO (TREC)": 0.386,
|
| 856 |
-
"CNN/DailyMail": 0.08,
|
| 857 |
-
"XSUM": 0.03,
|
| 858 |
-
"IMDB": 0.945,
|
| 859 |
-
"CivilComments": 0.62,
|
| 860 |
-
"RAFT": 0.592
|
| 861 |
-
}
|
| 862 |
-
},
|
| 863 |
-
{
|
| 864 |
-
"model_id": "bigscience/T0pp-11B",
|
| 865 |
-
"name": "T0pp 11B",
|
| 866 |
-
"developer": "bigscience",
|
| 867 |
-
"scores": {
|
| 868 |
-
"Mean win rate": 0.197,
|
| 869 |
-
"MMLU": 0.407,
|
| 870 |
-
"BoolQ": 0,
|
| 871 |
-
"NarrativeQA": 0.151,
|
| 872 |
-
"NaturalQuestions (open-book)": 0.19,
|
| 873 |
-
"QuAC": 0.121,
|
| 874 |
-
"HellaSwag": -1,
|
| 875 |
-
"OpenbookQA": -1,
|
| 876 |
-
"TruthfulQA": 0.377,
|
| 877 |
-
"MS MARCO (TREC)": -1,
|
| 878 |
-
"CNN/DailyMail": 0.122,
|
| 879 |
-
"XSUM": 0.09,
|
| 880 |
-
"IMDB": 0.207,
|
| 881 |
-
"CivilComments": 0.234,
|
| 882 |
-
"RAFT": 0.118
|
| 883 |
-
}
|
| 884 |
-
},
|
| 885 |
-
{
|
| 886 |
-
"model_id": "cohere/Cohere-Command-beta-52.4B",
|
| 887 |
-
"name": "Cohere Command beta 52.4B",
|
| 888 |
-
"developer": "cohere",
|
| 889 |
-
"scores": {
|
| 890 |
-
"Mean win rate": 0.874,
|
| 891 |
-
"MMLU": 0.452,
|
| 892 |
-
"BoolQ": 0.856,
|
| 893 |
-
"NarrativeQA": 0.752,
|
| 894 |
-
"NaturalQuestions (open-book)": 0.76,
|
| 895 |
-
"QuAC": 0.432,
|
| 896 |
-
"HellaSwag": 0.811,
|
| 897 |
-
"OpenbookQA": 0.582,
|
| 898 |
-
"TruthfulQA": 0.269,
|
| 899 |
-
"MS MARCO (TREC)": 0.762,
|
| 900 |
-
"CNN/DailyMail": 0.161,
|
| 901 |
-
"XSUM": 0.152,
|
| 902 |
-
"IMDB": 0.96,
|
| 903 |
-
"CivilComments": 0.601,
|
| 904 |
-
"RAFT": 0.667
|
| 905 |
-
}
|
| 906 |
-
},
|
| 907 |
-
{
|
| 908 |
-
"model_id": "cohere/Cohere-Command-beta-6.1B",
|
| 909 |
-
"name": "Cohere Command beta 6.1B",
|
| 910 |
-
"developer": "cohere",
|
| 911 |
-
"scores": {
|
| 912 |
-
"Mean win rate": 0.675,
|
| 913 |
-
"MMLU": 0.406,
|
| 914 |
-
"BoolQ": 0.798,
|
| 915 |
-
"NarrativeQA": 0.709,
|
| 916 |
-
"NaturalQuestions (open-book)": 0.717,
|
| 917 |
-
"QuAC": 0.375,
|
| 918 |
-
"HellaSwag": 0.752,
|
| 919 |
-
"OpenbookQA": 0.55,
|
| 920 |
-
"TruthfulQA": 0.203,
|
| 921 |
-
"MS MARCO (TREC)": 0.709,
|
| 922 |
-
"CNN/DailyMail": 0.153,
|
| 923 |
-
"XSUM": 0.122,
|
| 924 |
-
"IMDB": 0.961,
|
| 925 |
-
"CivilComments": 0.54,
|
| 926 |
-
"RAFT": 0.634
|
| 927 |
-
}
|
| 928 |
-
},
|
| 929 |
-
{
|
| 930 |
-
"model_id": "cohere/Cohere-large-v20220720-13.1B",
|
| 931 |
-
"name": "Cohere large v20220720 13.1B",
|
| 932 |
-
"developer": "cohere",
|
| 933 |
-
"scores": {
|
| 934 |
-
"Mean win rate": 0.372,
|
| 935 |
-
"MMLU": 0.324,
|
| 936 |
-
"BoolQ": 0.725,
|
| 937 |
-
"NarrativeQA": 0.625,
|
| 938 |
-
"NaturalQuestions (open-book)": 0.573,
|
| 939 |
-
"QuAC": 0.338,
|
| 940 |
-
"HellaSwag": 0.736,
|
| 941 |
-
"OpenbookQA": 0.542,
|
| 942 |
-
"TruthfulQA": 0.181,
|
| 943 |
-
"MS MARCO (TREC)": 0.33,
|
| 944 |
-
"CNN/DailyMail": 0.126,
|
| 945 |
-
"XSUM": 0.108,
|
| 946 |
-
"IMDB": 0.933,
|
| 947 |
-
"CivilComments": 0.507,
|
| 948 |
-
"RAFT": 0.596
|
| 949 |
-
}
|
| 950 |
-
},
|
| 951 |
-
{
|
| 952 |
-
"model_id": "cohere/Cohere-medium-v20220720-6.1B",
|
| 953 |
-
"name": "Cohere medium v20220720 6.1B",
|
| 954 |
-
"developer": "cohere",
|
| 955 |
-
"scores": {
|
| 956 |
-
"Mean win rate": 0.23,
|
| 957 |
-
"MMLU": 0.279,
|
| 958 |
-
"BoolQ": 0.659,
|
| 959 |
-
"NarrativeQA": 0.559,
|
| 960 |
-
"NaturalQuestions (open-book)": 0.504,
|
| 961 |
-
"QuAC": 0.279,
|
| 962 |
-
"HellaSwag": 0.706,
|
| 963 |
-
"OpenbookQA": 0.496,
|
| 964 |
-
"TruthfulQA": 0.19,
|
| 965 |
-
"MS MARCO (TREC)": 0.374,
|
| 966 |
-
"CNN/DailyMail": 0.077,
|
| 967 |
-
"XSUM": 0.087,
|
| 968 |
-
"IMDB": 0.935,
|
| 969 |
-
"CivilComments": 0.504,
|
| 970 |
-
"RAFT": 0.52
|
| 971 |
-
}
|
| 972 |
-
},
|
| 973 |
-
{
|
| 974 |
-
"model_id": "cohere/Cohere-medium-v20221108-6.1B",
|
| 975 |
-
"name": "Cohere medium v20221108 6.1B",
|
| 976 |
-
"developer": "cohere",
|
| 977 |
-
"scores": {
|
| 978 |
-
"Mean win rate": 0.312,
|
| 979 |
-
"MMLU": 0.254,
|
| 980 |
-
"BoolQ": 0.7,
|
| 981 |
-
"NarrativeQA": 0.61,
|
| 982 |
-
"NaturalQuestions (open-book)": 0.517,
|
| 983 |
-
"QuAC": 0.314,
|
| 984 |
-
"HellaSwag": 0.726,
|
| 985 |
-
"OpenbookQA": 0.538,
|
| 986 |
-
"TruthfulQA": 0.215,
|
| 987 |
-
"MS MARCO (TREC)": 0.373,
|
| 988 |
-
"CNN/DailyMail": 0.121,
|
| 989 |
-
"XSUM": 0.099,
|
| 990 |
-
"IMDB": 0.935,
|
| 991 |
-
"CivilComments": 0.5,
|
| 992 |
-
"RAFT": 0.591
|
| 993 |
-
}
|
| 994 |
-
},
|
| 995 |
-
{
|
| 996 |
-
"model_id": "cohere/Cohere-small-v20220720-410M",
|
| 997 |
-
"name": "Cohere small v20220720 410M",
|
| 998 |
-
"developer": "cohere",
|
| 999 |
-
"scores": {
|
| 1000 |
-
"Mean win rate": 0.109,
|
| 1001 |
-
"MMLU": 0.264,
|
| 1002 |
-
"BoolQ": 0.457,
|
| 1003 |
-
"NarrativeQA": 0.294,
|
| 1004 |
-
"NaturalQuestions (open-book)": 0.309,
|
| 1005 |
-
"QuAC": 0.219,
|
| 1006 |
-
"HellaSwag": 0.483,
|
| 1007 |
-
"OpenbookQA": 0.348,
|
| 1008 |
-
"TruthfulQA": 0.217,
|
| 1009 |
-
"MS MARCO (TREC)": 0.304,
|
| 1010 |
-
"CNN/DailyMail": 0.063,
|
| 1011 |
-
"XSUM": 0.033,
|
| 1012 |
-
"IMDB": 0.578,
|
| 1013 |
-
"CivilComments": 0.501,
|
| 1014 |
-
"RAFT": 0.492
|
| 1015 |
-
}
|
| 1016 |
-
},
|
| 1017 |
-
{
|
| 1018 |
-
"model_id": "cohere/Cohere-xlarge-v20220609-52.4B",
|
| 1019 |
-
"name": "Cohere xlarge v20220609 52.4B",
|
| 1020 |
-
"developer": "cohere",
|
| 1021 |
-
"scores": {
|
| 1022 |
-
"Mean win rate": 0.56,
|
| 1023 |
-
"MMLU": 0.353,
|
| 1024 |
-
"BoolQ": 0.718,
|
| 1025 |
-
"NarrativeQA": 0.65,
|
| 1026 |
-
"NaturalQuestions (open-book)": 0.595,
|
| 1027 |
-
"QuAC": 0.361,
|
| 1028 |
-
"HellaSwag": 0.811,
|
| 1029 |
-
"OpenbookQA": 0.55,
|
| 1030 |
-
"TruthfulQA": 0.198,
|
| 1031 |
-
"MS MARCO (TREC)": 0.459,
|
| 1032 |
-
"CNN/DailyMail": 0.144,
|
| 1033 |
-
"XSUM": 0.129,
|
| 1034 |
-
"IMDB": 0.956,
|
| 1035 |
-
"CivilComments": 0.532,
|
| 1036 |
-
"RAFT": 0.633
|
| 1037 |
-
}
|
| 1038 |
-
},
|
| 1039 |
-
{
|
| 1040 |
-
"model_id": "cohere/Cohere-xlarge-v20221108-52.4B",
|
| 1041 |
-
"name": "Cohere xlarge v20221108 52.4B",
|
| 1042 |
-
"developer": "cohere",
|
| 1043 |
-
"scores": {
|
| 1044 |
-
"Mean win rate": 0.664,
|
| 1045 |
-
"MMLU": 0.382,
|
| 1046 |
-
"BoolQ": 0.762,
|
| 1047 |
-
"NarrativeQA": 0.672,
|
| 1048 |
-
"NaturalQuestions (open-book)": 0.628,
|
| 1049 |
-
"QuAC": 0.374,
|
| 1050 |
-
"HellaSwag": 0.81,
|
| 1051 |
-
"OpenbookQA": 0.588,
|
| 1052 |
-
"TruthfulQA": 0.169,
|
| 1053 |
-
"MS MARCO (TREC)": 0.55,
|
| 1054 |
-
"CNN/DailyMail": 0.153,
|
| 1055 |
-
"XSUM": 0.153,
|
| 1056 |
-
"IMDB": 0.956,
|
| 1057 |
-
"CivilComments": 0.524,
|
| 1058 |
-
"RAFT": 0.624
|
| 1059 |
-
}
|
| 1060 |
-
},
|
| 1061 |
-
{
|
| 1062 |
-
"model_id": "google/Palmyra-X-43B",
|
| 1063 |
-
"name": "Palmyra X 43B",
|
| 1064 |
-
"developer": "Google",
|
| 1065 |
-
"scores": {
|
| 1066 |
-
"Mean win rate": 0.732,
|
| 1067 |
-
"MMLU": 0.609,
|
| 1068 |
-
"BoolQ": 0.896,
|
| 1069 |
-
"NarrativeQA": 0.742,
|
| 1070 |
-
"NaturalQuestions (open-book)": -1,
|
| 1071 |
-
"QuAC": 0.473,
|
| 1072 |
-
"HellaSwag": -1,
|
| 1073 |
-
"OpenbookQA": -1,
|
| 1074 |
-
"TruthfulQA": 0.616,
|
| 1075 |
-
"MS MARCO (TREC)": -1,
|
| 1076 |
-
"CNN/DailyMail": 0.049,
|
| 1077 |
-
"XSUM": 0.149,
|
| 1078 |
-
"IMDB": 0.935,
|
| 1079 |
-
"CivilComments": 0.008,
|
| 1080 |
-
"RAFT": 0.701
|
| 1081 |
-
}
|
| 1082 |
-
},
|
| 1083 |
-
{
|
| 1084 |
-
"model_id": "google/T5-11B",
|
| 1085 |
-
"name": "T5 11B",
|
| 1086 |
-
"developer": "Google",
|
| 1087 |
-
"scores": {
|
| 1088 |
-
"Mean win rate": 0.131,
|
| 1089 |
-
"MMLU": 0.29,
|
| 1090 |
-
"BoolQ": 0.761,
|
| 1091 |
-
"NarrativeQA": 0.086,
|
| 1092 |
-
"NaturalQuestions (open-book)": 0.477,
|
| 1093 |
-
"QuAC": 0.116,
|
| 1094 |
-
"HellaSwag": -1,
|
| 1095 |
-
"OpenbookQA": -1,
|
| 1096 |
-
"TruthfulQA": 0.133,
|
| 1097 |
-
"MS MARCO (TREC)": -1,
|
| 1098 |
-
"CNN/DailyMail": 0.043,
|
| 1099 |
-
"XSUM": 0.015,
|
| 1100 |
-
"IMDB": 0.379,
|
| 1101 |
-
"CivilComments": 0.509,
|
| 1102 |
-
"RAFT": 0.37
|
| 1103 |
-
}
|
| 1104 |
-
},
|
| 1105 |
-
{
|
| 1106 |
-
"model_id": "google/UL2-20B",
|
| 1107 |
-
"name": "UL2 20B",
|
| 1108 |
-
"developer": "Google",
|
| 1109 |
-
"scores": {
|
| 1110 |
-
"Mean win rate": 0.167,
|
| 1111 |
-
"MMLU": 0.291,
|
| 1112 |
-
"BoolQ": 0.746,
|
| 1113 |
-
"NarrativeQA": 0.083,
|
| 1114 |
-
"NaturalQuestions (open-book)": 0.349,
|
| 1115 |
-
"QuAC": 0.144,
|
| 1116 |
-
"HellaSwag": -1,
|
| 1117 |
-
"OpenbookQA": -1,
|
| 1118 |
-
"TruthfulQA": 0.193,
|
| 1119 |
-
"MS MARCO (TREC)": -1,
|
| 1120 |
-
"CNN/DailyMail": 0.03,
|
| 1121 |
-
"XSUM": 0.058,
|
| 1122 |
-
"IMDB": 0.337,
|
| 1123 |
-
"CivilComments": 0.521,
|
| 1124 |
-
"RAFT": 0.404
|
| 1125 |
-
}
|
| 1126 |
-
},
|
| 1127 |
-
{
|
| 1128 |
-
"model_id": "lmsys/Vicuna-v1.3-13B",
|
| 1129 |
-
"name": "Vicuna v1.3 13B",
|
| 1130 |
-
"developer": "lmsys",
|
| 1131 |
-
"scores": {
|
| 1132 |
-
"Mean win rate": 0.706,
|
| 1133 |
-
"MMLU": 0.462,
|
| 1134 |
-
"BoolQ": 0.808,
|
| 1135 |
-
"NarrativeQA": 0.691,
|
| 1136 |
-
"NaturalQuestions (open-book)": 0.686,
|
| 1137 |
-
"QuAC": 0.403,
|
| 1138 |
-
"HellaSwag": -1,
|
| 1139 |
-
"OpenbookQA": -1,
|
| 1140 |
-
"TruthfulQA": 0.385,
|
| 1141 |
-
"MS MARCO (TREC)": -1,
|
| 1142 |
-
"CNN/DailyMail": -1,
|
| 1143 |
-
"XSUM": -1,
|
| 1144 |
-
"IMDB": 0.762,
|
| 1145 |
-
"CivilComments": 0.645,
|
| 1146 |
-
"RAFT": 0.657
|
| 1147 |
-
}
|
| 1148 |
-
},
|
| 1149 |
-
{
|
| 1150 |
-
"model_id": "lmsys/Vicuna-v1.3-7B",
|
| 1151 |
-
"name": "Vicuna v1.3 7B",
|
| 1152 |
-
"developer": "lmsys",
|
| 1153 |
-
"scores": {
|
| 1154 |
-
"Mean win rate": 0.625,
|
| 1155 |
-
"MMLU": 0.434,
|
| 1156 |
-
"BoolQ": 0.76,
|
| 1157 |
-
"NarrativeQA": 0.643,
|
| 1158 |
-
"NaturalQuestions (open-book)": 0.634,
|
| 1159 |
-
"QuAC": 0.392,
|
| 1160 |
-
"HellaSwag": -1,
|
| 1161 |
-
"OpenbookQA": -1,
|
| 1162 |
-
"TruthfulQA": 0.292,
|
| 1163 |
-
"MS MARCO (TREC)": -1,
|
| 1164 |
-
"CNN/DailyMail": -1,
|
| 1165 |
-
"XSUM": -1,
|
| 1166 |
-
"IMDB": 0.916,
|
| 1167 |
-
"CivilComments": 0.62,
|
| 1168 |
-
"RAFT": 0.693
|
| 1169 |
-
}
|
| 1170 |
-
},
|
| 1171 |
-
{
|
| 1172 |
-
"model_id": "meta/LLaMA-13B",
|
| 1173 |
-
"name": "LLaMA 13B",
|
| 1174 |
-
"developer": "Meta",
|
| 1175 |
-
"scores": {
|
| 1176 |
-
"Mean win rate": 0.595,
|
| 1177 |
-
"MMLU": 0.422,
|
| 1178 |
-
"BoolQ": 0.714,
|
| 1179 |
-
"NarrativeQA": 0.711,
|
| 1180 |
-
"NaturalQuestions (open-book)": 0.614,
|
| 1181 |
-
"QuAC": 0.347,
|
| 1182 |
-
"HellaSwag": -1,
|
| 1183 |
-
"OpenbookQA": -1,
|
| 1184 |
-
"TruthfulQA": 0.324,
|
| 1185 |
-
"MS MARCO (TREC)": -1,
|
| 1186 |
-
"CNN/DailyMail": -1,
|
| 1187 |
-
"XSUM": -1,
|
| 1188 |
-
"IMDB": 0.928,
|
| 1189 |
-
"CivilComments": 0.6,
|
| 1190 |
-
"RAFT": 0.643
|
| 1191 |
-
}
|
| 1192 |
-
},
|
| 1193 |
-
{
|
| 1194 |
-
"model_id": "meta/LLaMA-30B",
|
| 1195 |
-
"name": "LLaMA 30B",
|
| 1196 |
-
"developer": "Meta",
|
| 1197 |
-
"scores": {
|
| 1198 |
-
"Mean win rate": 0.781,
|
| 1199 |
-
"MMLU": 0.531,
|
| 1200 |
-
"BoolQ": 0.861,
|
| 1201 |
-
"NarrativeQA": 0.752,
|
| 1202 |
-
"NaturalQuestions (open-book)": 0.666,
|
| 1203 |
-
"QuAC": 0.39,
|
| 1204 |
-
"HellaSwag": -1,
|
| 1205 |
-
"OpenbookQA": -1,
|
| 1206 |
-
"TruthfulQA": 0.344,
|
| 1207 |
-
"MS MARCO (TREC)": -1,
|
| 1208 |
-
"CNN/DailyMail": -1,
|
| 1209 |
-
"XSUM": -1,
|
| 1210 |
-
"IMDB": 0.927,
|
| 1211 |
-
"CivilComments": 0.549,
|
| 1212 |
-
"RAFT": 0.752
|
| 1213 |
-
}
|
| 1214 |
-
},
|
| 1215 |
-
{
|
| 1216 |
-
"model_id": "meta/LLaMA-65B",
|
| 1217 |
-
"name": "LLaMA 65B",
|
| 1218 |
-
"developer": "Meta",
|
| 1219 |
-
"scores": {
|
| 1220 |
-
"Mean win rate": 0.908,
|
| 1221 |
-
"MMLU": 0.584,
|
| 1222 |
-
"BoolQ": 0.871,
|
| 1223 |
-
"NarrativeQA": 0.755,
|
| 1224 |
-
"NaturalQuestions (open-book)": 0.672,
|
| 1225 |
-
"QuAC": 0.401,
|
| 1226 |
-
"HellaSwag": -1,
|
| 1227 |
-
"OpenbookQA": -1,
|
| 1228 |
-
"TruthfulQA": 0.508,
|
| 1229 |
-
"MS MARCO (TREC)": -1,
|
| 1230 |
-
"CNN/DailyMail": -1,
|
| 1231 |
-
"XSUM": -1,
|
| 1232 |
-
"IMDB": 0.962,
|
| 1233 |
-
"CivilComments": 0.655,
|
| 1234 |
-
"RAFT": 0.702
|
| 1235 |
-
}
|
| 1236 |
-
},
|
| 1237 |
-
{
|
| 1238 |
-
"model_id": "meta/LLaMA-7B",
|
| 1239 |
-
"name": "LLaMA 7B",
|
| 1240 |
-
"developer": "Meta",
|
| 1241 |
-
"scores": {
|
| 1242 |
-
"Mean win rate": 0.533,
|
| 1243 |
-
"MMLU": 0.321,
|
| 1244 |
-
"BoolQ": 0.756,
|
| 1245 |
-
"NarrativeQA": 0.669,
|
| 1246 |
-
"NaturalQuestions (open-book)": 0.589,
|
| 1247 |
-
"QuAC": 0.338,
|
| 1248 |
-
"HellaSwag": -1,
|
| 1249 |
-
"OpenbookQA": -1,
|
| 1250 |
-
"TruthfulQA": 0.28,
|
| 1251 |
-
"MS MARCO (TREC)": -1,
|
| 1252 |
-
"CNN/DailyMail": -1,
|
| 1253 |
-
"XSUM": -1,
|
| 1254 |
-
"IMDB": 0.947,
|
| 1255 |
-
"CivilComments": 0.563,
|
| 1256 |
-
"RAFT": 0.573
|
| 1257 |
-
}
|
| 1258 |
-
},
|
| 1259 |
-
{
|
| 1260 |
-
"model_id": "meta/OPT-175B",
|
| 1261 |
-
"name": "OPT 175B",
|
| 1262 |
-
"developer": "Meta",
|
| 1263 |
-
"scores": {
|
| 1264 |
-
"Mean win rate": 0.609,
|
| 1265 |
-
"MMLU": 0.318,
|
| 1266 |
-
"BoolQ": 0.793,
|
| 1267 |
-
"NarrativeQA": 0.671,
|
| 1268 |
-
"NaturalQuestions (open-book)": 0.615,
|
| 1269 |
-
"QuAC": 0.36,
|
| 1270 |
-
"HellaSwag": 0.791,
|
| 1271 |
-
"OpenbookQA": 0.586,
|
| 1272 |
-
"TruthfulQA": 0.25,
|
| 1273 |
-
"MS MARCO (TREC)": 0.448,
|
| 1274 |
-
"CNN/DailyMail": 0.146,
|
| 1275 |
-
"XSUM": 0.155,
|
| 1276 |
-
"IMDB": 0.947,
|
| 1277 |
-
"CivilComments": 0.505,
|
| 1278 |
-
"RAFT": 0.606
|
| 1279 |
-
}
|
| 1280 |
-
},
|
| 1281 |
-
{
|
| 1282 |
-
"model_id": "meta/OPT-66B",
|
| 1283 |
-
"name": "OPT 66B",
|
| 1284 |
-
"developer": "Meta",
|
| 1285 |
-
"scores": {
|
| 1286 |
-
"Mean win rate": 0.448,
|
| 1287 |
-
"MMLU": 0.276,
|
| 1288 |
-
"BoolQ": 0.76,
|
| 1289 |
-
"NarrativeQA": 0.638,
|
| 1290 |
-
"NaturalQuestions (open-book)": 0.596,
|
| 1291 |
-
"QuAC": 0.357,
|
| 1292 |
-
"HellaSwag": 0.745,
|
| 1293 |
-
"OpenbookQA": 0.534,
|
| 1294 |
-
"TruthfulQA": 0.201,
|
| 1295 |
-
"MS MARCO (TREC)": 0.482,
|
| 1296 |
-
"CNN/DailyMail": 0.136,
|
| 1297 |
-
"XSUM": 0.126,
|
| 1298 |
-
"IMDB": 0.917,
|
| 1299 |
-
"CivilComments": 0.506,
|
| 1300 |
-
"RAFT": 0.557
|
| 1301 |
-
}
|
| 1302 |
-
},
|
| 1303 |
-
{
|
| 1304 |
-
"model_id": "meta/llama-2-13b",
|
| 1305 |
-
"name": "Llama 2 13B",
|
| 1306 |
-
"developer": "Meta",
|
| 1307 |
-
"scores": {
|
| 1308 |
-
"Mean win rate": 0.823,
|
| 1309 |
-
"MMLU": 0.507,
|
| 1310 |
-
"BoolQ": 0.811,
|
| 1311 |
-
"NarrativeQA": 0.744,
|
| 1312 |
-
"NaturalQuestions (open-book)": 0.637,
|
| 1313 |
-
"QuAC": 0.424,
|
| 1314 |
-
"HellaSwag": -1,
|
| 1315 |
-
"OpenbookQA": -1,
|
| 1316 |
-
"TruthfulQA": 0.33,
|
| 1317 |
-
"MS MARCO (TREC)": -1,
|
| 1318 |
-
"CNN/DailyMail": -1,
|
| 1319 |
-
"XSUM": -1,
|
| 1320 |
-
"IMDB": 0.962,
|
| 1321 |
-
"CivilComments": 0.588,
|
| 1322 |
-
"RAFT": 0.707
|
| 1323 |
-
}
|
| 1324 |
-
},
|
| 1325 |
-
{
|
| 1326 |
-
"model_id": "meta/llama-2-70b",
|
| 1327 |
-
"name": "Llama 2 70B",
|
| 1328 |
-
"developer": "Meta",
|
| 1329 |
-
"scores": {
|
| 1330 |
-
"Mean win rate": 0.944,
|
| 1331 |
-
"MMLU": 0.582,
|
| 1332 |
-
"BoolQ": 0.886,
|
| 1333 |
-
"NarrativeQA": 0.77,
|
| 1334 |
-
"NaturalQuestions (open-book)": 0.674,
|
| 1335 |
-
"QuAC": 0.484,
|
| 1336 |
-
"HellaSwag": -1,
|
| 1337 |
-
"OpenbookQA": -1,
|
| 1338 |
-
"TruthfulQA": 0.554,
|
| 1339 |
-
"MS MARCO (TREC)": -1,
|
| 1340 |
-
"CNN/DailyMail": -1,
|
| 1341 |
-
"XSUM": -1,
|
| 1342 |
-
"IMDB": 0.961,
|
| 1343 |
-
"CivilComments": 0.652,
|
| 1344 |
-
"RAFT": 0.727
|
| 1345 |
-
}
|
| 1346 |
-
},
|
| 1347 |
-
{
|
| 1348 |
-
"model_id": "meta/llama-2-7b",
|
| 1349 |
-
"name": "Llama 2 7B",
|
| 1350 |
-
"developer": "Meta",
|
| 1351 |
-
"scores": {
|
| 1352 |
-
"Mean win rate": 0.607,
|
| 1353 |
-
"MMLU": 0.431,
|
| 1354 |
-
"BoolQ": 0.762,
|
| 1355 |
-
"NarrativeQA": 0.691,
|
| 1356 |
-
"NaturalQuestions (open-book)": 0.611,
|
| 1357 |
-
"QuAC": 0.406,
|
| 1358 |
-
"HellaSwag": -1,
|
| 1359 |
-
"OpenbookQA": -1,
|
| 1360 |
-
"TruthfulQA": 0.272,
|
| 1361 |
-
"MS MARCO (TREC)": -1,
|
| 1362 |
-
"CNN/DailyMail": -1,
|
| 1363 |
-
"XSUM": -1,
|
| 1364 |
-
"IMDB": 0.907,
|
| 1365 |
-
"CivilComments": 0.562,
|
| 1366 |
-
"RAFT": 0.643
|
| 1367 |
-
}
|
| 1368 |
-
},
|
| 1369 |
-
{
|
| 1370 |
-
"model_id": "microsoft/TNLG-v2-530B",
|
| 1371 |
-
"name": "TNLG v2 530B",
|
| 1372 |
-
"developer": "microsoft",
|
| 1373 |
-
"scores": {
|
| 1374 |
-
"Mean win rate": 0.787,
|
| 1375 |
-
"MMLU": 0.469,
|
| 1376 |
-
"BoolQ": 0.809,
|
| 1377 |
-
"NarrativeQA": 0.722,
|
| 1378 |
-
"NaturalQuestions (open-book)": 0.642,
|
| 1379 |
-
"QuAC": 0.39,
|
| 1380 |
-
"HellaSwag": 0.799,
|
| 1381 |
-
"OpenbookQA": 0.562,
|
| 1382 |
-
"TruthfulQA": 0.251,
|
| 1383 |
-
"MS MARCO (TREC)": 0.643,
|
| 1384 |
-
"CNN/DailyMail": 0.161,
|
| 1385 |
-
"XSUM": 0.169,
|
| 1386 |
-
"IMDB": 0.941,
|
| 1387 |
-
"CivilComments": 0.601,
|
| 1388 |
-
"RAFT": 0.679
|
| 1389 |
-
}
|
| 1390 |
-
},
|
| 1391 |
-
{
|
| 1392 |
-
"model_id": "microsoft/TNLG-v2-6.7B",
|
| 1393 |
-
"name": "TNLG v2 6.7B",
|
| 1394 |
-
"developer": "microsoft",
|
| 1395 |
-
"scores": {
|
| 1396 |
-
"Mean win rate": 0.309,
|
| 1397 |
-
"MMLU": 0.242,
|
| 1398 |
-
"BoolQ": 0.698,
|
| 1399 |
-
"NarrativeQA": 0.631,
|
| 1400 |
-
"NaturalQuestions (open-book)": 0.561,
|
| 1401 |
-
"QuAC": 0.345,
|
| 1402 |
-
"HellaSwag": 0.704,
|
| 1403 |
-
"OpenbookQA": 0.478,
|
| 1404 |
-
"TruthfulQA": 0.167,
|
| 1405 |
-
"MS MARCO (TREC)": 0.332,
|
| 1406 |
-
"CNN/DailyMail": 0.146,
|
| 1407 |
-
"XSUM": 0.11,
|
| 1408 |
-
"IMDB": 0.927,
|
| 1409 |
-
"CivilComments": 0.532,
|
| 1410 |
-
"RAFT": 0.525
|
| 1411 |
-
}
|
| 1412 |
-
},
|
| 1413 |
-
{
|
| 1414 |
-
"model_id": "mistralai/Mistral-v0.1-7B",
|
| 1415 |
-
"name": "Mistral v0.1 7B",
|
| 1416 |
-
"developer": "mistralai",
|
| 1417 |
-
"scores": {
|
| 1418 |
-
"Mean win rate": 0.884,
|
| 1419 |
-
"MMLU": 0.572,
|
| 1420 |
-
"BoolQ": 0.874,
|
| 1421 |
-
"NarrativeQA": 0.716,
|
| 1422 |
-
"NaturalQuestions (open-book)": 0.687,
|
| 1423 |
-
"QuAC": 0.423,
|
| 1424 |
-
"HellaSwag": -1,
|
| 1425 |
-
"OpenbookQA": -1,
|
| 1426 |
-
"TruthfulQA": 0.422,
|
| 1427 |
-
"MS MARCO (TREC)": -1,
|
| 1428 |
-
"CNN/DailyMail": -1,
|
| 1429 |
-
"XSUM": -1,
|
| 1430 |
-
"IMDB": 0.962,
|
| 1431 |
-
"CivilComments": 0.624,
|
| 1432 |
-
"RAFT": 0.707
|
| 1433 |
-
}
|
| 1434 |
-
},
|
| 1435 |
-
{
|
| 1436 |
-
"model_id": "mosaicml/MPT-30B",
|
| 1437 |
-
"name": "MPT 30B",
|
| 1438 |
-
"developer": "mosaicml",
|
| 1439 |
-
"scores": {
|
| 1440 |
-
"Mean win rate": 0.714,
|
| 1441 |
-
"MMLU": 0.437,
|
| 1442 |
-
"BoolQ": 0.704,
|
| 1443 |
-
"NarrativeQA": 0.732,
|
| 1444 |
-
"NaturalQuestions (open-book)": 0.673,
|
| 1445 |
-
"QuAC": 0.393,
|
| 1446 |
-
"HellaSwag": -1,
|
| 1447 |
-
"OpenbookQA": -1,
|
| 1448 |
-
"TruthfulQA": 0.231,
|
| 1449 |
-
"MS MARCO (TREC)": -1,
|
| 1450 |
-
"CNN/DailyMail": -1,
|
| 1451 |
-
"XSUM": -1,
|
| 1452 |
-
"IMDB": 0.959,
|
| 1453 |
-
"CivilComments": 0.599,
|
| 1454 |
-
"RAFT": 0.723
|
| 1455 |
-
}
|
| 1456 |
-
},
|
| 1457 |
-
{
|
| 1458 |
-
"model_id": "mosaicml/MPT-Instruct-30B",
|
| 1459 |
-
"name": "MPT-Instruct 30B",
|
| 1460 |
-
"developer": "mosaicml",
|
| 1461 |
-
"scores": {
|
| 1462 |
-
"Mean win rate": 0.716,
|
| 1463 |
-
"MMLU": 0.444,
|
| 1464 |
-
"BoolQ": 0.85,
|
| 1465 |
-
"NarrativeQA": 0.733,
|
| 1466 |
-
"NaturalQuestions (open-book)": 0.697,
|
| 1467 |
-
"QuAC": 0.327,
|
| 1468 |
-
"HellaSwag": -1,
|
| 1469 |
-
"OpenbookQA": -1,
|
| 1470 |
-
"TruthfulQA": 0.234,
|
| 1471 |
-
"MS MARCO (TREC)": -1,
|
| 1472 |
-
"CNN/DailyMail": -1,
|
| 1473 |
-
"XSUM": -1,
|
| 1474 |
-
"IMDB": 0.956,
|
| 1475 |
-
"CivilComments": 0.573,
|
| 1476 |
-
"RAFT": 0.68
|
| 1477 |
-
}
|
| 1478 |
-
},
|
| 1479 |
-
{
|
| 1480 |
-
"model_id": "openai/GPT-J-6B",
|
| 1481 |
-
"name": "GPT-J 6B",
|
| 1482 |
-
"developer": "OpenAI",
|
| 1483 |
-
"scores": {
|
| 1484 |
-
"Mean win rate": 0.273,
|
| 1485 |
-
"MMLU": 0.249,
|
| 1486 |
-
"BoolQ": 0.649,
|
| 1487 |
-
"NarrativeQA": 0.545,
|
| 1488 |
-
"NaturalQuestions (open-book)": 0.559,
|
| 1489 |
-
"QuAC": 0.33,
|
| 1490 |
-
"HellaSwag": 0.663,
|
| 1491 |
-
"OpenbookQA": 0.514,
|
| 1492 |
-
"TruthfulQA": 0.199,
|
| 1493 |
-
"MS MARCO (TREC)": 0.345,
|
| 1494 |
-
"CNN/DailyMail": 0.131,
|
| 1495 |
-
"XSUM": 0.096,
|
| 1496 |
-
"IMDB": 0.939,
|
| 1497 |
-
"CivilComments": 0.52,
|
| 1498 |
-
"RAFT": 0.619
|
| 1499 |
-
}
|
| 1500 |
-
},
|
| 1501 |
-
{
|
| 1502 |
-
"model_id": "openai/GPT-NeoX-20B",
|
| 1503 |
-
"name": "GPT-NeoX 20B",
|
| 1504 |
-
"developer": "OpenAI",
|
| 1505 |
-
"scores": {
|
| 1506 |
-
"Mean win rate": 0.351,
|
| 1507 |
-
"MMLU": 0.276,
|
| 1508 |
-
"BoolQ": 0.683,
|
| 1509 |
-
"NarrativeQA": 0.599,
|
| 1510 |
-
"NaturalQuestions (open-book)": 0.596,
|
| 1511 |
-
"QuAC": 0.326,
|
| 1512 |
-
"HellaSwag": 0.718,
|
| 1513 |
-
"OpenbookQA": 0.524,
|
| 1514 |
-
"TruthfulQA": 0.216,
|
| 1515 |
-
"MS MARCO (TREC)": 0.398,
|
| 1516 |
-
"CNN/DailyMail": 0.123,
|
| 1517 |
-
"XSUM": 0.102,
|
| 1518 |
-
"IMDB": 0.948,
|
| 1519 |
-
"CivilComments": 0.516,
|
| 1520 |
-
"RAFT": 0.505
|
| 1521 |
-
}
|
| 1522 |
-
},
|
| 1523 |
-
{
|
| 1524 |
-
"model_id": "openai/ada-350M",
|
| 1525 |
-
"name": "ada 350M",
|
| 1526 |
-
"developer": "OpenAI",
|
| 1527 |
-
"scores": {
|
| 1528 |
-
"Mean win rate": 0.108,
|
| 1529 |
-
"MMLU": 0.243,
|
| 1530 |
-
"BoolQ": 0.581,
|
| 1531 |
-
"NarrativeQA": 0.326,
|
| 1532 |
-
"NaturalQuestions (open-book)": 0.365,
|
| 1533 |
-
"QuAC": 0.242,
|
| 1534 |
-
"HellaSwag": 0.435,
|
| 1535 |
-
"OpenbookQA": 0.38,
|
| 1536 |
-
"TruthfulQA": 0.215,
|
| 1537 |
-
"MS MARCO (TREC)": 0.29,
|
| 1538 |
-
"CNN/DailyMail": 0.09,
|
| 1539 |
-
"XSUM": 0.022,
|
| 1540 |
-
"IMDB": 0.849,
|
| 1541 |
-
"CivilComments": 0.517,
|
| 1542 |
-
"RAFT": 0.423
|
| 1543 |
-
}
|
| 1544 |
-
},
|
| 1545 |
-
{
|
| 1546 |
-
"model_id": "openai/babbage-1.3B",
|
| 1547 |
-
"name": "babbage 1.3B",
|
| 1548 |
-
"developer": "OpenAI",
|
| 1549 |
-
"scores": {
|
| 1550 |
-
"Mean win rate": 0.114,
|
| 1551 |
-
"MMLU": 0.235,
|
| 1552 |
-
"BoolQ": 0.574,
|
| 1553 |
-
"NarrativeQA": 0.491,
|
| 1554 |
-
"NaturalQuestions (open-book)": 0.451,
|
| 1555 |
-
"QuAC": 0.273,
|
| 1556 |
-
"HellaSwag": 0.555,
|
| 1557 |
-
"OpenbookQA": 0.438,
|
| 1558 |
-
"TruthfulQA": 0.188,
|
| 1559 |
-
"MS MARCO (TREC)": 0.317,
|
| 1560 |
-
"CNN/DailyMail": 0.079,
|
| 1561 |
-
"XSUM": 0.045,
|
| 1562 |
-
"IMDB": 0.597,
|
| 1563 |
-
"CivilComments": 0.519,
|
| 1564 |
-
"RAFT": 0.455
|
| 1565 |
-
}
|
| 1566 |
-
},
|
| 1567 |
-
{
|
| 1568 |
-
"model_id": "openai/curie-6.7B",
|
| 1569 |
-
"name": "curie 6.7B",
|
| 1570 |
-
"developer": "OpenAI",
|
| 1571 |
-
"scores": {
|
| 1572 |
-
"Mean win rate": 0.247,
|
| 1573 |
-
"MMLU": 0.243,
|
| 1574 |
-
"BoolQ": 0.656,
|
| 1575 |
-
"NarrativeQA": 0.604,
|
| 1576 |
-
"NaturalQuestions (open-book)": 0.552,
|
| 1577 |
-
"QuAC": 0.321,
|
| 1578 |
-
"HellaSwag": 0.682,
|
| 1579 |
-
"OpenbookQA": 0.502,
|
| 1580 |
-
"TruthfulQA": 0.232,
|
| 1581 |
-
"MS MARCO (TREC)": 0.3,
|
| 1582 |
-
"CNN/DailyMail": 0.113,
|
| 1583 |
-
"XSUM": 0.091,
|
| 1584 |
-
"IMDB": 0.889,
|
| 1585 |
-
"CivilComments": 0.539,
|
| 1586 |
-
"RAFT": 0.49
|
| 1587 |
-
}
|
| 1588 |
-
},
|
| 1589 |
-
{
|
| 1590 |
-
"model_id": "openai/davinci-175B",
|
| 1591 |
-
"name": "davinci 175B",
|
| 1592 |
-
"developer": "OpenAI",
|
| 1593 |
-
"scores": {
|
| 1594 |
-
"Mean win rate": 0.538,
|
| 1595 |
-
"MMLU": 0.422,
|
| 1596 |
-
"BoolQ": 0.722,
|
| 1597 |
-
"NarrativeQA": 0.687,
|
| 1598 |
-
"NaturalQuestions (open-book)": 0.625,
|
| 1599 |
-
"QuAC": 0.36,
|
| 1600 |
-
"HellaSwag": 0.775,
|
| 1601 |
-
"OpenbookQA": 0.586,
|
| 1602 |
-
"TruthfulQA": 0.194,
|
| 1603 |
-
"MS MARCO (TREC)": 0.378,
|
| 1604 |
-
"CNN/DailyMail": 0.127,
|
| 1605 |
-
"XSUM": 0.126,
|
| 1606 |
-
"IMDB": 0.933,
|
| 1607 |
-
"CivilComments": 0.532,
|
| 1608 |
-
"RAFT": 0.642
|
| 1609 |
-
}
|
| 1610 |
-
},
|
| 1611 |
-
{
|
| 1612 |
-
"model_id": "openai/gpt-3.5-turbo-0301",
|
| 1613 |
-
"name": "gpt-3.5-turbo-0301",
|
| 1614 |
-
"developer": "OpenAI",
|
| 1615 |
-
"scores": {
|
| 1616 |
-
"Mean win rate": 0.76,
|
| 1617 |
-
"MMLU": 0.59,
|
| 1618 |
-
"BoolQ": 0.74,
|
| 1619 |
-
"NarrativeQA": 0.663,
|
| 1620 |
-
"NaturalQuestions (open-book)": 0.624,
|
| 1621 |
-
"QuAC": 0.512,
|
| 1622 |
-
"HellaSwag": -1,
|
| 1623 |
-
"OpenbookQA": -1,
|
| 1624 |
-
"TruthfulQA": 0.609,
|
| 1625 |
-
"MS MARCO (TREC)": -1,
|
| 1626 |
-
"CNN/DailyMail": -1,
|
| 1627 |
-
"XSUM": -1,
|
| 1628 |
-
"IMDB": 0.899,
|
| 1629 |
-
"CivilComments": 0.674,
|
| 1630 |
-
"RAFT": 0.768
|
| 1631 |
-
}
|
| 1632 |
-
},
|
| 1633 |
-
{
|
| 1634 |
-
"model_id": "openai/gpt-3.5-turbo-0613",
|
| 1635 |
-
"name": "GPT-3.5 Turbo 0613",
|
| 1636 |
-
"developer": "OpenAI",
|
| 1637 |
-
"scores": {
|
| 1638 |
-
"Mean win rate": 0.783,
|
| 1639 |
-
"MMLU": 0.391,
|
| 1640 |
-
"BoolQ": 0.87,
|
| 1641 |
-
"NarrativeQA": 0.625,
|
| 1642 |
-
"NaturalQuestions (open-book)": 0.675,
|
| 1643 |
-
"QuAC": 0.485,
|
| 1644 |
-
"HellaSwag": -1,
|
| 1645 |
-
"OpenbookQA": -1,
|
| 1646 |
-
"TruthfulQA": 0.339,
|
| 1647 |
-
"MS MARCO (TREC)": -1,
|
| 1648 |
-
"CNN/DailyMail": -1,
|
| 1649 |
-
"XSUM": -1,
|
| 1650 |
-
"IMDB": 0.943,
|
| 1651 |
-
"CivilComments": 0.696,
|
| 1652 |
-
"RAFT": 0.748
|
| 1653 |
-
}
|
| 1654 |
-
},
|
| 1655 |
-
{
|
| 1656 |
-
"model_id": "openai/text-ada-001",
|
| 1657 |
-
"name": "text-ada-001",
|
| 1658 |
-
"developer": "OpenAI",
|
| 1659 |
-
"scores": {
|
| 1660 |
-
"Mean win rate": 0.107,
|
| 1661 |
-
"MMLU": 0.238,
|
| 1662 |
-
"BoolQ": 0.464,
|
| 1663 |
-
"NarrativeQA": 0.238,
|
| 1664 |
-
"NaturalQuestions (open-book)": 0.149,
|
| 1665 |
-
"QuAC": 0.176,
|
| 1666 |
-
"HellaSwag": 0.429,
|
| 1667 |
-
"OpenbookQA": 0.346,
|
| 1668 |
-
"TruthfulQA": 0.232,
|
| 1669 |
-
"MS MARCO (TREC)": 0.302,
|
| 1670 |
-
"CNN/DailyMail": 0.136,
|
| 1671 |
-
"XSUM": 0.034,
|
| 1672 |
-
"IMDB": 0.822,
|
| 1673 |
-
"CivilComments": 0.503,
|
| 1674 |
-
"RAFT": 0.406
|
| 1675 |
-
}
|
| 1676 |
-
},
|
| 1677 |
-
{
|
| 1678 |
-
"model_id": "openai/text-babbage-001",
|
| 1679 |
-
"name": "text-babbage-001",
|
| 1680 |
-
"developer": "OpenAI",
|
| 1681 |
-
"scores": {
|
| 1682 |
-
"Mean win rate": 0.229,
|
| 1683 |
-
"MMLU": 0.229,
|
| 1684 |
-
"BoolQ": 0.451,
|
| 1685 |
-
"NarrativeQA": 0.429,
|
| 1686 |
-
"NaturalQuestions (open-book)": 0.33,
|
| 1687 |
-
"QuAC": 0.284,
|
| 1688 |
-
"HellaSwag": 0.561,
|
| 1689 |
-
"OpenbookQA": 0.452,
|
| 1690 |
-
"TruthfulQA": 0.233,
|
| 1691 |
-
"MS MARCO (TREC)": 0.449,
|
| 1692 |
-
"CNN/DailyMail": 0.151,
|
| 1693 |
-
"XSUM": 0.046,
|
| 1694 |
-
"IMDB": 0.913,
|
| 1695 |
-
"CivilComments": 0.499,
|
| 1696 |
-
"RAFT": 0.509
|
| 1697 |
-
}
|
| 1698 |
-
},
|
| 1699 |
-
{
|
| 1700 |
-
"model_id": "openai/text-curie-001",
|
| 1701 |
-
"name": "text-curie-001",
|
| 1702 |
-
"developer": "OpenAI",
|
| 1703 |
-
"scores": {
|
| 1704 |
-
"Mean win rate": 0.36,
|
| 1705 |
-
"MMLU": 0.237,
|
| 1706 |
-
"BoolQ": 0.62,
|
| 1707 |
-
"NarrativeQA": 0.582,
|
| 1708 |
-
"NaturalQuestions (open-book)": 0.571,
|
| 1709 |
-
"QuAC": 0.358,
|
| 1710 |
-
"HellaSwag": 0.676,
|
| 1711 |
-
"OpenbookQA": 0.514,
|
| 1712 |
-
"TruthfulQA": 0.257,
|
| 1713 |
-
"MS MARCO (TREC)": 0.507,
|
| 1714 |
-
"CNN/DailyMail": 0.152,
|
| 1715 |
-
"XSUM": 0.076,
|
| 1716 |
-
"IMDB": 0.923,
|
| 1717 |
-
"CivilComments": 0.537,
|
| 1718 |
-
"RAFT": 0.489
|
| 1719 |
-
}
|
| 1720 |
-
},
|
| 1721 |
-
{
|
| 1722 |
-
"model_id": "openai/text-davinci-002",
|
| 1723 |
-
"name": "GPT-3.5 text-davinci-002",
|
| 1724 |
-
"developer": "OpenAI",
|
| 1725 |
-
"scores": {
|
| 1726 |
-
"Mean win rate": 0.905,
|
| 1727 |
-
"MMLU": 0.568,
|
| 1728 |
-
"BoolQ": 0.877,
|
| 1729 |
-
"NarrativeQA": 0.727,
|
| 1730 |
-
"NaturalQuestions (open-book)": 0.713,
|
| 1731 |
-
"QuAC": 0.445,
|
| 1732 |
-
"HellaSwag": 0.815,
|
| 1733 |
-
"OpenbookQA": 0.594,
|
| 1734 |
-
"TruthfulQA": 0.61,
|
| 1735 |
-
"MS MARCO (TREC)": 0.664,
|
| 1736 |
-
"CNN/DailyMail": 0.153,
|
| 1737 |
-
"XSUM": 0.144,
|
| 1738 |
-
"IMDB": 0.948,
|
| 1739 |
-
"CivilComments": 0.668,
|
| 1740 |
-
"RAFT": 0.733
|
| 1741 |
-
}
|
| 1742 |
-
},
|
| 1743 |
-
{
|
| 1744 |
-
"model_id": "openai/text-davinci-003",
|
| 1745 |
-
"name": "GPT-3.5 text-davinci-003",
|
| 1746 |
-
"developer": "OpenAI",
|
| 1747 |
-
"scores": {
|
| 1748 |
-
"Mean win rate": 0.872,
|
| 1749 |
-
"MMLU": 0.569,
|
| 1750 |
-
"BoolQ": 0.881,
|
| 1751 |
-
"NarrativeQA": 0.727,
|
| 1752 |
-
"NaturalQuestions (open-book)": 0.77,
|
| 1753 |
-
"QuAC": 0.525,
|
| 1754 |
-
"HellaSwag": 0.822,
|
| 1755 |
-
"OpenbookQA": 0.646,
|
| 1756 |
-
"TruthfulQA": 0.593,
|
| 1757 |
-
"MS MARCO (TREC)": 0.644,
|
| 1758 |
-
"CNN/DailyMail": 0.156,
|
| 1759 |
-
"XSUM": 0.124,
|
| 1760 |
-
"IMDB": 0.848,
|
| 1761 |
-
"CivilComments": 0.684,
|
| 1762 |
-
"RAFT": 0.759
|
| 1763 |
-
}
|
| 1764 |
-
},
|
| 1765 |
-
{
|
| 1766 |
-
"model_id": "stanford/Alpaca-7B",
|
| 1767 |
-
"name": "Alpaca 7B",
|
| 1768 |
-
"developer": "stanford",
|
| 1769 |
-
"scores": {
|
| 1770 |
-
"Mean win rate": 0.381,
|
| 1771 |
-
"MMLU": 0.385,
|
| 1772 |
-
"BoolQ": 0.778,
|
| 1773 |
-
"NarrativeQA": 0.396,
|
| 1774 |
-
"NaturalQuestions (open-book)": 0.592,
|
| 1775 |
-
"QuAC": 0.27,
|
| 1776 |
-
"HellaSwag": -1,
|
| 1777 |
-
"OpenbookQA": -1,
|
| 1778 |
-
"TruthfulQA": 0.243,
|
| 1779 |
-
"MS MARCO (TREC)": -1,
|
| 1780 |
-
"CNN/DailyMail": -1,
|
| 1781 |
-
"XSUM": -1,
|
| 1782 |
-
"IMDB": 0.738,
|
| 1783 |
-
"CivilComments": 0.566,
|
| 1784 |
-
"RAFT": 0.486
|
| 1785 |
-
}
|
| 1786 |
-
},
|
| 1787 |
-
{
|
| 1788 |
-
"model_id": "tiiuae/Falcon-Instruct-40B",
|
| 1789 |
-
"name": "Falcon-Instruct 40B",
|
| 1790 |
-
"developer": "tiiuae",
|
| 1791 |
-
"scores": {
|
| 1792 |
-
"Mean win rate": 0.727,
|
| 1793 |
-
"MMLU": 0.497,
|
| 1794 |
-
"BoolQ": 0.829,
|
| 1795 |
-
"NarrativeQA": 0.625,
|
| 1796 |
-
"NaturalQuestions (open-book)": 0.666,
|
| 1797 |
-
"QuAC": 0.371,
|
| 1798 |
-
"HellaSwag": -1,
|
| 1799 |
-
"OpenbookQA": -1,
|
| 1800 |
-
"TruthfulQA": 0.384,
|
| 1801 |
-
"MS MARCO (TREC)": -1,
|
| 1802 |
-
"CNN/DailyMail": -1,
|
| 1803 |
-
"XSUM": -1,
|
| 1804 |
-
"IMDB": 0.959,
|
| 1805 |
-
"CivilComments": 0.603,
|
| 1806 |
-
"RAFT": 0.586
|
| 1807 |
-
}
|
| 1808 |
-
},
|
| 1809 |
-
{
|
| 1810 |
-
"model_id": "tiiuae/Falcon-Instruct-7B",
|
| 1811 |
-
"name": "Falcon-Instruct 7B",
|
| 1812 |
-
"developer": "tiiuae",
|
| 1813 |
-
"scores": {
|
| 1814 |
-
"Mean win rate": 0.244,
|
| 1815 |
-
"MMLU": 0.275,
|
| 1816 |
-
"BoolQ": 0.72,
|
| 1817 |
-
"NarrativeQA": 0.476,
|
| 1818 |
-
"NaturalQuestions (open-book)": 0.449,
|
| 1819 |
-
"QuAC": 0.311,
|
| 1820 |
-
"HellaSwag": -1,
|
| 1821 |
-
"OpenbookQA": -1,
|
| 1822 |
-
"TruthfulQA": 0.213,
|
| 1823 |
-
"MS MARCO (TREC)": -1,
|
| 1824 |
-
"CNN/DailyMail": -1,
|
| 1825 |
-
"XSUM": -1,
|
| 1826 |
-
"IMDB": 0.852,
|
| 1827 |
-
"CivilComments": 0.511,
|
| 1828 |
-
"RAFT": 0.523
|
| 1829 |
-
}
|
| 1830 |
-
},
|
| 1831 |
-
{
|
| 1832 |
-
"model_id": "tiiuae/falcon-40b",
|
| 1833 |
-
"name": "Falcon 40B",
|
| 1834 |
-
"developer": "tiiuae",
|
| 1835 |
-
"scores": {
|
| 1836 |
-
"Mean win rate": 0.729,
|
| 1837 |
-
"MMLU": 0.509,
|
| 1838 |
-
"BoolQ": 0.819,
|
| 1839 |
-
"NarrativeQA": 0.673,
|
| 1840 |
-
"NaturalQuestions (open-book)": 0.675,
|
| 1841 |
-
"QuAC": 0.307,
|
| 1842 |
-
"HellaSwag": -1,
|
| 1843 |
-
"OpenbookQA": -1,
|
| 1844 |
-
"TruthfulQA": 0.353,
|
| 1845 |
-
"MS MARCO (TREC)": -1,
|
| 1846 |
-
"CNN/DailyMail": -1,
|
| 1847 |
-
"XSUM": -1,
|
| 1848 |
-
"IMDB": 0.959,
|
| 1849 |
-
"CivilComments": 0.552,
|
| 1850 |
-
"RAFT": 0.661
|
| 1851 |
-
}
|
| 1852 |
-
},
|
| 1853 |
-
{
|
| 1854 |
-
"model_id": "tiiuae/falcon-7b",
|
| 1855 |
-
"name": "Falcon 7B",
|
| 1856 |
-
"developer": "tiiuae",
|
| 1857 |
-
"scores": {
|
| 1858 |
-
"Mean win rate": 0.378,
|
| 1859 |
-
"MMLU": 0.286,
|
| 1860 |
-
"BoolQ": 0.753,
|
| 1861 |
-
"NarrativeQA": 0.621,
|
| 1862 |
-
"NaturalQuestions (open-book)": 0.579,
|
| 1863 |
-
"QuAC": 0.332,
|
| 1864 |
-
"HellaSwag": -1,
|
| 1865 |
-
"OpenbookQA": -1,
|
| 1866 |
-
"TruthfulQA": 0.234,
|
| 1867 |
-
"MS MARCO (TREC)": -1,
|
| 1868 |
-
"CNN/DailyMail": -1,
|
| 1869 |
-
"XSUM": -1,
|
| 1870 |
-
"IMDB": 0.836,
|
| 1871 |
-
"CivilComments": 0.514,
|
| 1872 |
-
"RAFT": 0.602
|
| 1873 |
-
}
|
| 1874 |
-
},
|
| 1875 |
-
{
|
| 1876 |
-
"model_id": "together/RedPajama-INCITE-Base-7B",
|
| 1877 |
-
"name": "RedPajama-INCITE-Base 7B",
|
| 1878 |
-
"developer": "together",
|
| 1879 |
-
"scores": {
|
| 1880 |
-
"Mean win rate": 0.378,
|
| 1881 |
-
"MMLU": 0.302,
|
| 1882 |
-
"BoolQ": 0.713,
|
| 1883 |
-
"NarrativeQA": 0.617,
|
| 1884 |
-
"NaturalQuestions (open-book)": 0.586,
|
| 1885 |
-
"QuAC": 0.336,
|
| 1886 |
-
"HellaSwag": -1,
|
| 1887 |
-
"OpenbookQA": -1,
|
| 1888 |
-
"TruthfulQA": 0.205,
|
| 1889 |
-
"MS MARCO (TREC)": -1,
|
| 1890 |
-
"CNN/DailyMail": -1,
|
| 1891 |
-
"XSUM": -1,
|
| 1892 |
-
"IMDB": 0.752,
|
| 1893 |
-
"CivilComments": 0.547,
|
| 1894 |
-
"RAFT": 0.648
|
| 1895 |
-
}
|
| 1896 |
-
},
|
| 1897 |
-
{
|
| 1898 |
-
"model_id": "together/RedPajama-INCITE-Base-v1-3B",
|
| 1899 |
-
"name": "RedPajama-INCITE-Base-v1 3B",
|
| 1900 |
-
"developer": "together",
|
| 1901 |
-
"scores": {
|
| 1902 |
-
"Mean win rate": 0.311,
|
| 1903 |
-
"MMLU": 0.263,
|
| 1904 |
-
"BoolQ": 0.685,
|
| 1905 |
-
"NarrativeQA": 0.555,
|
| 1906 |
-
"NaturalQuestions (open-book)": 0.52,
|
| 1907 |
-
"QuAC": 0.309,
|
| 1908 |
-
"HellaSwag": -1,
|
| 1909 |
-
"OpenbookQA": -1,
|
| 1910 |
-
"TruthfulQA": 0.277,
|
| 1911 |
-
"MS MARCO (TREC)": -1,
|
| 1912 |
-
"CNN/DailyMail": -1,
|
| 1913 |
-
"XSUM": -1,
|
| 1914 |
-
"IMDB": 0.907,
|
| 1915 |
-
"CivilComments": 0.549,
|
| 1916 |
-
"RAFT": 0.502
|
| 1917 |
-
}
|
| 1918 |
-
},
|
| 1919 |
-
{
|
| 1920 |
-
"model_id": "together/RedPajama-INCITE-Instruct-7B",
|
| 1921 |
-
"name": "RedPajama-INCITE-Instruct 7B",
|
| 1922 |
-
"developer": "together",
|
| 1923 |
-
"scores": {
|
| 1924 |
-
"Mean win rate": 0.524,
|
| 1925 |
-
"MMLU": 0.363,
|
| 1926 |
-
"BoolQ": 0.705,
|
| 1927 |
-
"NarrativeQA": 0.638,
|
| 1928 |
-
"NaturalQuestions (open-book)": 0.659,
|
| 1929 |
-
"QuAC": 0.26,
|
| 1930 |
-
"HellaSwag": -1,
|
| 1931 |
-
"OpenbookQA": -1,
|
| 1932 |
-
"TruthfulQA": 0.243,
|
| 1933 |
-
"MS MARCO (TREC)": -1,
|
| 1934 |
-
"CNN/DailyMail": -1,
|
| 1935 |
-
"XSUM": -1,
|
| 1936 |
-
"IMDB": 0.927,
|
| 1937 |
-
"CivilComments": 0.664,
|
| 1938 |
-
"RAFT": 0.695
|
| 1939 |
-
}
|
| 1940 |
-
},
|
| 1941 |
-
{
|
| 1942 |
-
"model_id": "together/RedPajama-INCITE-Instruct-v1-3B",
|
| 1943 |
-
"name": "RedPajama-INCITE-Instruct-v1 3B",
|
| 1944 |
-
"developer": "together",
|
| 1945 |
-
"scores": {
|
| 1946 |
-
"Mean win rate": 0.366,
|
| 1947 |
-
"MMLU": 0.257,
|
| 1948 |
-
"BoolQ": 0.677,
|
| 1949 |
-
"NarrativeQA": 0.638,
|
| 1950 |
-
"NaturalQuestions (open-book)": 0.637,
|
| 1951 |
-
"QuAC": 0.259,
|
| 1952 |
-
"HellaSwag": -1,
|
| 1953 |
-
"OpenbookQA": -1,
|
| 1954 |
-
"TruthfulQA": 0.208,
|
| 1955 |
-
"MS MARCO (TREC)": -1,
|
| 1956 |
-
"CNN/DailyMail": -1,
|
| 1957 |
-
"XSUM": -1,
|
| 1958 |
-
"IMDB": 0.894,
|
| 1959 |
-
"CivilComments": 0.549,
|
| 1960 |
-
"RAFT": 0.661
|
| 1961 |
-
}
|
| 1962 |
-
},
|
| 1963 |
-
{
|
| 1964 |
-
"model_id": "writer/InstructPalmyra-30B",
|
| 1965 |
-
"name": "InstructPalmyra 30B",
|
| 1966 |
-
"developer": "writer",
|
| 1967 |
-
"scores": {
|
| 1968 |
-
"Mean win rate": 0.568,
|
| 1969 |
-
"MMLU": 0.403,
|
| 1970 |
-
"BoolQ": 0.751,
|
| 1971 |
-
"NarrativeQA": 0.496,
|
| 1972 |
-
"NaturalQuestions (open-book)": 0.682,
|
| 1973 |
-
"QuAC": 0.433,
|
| 1974 |
-
"HellaSwag": -1,
|
| 1975 |
-
"OpenbookQA": -1,
|
| 1976 |
-
"TruthfulQA": 0.185,
|
| 1977 |
-
"MS MARCO (TREC)": -1,
|
| 1978 |
-
"CNN/DailyMail": 0.152,
|
| 1979 |
-
"XSUM": 0.104,
|
| 1980 |
-
"IMDB": 0.94,
|
| 1981 |
-
"CivilComments": 0.555,
|
| 1982 |
-
"RAFT": 0.652
|
| 1983 |
-
}
|
| 1984 |
-
},
|
| 1985 |
-
{
|
| 1986 |
-
"model_id": "yandex/YaLM-100B",
|
| 1987 |
-
"name": "YaLM 100B",
|
| 1988 |
-
"developer": "yandex",
|
| 1989 |
-
"scores": {
|
| 1990 |
-
"Mean win rate": 0.075,
|
| 1991 |
-
"MMLU": 0.243,
|
| 1992 |
-
"BoolQ": 0.634,
|
| 1993 |
-
"NarrativeQA": 0.252,
|
| 1994 |
-
"NaturalQuestions (open-book)": 0.227,
|
| 1995 |
-
"QuAC": 0.162,
|
| 1996 |
-
"HellaSwag": -1,
|
| 1997 |
-
"OpenbookQA": -1,
|
| 1998 |
-
"TruthfulQA": 0.202,
|
| 1999 |
-
"MS MARCO (TREC)": -1,
|
| 2000 |
-
"CNN/DailyMail": 0.017,
|
| 2001 |
-
"XSUM": 0.021,
|
| 2002 |
-
"IMDB": 0.836,
|
| 2003 |
-
"CivilComments": 0.49,
|
| 2004 |
-
"RAFT": 0.395
|
| 2005 |
-
}
|
| 2006 |
-
},
|
| 2007 |
-
{
|
| 2008 |
-
"model_id": "zhipu-ai/GLM-130B",
|
| 2009 |
-
"name": "GLM 130B",
|
| 2010 |
-
"developer": "zhipu-ai",
|
| 2011 |
-
"scores": {
|
| 2012 |
-
"Mean win rate": 0.512,
|
| 2013 |
-
"MMLU": 0.344,
|
| 2014 |
-
"BoolQ": 0.784,
|
| 2015 |
-
"NarrativeQA": 0.706,
|
| 2016 |
-
"NaturalQuestions (open-book)": 0.642,
|
| 2017 |
-
"QuAC": 0.272,
|
| 2018 |
-
"HellaSwag": -1,
|
| 2019 |
-
"OpenbookQA": -1,
|
| 2020 |
-
"TruthfulQA": 0.218,
|
| 2021 |
-
"MS MARCO (TREC)": -1,
|
| 2022 |
-
"CNN/DailyMail": 0.154,
|
| 2023 |
-
"XSUM": 0.132,
|
| 2024 |
-
"IMDB": 0.955,
|
| 2025 |
-
"CivilComments": 0.5,
|
| 2026 |
-
"RAFT": 0.598
|
| 2027 |
-
}
|
| 2028 |
-
}
|
| 2029 |
-
]
|
| 2030 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/helm_instruct.json
DELETED
|
@@ -1,60 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "anthropic/claude-v1.3",
|
| 5 |
-
"name": "Anthropic Claude v1.3",
|
| 6 |
-
"developer": "Anthropic",
|
| 7 |
-
"scores": {
|
| 8 |
-
"Mean win rate": 0.611,
|
| 9 |
-
"Anthropic RLHF dataset": 4.965,
|
| 10 |
-
"Best ChatGPT Prompts": 4.995,
|
| 11 |
-
"Koala test dataset": 4.981,
|
| 12 |
-
"Open Assistant": 4.975,
|
| 13 |
-
"Self Instruct": 4.992,
|
| 14 |
-
"Vicuna": 4.989
|
| 15 |
-
}
|
| 16 |
-
},
|
| 17 |
-
{
|
| 18 |
-
"model_id": "cohere/command-xlarge-beta",
|
| 19 |
-
"name": "Cohere Command beta 52.4B",
|
| 20 |
-
"developer": "cohere",
|
| 21 |
-
"scores": {
|
| 22 |
-
"Mean win rate": 0.089,
|
| 23 |
-
"Anthropic RLHF dataset": 4.214,
|
| 24 |
-
"Best ChatGPT Prompts": 4.988,
|
| 25 |
-
"Koala test dataset": 4.969,
|
| 26 |
-
"Open Assistant": 4.967,
|
| 27 |
-
"Self Instruct": 4.971,
|
| 28 |
-
"Vicuna": 4.995
|
| 29 |
-
}
|
| 30 |
-
},
|
| 31 |
-
{
|
| 32 |
-
"model_id": "openai/gpt-3.5-turbo-0613",
|
| 33 |
-
"name": "GPT-3.5 Turbo 0613",
|
| 34 |
-
"developer": "OpenAI",
|
| 35 |
-
"scores": {
|
| 36 |
-
"Mean win rate": 0.689,
|
| 37 |
-
"Anthropic RLHF dataset": 4.964,
|
| 38 |
-
"Best ChatGPT Prompts": 4.986,
|
| 39 |
-
"Koala test dataset": 4.987,
|
| 40 |
-
"Open Assistant": 4.987,
|
| 41 |
-
"Self Instruct": 4.99,
|
| 42 |
-
"Vicuna": 4.992
|
| 43 |
-
}
|
| 44 |
-
},
|
| 45 |
-
{
|
| 46 |
-
"model_id": "openai/gpt-4-0314",
|
| 47 |
-
"name": "GPT-4 0314",
|
| 48 |
-
"developer": "OpenAI",
|
| 49 |
-
"scores": {
|
| 50 |
-
"Mean win rate": 0.611,
|
| 51 |
-
"Anthropic RLHF dataset": 4.934,
|
| 52 |
-
"Best ChatGPT Prompts": 4.973,
|
| 53 |
-
"Koala test dataset": 4.966,
|
| 54 |
-
"Open Assistant": 4.986,
|
| 55 |
-
"Self Instruct": 4.976,
|
| 56 |
-
"Vicuna": 4.995
|
| 57 |
-
}
|
| 58 |
-
}
|
| 59 |
-
]
|
| 60 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/helm_lite.json
DELETED
|
@@ -1,1893 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"benchmark_cards": {
|
| 3 |
-
"GSM8K": {
|
| 4 |
-
"benchmark_details": {
|
| 5 |
-
"name": "GSM8K",
|
| 6 |
-
"overview": "GSM8K is a benchmark that measures the ability of language models to perform multi-step mathematical reasoning. It consists of 8.5K high-quality, linguistically diverse grade school math word problems. The problems are distinctive because they require 2 to 8 steps to solve using basic arithmetic, and even the largest transformer models struggle to achieve high test performance on them. Solutions are provided in natural language with step-by-step reasoning.",
|
| 7 |
-
"data_type": "text",
|
| 8 |
-
"domains": [
|
| 9 |
-
"grade school mathematics",
|
| 10 |
-
"math word problems"
|
| 11 |
-
],
|
| 12 |
-
"languages": [
|
| 13 |
-
"English"
|
| 14 |
-
],
|
| 15 |
-
"similar_benchmarks": [
|
| 16 |
-
"Not specified"
|
| 17 |
-
],
|
| 18 |
-
"resources": [
|
| 19 |
-
"https://arxiv.org/abs/2110.14168",
|
| 20 |
-
"https://huggingface.co/datasets/openai/gsm8k",
|
| 21 |
-
"https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
|
| 22 |
-
]
|
| 23 |
-
},
|
| 24 |
-
"purpose_and_intended_users": {
|
| 25 |
-
"goal": "To diagnose the failures of current language models in robust multi-step mathematical reasoning and to support research, particularly in methods like training verifiers to judge solution correctness. It also aims to shed light on the properties of large language models' reasoning processes.",
|
| 26 |
-
"audience": [
|
| 27 |
-
"Researchers working on language model capabilities and mathematical reasoning"
|
| 28 |
-
],
|
| 29 |
-
"tasks": [
|
| 30 |
-
"Solving grade school math word problems",
|
| 31 |
-
"Text generation for question answering"
|
| 32 |
-
],
|
| 33 |
-
"limitations": "Even the largest models struggle with high test performance on this dataset, and autoregressive models have no mechanism to correct their own errors during solution generation.",
|
| 34 |
-
"out_of_scope_uses": [
|
| 35 |
-
"Not specified"
|
| 36 |
-
]
|
| 37 |
-
},
|
| 38 |
-
"data": {
|
| 39 |
-
"source": "The dataset was created by hiring freelance contractors via Upwork and then scaled using the NLP data labeling platform Surge AI. Problems and solutions were written by these contractors.",
|
| 40 |
-
"size": "8.5K (8,500) problems, with a size category of 10K<n<100K. The training set contains 7,473 examples and the test set contains 1,319 examples.",
|
| 41 |
-
"format": "parquet. The data is structured with a 'Problem:' field followed by a 'Solution:' field, where the solution includes step-by-step reasoning with intermediate calculations in special tags (e.g., `<<4*2=8>>`) and ends with a 'Final Answer:'.",
|
| 42 |
-
"annotation": "Contractors wrote the problems and solutions. For verification, different workers re-solved all problems to check agreement with the original solutions; problematic problems were either repaired or discarded. The annotators were from Surge AI."
|
| 43 |
-
},
|
| 44 |
-
"methodology": {
|
| 45 |
-
"methods": [
|
| 46 |
-
"Models are evaluated by generating step-by-step solutions to math word problems. The dataset provides two answer formats: a standard step-by-step solution and a solution structured with Socratic sub-questions.",
|
| 47 |
-
"The paper proposes a verification method where a separate verifier model is trained to judge the correctness of generated solutions. At test time, multiple candidate solutions are generated, and the one ranked highest by the verifier is selected."
|
| 48 |
-
],
|
| 49 |
-
"metrics": [
|
| 50 |
-
"GSM8K"
|
| 51 |
-
],
|
| 52 |
-
"calculation": "The GSM8K metric is a continuous score where higher values are better. It is described as 'EM on GSM8K', indicating it measures exact match accuracy.",
|
| 53 |
-
"interpretation": "Higher scores indicate better performance. The score is not bounded, but typical model performance ranges from low to high, with the highest reported score being 75.2.",
|
| 54 |
-
"baseline_results": "Paper baselines: The paper notes that even the largest transformer models fail to achieve high test performance, but does not report specific scores. EEE results: Llama 3.1 8B Instruct scored 75.2, and Yi 34B scored 0.648 on GSM8K.",
|
| 55 |
-
"validation": "The paper provides empirical evidence that the verification method scales more effectively with data than a finetuning baseline and remains effective even with a verifier much smaller than the generator."
|
| 56 |
-
},
|
| 57 |
-
"ethical_and_legal_considerations": {
|
| 58 |
-
"privacy_and_anonymity": "Not specified",
|
| 59 |
-
"data_licensing": "MIT License",
|
| 60 |
-
"consent_procedures": "Not specified",
|
| 61 |
-
"compliance_with_regulations": "Not specified"
|
| 62 |
-
},
|
| 63 |
-
"possible_risks": [
|
| 64 |
-
{
|
| 65 |
-
"category": "Over- or under-reliance",
|
| 66 |
-
"description": [
|
| 67 |
-
"In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
|
| 68 |
-
],
|
| 69 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
|
| 70 |
-
},
|
| 71 |
-
{
|
| 72 |
-
"category": "Data bias",
|
| 73 |
-
"description": [
|
| 74 |
-
"Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
|
| 75 |
-
],
|
| 76 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
|
| 77 |
-
},
|
| 78 |
-
{
|
| 79 |
-
"category": "Reproducibility",
|
| 80 |
-
"description": [
|
| 81 |
-
"Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI."
|
| 82 |
-
],
|
| 83 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html"
|
| 84 |
-
},
|
| 85 |
-
{
|
| 86 |
-
"category": "Incomplete advice",
|
| 87 |
-
"description": [
|
| 88 |
-
"When a model provides advice without having enough information, resulting in possible harm if the advice is followed."
|
| 89 |
-
],
|
| 90 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-advice.html"
|
| 91 |
-
},
|
| 92 |
-
{
|
| 93 |
-
"category": "Improper usage",
|
| 94 |
-
"description": [
|
| 95 |
-
"Improper usage occurs when a model is used for a purpose that it was not originally designed for."
|
| 96 |
-
],
|
| 97 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html"
|
| 98 |
-
}
|
| 99 |
-
],
|
| 100 |
-
"flagged_fields": {},
|
| 101 |
-
"missing_fields": [
|
| 102 |
-
"benchmark_details.similar_benchmarks",
|
| 103 |
-
"purpose_and_intended_users.out_of_scope_uses",
|
| 104 |
-
"ethical_and_legal_considerations.privacy_and_anonymity",
|
| 105 |
-
"ethical_and_legal_considerations.consent_procedures",
|
| 106 |
-
"ethical_and_legal_considerations.compliance_with_regulations"
|
| 107 |
-
],
|
| 108 |
-
"card_info": {
|
| 109 |
-
"created_at": "2026-03-17T15:37:16.459776",
|
| 110 |
-
"llm": "deepseek-ai/DeepSeek-V3.2"
|
| 111 |
-
}
|
| 112 |
-
},
|
| 113 |
-
"LegalBench": {
|
| 114 |
-
"benchmark_details": {
|
| 115 |
-
"name": "LEGALBENCH",
|
| 116 |
-
"overview": "LEGALBENCH is a benchmark designed to measure the legal reasoning capabilities of large language models. It comprises 162 tasks collaboratively constructed and hand-crafted by legal professionals, covering six distinct types of legal reasoning.",
|
| 117 |
-
"data_type": "text",
|
| 118 |
-
"domains": [
|
| 119 |
-
"legal",
|
| 120 |
-
"law",
|
| 121 |
-
"finance"
|
| 122 |
-
],
|
| 123 |
-
"languages": [
|
| 124 |
-
"English"
|
| 125 |
-
],
|
| 126 |
-
"similar_benchmarks": [
|
| 127 |
-
"GLUE",
|
| 128 |
-
"HELM",
|
| 129 |
-
"BigBench",
|
| 130 |
-
"RAFT"
|
| 131 |
-
],
|
| 132 |
-
"resources": [
|
| 133 |
-
"https://arxiv.org/abs/2308.11462",
|
| 134 |
-
"https://huggingface.co/datasets/nguha/legalbench",
|
| 135 |
-
"https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
|
| 136 |
-
]
|
| 137 |
-
},
|
| 138 |
-
"purpose_and_intended_users": {
|
| 139 |
-
"goal": "To enable greater study of what types of legal reasoning large language models (LLMs) can perform.",
|
| 140 |
-
"audience": [
|
| 141 |
-
"Practitioners (to integrate LLMs into workflows)",
|
| 142 |
-
"Legal academics",
|
| 143 |
-
"Computer scientists"
|
| 144 |
-
],
|
| 145 |
-
"tasks": [
|
| 146 |
-
"Text classification",
|
| 147 |
-
"Question answering",
|
| 148 |
-
"Text generation",
|
| 149 |
-
"Rule-application tasks"
|
| 150 |
-
],
|
| 151 |
-
"limitations": "The tasks do not generalize to all legal reasoning tasks or all types of legal documents, offering only a preliminary understanding of LLM performance.",
|
| 152 |
-
"out_of_scope_uses": [
|
| 153 |
-
"Predicting the legality of real-world events",
|
| 154 |
-
"Predicting the outcome of lawsuits",
|
| 155 |
-
"Providing legal advice"
|
| 156 |
-
]
|
| 157 |
-
},
|
| 158 |
-
"data": {
|
| 159 |
-
"source": "Data is drawn from three categories: existing publicly available datasets and corpora (some reformatted), datasets previously created by legal professionals but not released, and tasks developed specifically for LegalBench. The tasks originate from 36 distinct corpora.",
|
| 160 |
-
"size": "The benchmark comprises 162 tasks. The distribution of tasks by sample count is: 28 tasks have 50-100 samples, 97 have 100-500 samples, 29 have 500-2000 samples, and 8 have 2000+ samples. The overall dataset falls into the size category of 10,000 to 100,000 examples.",
|
| 161 |
-
"format": "Examples are presented in a structured format with fields such as 'Task name', 'Question', 'Options', and 'Answer'.",
|
| 162 |
-
"annotation": "Annotation procedures are task-dependent. For certain tasks, each data point was manually validated by a law-trained expert. Detailed annotation methodology for each task is documented in a separate section of the paper."
|
| 163 |
-
},
|
| 164 |
-
"methodology": {
|
| 165 |
-
"methods": [
|
| 166 |
-
"Models are evaluated in a few-shot setting. Train splits consist of a small random sample of between 2 to 8 instances to capture a true few-shot learning scenario.",
|
| 167 |
-
"For rule-application tasks, a law-trained expert manually validates each model generation."
|
| 168 |
-
],
|
| 169 |
-
"metrics": [
|
| 170 |
-
"LegalBench",
|
| 171 |
-
"Correctness",
|
| 172 |
-
"Analysis"
|
| 173 |
-
],
|
| 174 |
-
"calculation": "The primary benchmark metric is Exact Match (EM) on LegalBench. For rule-application tasks, two separate metrics are computed: 'correctness' (the proportion of generations without errors) and 'analysis'.",
|
| 175 |
-
"interpretation": "Higher scores on the LegalBench metric indicate better performance. The metric is continuous and lower scores are not better.",
|
| 176 |
-
"baseline_results": "The original paper evaluated 20 LLMs from 11 different families but did not provide specific scores. In a separate evaluation suite, the Yi 34B model achieved a score of 0.618.",
|
| 177 |
-
"validation": "For rule-application tasks, a law-trained expert manually validated each model generation. For datasets reused or adapted from other sources, the original data sheets document any redactions or missing data."
|
| 178 |
-
},
|
| 179 |
-
"ethical_and_legal_considerations": {
|
| 180 |
-
"privacy_and_anonymity": "For tasks that reuse or adapt existing datasets, the benchmark refers to the original data sheets for details on any data redactions or missing information.",
|
| 181 |
-
"data_licensing": "other",
|
| 182 |
-
"consent_procedures": "Not specified.",
|
| 183 |
-
"compliance_with_regulations": "The benchmark includes a section for each task intended to provide information relevant to ethical review processes, but specific details are not provided in the available facts."
|
| 184 |
-
},
|
| 185 |
-
"possible_risks": [
|
| 186 |
-
{
|
| 187 |
-
"category": "Over- or under-reliance",
|
| 188 |
-
"description": [
|
| 189 |
-
"In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
|
| 190 |
-
],
|
| 191 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
|
| 192 |
-
},
|
| 193 |
-
{
|
| 194 |
-
"category": "Unrepresentative data",
|
| 195 |
-
"description": [
|
| 196 |
-
"Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
|
| 197 |
-
],
|
| 198 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
|
| 199 |
-
},
|
| 200 |
-
{
|
| 201 |
-
"category": "Data bias",
|
| 202 |
-
"description": [
|
| 203 |
-
"Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
|
| 204 |
-
],
|
| 205 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
|
| 206 |
-
},
|
| 207 |
-
{
|
| 208 |
-
"category": "Lack of data transparency",
|
| 209 |
-
"description": [
|
| 210 |
-
"Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
|
| 211 |
-
],
|
| 212 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
|
| 213 |
-
},
|
| 214 |
-
{
|
| 215 |
-
"category": "Improper usage",
|
| 216 |
-
"description": [
|
| 217 |
-
"Improper usage occurs when a model is used for a purpose that it was not originally designed for."
|
| 218 |
-
],
|
| 219 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html"
|
| 220 |
-
}
|
| 221 |
-
],
|
| 222 |
-
"flagged_fields": {},
|
| 223 |
-
"missing_fields": [
|
| 224 |
-
"ethical_and_legal_considerations.consent_procedures"
|
| 225 |
-
],
|
| 226 |
-
"card_info": {
|
| 227 |
-
"created_at": "2026-03-17T12:59:10.203815",
|
| 228 |
-
"llm": "deepseek-ai/DeepSeek-V3.2"
|
| 229 |
-
}
|
| 230 |
-
},
|
| 231 |
-
"MedQA": {
|
| 232 |
-
"benchmark_details": {
|
| 233 |
-
"name": "MEDQA",
|
| 234 |
-
"overview": "MEDQA is a free-form multiple-choice open-domain question answering (OpenQA) benchmark designed to measure a model's ability to solve medical problems. It is distinctive as the first such dataset sourced from professional medical board exams, covering multiple languages and presenting a challenging real-world scenario.",
|
| 235 |
-
"data_type": "text",
|
| 236 |
-
"domains": [
|
| 237 |
-
"medical knowledge",
|
| 238 |
-
"professional medical exams"
|
| 239 |
-
],
|
| 240 |
-
"languages": [
|
| 241 |
-
"English"
|
| 242 |
-
],
|
| 243 |
-
"similar_benchmarks": [
|
| 244 |
-
"ARC",
|
| 245 |
-
"OpenBookQA"
|
| 246 |
-
],
|
| 247 |
-
"resources": [
|
| 248 |
-
"https://github.com/jind11/MedQA",
|
| 249 |
-
"https://arxiv.org/abs/2009.13081",
|
| 250 |
-
"https://huggingface.co/datasets/GBaker/MedQA-USMLE-4-options",
|
| 251 |
-
"https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json"
|
| 252 |
-
]
|
| 253 |
-
},
|
| 254 |
-
"purpose_and_intended_users": {
|
| 255 |
-
"goal": "To present a challenging open-domain question answering dataset to promote the development of stronger models capable of handling sophisticated real-world medical scenarios, specifically evaluating performance on medical knowledge as tested in professional exams.",
|
| 256 |
-
"audience": [
|
| 257 |
-
"The natural language processing (NLP) community"
|
| 258 |
-
],
|
| 259 |
-
"tasks": [
|
| 260 |
-
"Free-form multiple-choice question answering",
|
| 261 |
-
"Open-domain question answering"
|
| 262 |
-
],
|
| 263 |
-
"limitations": "Even the best current methods achieve relatively low accuracy (36.7% to 70.1% across languages), indicating the benchmark's difficulty and the limitations of existing models.",
|
| 264 |
-
"out_of_scope_uses": [
|
| 265 |
-
"Not specified"
|
| 266 |
-
]
|
| 267 |
-
},
|
| 268 |
-
"data": {
|
| 269 |
-
"source": "The data is collected from professional medical board exams.",
|
| 270 |
-
"size": "The dataset contains 12,723 questions in English, 34,251 in simplified Chinese, and 14,123 in traditional Chinese. The total number of examples falls within the 10K to 100K range.",
|
| 271 |
-
"format": "JSON",
|
| 272 |
-
"annotation": "The answer labels are the correct answers from the professional exams. No additional annotation process is described."
|
| 273 |
-
},
|
| 274 |
-
"methodology": {
|
| 275 |
-
"methods": [
|
| 276 |
-
"The benchmark uses a sequential combination of a document retriever and a machine comprehension model. It includes both rule-based and neural methods.",
|
| 277 |
-
"The evaluation is a standard question-answering task, though the specific learning setting (e.g., zero-shot, few-shot, fine-tuning) is not explicitly defined."
|
| 278 |
-
],
|
| 279 |
-
"metrics": [
|
| 280 |
-
"Accuracy"
|
| 281 |
-
],
|
| 282 |
-
"calculation": "The overall score is the accuracy on the test set.",
|
| 283 |
-
"interpretation": "Higher accuracy indicates better performance. The best reported accuracies are 36.7% for English, 42.0% for traditional Chinese, and 70.1% for simplified Chinese questions.",
|
| 284 |
-
"baseline_results": "Original paper baselines: The best method reported achieves 36.7% accuracy on English, 42.0% on traditional Chinese, and 70.1% on simplified Chinese questions. Model names are not specified. EEE results: Yi 34B achieves a score of 0.656 (65.6%).",
|
| 285 |
-
"validation": "Not specified"
|
| 286 |
-
},
|
| 287 |
-
"ethical_and_legal_considerations": {
|
| 288 |
-
"privacy_and_anonymity": "Not specified",
|
| 289 |
-
"data_licensing": "Creative Commons Attribution 4.0",
|
| 290 |
-
"consent_procedures": "Not specified",
|
| 291 |
-
"compliance_with_regulations": "Not specified"
|
| 292 |
-
},
|
| 293 |
-
"possible_risks": [
|
| 294 |
-
{
|
| 295 |
-
"category": "Over- or under-reliance",
|
| 296 |
-
"description": [
|
| 297 |
-
"In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should."
|
| 298 |
-
],
|
| 299 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html"
|
| 300 |
-
},
|
| 301 |
-
{
|
| 302 |
-
"category": "Unrepresentative data",
|
| 303 |
-
"description": [
|
| 304 |
-
"Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios."
|
| 305 |
-
],
|
| 306 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html"
|
| 307 |
-
},
|
| 308 |
-
{
|
| 309 |
-
"category": "Uncertain data provenance",
|
| 310 |
-
"description": [
|
| 311 |
-
"Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation."
|
| 312 |
-
],
|
| 313 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html"
|
| 314 |
-
},
|
| 315 |
-
{
|
| 316 |
-
"category": "Data bias",
|
| 317 |
-
"description": [
|
| 318 |
-
"Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods."
|
| 319 |
-
],
|
| 320 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html"
|
| 321 |
-
},
|
| 322 |
-
{
|
| 323 |
-
"category": "Lack of data transparency",
|
| 324 |
-
"description": [
|
| 325 |
-
"Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. "
|
| 326 |
-
],
|
| 327 |
-
"url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html"
|
| 328 |
-
}
|
| 329 |
-
],
|
| 330 |
-
"flagged_fields": {},
|
| 331 |
-
"missing_fields": [
|
| 332 |
-
"purpose_and_intended_users.out_of_scope_uses",
|
| 333 |
-
"methodology.validation",
|
| 334 |
-
"ethical_and_legal_considerations.privacy_and_anonymity",
|
| 335 |
-
"ethical_and_legal_considerations.consent_procedures",
|
| 336 |
-
"ethical_and_legal_considerations.compliance_with_regulations"
|
| 337 |
-
],
|
| 338 |
-
"card_info": {
|
| 339 |
-
"created_at": "2026-03-17T13:23:29.822123",
|
| 340 |
-
"llm": "deepseek-ai/DeepSeek-V3.2"
|
| 341 |
-
}
|
| 342 |
-
}
|
| 343 |
-
},
|
| 344 |
-
"models": [
|
| 345 |
-
{
|
| 346 |
-
"model_id": "01-ai/yi-34b",
|
| 347 |
-
"name": "Yi 34B",
|
| 348 |
-
"developer": "01-ai",
|
| 349 |
-
"scores": {
|
| 350 |
-
"Mean win rate": 0.57,
|
| 351 |
-
"NarrativeQA": 0.782,
|
| 352 |
-
"NaturalQuestions (closed-book)": 0.443,
|
| 353 |
-
"OpenbookQA": 0.92,
|
| 354 |
-
"MMLU": 0.65,
|
| 355 |
-
"MATH": 0.375,
|
| 356 |
-
"GSM8K": 0.648,
|
| 357 |
-
"LegalBench": 0.618,
|
| 358 |
-
"MedQA": 0.656,
|
| 359 |
-
"WMT 2014": 0.172
|
| 360 |
-
}
|
| 361 |
-
},
|
| 362 |
-
{
|
| 363 |
-
"model_id": "01-ai/yi-6b",
|
| 364 |
-
"name": "Yi 6B",
|
| 365 |
-
"developer": "01-ai",
|
| 366 |
-
"scores": {
|
| 367 |
-
"Mean win rate": 0.253,
|
| 368 |
-
"NarrativeQA": 0.702,
|
| 369 |
-
"NaturalQuestions (closed-book)": 0.31,
|
| 370 |
-
"OpenbookQA": 0.8,
|
| 371 |
-
"MMLU": 0.53,
|
| 372 |
-
"MATH": 0.126,
|
| 373 |
-
"GSM8K": 0.375,
|
| 374 |
-
"LegalBench": 0.519,
|
| 375 |
-
"MedQA": 0.497,
|
| 376 |
-
"WMT 2014": 0.117
|
| 377 |
-
}
|
| 378 |
-
},
|
| 379 |
-
{
|
| 380 |
-
"model_id": "01-ai/yi-large-preview",
|
| 381 |
-
"name": "Yi Large Preview",
|
| 382 |
-
"developer": "01-ai",
|
| 383 |
-
"scores": {
|
| 384 |
-
"Mean win rate": 0.471,
|
| 385 |
-
"NarrativeQA": 0.373,
|
| 386 |
-
"NaturalQuestions (closed-book)": 0.428,
|
| 387 |
-
"OpenbookQA": 0.946,
|
| 388 |
-
"MMLU": 0.712,
|
| 389 |
-
"MATH": 0.712,
|
| 390 |
-
"GSM8K": 0.69,
|
| 391 |
-
"LegalBench": 0.519,
|
| 392 |
-
"MedQA": 0.66,
|
| 393 |
-
"WMT 2014": 0.176
|
| 394 |
-
}
|
| 395 |
-
},
|
| 396 |
-
{
|
| 397 |
-
"model_id": "AlephAlpha/luminous-base",
|
| 398 |
-
"name": "Luminous Base 13B",
|
| 399 |
-
"developer": "AlephAlpha",
|
| 400 |
-
"scores": {
|
| 401 |
-
"Mean win rate": 0.041,
|
| 402 |
-
"NarrativeQA": 0.633,
|
| 403 |
-
"NaturalQuestions (closed-book)": 0.197,
|
| 404 |
-
"OpenbookQA": 0.286,
|
| 405 |
-
"MMLU": 0.243,
|
| 406 |
-
"MATH": 0.026,
|
| 407 |
-
"GSM8K": 0.028,
|
| 408 |
-
"LegalBench": 0.332,
|
| 409 |
-
"MedQA": 0.26,
|
| 410 |
-
"WMT 2014": 0.066
|
| 411 |
-
}
|
| 412 |
-
},
|
| 413 |
-
{
|
| 414 |
-
"model_id": "AlephAlpha/luminous-extended",
|
| 415 |
-
"name": "Luminous Extended 30B",
|
| 416 |
-
"developer": "AlephAlpha",
|
| 417 |
-
"scores": {
|
| 418 |
-
"Mean win rate": 0.078,
|
| 419 |
-
"NarrativeQA": 0.684,
|
| 420 |
-
"NaturalQuestions (closed-book)": 0.253,
|
| 421 |
-
"OpenbookQA": 0.272,
|
| 422 |
-
"MMLU": 0.248,
|
| 423 |
-
"MATH": 0.04,
|
| 424 |
-
"GSM8K": 0.075,
|
| 425 |
-
"LegalBench": 0.421,
|
| 426 |
-
"MedQA": 0.276,
|
| 427 |
-
"WMT 2014": 0.083
|
| 428 |
-
}
|
| 429 |
-
},
|
| 430 |
-
{
|
| 431 |
-
"model_id": "AlephAlpha/luminous-supreme",
|
| 432 |
-
"name": "Luminous Supreme 70B",
|
| 433 |
-
"developer": "AlephAlpha",
|
| 434 |
-
"scores": {
|
| 435 |
-
"Mean win rate": 0.145,
|
| 436 |
-
"NarrativeQA": 0.743,
|
| 437 |
-
"NaturalQuestions (closed-book)": 0.299,
|
| 438 |
-
"OpenbookQA": 0.284,
|
| 439 |
-
"MMLU": 0.316,
|
| 440 |
-
"MATH": 0.078,
|
| 441 |
-
"GSM8K": 0.137,
|
| 442 |
-
"LegalBench": 0.452,
|
| 443 |
-
"MedQA": 0.276,
|
| 444 |
-
"WMT 2014": 0.102
|
| 445 |
-
}
|
| 446 |
-
},
|
| 447 |
-
{
|
| 448 |
-
"model_id": "ai21/j2-grande",
|
| 449 |
-
"name": "Jurassic-2 Grande 17B",
|
| 450 |
-
"developer": "ai21",
|
| 451 |
-
"scores": {
|
| 452 |
-
"Mean win rate": 0.172,
|
| 453 |
-
"NarrativeQA": 0.744,
|
| 454 |
-
"NaturalQuestions (closed-book)": 0.35,
|
| 455 |
-
"OpenbookQA": 0.614,
|
| 456 |
-
"MMLU": 0.471,
|
| 457 |
-
"MATH": 0.064,
|
| 458 |
-
"GSM8K": 0.159,
|
| 459 |
-
"LegalBench": 0.468,
|
| 460 |
-
"MedQA": 0.39,
|
| 461 |
-
"WMT 2014": 0.102
|
| 462 |
-
}
|
| 463 |
-
},
|
| 464 |
-
{
|
| 465 |
-
"model_id": "ai21/j2-jumbo",
|
| 466 |
-
"name": "Jurassic-2 Jumbo 178B",
|
| 467 |
-
"developer": "ai21",
|
| 468 |
-
"scores": {
|
| 469 |
-
"Mean win rate": 0.215,
|
| 470 |
-
"NarrativeQA": 0.728,
|
| 471 |
-
"NaturalQuestions (closed-book)": 0.385,
|
| 472 |
-
"OpenbookQA": 0.688,
|
| 473 |
-
"MMLU": 0.483,
|
| 474 |
-
"MATH": 0.103,
|
| 475 |
-
"GSM8K": 0.239,
|
| 476 |
-
"LegalBench": 0.533,
|
| 477 |
-
"MedQA": 0.431,
|
| 478 |
-
"WMT 2014": 0.114
|
| 479 |
-
}
|
| 480 |
-
},
|
| 481 |
-
{
|
| 482 |
-
"model_id": "ai21/jamba-1.5-large",
|
| 483 |
-
"name": "Jamba 1.5 Large",
|
| 484 |
-
"developer": "ai21",
|
| 485 |
-
"scores": {
|
| 486 |
-
"Mean win rate": 0.637,
|
| 487 |
-
"NarrativeQA": 0.664,
|
| 488 |
-
"NaturalQuestions (closed-book)": 0.394,
|
| 489 |
-
"OpenbookQA": 0.948,
|
| 490 |
-
"MMLU": 0.683,
|
| 491 |
-
"MATH": 0.692,
|
| 492 |
-
"GSM8K": 0.846,
|
| 493 |
-
"LegalBench": 0.675,
|
| 494 |
-
"MedQA": 0.698,
|
| 495 |
-
"WMT 2014": 0.203
|
| 496 |
-
}
|
| 497 |
-
},
|
| 498 |
-
{
|
| 499 |
-
"model_id": "ai21/jamba-1.5-mini",
|
| 500 |
-
"name": "Jamba 1.5 Mini",
|
| 501 |
-
"developer": "ai21",
|
| 502 |
-
"scores": {
|
| 503 |
-
"Mean win rate": 0.414,
|
| 504 |
-
"NarrativeQA": 0.746,
|
| 505 |
-
"NaturalQuestions (closed-book)": 0.388,
|
| 506 |
-
"OpenbookQA": 0.89,
|
| 507 |
-
"MMLU": 0.582,
|
| 508 |
-
"MATH": 0.318,
|
| 509 |
-
"GSM8K": 0.691,
|
| 510 |
-
"LegalBench": 0.503,
|
| 511 |
-
"MedQA": 0.632,
|
| 512 |
-
"WMT 2014": 0.179
|
| 513 |
-
}
|
| 514 |
-
},
|
| 515 |
-
{
|
| 516 |
-
"model_id": "ai21/jamba-instruct",
|
| 517 |
-
"name": "Jamba Instruct",
|
| 518 |
-
"developer": "ai21",
|
| 519 |
-
"scores": {
|
| 520 |
-
"Mean win rate": 0.287,
|
| 521 |
-
"NarrativeQA": 0.658,
|
| 522 |
-
"NaturalQuestions (closed-book)": 0.384,
|
| 523 |
-
"OpenbookQA": 0.796,
|
| 524 |
-
"MMLU": 0.582,
|
| 525 |
-
"MATH": 0.38,
|
| 526 |
-
"GSM8K": 0.67,
|
| 527 |
-
"LegalBench": 0.54,
|
| 528 |
-
"MedQA": 0.519,
|
| 529 |
-
"WMT 2014": 0.164
|
| 530 |
-
}
|
| 531 |
-
},
|
| 532 |
-
{
|
| 533 |
-
"model_id": "allenai/olmo-7b",
|
| 534 |
-
"name": "OLMo 7B",
|
| 535 |
-
"developer": "allenai",
|
| 536 |
-
"scores": {
|
| 537 |
-
"Mean win rate": 0.052,
|
| 538 |
-
"NarrativeQA": 0.597,
|
| 539 |
-
"NaturalQuestions (closed-book)": 0.259,
|
| 540 |
-
"OpenbookQA": 0.222,
|
| 541 |
-
"MMLU": 0.305,
|
| 542 |
-
"MATH": 0.029,
|
| 543 |
-
"GSM8K": 0.044,
|
| 544 |
-
"LegalBench": 0.341,
|
| 545 |
-
"MedQA": 0.229,
|
| 546 |
-
"WMT 2014": 0.097
|
| 547 |
-
}
|
| 548 |
-
},
|
| 549 |
-
{
|
| 550 |
-
"model_id": "amazon/nova-lite-v1:0",
|
| 551 |
-
"name": "Amazon Nova Lite",
|
| 552 |
-
"developer": "amazon",
|
| 553 |
-
"scores": {
|
| 554 |
-
"Mean win rate": 0.708,
|
| 555 |
-
"NarrativeQA": 0.768,
|
| 556 |
-
"NaturalQuestions (closed-book)": 0.352,
|
| 557 |
-
"OpenbookQA": 0.928,
|
| 558 |
-
"MMLU": 0.693,
|
| 559 |
-
"MATH": 0.779,
|
| 560 |
-
"GSM8K": 0.829,
|
| 561 |
-
"LegalBench": 0.659,
|
| 562 |
-
"MedQA": 0.696,
|
| 563 |
-
"WMT 2014": 0.204
|
| 564 |
-
}
|
| 565 |
-
},
|
| 566 |
-
{
|
| 567 |
-
"model_id": "amazon/nova-micro-v1:0",
|
| 568 |
-
"name": "Amazon Nova Micro",
|
| 569 |
-
"developer": "amazon",
|
| 570 |
-
"scores": {
|
| 571 |
-
"Mean win rate": 0.524,
|
| 572 |
-
"NarrativeQA": 0.744,
|
| 573 |
-
"NaturalQuestions (closed-book)": 0.285,
|
| 574 |
-
"OpenbookQA": 0.888,
|
| 575 |
-
"MMLU": 0.64,
|
| 576 |
-
"MATH": 0.76,
|
| 577 |
-
"GSM8K": 0.794,
|
| 578 |
-
"LegalBench": 0.615,
|
| 579 |
-
"MedQA": 0.608,
|
| 580 |
-
"WMT 2014": 0.192
|
| 581 |
-
}
|
| 582 |
-
},
|
| 583 |
-
{
|
| 584 |
-
"model_id": "amazon/nova-pro-v1:0",
|
| 585 |
-
"name": "Amazon Nova Pro",
|
| 586 |
-
"developer": "amazon",
|
| 587 |
-
"scores": {
|
| 588 |
-
"Mean win rate": 0.885,
|
| 589 |
-
"NarrativeQA": 0.791,
|
| 590 |
-
"NaturalQuestions (closed-book)": 0.405,
|
| 591 |
-
"OpenbookQA": 0.96,
|
| 592 |
-
"MMLU": 0.758,
|
| 593 |
-
"MATH": 0.821,
|
| 594 |
-
"GSM8K": 0.87,
|
| 595 |
-
"LegalBench": 0.736,
|
| 596 |
-
"MedQA": 0.811,
|
| 597 |
-
"WMT 2014": 0.229
|
| 598 |
-
}
|
| 599 |
-
},
|
| 600 |
-
{
|
| 601 |
-
"model_id": "anthropic/claude-2.0",
|
| 602 |
-
"name": "Claude 2.0",
|
| 603 |
-
"developer": "Anthropic",
|
| 604 |
-
"scores": {
|
| 605 |
-
"Mean win rate": 0.489,
|
| 606 |
-
"NarrativeQA": 0.718,
|
| 607 |
-
"NaturalQuestions (closed-book)": 0.428,
|
| 608 |
-
"OpenbookQA": 0.862,
|
| 609 |
-
"MMLU": 0.639,
|
| 610 |
-
"MATH": 0.603,
|
| 611 |
-
"GSM8K": 0.583,
|
| 612 |
-
"LegalBench": 0.643,
|
| 613 |
-
"MedQA": 0.652,
|
| 614 |
-
"WMT 2014": 0.219
|
| 615 |
-
}
|
| 616 |
-
},
|
| 617 |
-
{
|
| 618 |
-
"model_id": "anthropic/claude-2.1",
|
| 619 |
-
"name": "Claude 2.1",
|
| 620 |
-
"developer": "Anthropic",
|
| 621 |
-
"scores": {
|
| 622 |
-
"Mean win rate": 0.437,
|
| 623 |
-
"NarrativeQA": 0.677,
|
| 624 |
-
"NaturalQuestions (closed-book)": 0.375,
|
| 625 |
-
"OpenbookQA": 0.872,
|
| 626 |
-
"MMLU": 0.643,
|
| 627 |
-
"MATH": 0.632,
|
| 628 |
-
"GSM8K": 0.604,
|
| 629 |
-
"LegalBench": 0.643,
|
| 630 |
-
"MedQA": 0.644,
|
| 631 |
-
"WMT 2014": 0.204
|
| 632 |
-
}
|
| 633 |
-
},
|
| 634 |
-
{
|
| 635 |
-
"model_id": "anthropic/claude-3-5-haiku-20241022",
|
| 636 |
-
"name": "Claude 3.5 Haiku 20241022",
|
| 637 |
-
"developer": "Anthropic",
|
| 638 |
-
"scores": {
|
| 639 |
-
"Mean win rate": 0.531,
|
| 640 |
-
"NarrativeQA": 0.763,
|
| 641 |
-
"NaturalQuestions (closed-book)": 0.344,
|
| 642 |
-
"OpenbookQA": 0.854,
|
| 643 |
-
"MMLU": 0.671,
|
| 644 |
-
"MATH": 0.872,
|
| 645 |
-
"GSM8K": 0.815,
|
| 646 |
-
"LegalBench": 0.631,
|
| 647 |
-
"MedQA": 0.722,
|
| 648 |
-
"WMT 2014": 0.135
|
| 649 |
-
}
|
| 650 |
-
},
|
| 651 |
-
{
|
| 652 |
-
"model_id": "anthropic/claude-3-5-sonnet-20240620",
|
| 653 |
-
"name": "Claude 3.5 Sonnet 20240620",
|
| 654 |
-
"developer": "Anthropic",
|
| 655 |
-
"scores": {
|
| 656 |
-
"Mean win rate": 0.885,
|
| 657 |
-
"NarrativeQA": 0.746,
|
| 658 |
-
"NaturalQuestions (closed-book)": 0.502,
|
| 659 |
-
"OpenbookQA": 0.972,
|
| 660 |
-
"MMLU": 0.799,
|
| 661 |
-
"MATH": 0.813,
|
| 662 |
-
"GSM8K": 0.949,
|
| 663 |
-
"LegalBench": 0.707,
|
| 664 |
-
"MedQA": 0.825,
|
| 665 |
-
"WMT 2014": 0.229
|
| 666 |
-
}
|
| 667 |
-
},
|
| 668 |
-
{
|
| 669 |
-
"model_id": "anthropic/claude-3-5-sonnet-20241022",
|
| 670 |
-
"name": "Claude 3.5 Sonnet 20241022",
|
| 671 |
-
"developer": "Anthropic",
|
| 672 |
-
"scores": {
|
| 673 |
-
"Mean win rate": 0.846,
|
| 674 |
-
"NarrativeQA": 0.77,
|
| 675 |
-
"NaturalQuestions (closed-book)": 0.467,
|
| 676 |
-
"OpenbookQA": 0.966,
|
| 677 |
-
"MMLU": 0.809,
|
| 678 |
-
"MATH": 0.904,
|
| 679 |
-
"GSM8K": 0.956,
|
| 680 |
-
"LegalBench": 0.647,
|
| 681 |
-
"MedQA": 0.859,
|
| 682 |
-
"WMT 2014": 0.226
|
| 683 |
-
}
|
| 684 |
-
},
|
| 685 |
-
{
|
| 686 |
-
"model_id": "anthropic/claude-3-haiku-20240307",
|
| 687 |
-
"name": "Claude 3 Haiku 20240307",
|
| 688 |
-
"developer": "Anthropic",
|
| 689 |
-
"scores": {
|
| 690 |
-
"Mean win rate": 0.263,
|
| 691 |
-
"NarrativeQA": 0.244,
|
| 692 |
-
"NaturalQuestions (closed-book)": 0.144,
|
| 693 |
-
"OpenbookQA": 0.838,
|
| 694 |
-
"MMLU": 0.662,
|
| 695 |
-
"MATH": 0.131,
|
| 696 |
-
"GSM8K": 0.699,
|
| 697 |
-
"LegalBench": 0.46,
|
| 698 |
-
"MedQA": 0.702,
|
| 699 |
-
"WMT 2014": 0.148
|
| 700 |
-
}
|
| 701 |
-
},
|
| 702 |
-
{
|
| 703 |
-
"model_id": "anthropic/claude-3-opus-20240229",
|
| 704 |
-
"name": "Claude 3 Opus 20240229",
|
| 705 |
-
"developer": "Anthropic",
|
| 706 |
-
"scores": {
|
| 707 |
-
"Mean win rate": 0.683,
|
| 708 |
-
"NarrativeQA": 0.351,
|
| 709 |
-
"NaturalQuestions (closed-book)": 0.441,
|
| 710 |
-
"OpenbookQA": 0.956,
|
| 711 |
-
"MMLU": 0.768,
|
| 712 |
-
"MATH": 0.76,
|
| 713 |
-
"GSM8K": 0.924,
|
| 714 |
-
"LegalBench": 0.662,
|
| 715 |
-
"MedQA": 0.775,
|
| 716 |
-
"WMT 2014": 0.24
|
| 717 |
-
}
|
| 718 |
-
},
|
| 719 |
-
{
|
| 720 |
-
"model_id": "anthropic/claude-3-sonnet-20240229",
|
| 721 |
-
"name": "Claude 3 Sonnet 20240229",
|
| 722 |
-
"developer": "Anthropic",
|
| 723 |
-
"scores": {
|
| 724 |
-
"Mean win rate": 0.377,
|
| 725 |
-
"NarrativeQA": 0.111,
|
| 726 |
-
"NaturalQuestions (closed-book)": 0.028,
|
| 727 |
-
"OpenbookQA": 0.918,
|
| 728 |
-
"MMLU": 0.652,
|
| 729 |
-
"MATH": 0.084,
|
| 730 |
-
"GSM8K": 0.907,
|
| 731 |
-
"LegalBench": 0.49,
|
| 732 |
-
"MedQA": 0.684,
|
| 733 |
-
"WMT 2014": 0.218
|
| 734 |
-
}
|
| 735 |
-
},
|
| 736 |
-
{
|
| 737 |
-
"model_id": "anthropic/claude-instant-1.2",
|
| 738 |
-
"name": "Claude Instant 1.2",
|
| 739 |
-
"developer": "Anthropic",
|
| 740 |
-
"scores": {
|
| 741 |
-
"Mean win rate": 0.399,
|
| 742 |
-
"NarrativeQA": 0.616,
|
| 743 |
-
"NaturalQuestions (closed-book)": 0.343,
|
| 744 |
-
"OpenbookQA": 0.844,
|
| 745 |
-
"MMLU": 0.631,
|
| 746 |
-
"MATH": 0.499,
|
| 747 |
-
"GSM8K": 0.721,
|
| 748 |
-
"LegalBench": 0.586,
|
| 749 |
-
"MedQA": 0.559,
|
| 750 |
-
"WMT 2014": 0.194
|
| 751 |
-
}
|
| 752 |
-
},
|
| 753 |
-
{
|
| 754 |
-
"model_id": "anthropic/claude-v1.3",
|
| 755 |
-
"name": "Anthropic Claude v1.3",
|
| 756 |
-
"developer": "Anthropic",
|
| 757 |
-
"scores": {
|
| 758 |
-
"Mean win rate": 0.518,
|
| 759 |
-
"NarrativeQA": 0.723,
|
| 760 |
-
"NaturalQuestions (closed-book)": 0.409,
|
| 761 |
-
"OpenbookQA": 0.908,
|
| 762 |
-
"MMLU": 0.631,
|
| 763 |
-
"MATH": 0.54,
|
| 764 |
-
"GSM8K": 0.784,
|
| 765 |
-
"LegalBench": 0.629,
|
| 766 |
-
"MedQA": 0.618,
|
| 767 |
-
"WMT 2014": 0.219
|
| 768 |
-
}
|
| 769 |
-
},
|
| 770 |
-
{
|
| 771 |
-
"model_id": "cohere/command",
|
| 772 |
-
"name": "Command",
|
| 773 |
-
"developer": "cohere",
|
| 774 |
-
"scores": {
|
| 775 |
-
"Mean win rate": 0.327,
|
| 776 |
-
"NarrativeQA": 0.749,
|
| 777 |
-
"NaturalQuestions (closed-book)": 0.391,
|
| 778 |
-
"OpenbookQA": 0.774,
|
| 779 |
-
"MMLU": 0.525,
|
| 780 |
-
"MATH": 0.236,
|
| 781 |
-
"GSM8K": 0.452,
|
| 782 |
-
"LegalBench": 0.578,
|
| 783 |
-
"MedQA": 0.445,
|
| 784 |
-
"WMT 2014": 0.088
|
| 785 |
-
}
|
| 786 |
-
},
|
| 787 |
-
{
|
| 788 |
-
"model_id": "cohere/command-light",
|
| 789 |
-
"name": "Command Light",
|
| 790 |
-
"developer": "cohere",
|
| 791 |
-
"scores": {
|
| 792 |
-
"Mean win rate": 0.105,
|
| 793 |
-
"NarrativeQA": 0.629,
|
| 794 |
-
"NaturalQuestions (closed-book)": 0.195,
|
| 795 |
-
"OpenbookQA": 0.398,
|
| 796 |
-
"MMLU": 0.386,
|
| 797 |
-
"MATH": 0.098,
|
| 798 |
-
"GSM8K": 0.149,
|
| 799 |
-
"LegalBench": 0.397,
|
| 800 |
-
"MedQA": 0.312,
|
| 801 |
-
"WMT 2014": 0.023
|
| 802 |
-
}
|
| 803 |
-
},
|
| 804 |
-
{
|
| 805 |
-
"model_id": "cohere/command-r",
|
| 806 |
-
"name": "Command R",
|
| 807 |
-
"developer": "cohere",
|
| 808 |
-
"scores": {
|
| 809 |
-
"Mean win rate": 0.299,
|
| 810 |
-
"NarrativeQA": 0.742,
|
| 811 |
-
"NaturalQuestions (closed-book)": 0.352,
|
| 812 |
-
"OpenbookQA": 0.782,
|
| 813 |
-
"MMLU": 0.567,
|
| 814 |
-
"MATH": 0.266,
|
| 815 |
-
"GSM8K": 0.551,
|
| 816 |
-
"LegalBench": 0.507,
|
| 817 |
-
"MedQA": 0.555,
|
| 818 |
-
"WMT 2014": 0.149
|
| 819 |
-
}
|
| 820 |
-
},
|
| 821 |
-
{
|
| 822 |
-
"model_id": "cohere/command-r-plus",
|
| 823 |
-
"name": "Command R Plus",
|
| 824 |
-
"developer": "cohere",
|
| 825 |
-
"scores": {
|
| 826 |
-
"Mean win rate": 0.441,
|
| 827 |
-
"NarrativeQA": 0.735,
|
| 828 |
-
"NaturalQuestions (closed-book)": 0.343,
|
| 829 |
-
"OpenbookQA": 0.828,
|
| 830 |
-
"MMLU": 0.59,
|
| 831 |
-
"MATH": 0.403,
|
| 832 |
-
"GSM8K": 0.738,
|
| 833 |
-
"LegalBench": 0.672,
|
| 834 |
-
"MedQA": 0.567,
|
| 835 |
-
"WMT 2014": 0.203
|
| 836 |
-
}
|
| 837 |
-
},
|
| 838 |
-
{
|
| 839 |
-
"model_id": "databricks/dbrx-instruct",
|
| 840 |
-
"name": "DBRX Instruct",
|
| 841 |
-
"developer": "databricks",
|
| 842 |
-
"scores": {
|
| 843 |
-
"Mean win rate": 0.289,
|
| 844 |
-
"NarrativeQA": 0.488,
|
| 845 |
-
"NaturalQuestions (closed-book)": 0.284,
|
| 846 |
-
"OpenbookQA": 0.91,
|
| 847 |
-
"MMLU": 0.643,
|
| 848 |
-
"MATH": 0.358,
|
| 849 |
-
"GSM8K": 0.671,
|
| 850 |
-
"LegalBench": 0.426,
|
| 851 |
-
"MedQA": 0.694,
|
| 852 |
-
"WMT 2014": 0.131
|
| 853 |
-
}
|
| 854 |
-
},
|
| 855 |
-
{
|
| 856 |
-
"model_id": "deepseek-ai/deepseek-llm-67b-chat",
|
| 857 |
-
"name": "DeepSeek LLM Chat 67B",
|
| 858 |
-
"developer": "deepseek-ai",
|
| 859 |
-
"scores": {
|
| 860 |
-
"Mean win rate": 0.488,
|
| 861 |
-
"NarrativeQA": 0.581,
|
| 862 |
-
"NaturalQuestions (closed-book)": 0.412,
|
| 863 |
-
"OpenbookQA": 0.88,
|
| 864 |
-
"MMLU": 0.641,
|
| 865 |
-
"MATH": 0.615,
|
| 866 |
-
"GSM8K": 0.795,
|
| 867 |
-
"LegalBench": 0.637,
|
| 868 |
-
"MedQA": 0.628,
|
| 869 |
-
"WMT 2014": 0.186
|
| 870 |
-
}
|
| 871 |
-
},
|
| 872 |
-
{
|
| 873 |
-
"model_id": "deepseek-ai/deepseek-v3",
|
| 874 |
-
"name": "DeepSeek v3",
|
| 875 |
-
"developer": "deepseek-ai",
|
| 876 |
-
"scores": {
|
| 877 |
-
"Mean win rate": 0.908,
|
| 878 |
-
"NarrativeQA": 0.796,
|
| 879 |
-
"NaturalQuestions (closed-book)": 0.467,
|
| 880 |
-
"OpenbookQA": 0.954,
|
| 881 |
-
"MMLU": 0.803,
|
| 882 |
-
"MATH": 0.912,
|
| 883 |
-
"GSM8K": 0.94,
|
| 884 |
-
"LegalBench": 0.718,
|
| 885 |
-
"MedQA": 0.809,
|
| 886 |
-
"WMT 2014": 0.209
|
| 887 |
-
}
|
| 888 |
-
},
|
| 889 |
-
{
|
| 890 |
-
"model_id": "google/gemini-1.0-pro-002",
|
| 891 |
-
"name": "Gemini 1.0 Pro 002",
|
| 892 |
-
"developer": "Google",
|
| 893 |
-
"scores": {
|
| 894 |
-
"Mean win rate": 0.422,
|
| 895 |
-
"NarrativeQA": 0.751,
|
| 896 |
-
"NaturalQuestions (closed-book)": 0.391,
|
| 897 |
-
"OpenbookQA": 0.788,
|
| 898 |
-
"MMLU": 0.534,
|
| 899 |
-
"MATH": 0.665,
|
| 900 |
-
"GSM8K": 0.816,
|
| 901 |
-
"LegalBench": 0.475,
|
| 902 |
-
"MedQA": 0.483,
|
| 903 |
-
"WMT 2014": 0.194
|
| 904 |
-
}
|
| 905 |
-
},
|
| 906 |
-
{
|
| 907 |
-
"model_id": "google/gemini-1.5-flash-001",
|
| 908 |
-
"name": "Gemini 1.5 Flash 001",
|
| 909 |
-
"developer": "Google",
|
| 910 |
-
"scores": {
|
| 911 |
-
"Mean win rate": 0.667,
|
| 912 |
-
"NarrativeQA": 0.783,
|
| 913 |
-
"NaturalQuestions (closed-book)": 0.332,
|
| 914 |
-
"OpenbookQA": 0.928,
|
| 915 |
-
"MMLU": 0.703,
|
| 916 |
-
"MATH": 0.753,
|
| 917 |
-
"GSM8K": 0.785,
|
| 918 |
-
"LegalBench": 0.661,
|
| 919 |
-
"MedQA": 0.68,
|
| 920 |
-
"WMT 2014": 0.225
|
| 921 |
-
}
|
| 922 |
-
},
|
| 923 |
-
{
|
| 924 |
-
"model_id": "google/gemini-1.5-flash-002",
|
| 925 |
-
"name": "Gemini 1.5 Flash 002",
|
| 926 |
-
"developer": "Google",
|
| 927 |
-
"scores": {
|
| 928 |
-
"Mean win rate": 0.573,
|
| 929 |
-
"NarrativeQA": 0.746,
|
| 930 |
-
"NaturalQuestions (closed-book)": 0.323,
|
| 931 |
-
"OpenbookQA": 0.914,
|
| 932 |
-
"MMLU": 0.679,
|
| 933 |
-
"MATH": 0.908,
|
| 934 |
-
"GSM8K": 0.328,
|
| 935 |
-
"LegalBench": 0.67,
|
| 936 |
-
"MedQA": 0.656,
|
| 937 |
-
"WMT 2014": 0.212
|
| 938 |
-
}
|
| 939 |
-
},
|
| 940 |
-
{
|
| 941 |
-
"model_id": "google/gemini-1.5-pro-001",
|
| 942 |
-
"name": "Gemini 1.5 Pro 001",
|
| 943 |
-
"developer": "Google",
|
| 944 |
-
"scores": {
|
| 945 |
-
"Mean win rate": 0.739,
|
| 946 |
-
"NarrativeQA": 0.783,
|
| 947 |
-
"NaturalQuestions (closed-book)": 0.378,
|
| 948 |
-
"OpenbookQA": 0.902,
|
| 949 |
-
"MMLU": 0.772,
|
| 950 |
-
"MATH": 0.825,
|
| 951 |
-
"GSM8K": 0.836,
|
| 952 |
-
"LegalBench": 0.757,
|
| 953 |
-
"MedQA": 0.692,
|
| 954 |
-
"WMT 2014": 0.189
|
| 955 |
-
}
|
| 956 |
-
},
|
| 957 |
-
{
|
| 958 |
-
"model_id": "google/gemini-1.5-pro-002",
|
| 959 |
-
"name": "Gemini 1.5 Pro 002",
|
| 960 |
-
"developer": "Google",
|
| 961 |
-
"scores": {
|
| 962 |
-
"Mean win rate": 0.842,
|
| 963 |
-
"NarrativeQA": 0.756,
|
| 964 |
-
"NaturalQuestions (closed-book)": 0.455,
|
| 965 |
-
"OpenbookQA": 0.952,
|
| 966 |
-
"MMLU": 0.795,
|
| 967 |
-
"MATH": 0.92,
|
| 968 |
-
"GSM8K": 0.817,
|
| 969 |
-
"LegalBench": 0.747,
|
| 970 |
-
"MedQA": 0.771,
|
| 971 |
-
"WMT 2014": 0.231
|
| 972 |
-
}
|
| 973 |
-
},
|
| 974 |
-
{
|
| 975 |
-
"model_id": "google/gemini-2.0-flash-exp",
|
| 976 |
-
"name": "Gemini 2.0 Flash Experimental",
|
| 977 |
-
"developer": "Google",
|
| 978 |
-
"scores": {
|
| 979 |
-
"Mean win rate": 0.813,
|
| 980 |
-
"NarrativeQA": 0.783,
|
| 981 |
-
"NaturalQuestions (closed-book)": 0.443,
|
| 982 |
-
"OpenbookQA": 0.946,
|
| 983 |
-
"MMLU": 0.717,
|
| 984 |
-
"MATH": 0.901,
|
| 985 |
-
"GSM8K": 0.946,
|
| 986 |
-
"LegalBench": 0.674,
|
| 987 |
-
"MedQA": 0.73,
|
| 988 |
-
"WMT 2014": 0.212
|
| 989 |
-
}
|
| 990 |
-
},
|
| 991 |
-
{
|
| 992 |
-
"model_id": "google/gemma-2-27b-it",
|
| 993 |
-
"name": "Gemma 2 Instruct 27B",
|
| 994 |
-
"developer": "Google",
|
| 995 |
-
"scores": {
|
| 996 |
-
"Mean win rate": 0.675,
|
| 997 |
-
"NarrativeQA": 0.79,
|
| 998 |
-
"NaturalQuestions (closed-book)": 0.353,
|
| 999 |
-
"OpenbookQA": 0.918,
|
| 1000 |
-
"MMLU": 0.664,
|
| 1001 |
-
"MATH": 0.746,
|
| 1002 |
-
"GSM8K": 0.812,
|
| 1003 |
-
"LegalBench": 0.7,
|
| 1004 |
-
"MedQA": 0.684,
|
| 1005 |
-
"WMT 2014": 0.214
|
| 1006 |
-
}
|
| 1007 |
-
},
|
| 1008 |
-
{
|
| 1009 |
-
"model_id": "google/gemma-2-9b-it",
|
| 1010 |
-
"name": "Gemma 2 Instruct 9B",
|
| 1011 |
-
"developer": "Google",
|
| 1012 |
-
"scores": {
|
| 1013 |
-
"Mean win rate": 0.562,
|
| 1014 |
-
"NarrativeQA": 0.768,
|
| 1015 |
-
"NaturalQuestions (closed-book)": 0.328,
|
| 1016 |
-
"OpenbookQA": 0.91,
|
| 1017 |
-
"MMLU": 0.645,
|
| 1018 |
-
"MATH": 0.724,
|
| 1019 |
-
"GSM8K": 0.762,
|
| 1020 |
-
"LegalBench": 0.639,
|
| 1021 |
-
"MedQA": 0.63,
|
| 1022 |
-
"WMT 2014": 0.201
|
| 1023 |
-
}
|
| 1024 |
-
},
|
| 1025 |
-
{
|
| 1026 |
-
"model_id": "google/gemma-7b",
|
| 1027 |
-
"name": "Gemma 7B",
|
| 1028 |
-
"developer": "Google",
|
| 1029 |
-
"scores": {
|
| 1030 |
-
"Mean win rate": 0.336,
|
| 1031 |
-
"NarrativeQA": 0.752,
|
| 1032 |
-
"NaturalQuestions (closed-book)": 0.336,
|
| 1033 |
-
"OpenbookQA": 0.808,
|
| 1034 |
-
"MMLU": 0.571,
|
| 1035 |
-
"MATH": 0.5,
|
| 1036 |
-
"GSM8K": 0.559,
|
| 1037 |
-
"LegalBench": 0.581,
|
| 1038 |
-
"MedQA": 0.513,
|
| 1039 |
-
"WMT 2014": 0.187
|
| 1040 |
-
}
|
| 1041 |
-
},
|
| 1042 |
-
{
|
| 1043 |
-
"model_id": "google/text-bison@001",
|
| 1044 |
-
"name": "PaLM-2 Bison",
|
| 1045 |
-
"developer": "Google",
|
| 1046 |
-
"scores": {
|
| 1047 |
-
"Mean win rate": 0.526,
|
| 1048 |
-
"NarrativeQA": 0.718,
|
| 1049 |
-
"NaturalQuestions (closed-book)": 0.39,
|
| 1050 |
-
"OpenbookQA": 0.878,
|
| 1051 |
-
"MMLU": 0.608,
|
| 1052 |
-
"MATH": 0.421,
|
| 1053 |
-
"GSM8K": 0.61,
|
| 1054 |
-
"LegalBench": 0.645,
|
| 1055 |
-
"MedQA": 0.547,
|
| 1056 |
-
"WMT 2014": 0.241
|
| 1057 |
-
}
|
| 1058 |
-
},
|
| 1059 |
-
{
|
| 1060 |
-
"model_id": "google/text-unicorn@001",
|
| 1061 |
-
"name": "PaLM-2 Unicorn",
|
| 1062 |
-
"developer": "Google",
|
| 1063 |
-
"scores": {
|
| 1064 |
-
"Mean win rate": 0.644,
|
| 1065 |
-
"NarrativeQA": 0.583,
|
| 1066 |
-
"NaturalQuestions (closed-book)": 0.435,
|
| 1067 |
-
"OpenbookQA": 0.938,
|
| 1068 |
-
"MMLU": 0.702,
|
| 1069 |
-
"MATH": 0.674,
|
| 1070 |
-
"GSM8K": 0.831,
|
| 1071 |
-
"LegalBench": 0.677,
|
| 1072 |
-
"MedQA": 0.684,
|
| 1073 |
-
"WMT 2014": 0.26
|
| 1074 |
-
}
|
| 1075 |
-
},
|
| 1076 |
-
{
|
| 1077 |
-
"model_id": "meta/LLaMA-65B",
|
| 1078 |
-
"name": "LLaMA 65B",
|
| 1079 |
-
"developer": "Meta",
|
| 1080 |
-
"scores": {
|
| 1081 |
-
"Mean win rate": 0.345,
|
| 1082 |
-
"NarrativeQA": 0.755,
|
| 1083 |
-
"NaturalQuestions (closed-book)": 0.433,
|
| 1084 |
-
"OpenbookQA": 0.754,
|
| 1085 |
-
"MMLU": 0.584,
|
| 1086 |
-
"MATH": 0.257,
|
| 1087 |
-
"GSM8K": 0.489,
|
| 1088 |
-
"LegalBench": 0.48,
|
| 1089 |
-
"MedQA": 0.507,
|
| 1090 |
-
"WMT 2014": 0.189
|
| 1091 |
-
}
|
| 1092 |
-
},
|
| 1093 |
-
{
|
| 1094 |
-
"model_id": "meta/llama-2-13b",
|
| 1095 |
-
"name": "Llama 2 13B",
|
| 1096 |
-
"developer": "Meta",
|
| 1097 |
-
"scores": {
|
| 1098 |
-
"Mean win rate": 0.233,
|
| 1099 |
-
"NarrativeQA": 0.741,
|
| 1100 |
-
"NaturalQuestions (closed-book)": 0.371,
|
| 1101 |
-
"OpenbookQA": 0.634,
|
| 1102 |
-
"MMLU": 0.505,
|
| 1103 |
-
"MATH": 0.102,
|
| 1104 |
-
"GSM8K": 0.266,
|
| 1105 |
-
"LegalBench": 0.591,
|
| 1106 |
-
"MedQA": 0.392,
|
| 1107 |
-
"WMT 2014": 0.167
|
| 1108 |
-
}
|
| 1109 |
-
},
|
| 1110 |
-
{
|
| 1111 |
-
"model_id": "meta/llama-2-70b",
|
| 1112 |
-
"name": "Llama 2 70B",
|
| 1113 |
-
"developer": "Meta",
|
| 1114 |
-
"scores": {
|
| 1115 |
-
"Mean win rate": 0.482,
|
| 1116 |
-
"NarrativeQA": 0.763,
|
| 1117 |
-
"NaturalQuestions (closed-book)": 0.46,
|
| 1118 |
-
"OpenbookQA": 0.838,
|
| 1119 |
-
"MMLU": 0.58,
|
| 1120 |
-
"MATH": 0.323,
|
| 1121 |
-
"GSM8K": 0.567,
|
| 1122 |
-
"LegalBench": 0.673,
|
| 1123 |
-
"MedQA": 0.618,
|
| 1124 |
-
"WMT 2014": 0.196
|
| 1125 |
-
}
|
| 1126 |
-
},
|
| 1127 |
-
{
|
| 1128 |
-
"model_id": "meta/llama-2-7b",
|
| 1129 |
-
"name": "Llama 2 7B",
|
| 1130 |
-
"developer": "Meta",
|
| 1131 |
-
"scores": {
|
| 1132 |
-
"Mean win rate": 0.152,
|
| 1133 |
-
"NarrativeQA": 0.686,
|
| 1134 |
-
"NaturalQuestions (closed-book)": 0.333,
|
| 1135 |
-
"OpenbookQA": 0.544,
|
| 1136 |
-
"MMLU": 0.425,
|
| 1137 |
-
"MATH": 0.097,
|
| 1138 |
-
"GSM8K": 0.154,
|
| 1139 |
-
"LegalBench": 0.502,
|
| 1140 |
-
"MedQA": 0.392,
|
| 1141 |
-
"WMT 2014": 0.144
|
| 1142 |
-
}
|
| 1143 |
-
},
|
| 1144 |
-
{
|
| 1145 |
-
"model_id": "meta/llama-3-70b",
|
| 1146 |
-
"name": "Llama 3 70B",
|
| 1147 |
-
"developer": "Meta",
|
| 1148 |
-
"scores": {
|
| 1149 |
-
"Mean win rate": 0.793,
|
| 1150 |
-
"NarrativeQA": 0.798,
|
| 1151 |
-
"NaturalQuestions (closed-book)": 0.475,
|
| 1152 |
-
"OpenbookQA": 0.934,
|
| 1153 |
-
"MMLU": 0.695,
|
| 1154 |
-
"MATH": 0.663,
|
| 1155 |
-
"GSM8K": 0.805,
|
| 1156 |
-
"LegalBench": 0.733,
|
| 1157 |
-
"MedQA": 0.777,
|
| 1158 |
-
"WMT 2014": 0.225
|
| 1159 |
-
}
|
| 1160 |
-
},
|
| 1161 |
-
{
|
| 1162 |
-
"model_id": "meta/llama-3-8b",
|
| 1163 |
-
"name": "Llama 3 8B",
|
| 1164 |
-
"developer": "Meta",
|
| 1165 |
-
"scores": {
|
| 1166 |
-
"Mean win rate": 0.387,
|
| 1167 |
-
"NarrativeQA": 0.754,
|
| 1168 |
-
"NaturalQuestions (closed-book)": 0.378,
|
| 1169 |
-
"OpenbookQA": 0.766,
|
| 1170 |
-
"MMLU": 0.602,
|
| 1171 |
-
"MATH": 0.391,
|
| 1172 |
-
"GSM8K": 0.499,
|
| 1173 |
-
"LegalBench": 0.637,
|
| 1174 |
-
"MedQA": 0.581,
|
| 1175 |
-
"WMT 2014": 0.183
|
| 1176 |
-
}
|
| 1177 |
-
},
|
| 1178 |
-
{
|
| 1179 |
-
"model_id": "meta/llama-3.1-405b-instruct-turbo",
|
| 1180 |
-
"name": "Llama 3.1 Instruct Turbo 405B",
|
| 1181 |
-
"developer": "Meta",
|
| 1182 |
-
"scores": {
|
| 1183 |
-
"Mean win rate": 0.854,
|
| 1184 |
-
"NarrativeQA": 0.749,
|
| 1185 |
-
"NaturalQuestions (closed-book)": 0.456,
|
| 1186 |
-
"OpenbookQA": 0.94,
|
| 1187 |
-
"MMLU": 0.759,
|
| 1188 |
-
"MATH": 0.827,
|
| 1189 |
-
"GSM8K": 0.949,
|
| 1190 |
-
"LegalBench": 0.707,
|
| 1191 |
-
"MedQA": 0.805,
|
| 1192 |
-
"WMT 2014": 0.238
|
| 1193 |
-
}
|
| 1194 |
-
},
|
| 1195 |
-
{
|
| 1196 |
-
"model_id": "meta/llama-3.1-70b-instruct-turbo",
|
| 1197 |
-
"name": "Llama 3.1 Instruct Turbo 70B",
|
| 1198 |
-
"developer": "Meta",
|
| 1199 |
-
"scores": {
|
| 1200 |
-
"Mean win rate": 0.808,
|
| 1201 |
-
"NarrativeQA": 0.772,
|
| 1202 |
-
"NaturalQuestions (closed-book)": 0.452,
|
| 1203 |
-
"OpenbookQA": 0.938,
|
| 1204 |
-
"MMLU": 0.709,
|
| 1205 |
-
"MATH": 0.783,
|
| 1206 |
-
"GSM8K": 0.938,
|
| 1207 |
-
"LegalBench": 0.687,
|
| 1208 |
-
"MedQA": 0.769,
|
| 1209 |
-
"WMT 2014": 0.223
|
| 1210 |
-
}
|
| 1211 |
-
},
|
| 1212 |
-
{
|
| 1213 |
-
"model_id": "meta/llama-3.1-8b-instruct-turbo",
|
| 1214 |
-
"name": "Llama 3.1 Instruct Turbo 8B",
|
| 1215 |
-
"developer": "Meta",
|
| 1216 |
-
"scores": {
|
| 1217 |
-
"Mean win rate": 0.303,
|
| 1218 |
-
"NarrativeQA": 0.756,
|
| 1219 |
-
"NaturalQuestions (closed-book)": 0.209,
|
| 1220 |
-
"OpenbookQA": 0.74,
|
| 1221 |
-
"MMLU": 0.5,
|
| 1222 |
-
"MATH": 0.703,
|
| 1223 |
-
"GSM8K": 0.798,
|
| 1224 |
-
"LegalBench": 0.342,
|
| 1225 |
-
"MedQA": 0.245,
|
| 1226 |
-
"WMT 2014": 0.181
|
| 1227 |
-
}
|
| 1228 |
-
},
|
| 1229 |
-
{
|
| 1230 |
-
"model_id": "meta/llama-3.2-11b-vision-instruct-turbo",
|
| 1231 |
-
"name": "Llama 3.2 Vision Instruct Turbo 11B",
|
| 1232 |
-
"developer": "Meta",
|
| 1233 |
-
"scores": {
|
| 1234 |
-
"Mean win rate": 0.325,
|
| 1235 |
-
"NarrativeQA": 0.756,
|
| 1236 |
-
"NaturalQuestions (closed-book)": 0.234,
|
| 1237 |
-
"OpenbookQA": 0.724,
|
| 1238 |
-
"MMLU": 0.511,
|
| 1239 |
-
"MATH": 0.739,
|
| 1240 |
-
"GSM8K": 0.823,
|
| 1241 |
-
"LegalBench": 0.435,
|
| 1242 |
-
"MedQA": 0.27,
|
| 1243 |
-
"WMT 2014": 0.179
|
| 1244 |
-
}
|
| 1245 |
-
},
|
| 1246 |
-
{
|
| 1247 |
-
"model_id": "meta/llama-3.2-90b-vision-instruct-turbo",
|
| 1248 |
-
"name": "Llama 3.2 Vision Instruct Turbo 90B",
|
| 1249 |
-
"developer": "Meta",
|
| 1250 |
-
"scores": {
|
| 1251 |
-
"Mean win rate": 0.819,
|
| 1252 |
-
"NarrativeQA": 0.777,
|
| 1253 |
-
"NaturalQuestions (closed-book)": 0.457,
|
| 1254 |
-
"OpenbookQA": 0.942,
|
| 1255 |
-
"MMLU": 0.703,
|
| 1256 |
-
"MATH": 0.791,
|
| 1257 |
-
"GSM8K": 0.936,
|
| 1258 |
-
"LegalBench": 0.68,
|
| 1259 |
-
"MedQA": 0.769,
|
| 1260 |
-
"WMT 2014": 0.224
|
| 1261 |
-
}
|
| 1262 |
-
},
|
| 1263 |
-
{
|
| 1264 |
-
"model_id": "meta/llama-3.3-70b-instruct-turbo",
|
| 1265 |
-
"name": "Llama 3.3 Instruct Turbo 70B",
|
| 1266 |
-
"developer": "Meta",
|
| 1267 |
-
"scores": {
|
| 1268 |
-
"Mean win rate": 0.812,
|
| 1269 |
-
"NarrativeQA": 0.791,
|
| 1270 |
-
"NaturalQuestions (closed-book)": 0.431,
|
| 1271 |
-
"OpenbookQA": 0.928,
|
| 1272 |
-
"MMLU": 0.7,
|
| 1273 |
-
"MATH": 0.808,
|
| 1274 |
-
"GSM8K": 0.942,
|
| 1275 |
-
"LegalBench": 0.725,
|
| 1276 |
-
"MedQA": 0.761,
|
| 1277 |
-
"WMT 2014": 0.219
|
| 1278 |
-
}
|
| 1279 |
-
},
|
| 1280 |
-
{
|
| 1281 |
-
"model_id": "microsoft/phi-2",
|
| 1282 |
-
"name": "Phi-2",
|
| 1283 |
-
"developer": "microsoft",
|
| 1284 |
-
"scores": {
|
| 1285 |
-
"Mean win rate": 0.169,
|
| 1286 |
-
"NarrativeQA": 0.703,
|
| 1287 |
-
"NaturalQuestions (closed-book)": 0.155,
|
| 1288 |
-
"OpenbookQA": 0.798,
|
| 1289 |
-
"MMLU": 0.518,
|
| 1290 |
-
"MATH": 0.255,
|
| 1291 |
-
"GSM8K": 0.581,
|
| 1292 |
-
"LegalBench": 0.334,
|
| 1293 |
-
"MedQA": 0.41,
|
| 1294 |
-
"WMT 2014": 0.038
|
| 1295 |
-
}
|
| 1296 |
-
},
|
| 1297 |
-
{
|
| 1298 |
-
"model_id": "microsoft/phi-3-medium-4k-instruct",
|
| 1299 |
-
"name": "Phi-3 14B",
|
| 1300 |
-
"developer": "microsoft",
|
| 1301 |
-
"scores": {
|
| 1302 |
-
"Mean win rate": 0.509,
|
| 1303 |
-
"NarrativeQA": 0.724,
|
| 1304 |
-
"NaturalQuestions (closed-book)": 0.278,
|
| 1305 |
-
"OpenbookQA": 0.916,
|
| 1306 |
-
"MMLU": 0.675,
|
| 1307 |
-
"MATH": 0.611,
|
| 1308 |
-
"GSM8K": 0.878,
|
| 1309 |
-
"LegalBench": 0.593,
|
| 1310 |
-
"MedQA": 0.696,
|
| 1311 |
-
"WMT 2014": 0.17
|
| 1312 |
-
}
|
| 1313 |
-
},
|
| 1314 |
-
{
|
| 1315 |
-
"model_id": "microsoft/phi-3-small-8k-instruct",
|
| 1316 |
-
"name": "Phi-3 7B",
|
| 1317 |
-
"developer": "microsoft",
|
| 1318 |
-
"scores": {
|
| 1319 |
-
"Mean win rate": 0.473,
|
| 1320 |
-
"NarrativeQA": 0.754,
|
| 1321 |
-
"NaturalQuestions (closed-book)": 0.324,
|
| 1322 |
-
"OpenbookQA": 0.912,
|
| 1323 |
-
"MMLU": 0.659,
|
| 1324 |
-
"MATH": 0.703,
|
| 1325 |
-
"GSM8K": -1,
|
| 1326 |
-
"LegalBench": 0.584,
|
| 1327 |
-
"MedQA": 0.672,
|
| 1328 |
-
"WMT 2014": 0.154
|
| 1329 |
-
}
|
| 1330 |
-
},
|
| 1331 |
-
{
|
| 1332 |
-
"model_id": "mistralai/mistral-7b-instruct-v0.3",
|
| 1333 |
-
"name": "Mistral Instruct v0.3 7B",
|
| 1334 |
-
"developer": "mistralai",
|
| 1335 |
-
"scores": {
|
| 1336 |
-
"Mean win rate": 0.196,
|
| 1337 |
-
"NarrativeQA": 0.716,
|
| 1338 |
-
"NaturalQuestions (closed-book)": 0.253,
|
| 1339 |
-
"OpenbookQA": 0.79,
|
| 1340 |
-
"MMLU": 0.51,
|
| 1341 |
-
"MATH": 0.289,
|
| 1342 |
-
"GSM8K": 0.538,
|
| 1343 |
-
"LegalBench": 0.331,
|
| 1344 |
-
"MedQA": 0.517,
|
| 1345 |
-
"WMT 2014": 0.142
|
| 1346 |
-
}
|
| 1347 |
-
},
|
| 1348 |
-
{
|
| 1349 |
-
"model_id": "mistralai/mistral-7b-v0.1",
|
| 1350 |
-
"name": "Mistral v0.1 7B",
|
| 1351 |
-
"developer": "mistralai",
|
| 1352 |
-
"scores": {
|
| 1353 |
-
"Mean win rate": 0.292,
|
| 1354 |
-
"NarrativeQA": 0.716,
|
| 1355 |
-
"NaturalQuestions (closed-book)": 0.367,
|
| 1356 |
-
"OpenbookQA": 0.776,
|
| 1357 |
-
"MMLU": 0.584,
|
| 1358 |
-
"MATH": 0.297,
|
| 1359 |
-
"GSM8K": 0.377,
|
| 1360 |
-
"LegalBench": 0.58,
|
| 1361 |
-
"MedQA": 0.525,
|
| 1362 |
-
"WMT 2014": 0.16
|
| 1363 |
-
}
|
| 1364 |
-
},
|
| 1365 |
-
{
|
| 1366 |
-
"model_id": "mistralai/mistral-large-2402",
|
| 1367 |
-
"name": "Mistral Large 2402",
|
| 1368 |
-
"developer": "mistralai",
|
| 1369 |
-
"scores": {
|
| 1370 |
-
"Mean win rate": 0.328,
|
| 1371 |
-
"NarrativeQA": 0.454,
|
| 1372 |
-
"NaturalQuestions (closed-book)": 0.311,
|
| 1373 |
-
"OpenbookQA": 0.894,
|
| 1374 |
-
"MMLU": 0.638,
|
| 1375 |
-
"MATH": 0.75,
|
| 1376 |
-
"GSM8K": 0.694,
|
| 1377 |
-
"LegalBench": 0.479,
|
| 1378 |
-
"MedQA": 0.499,
|
| 1379 |
-
"WMT 2014": 0.182
|
| 1380 |
-
}
|
| 1381 |
-
},
|
| 1382 |
-
{
|
| 1383 |
-
"model_id": "mistralai/mistral-large-2407",
|
| 1384 |
-
"name": "Mistral Large 2 2407",
|
| 1385 |
-
"developer": "mistralai",
|
| 1386 |
-
"scores": {
|
| 1387 |
-
"Mean win rate": 0.744,
|
| 1388 |
-
"NarrativeQA": 0.779,
|
| 1389 |
-
"NaturalQuestions (closed-book)": 0.453,
|
| 1390 |
-
"OpenbookQA": 0.932,
|
| 1391 |
-
"MMLU": 0.725,
|
| 1392 |
-
"MATH": 0.677,
|
| 1393 |
-
"GSM8K": 0.912,
|
| 1394 |
-
"LegalBench": 0.646,
|
| 1395 |
-
"MedQA": 0.775,
|
| 1396 |
-
"WMT 2014": 0.192
|
| 1397 |
-
}
|
| 1398 |
-
},
|
| 1399 |
-
{
|
| 1400 |
-
"model_id": "mistralai/mistral-medium-2312",
|
| 1401 |
-
"name": "Mistral Medium 2312",
|
| 1402 |
-
"developer": "mistralai",
|
| 1403 |
-
"scores": {
|
| 1404 |
-
"Mean win rate": 0.268,
|
| 1405 |
-
"NarrativeQA": 0.449,
|
| 1406 |
-
"NaturalQuestions (closed-book)": 0.29,
|
| 1407 |
-
"OpenbookQA": 0.83,
|
| 1408 |
-
"MMLU": 0.618,
|
| 1409 |
-
"MATH": 0.565,
|
| 1410 |
-
"GSM8K": 0.706,
|
| 1411 |
-
"LegalBench": 0.452,
|
| 1412 |
-
"MedQA": 0.61,
|
| 1413 |
-
"WMT 2014": 0.169
|
| 1414 |
-
}
|
| 1415 |
-
},
|
| 1416 |
-
{
|
| 1417 |
-
"model_id": "mistralai/mistral-small-2402",
|
| 1418 |
-
"name": "Mistral Small 2402",
|
| 1419 |
-
"developer": "mistralai",
|
| 1420 |
-
"scores": {
|
| 1421 |
-
"Mean win rate": 0.288,
|
| 1422 |
-
"NarrativeQA": 0.519,
|
| 1423 |
-
"NaturalQuestions (closed-book)": 0.304,
|
| 1424 |
-
"OpenbookQA": 0.862,
|
| 1425 |
-
"MMLU": 0.593,
|
| 1426 |
-
"MATH": 0.621,
|
| 1427 |
-
"GSM8K": 0.734,
|
| 1428 |
-
"LegalBench": 0.389,
|
| 1429 |
-
"MedQA": 0.616,
|
| 1430 |
-
"WMT 2014": 0.169
|
| 1431 |
-
}
|
| 1432 |
-
},
|
| 1433 |
-
{
|
| 1434 |
-
"model_id": "mistralai/mixtral-8x22b",
|
| 1435 |
-
"name": "Mixtral 8x22B",
|
| 1436 |
-
"developer": "mistralai",
|
| 1437 |
-
"scores": {
|
| 1438 |
-
"Mean win rate": 0.705,
|
| 1439 |
-
"NarrativeQA": 0.779,
|
| 1440 |
-
"NaturalQuestions (closed-book)": 0.478,
|
| 1441 |
-
"OpenbookQA": 0.882,
|
| 1442 |
-
"MMLU": 0.701,
|
| 1443 |
-
"MATH": 0.656,
|
| 1444 |
-
"GSM8K": 0.8,
|
| 1445 |
-
"LegalBench": 0.708,
|
| 1446 |
-
"MedQA": 0.704,
|
| 1447 |
-
"WMT 2014": 0.209
|
| 1448 |
-
}
|
| 1449 |
-
},
|
| 1450 |
-
{
|
| 1451 |
-
"model_id": "mistralai/mixtral-8x7b-32kseqlen",
|
| 1452 |
-
"name": "Mixtral 8x7B 32K seqlen",
|
| 1453 |
-
"developer": "mistralai",
|
| 1454 |
-
"scores": {
|
| 1455 |
-
"Mean win rate": 0.51,
|
| 1456 |
-
"NarrativeQA": 0.767,
|
| 1457 |
-
"NaturalQuestions (closed-book)": 0.427,
|
| 1458 |
-
"OpenbookQA": 0.868,
|
| 1459 |
-
"MMLU": 0.649,
|
| 1460 |
-
"MATH": 0.494,
|
| 1461 |
-
"GSM8K": 0.622,
|
| 1462 |
-
"LegalBench": 0.63,
|
| 1463 |
-
"MedQA": 0.652,
|
| 1464 |
-
"WMT 2014": 0.19
|
| 1465 |
-
}
|
| 1466 |
-
},
|
| 1467 |
-
{
|
| 1468 |
-
"model_id": "mistralai/open-mistral-nemo-2407",
|
| 1469 |
-
"name": "Mistral NeMo 2402",
|
| 1470 |
-
"developer": "mistralai",
|
| 1471 |
-
"scores": {
|
| 1472 |
-
"Mean win rate": 0.333,
|
| 1473 |
-
"NarrativeQA": 0.731,
|
| 1474 |
-
"NaturalQuestions (closed-book)": 0.265,
|
| 1475 |
-
"OpenbookQA": 0.822,
|
| 1476 |
-
"MMLU": 0.604,
|
| 1477 |
-
"MATH": 0.668,
|
| 1478 |
-
"GSM8K": 0.782,
|
| 1479 |
-
"LegalBench": 0.415,
|
| 1480 |
-
"MedQA": 0.59,
|
| 1481 |
-
"WMT 2014": 0.177
|
| 1482 |
-
}
|
| 1483 |
-
},
|
| 1484 |
-
{
|
| 1485 |
-
"model_id": "openai/gpt-3.5-turbo-0613",
|
| 1486 |
-
"name": "GPT-3.5 Turbo 0613",
|
| 1487 |
-
"developer": "OpenAI",
|
| 1488 |
-
"scores": {
|
| 1489 |
-
"Mean win rate": 0.358,
|
| 1490 |
-
"NarrativeQA": 0.655,
|
| 1491 |
-
"NaturalQuestions (closed-book)": 0.335,
|
| 1492 |
-
"OpenbookQA": 0.838,
|
| 1493 |
-
"MMLU": 0.614,
|
| 1494 |
-
"MATH": 0.667,
|
| 1495 |
-
"GSM8K": 0.501,
|
| 1496 |
-
"LegalBench": 0.528,
|
| 1497 |
-
"MedQA": 0.622,
|
| 1498 |
-
"WMT 2014": 0.187
|
| 1499 |
-
}
|
| 1500 |
-
},
|
| 1501 |
-
{
|
| 1502 |
-
"model_id": "openai/gpt-4-0613",
|
| 1503 |
-
"name": "GPT-4 0613",
|
| 1504 |
-
"developer": "OpenAI",
|
| 1505 |
-
"scores": {
|
| 1506 |
-
"Mean win rate": 0.867,
|
| 1507 |
-
"NarrativeQA": 0.768,
|
| 1508 |
-
"NaturalQuestions (closed-book)": 0.457,
|
| 1509 |
-
"OpenbookQA": 0.96,
|
| 1510 |
-
"MMLU": 0.735,
|
| 1511 |
-
"MATH": 0.802,
|
| 1512 |
-
"GSM8K": 0.932,
|
| 1513 |
-
"LegalBench": 0.713,
|
| 1514 |
-
"MedQA": 0.815,
|
| 1515 |
-
"WMT 2014": 0.211
|
| 1516 |
-
}
|
| 1517 |
-
},
|
| 1518 |
-
{
|
| 1519 |
-
"model_id": "openai/gpt-4-1106-preview",
|
| 1520 |
-
"name": "GPT-4 Turbo 1106 preview",
|
| 1521 |
-
"developer": "OpenAI",
|
| 1522 |
-
"scores": {
|
| 1523 |
-
"Mean win rate": 0.698,
|
| 1524 |
-
"NarrativeQA": 0.727,
|
| 1525 |
-
"NaturalQuestions (closed-book)": 0.435,
|
| 1526 |
-
"OpenbookQA": 0.95,
|
| 1527 |
-
"MMLU": 0.699,
|
| 1528 |
-
"MATH": 0.857,
|
| 1529 |
-
"GSM8K": 0.668,
|
| 1530 |
-
"LegalBench": 0.626,
|
| 1531 |
-
"MedQA": 0.817,
|
| 1532 |
-
"WMT 2014": 0.205
|
| 1533 |
-
}
|
| 1534 |
-
},
|
| 1535 |
-
{
|
| 1536 |
-
"model_id": "openai/gpt-4-turbo-2024-04-09",
|
| 1537 |
-
"name": "GPT-4 Turbo 2024-04-09",
|
| 1538 |
-
"developer": "OpenAI",
|
| 1539 |
-
"scores": {
|
| 1540 |
-
"Mean win rate": 0.864,
|
| 1541 |
-
"NarrativeQA": 0.761,
|
| 1542 |
-
"NaturalQuestions (closed-book)": 0.482,
|
| 1543 |
-
"OpenbookQA": 0.97,
|
| 1544 |
-
"MMLU": 0.711,
|
| 1545 |
-
"MATH": 0.833,
|
| 1546 |
-
"GSM8K": 0.824,
|
| 1547 |
-
"LegalBench": 0.727,
|
| 1548 |
-
"MedQA": 0.783,
|
| 1549 |
-
"WMT 2014": 0.218
|
| 1550 |
-
}
|
| 1551 |
-
},
|
| 1552 |
-
{
|
| 1553 |
-
"model_id": "openai/gpt-4o-2024-05-13",
|
| 1554 |
-
"name": "GPT-4o 2024-05-13",
|
| 1555 |
-
"developer": "OpenAI",
|
| 1556 |
-
"scores": {
|
| 1557 |
-
"Mean win rate": 0.938,
|
| 1558 |
-
"NarrativeQA": 0.804,
|
| 1559 |
-
"NaturalQuestions (closed-book)": 0.501,
|
| 1560 |
-
"OpenbookQA": 0.966,
|
| 1561 |
-
"MMLU": 0.748,
|
| 1562 |
-
"MATH": 0.829,
|
| 1563 |
-
"GSM8K": 0.905,
|
| 1564 |
-
"LegalBench": 0.733,
|
| 1565 |
-
"MedQA": 0.857,
|
| 1566 |
-
"WMT 2014": 0.231
|
| 1567 |
-
}
|
| 1568 |
-
},
|
| 1569 |
-
{
|
| 1570 |
-
"model_id": "openai/gpt-4o-2024-08-06",
|
| 1571 |
-
"name": "GPT-4o 2024-08-06",
|
| 1572 |
-
"developer": "OpenAI",
|
| 1573 |
-
"scores": {
|
| 1574 |
-
"Mean win rate": 0.928,
|
| 1575 |
-
"NarrativeQA": 0.795,
|
| 1576 |
-
"NaturalQuestions (closed-book)": 0.496,
|
| 1577 |
-
"OpenbookQA": 0.968,
|
| 1578 |
-
"MMLU": 0.738,
|
| 1579 |
-
"MATH": 0.853,
|
| 1580 |
-
"GSM8K": 0.909,
|
| 1581 |
-
"LegalBench": 0.721,
|
| 1582 |
-
"MedQA": 0.863,
|
| 1583 |
-
"WMT 2014": 0.225
|
| 1584 |
-
}
|
| 1585 |
-
},
|
| 1586 |
-
{
|
| 1587 |
-
"model_id": "openai/gpt-4o-mini-2024-07-18",
|
| 1588 |
-
"name": "GPT-4o mini 2024-07-18",
|
| 1589 |
-
"developer": "OpenAI",
|
| 1590 |
-
"scores": {
|
| 1591 |
-
"Mean win rate": 0.701,
|
| 1592 |
-
"NarrativeQA": 0.768,
|
| 1593 |
-
"NaturalQuestions (closed-book)": 0.386,
|
| 1594 |
-
"OpenbookQA": 0.92,
|
| 1595 |
-
"MMLU": 0.668,
|
| 1596 |
-
"MATH": 0.802,
|
| 1597 |
-
"GSM8K": 0.843,
|
| 1598 |
-
"LegalBench": 0.653,
|
| 1599 |
-
"MedQA": 0.748,
|
| 1600 |
-
"WMT 2014": 0.206
|
| 1601 |
-
}
|
| 1602 |
-
},
|
| 1603 |
-
{
|
| 1604 |
-
"model_id": "openai/text-davinci-002",
|
| 1605 |
-
"name": "GPT-3.5 text-davinci-002",
|
| 1606 |
-
"developer": "OpenAI",
|
| 1607 |
-
"scores": {
|
| 1608 |
-
"Mean win rate": 0.336,
|
| 1609 |
-
"NarrativeQA": 0.719,
|
| 1610 |
-
"NaturalQuestions (closed-book)": 0.394,
|
| 1611 |
-
"OpenbookQA": 0.796,
|
| 1612 |
-
"MMLU": 0.568,
|
| 1613 |
-
"MATH": 0.428,
|
| 1614 |
-
"GSM8K": 0.479,
|
| 1615 |
-
"LegalBench": 0.58,
|
| 1616 |
-
"MedQA": 0.525,
|
| 1617 |
-
"WMT 2014": 0.174
|
| 1618 |
-
}
|
| 1619 |
-
},
|
| 1620 |
-
{
|
| 1621 |
-
"model_id": "openai/text-davinci-003",
|
| 1622 |
-
"name": "GPT-3.5 text-davinci-003",
|
| 1623 |
-
"developer": "OpenAI",
|
| 1624 |
-
"scores": {
|
| 1625 |
-
"Mean win rate": 0.439,
|
| 1626 |
-
"NarrativeQA": 0.731,
|
| 1627 |
-
"NaturalQuestions (closed-book)": 0.413,
|
| 1628 |
-
"OpenbookQA": 0.828,
|
| 1629 |
-
"MMLU": 0.555,
|
| 1630 |
-
"MATH": 0.449,
|
| 1631 |
-
"GSM8K": 0.615,
|
| 1632 |
-
"LegalBench": 0.622,
|
| 1633 |
-
"MedQA": 0.531,
|
| 1634 |
-
"WMT 2014": 0.191
|
| 1635 |
-
}
|
| 1636 |
-
},
|
| 1637 |
-
{
|
| 1638 |
-
"model_id": "qwen/qwen1.5-110b-chat",
|
| 1639 |
-
"name": "Qwen1.5 Chat 110B",
|
| 1640 |
-
"developer": "qwen",
|
| 1641 |
-
"scores": {
|
| 1642 |
-
"Mean win rate": 0.55,
|
| 1643 |
-
"NarrativeQA": 0.721,
|
| 1644 |
-
"NaturalQuestions (closed-book)": 0.35,
|
| 1645 |
-
"OpenbookQA": 0.922,
|
| 1646 |
-
"MMLU": 0.704,
|
| 1647 |
-
"MATH": 0.568,
|
| 1648 |
-
"GSM8K": 0.815,
|
| 1649 |
-
"LegalBench": 0.624,
|
| 1650 |
-
"MedQA": 0.64,
|
| 1651 |
-
"WMT 2014": 0.192
|
| 1652 |
-
}
|
| 1653 |
-
},
|
| 1654 |
-
{
|
| 1655 |
-
"model_id": "qwen/qwen1.5-14b",
|
| 1656 |
-
"name": "Qwen1.5 14B",
|
| 1657 |
-
"developer": "qwen",
|
| 1658 |
-
"scores": {
|
| 1659 |
-
"Mean win rate": 0.425,
|
| 1660 |
-
"NarrativeQA": 0.711,
|
| 1661 |
-
"NaturalQuestions (closed-book)": 0.3,
|
| 1662 |
-
"OpenbookQA": 0.862,
|
| 1663 |
-
"MMLU": 0.626,
|
| 1664 |
-
"MATH": 0.686,
|
| 1665 |
-
"GSM8K": 0.693,
|
| 1666 |
-
"LegalBench": 0.593,
|
| 1667 |
-
"MedQA": 0.515,
|
| 1668 |
-
"WMT 2014": 0.178
|
| 1669 |
-
}
|
| 1670 |
-
},
|
| 1671 |
-
{
|
| 1672 |
-
"model_id": "qwen/qwen1.5-32b",
|
| 1673 |
-
"name": "Qwen1.5 32B",
|
| 1674 |
-
"developer": "qwen",
|
| 1675 |
-
"scores": {
|
| 1676 |
-
"Mean win rate": 0.546,
|
| 1677 |
-
"NarrativeQA": 0.589,
|
| 1678 |
-
"NaturalQuestions (closed-book)": 0.353,
|
| 1679 |
-
"OpenbookQA": 0.932,
|
| 1680 |
-
"MMLU": 0.628,
|
| 1681 |
-
"MATH": 0.733,
|
| 1682 |
-
"GSM8K": 0.773,
|
| 1683 |
-
"LegalBench": 0.636,
|
| 1684 |
-
"MedQA": 0.656,
|
| 1685 |
-
"WMT 2014": 0.193
|
| 1686 |
-
}
|
| 1687 |
-
},
|
| 1688 |
-
{
|
| 1689 |
-
"model_id": "qwen/qwen1.5-72b",
|
| 1690 |
-
"name": "Qwen1.5 72B",
|
| 1691 |
-
"developer": "qwen",
|
| 1692 |
-
"scores": {
|
| 1693 |
-
"Mean win rate": 0.608,
|
| 1694 |
-
"NarrativeQA": 0.601,
|
| 1695 |
-
"NaturalQuestions (closed-book)": 0.417,
|
| 1696 |
-
"OpenbookQA": 0.93,
|
| 1697 |
-
"MMLU": 0.647,
|
| 1698 |
-
"MATH": 0.683,
|
| 1699 |
-
"GSM8K": 0.799,
|
| 1700 |
-
"LegalBench": 0.694,
|
| 1701 |
-
"MedQA": 0.67,
|
| 1702 |
-
"WMT 2014": 0.201
|
| 1703 |
-
}
|
| 1704 |
-
},
|
| 1705 |
-
{
|
| 1706 |
-
"model_id": "qwen/qwen1.5-7b",
|
| 1707 |
-
"name": "Qwen1.5 7B",
|
| 1708 |
-
"developer": "qwen",
|
| 1709 |
-
"scores": {
|
| 1710 |
-
"Mean win rate": 0.275,
|
| 1711 |
-
"NarrativeQA": 0.448,
|
| 1712 |
-
"NaturalQuestions (closed-book)": 0.27,
|
| 1713 |
-
"OpenbookQA": 0.806,
|
| 1714 |
-
"MMLU": 0.569,
|
| 1715 |
-
"MATH": 0.561,
|
| 1716 |
-
"GSM8K": 0.6,
|
| 1717 |
-
"LegalBench": 0.523,
|
| 1718 |
-
"MedQA": 0.479,
|
| 1719 |
-
"WMT 2014": 0.153
|
| 1720 |
-
}
|
| 1721 |
-
},
|
| 1722 |
-
{
|
| 1723 |
-
"model_id": "qwen/qwen2-72b-instruct",
|
| 1724 |
-
"name": "Qwen2 Instruct 72B",
|
| 1725 |
-
"developer": "qwen",
|
| 1726 |
-
"scores": {
|
| 1727 |
-
"Mean win rate": 0.77,
|
| 1728 |
-
"NarrativeQA": 0.727,
|
| 1729 |
-
"NaturalQuestions (closed-book)": 0.39,
|
| 1730 |
-
"OpenbookQA": 0.954,
|
| 1731 |
-
"MMLU": 0.769,
|
| 1732 |
-
"MATH": 0.79,
|
| 1733 |
-
"GSM8K": 0.92,
|
| 1734 |
-
"LegalBench": 0.712,
|
| 1735 |
-
"MedQA": 0.746,
|
| 1736 |
-
"WMT 2014": 0.207
|
| 1737 |
-
}
|
| 1738 |
-
},
|
| 1739 |
-
{
|
| 1740 |
-
"model_id": "qwen/qwen2.5-72b-instruct-turbo",
|
| 1741 |
-
"name": "Qwen2.5 Instruct Turbo 72B",
|
| 1742 |
-
"developer": "qwen",
|
| 1743 |
-
"scores": {
|
| 1744 |
-
"Mean win rate": 0.745,
|
| 1745 |
-
"NarrativeQA": 0.745,
|
| 1746 |
-
"NaturalQuestions (closed-book)": 0.359,
|
| 1747 |
-
"OpenbookQA": 0.962,
|
| 1748 |
-
"MMLU": 0.77,
|
| 1749 |
-
"MATH": 0.884,
|
| 1750 |
-
"GSM8K": 0.9,
|
| 1751 |
-
"LegalBench": 0.74,
|
| 1752 |
-
"MedQA": 0.753,
|
| 1753 |
-
"WMT 2014": 0.207
|
| 1754 |
-
}
|
| 1755 |
-
},
|
| 1756 |
-
{
|
| 1757 |
-
"model_id": "qwen/qwen2.5-7b-instruct-turbo",
|
| 1758 |
-
"name": "Qwen2.5 Instruct Turbo 7B",
|
| 1759 |
-
"developer": "qwen",
|
| 1760 |
-
"scores": {
|
| 1761 |
-
"Mean win rate": 0.488,
|
| 1762 |
-
"NarrativeQA": 0.742,
|
| 1763 |
-
"NaturalQuestions (closed-book)": 0.205,
|
| 1764 |
-
"OpenbookQA": 0.862,
|
| 1765 |
-
"MMLU": 0.658,
|
| 1766 |
-
"MATH": 0.835,
|
| 1767 |
-
"GSM8K": 0.83,
|
| 1768 |
-
"LegalBench": 0.632,
|
| 1769 |
-
"MedQA": 0.6,
|
| 1770 |
-
"WMT 2014": 0.155
|
| 1771 |
-
}
|
| 1772 |
-
},
|
| 1773 |
-
{
|
| 1774 |
-
"model_id": "snowflake/snowflake-arctic-instruct",
|
| 1775 |
-
"name": "Arctic Instruct",
|
| 1776 |
-
"developer": "snowflake",
|
| 1777 |
-
"scores": {
|
| 1778 |
-
"Mean win rate": 0.338,
|
| 1779 |
-
"NarrativeQA": 0.654,
|
| 1780 |
-
"NaturalQuestions (closed-book)": 0.39,
|
| 1781 |
-
"OpenbookQA": 0.828,
|
| 1782 |
-
"MMLU": 0.575,
|
| 1783 |
-
"MATH": 0.519,
|
| 1784 |
-
"GSM8K": 0.768,
|
| 1785 |
-
"LegalBench": 0.588,
|
| 1786 |
-
"MedQA": 0.581,
|
| 1787 |
-
"WMT 2014": 0.172
|
| 1788 |
-
}
|
| 1789 |
-
},
|
| 1790 |
-
{
|
| 1791 |
-
"model_id": "tiiuae/falcon-40b",
|
| 1792 |
-
"name": "Falcon 40B",
|
| 1793 |
-
"developer": "tiiuae",
|
| 1794 |
-
"scores": {
|
| 1795 |
-
"Mean win rate": 0.217,
|
| 1796 |
-
"NarrativeQA": 0.671,
|
| 1797 |
-
"NaturalQuestions (closed-book)": 0.392,
|
| 1798 |
-
"OpenbookQA": 0.662,
|
| 1799 |
-
"MMLU": 0.507,
|
| 1800 |
-
"MATH": 0.128,
|
| 1801 |
-
"GSM8K": 0.267,
|
| 1802 |
-
"LegalBench": 0.442,
|
| 1803 |
-
"MedQA": 0.419,
|
| 1804 |
-
"WMT 2014": 0.162
|
| 1805 |
-
}
|
| 1806 |
-
},
|
| 1807 |
-
{
|
| 1808 |
-
"model_id": "tiiuae/falcon-7b",
|
| 1809 |
-
"name": "Falcon 7B",
|
| 1810 |
-
"developer": "tiiuae",
|
| 1811 |
-
"scores": {
|
| 1812 |
-
"Mean win rate": 0.064,
|
| 1813 |
-
"NarrativeQA": 0.621,
|
| 1814 |
-
"NaturalQuestions (closed-book)": 0.285,
|
| 1815 |
-
"OpenbookQA": 0.26,
|
| 1816 |
-
"MMLU": 0.288,
|
| 1817 |
-
"MATH": 0.044,
|
| 1818 |
-
"GSM8K": 0.055,
|
| 1819 |
-
"LegalBench": 0.346,
|
| 1820 |
-
"MedQA": 0.254,
|
| 1821 |
-
"WMT 2014": 0.094
|
| 1822 |
-
}
|
| 1823 |
-
},
|
| 1824 |
-
{
|
| 1825 |
-
"model_id": "upstage/solar-pro-241126",
|
| 1826 |
-
"name": "Solar Pro",
|
| 1827 |
-
"developer": "upstage",
|
| 1828 |
-
"scores": {
|
| 1829 |
-
"Mean win rate": 0.602,
|
| 1830 |
-
"NarrativeQA": 0.753,
|
| 1831 |
-
"NaturalQuestions (closed-book)": 0.297,
|
| 1832 |
-
"OpenbookQA": 0.922,
|
| 1833 |
-
"MMLU": 0.679,
|
| 1834 |
-
"MATH": 0.567,
|
| 1835 |
-
"GSM8K": 0.871,
|
| 1836 |
-
"LegalBench": 0.67,
|
| 1837 |
-
"MedQA": 0.698,
|
| 1838 |
-
"WMT 2014": 0.169
|
| 1839 |
-
}
|
| 1840 |
-
},
|
| 1841 |
-
{
|
| 1842 |
-
"model_id": "writer/palmyra-x-004",
|
| 1843 |
-
"name": "Palmyra-X-004",
|
| 1844 |
-
"developer": "writer",
|
| 1845 |
-
"scores": {
|
| 1846 |
-
"Mean win rate": 0.808,
|
| 1847 |
-
"NarrativeQA": 0.773,
|
| 1848 |
-
"NaturalQuestions (closed-book)": 0.457,
|
| 1849 |
-
"OpenbookQA": 0.926,
|
| 1850 |
-
"MMLU": 0.739,
|
| 1851 |
-
"MATH": 0.767,
|
| 1852 |
-
"GSM8K": 0.905,
|
| 1853 |
-
"LegalBench": 0.73,
|
| 1854 |
-
"MedQA": 0.775,
|
| 1855 |
-
"WMT 2014": 0.203
|
| 1856 |
-
}
|
| 1857 |
-
},
|
| 1858 |
-
{
|
| 1859 |
-
"model_id": "writer/palmyra-x-v2",
|
| 1860 |
-
"name": "Palmyra X V2 33B",
|
| 1861 |
-
"developer": "writer",
|
| 1862 |
-
"scores": {
|
| 1863 |
-
"Mean win rate": 0.589,
|
| 1864 |
-
"NarrativeQA": 0.752,
|
| 1865 |
-
"NaturalQuestions (closed-book)": 0.428,
|
| 1866 |
-
"OpenbookQA": 0.878,
|
| 1867 |
-
"MMLU": 0.621,
|
| 1868 |
-
"MATH": 0.58,
|
| 1869 |
-
"GSM8K": 0.735,
|
| 1870 |
-
"LegalBench": 0.644,
|
| 1871 |
-
"MedQA": 0.598,
|
| 1872 |
-
"WMT 2014": 0.239
|
| 1873 |
-
}
|
| 1874 |
-
},
|
| 1875 |
-
{
|
| 1876 |
-
"model_id": "writer/palmyra-x-v3",
|
| 1877 |
-
"name": "Palmyra X V3 72B",
|
| 1878 |
-
"developer": "writer",
|
| 1879 |
-
"scores": {
|
| 1880 |
-
"Mean win rate": 0.679,
|
| 1881 |
-
"NarrativeQA": 0.706,
|
| 1882 |
-
"NaturalQuestions (closed-book)": 0.407,
|
| 1883 |
-
"OpenbookQA": 0.938,
|
| 1884 |
-
"MMLU": 0.702,
|
| 1885 |
-
"MATH": 0.723,
|
| 1886 |
-
"GSM8K": 0.831,
|
| 1887 |
-
"LegalBench": 0.709,
|
| 1888 |
-
"MedQA": 0.684,
|
| 1889 |
-
"WMT 2014": 0.262
|
| 1890 |
-
}
|
| 1891 |
-
}
|
| 1892 |
-
]
|
| 1893 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/helm_mmlu.json
DELETED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/benchmarks/hfopenllm_v2.json
DELETED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/benchmarks/la_leaderboard.json
DELETED
|
@@ -1,44 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "Qwen/Qwen2.5-7B",
|
| 5 |
-
"name": "Qwen2.5-7B",
|
| 6 |
-
"developer": "Qwen",
|
| 7 |
-
"scores": {
|
| 8 |
-
"la_leaderboard": 27.61
|
| 9 |
-
}
|
| 10 |
-
},
|
| 11 |
-
{
|
| 12 |
-
"model_id": "google/gemma-2-9b-it",
|
| 13 |
-
"name": "Gemma 2 Instruct 9B",
|
| 14 |
-
"developer": "Google",
|
| 15 |
-
"scores": {
|
| 16 |
-
"la_leaderboard": 33.62
|
| 17 |
-
}
|
| 18 |
-
},
|
| 19 |
-
{
|
| 20 |
-
"model_id": "meta-llama/Meta-Llama-3.1-8B",
|
| 21 |
-
"name": "Meta Llama 3.1 8B",
|
| 22 |
-
"developer": "unknown",
|
| 23 |
-
"scores": {
|
| 24 |
-
"la_leaderboard": 27.04
|
| 25 |
-
}
|
| 26 |
-
},
|
| 27 |
-
{
|
| 28 |
-
"model_id": "meta-llama/Meta-Llama-3.1-8B-Instruct",
|
| 29 |
-
"name": "Meta Llama 3.1 8B Instruct",
|
| 30 |
-
"developer": "unknown",
|
| 31 |
-
"scores": {
|
| 32 |
-
"la_leaderboard": 30.23
|
| 33 |
-
}
|
| 34 |
-
},
|
| 35 |
-
{
|
| 36 |
-
"model_id": "utter-project/EuroLLM-9B",
|
| 37 |
-
"name": "EuroLLM 9B",
|
| 38 |
-
"developer": "unknown",
|
| 39 |
-
"scores": {
|
| 40 |
-
"la_leaderboard": 25.87
|
| 41 |
-
}
|
| 42 |
-
}
|
| 43 |
-
]
|
| 44 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/livecodebenchpro.json
DELETED
|
@@ -1,274 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "alibaba/qwen3-235b-a22b-thinking-2507",
|
| 5 |
-
"name": "qwen3-235b-a22b-thinking-2507",
|
| 6 |
-
"developer": "Alibaba",
|
| 7 |
-
"scores": {
|
| 8 |
-
"Hard Problems": 0.0,
|
| 9 |
-
"Medium Problems": 0.1267605633802817,
|
| 10 |
-
"Easy Problems": 0.7605633802816901
|
| 11 |
-
}
|
| 12 |
-
},
|
| 13 |
-
{
|
| 14 |
-
"model_id": "alibaba/qwen3-30b-a3b",
|
| 15 |
-
"name": "qwen3-30b-a3b",
|
| 16 |
-
"developer": "Alibaba",
|
| 17 |
-
"scores": {
|
| 18 |
-
"Hard Problems": 0.0,
|
| 19 |
-
"Medium Problems": 0.028169014084507043,
|
| 20 |
-
"Easy Problems": 0.5774647887323944
|
| 21 |
-
}
|
| 22 |
-
},
|
| 23 |
-
{
|
| 24 |
-
"model_id": "alibaba/qwen3-max",
|
| 25 |
-
"name": "alibaba/qwen3-max",
|
| 26 |
-
"developer": "Alibaba",
|
| 27 |
-
"scores": {
|
| 28 |
-
"Hard Problems": 0.0,
|
| 29 |
-
"Medium Problems": 0.04225352112676056,
|
| 30 |
-
"Easy Problems": 0.36619718309859156
|
| 31 |
-
}
|
| 32 |
-
},
|
| 33 |
-
{
|
| 34 |
-
"model_id": "alibaba/qwen3-next-80b-a3b-thinking",
|
| 35 |
-
"name": "qwen3-next-80b-a3b-thinking",
|
| 36 |
-
"developer": "Alibaba",
|
| 37 |
-
"scores": {
|
| 38 |
-
"Hard Problems": 0.0,
|
| 39 |
-
"Medium Problems": 0.14084507042253522,
|
| 40 |
-
"Easy Problems": 0.7464788732394366
|
| 41 |
-
}
|
| 42 |
-
},
|
| 43 |
-
{
|
| 44 |
-
"model_id": "aliyun/qwen3-next-80b-a3b-thinking",
|
| 45 |
-
"name": "qwen3-next-80b-a3b-thinking",
|
| 46 |
-
"developer": "aliyun",
|
| 47 |
-
"scores": {
|
| 48 |
-
"Hard Problems": 0.0,
|
| 49 |
-
"Medium Problems": 0.0704,
|
| 50 |
-
"Easy Problems": 0.6901
|
| 51 |
-
}
|
| 52 |
-
},
|
| 53 |
-
{
|
| 54 |
-
"model_id": "anthropic/claude-3-7-sonnet-20250219",
|
| 55 |
-
"name": "claude-3-7-sonnet-20250219",
|
| 56 |
-
"developer": "Anthropic",
|
| 57 |
-
"scores": {
|
| 58 |
-
"Hard Problems": 0.0,
|
| 59 |
-
"Medium Problems": 0.0,
|
| 60 |
-
"Easy Problems": 0.28169014084507044
|
| 61 |
-
}
|
| 62 |
-
},
|
| 63 |
-
{
|
| 64 |
-
"model_id": "anthropic/claude-3.7-sonnet",
|
| 65 |
-
"name": "anthropic/claude-3.7-sonnet",
|
| 66 |
-
"developer": "Anthropic",
|
| 67 |
-
"scores": {
|
| 68 |
-
"Hard Problems": 0.0,
|
| 69 |
-
"Medium Problems": 0.014084507042253521,
|
| 70 |
-
"Easy Problems": 0.15492957746478872
|
| 71 |
-
}
|
| 72 |
-
},
|
| 73 |
-
{
|
| 74 |
-
"model_id": "anthropic/claude-sonnet-4-5-20250929",
|
| 75 |
-
"name": "claude-sonnet-4-5-20250929",
|
| 76 |
-
"developer": "Anthropic",
|
| 77 |
-
"scores": {
|
| 78 |
-
"Hard Problems": 0.0,
|
| 79 |
-
"Medium Problems": 0.0,
|
| 80 |
-
"Easy Problems": 0.5352
|
| 81 |
-
}
|
| 82 |
-
},
|
| 83 |
-
{
|
| 84 |
-
"model_id": "ark/ep-20250603132404-cgpjm",
|
| 85 |
-
"name": "ep-20250603132404-cgpjm",
|
| 86 |
-
"developer": "ark",
|
| 87 |
-
"scores": {
|
| 88 |
-
"Hard Problems": 0.0,
|
| 89 |
-
"Medium Problems": 0.0141,
|
| 90 |
-
"Easy Problems": 0.507
|
| 91 |
-
}
|
| 92 |
-
},
|
| 93 |
-
{
|
| 94 |
-
"model_id": "bytedance/doubao-seed-1-6-thinking-250615",
|
| 95 |
-
"name": "doubao-seed-1-6-thinking-250615",
|
| 96 |
-
"developer": "ByteDance",
|
| 97 |
-
"scores": {
|
| 98 |
-
"Hard Problems": 0.0,
|
| 99 |
-
"Medium Problems": 0.07042253521126761,
|
| 100 |
-
"Easy Problems": 0.5774647887323944
|
| 101 |
-
}
|
| 102 |
-
},
|
| 103 |
-
{
|
| 104 |
-
"model_id": "deepseek/chat-v3-0324",
|
| 105 |
-
"name": "deepseek/chat-v3-0324",
|
| 106 |
-
"developer": "DeepSeek",
|
| 107 |
-
"scores": {
|
| 108 |
-
"Hard Problems": 0.0,
|
| 109 |
-
"Medium Problems": 0.0,
|
| 110 |
-
"Easy Problems": 0.19718309859154928
|
| 111 |
-
}
|
| 112 |
-
},
|
| 113 |
-
{
|
| 114 |
-
"model_id": "deepseek/ep-20250214004308-p7n89",
|
| 115 |
-
"name": "ep-20250214004308-p7n89",
|
| 116 |
-
"developer": "DeepSeek",
|
| 117 |
-
"scores": {
|
| 118 |
-
"Hard Problems": 0.0,
|
| 119 |
-
"Medium Problems": 0.014084507042253521,
|
| 120 |
-
"Easy Problems": 0.4225352112676056
|
| 121 |
-
}
|
| 122 |
-
},
|
| 123 |
-
{
|
| 124 |
-
"model_id": "deepseek/ep-20250228232227-z44x5",
|
| 125 |
-
"name": "ep-20250228232227-z44x5",
|
| 126 |
-
"developer": "DeepSeek",
|
| 127 |
-
"scores": {
|
| 128 |
-
"Hard Problems": 0.0,
|
| 129 |
-
"Medium Problems": 0.0,
|
| 130 |
-
"Easy Problems": 0.1267605633802817
|
| 131 |
-
}
|
| 132 |
-
},
|
| 133 |
-
{
|
| 134 |
-
"model_id": "deepseek/ep-20250603132404-cgpjm",
|
| 135 |
-
"name": "ep-20250603132404-cgpjm",
|
| 136 |
-
"developer": "DeepSeek",
|
| 137 |
-
"scores": {
|
| 138 |
-
"Hard Problems": 0.0,
|
| 139 |
-
"Medium Problems": 0.08450704225352113,
|
| 140 |
-
"Easy Problems": 0.5774647887323944
|
| 141 |
-
}
|
| 142 |
-
},
|
| 143 |
-
{
|
| 144 |
-
"model_id": "google/gemini-2.5-flash",
|
| 145 |
-
"name": "Gemini 2.5 Flash",
|
| 146 |
-
"developer": "Google",
|
| 147 |
-
"scores": {
|
| 148 |
-
"Hard Problems": 0.0,
|
| 149 |
-
"Medium Problems": 0.028169014084507043,
|
| 150 |
-
"Easy Problems": 0.38028169014084506
|
| 151 |
-
}
|
| 152 |
-
},
|
| 153 |
-
{
|
| 154 |
-
"model_id": "google/gemini-2.5-pro",
|
| 155 |
-
"name": "Gemini 2.5 Pro",
|
| 156 |
-
"developer": "Google",
|
| 157 |
-
"scores": {
|
| 158 |
-
"Hard Problems": 0.014084507042253521,
|
| 159 |
-
"Medium Problems": 0.2112676056338028,
|
| 160 |
-
"Easy Problems": 0.7183098591549296
|
| 161 |
-
}
|
| 162 |
-
},
|
| 163 |
-
{
|
| 164 |
-
"model_id": "kuaishou/kwaipilot-40b-0604",
|
| 165 |
-
"name": "kwaipilot-40b-0604",
|
| 166 |
-
"developer": "Kuaishou",
|
| 167 |
-
"scores": {
|
| 168 |
-
"Hard Problems": 0.0,
|
| 169 |
-
"Medium Problems": 0.07042253521126761,
|
| 170 |
-
"Easy Problems": 0.056338028169014086
|
| 171 |
-
}
|
| 172 |
-
},
|
| 173 |
-
{
|
| 174 |
-
"model_id": "meta/llama-4-maverick",
|
| 175 |
-
"name": "meta/llama-4-maverick",
|
| 176 |
-
"developer": "Meta",
|
| 177 |
-
"scores": {
|
| 178 |
-
"Hard Problems": 0.0,
|
| 179 |
-
"Medium Problems": 0.0,
|
| 180 |
-
"Easy Problems": 0.09859154929577464
|
| 181 |
-
}
|
| 182 |
-
},
|
| 183 |
-
{
|
| 184 |
-
"model_id": "openai/gpt-4.1",
|
| 185 |
-
"name": "openai/gpt-4.1",
|
| 186 |
-
"developer": "OpenAI",
|
| 187 |
-
"scores": {
|
| 188 |
-
"Hard Problems": 0.0,
|
| 189 |
-
"Medium Problems": 0.0,
|
| 190 |
-
"Easy Problems": 0.19718309859154928
|
| 191 |
-
}
|
| 192 |
-
},
|
| 193 |
-
{
|
| 194 |
-
"model_id": "openai/gpt-4o-2024-11-20",
|
| 195 |
-
"name": "GPT-4o 2024-11-20",
|
| 196 |
-
"developer": "OpenAI",
|
| 197 |
-
"scores": {
|
| 198 |
-
"Hard Problems": 0.0,
|
| 199 |
-
"Medium Problems": 0.0,
|
| 200 |
-
"Easy Problems": 0.07042253521126761
|
| 201 |
-
}
|
| 202 |
-
},
|
| 203 |
-
{
|
| 204 |
-
"model_id": "openai/gpt-5-2025-08-07",
|
| 205 |
-
"name": "gpt-5-2025-08-07",
|
| 206 |
-
"developer": "OpenAI",
|
| 207 |
-
"scores": {
|
| 208 |
-
"Hard Problems": 0.0423,
|
| 209 |
-
"Medium Problems": 0.4085,
|
| 210 |
-
"Easy Problems": 0.9014
|
| 211 |
-
}
|
| 212 |
-
},
|
| 213 |
-
{
|
| 214 |
-
"model_id": "openai/gpt-5.2-2025-12-11",
|
| 215 |
-
"name": "gpt-5.2-2025-12-11",
|
| 216 |
-
"developer": "OpenAI",
|
| 217 |
-
"scores": {
|
| 218 |
-
"Hard Problems": 0.1594,
|
| 219 |
-
"Medium Problems": 0.5211,
|
| 220 |
-
"Easy Problems": 0.9014
|
| 221 |
-
}
|
| 222 |
-
},
|
| 223 |
-
{
|
| 224 |
-
"model_id": "openai/gpt-oss-120b",
|
| 225 |
-
"name": "GPT-OSS-120B",
|
| 226 |
-
"developer": "OpenAI",
|
| 227 |
-
"scores": {
|
| 228 |
-
"Hard Problems": 0.0,
|
| 229 |
-
"Medium Problems": 0.11267605633802817,
|
| 230 |
-
"Easy Problems": 0.6619718309859155
|
| 231 |
-
}
|
| 232 |
-
},
|
| 233 |
-
{
|
| 234 |
-
"model_id": "openai/gpt-oss-20b",
|
| 235 |
-
"name": "GPT-OSS-20B",
|
| 236 |
-
"developer": "OpenAI",
|
| 237 |
-
"scores": {
|
| 238 |
-
"Hard Problems": 0.0,
|
| 239 |
-
"Medium Problems": 0.056338028169014086,
|
| 240 |
-
"Easy Problems": 0.5070422535211268
|
| 241 |
-
}
|
| 242 |
-
},
|
| 243 |
-
{
|
| 244 |
-
"model_id": "openai/o3-2025-04-16",
|
| 245 |
-
"name": "o3-2025-04-16",
|
| 246 |
-
"developer": "OpenAI",
|
| 247 |
-
"scores": {
|
| 248 |
-
"Hard Problems": 0.0,
|
| 249 |
-
"Medium Problems": 0.22535211267605634,
|
| 250 |
-
"Easy Problems": 0.7183098591549296
|
| 251 |
-
}
|
| 252 |
-
},
|
| 253 |
-
{
|
| 254 |
-
"model_id": "openai/o4-mini-2025-04-16",
|
| 255 |
-
"name": "o4-mini-2025-04-16",
|
| 256 |
-
"developer": "OpenAI",
|
| 257 |
-
"scores": {
|
| 258 |
-
"Hard Problems": 0.0143,
|
| 259 |
-
"Medium Problems": 0.2923,
|
| 260 |
-
"Easy Problems": 0.8571
|
| 261 |
-
}
|
| 262 |
-
},
|
| 263 |
-
{
|
| 264 |
-
"model_id": "z-ai/glm-4.5",
|
| 265 |
-
"name": "z-ai/glm-4.5",
|
| 266 |
-
"developer": "Z.ai",
|
| 267 |
-
"scores": {
|
| 268 |
-
"Hard Problems": 0.0,
|
| 269 |
-
"Medium Problems": 0.028169014084507043,
|
| 270 |
-
"Easy Problems": 0.1267605633802817
|
| 271 |
-
}
|
| 272 |
-
}
|
| 273 |
-
]
|
| 274 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/reward-bench.json
DELETED
|
The diff for this file is too large to render.
See raw diff
|
|
|
data/benchmarks/swe-bench.json
DELETED
|
@@ -1,28 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "anthropic/claude-opus-4-5",
|
| 5 |
-
"name": "claude-opus-4-5",
|
| 6 |
-
"developer": "Anthropic",
|
| 7 |
-
"scores": {
|
| 8 |
-
"swe-bench": 0.65
|
| 9 |
-
}
|
| 10 |
-
},
|
| 11 |
-
{
|
| 12 |
-
"model_id": "google/gemini-3-pro-preview",
|
| 13 |
-
"name": "gemini-3-pro-preview",
|
| 14 |
-
"developer": "Google",
|
| 15 |
-
"scores": {
|
| 16 |
-
"swe-bench": 0.71
|
| 17 |
-
}
|
| 18 |
-
},
|
| 19 |
-
{
|
| 20 |
-
"model_id": "openai/gpt-5.2-2025-12-11",
|
| 21 |
-
"name": "gpt-5.2-2025-12-11",
|
| 22 |
-
"developer": "OpenAI",
|
| 23 |
-
"scores": {
|
| 24 |
-
"swe-bench": 0.57
|
| 25 |
-
}
|
| 26 |
-
}
|
| 27 |
-
]
|
| 28 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/tau-bench-2_airline.json
DELETED
|
@@ -1,28 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "anthropic/claude-opus-4-5",
|
| 5 |
-
"name": "claude-opus-4-5",
|
| 6 |
-
"developer": "Anthropic",
|
| 7 |
-
"scores": {
|
| 8 |
-
"tau-bench-2/airline": 0.66
|
| 9 |
-
}
|
| 10 |
-
},
|
| 11 |
-
{
|
| 12 |
-
"model_id": "google/gemini-3-pro-preview",
|
| 13 |
-
"name": "gemini-3-pro-preview",
|
| 14 |
-
"developer": "Google",
|
| 15 |
-
"scores": {
|
| 16 |
-
"tau-bench-2/airline": 0.62
|
| 17 |
-
}
|
| 18 |
-
},
|
| 19 |
-
{
|
| 20 |
-
"model_id": "openai/gpt-5.2-2025-12-11",
|
| 21 |
-
"name": "gpt-5.2-2025-12-11",
|
| 22 |
-
"developer": "OpenAI",
|
| 23 |
-
"scores": {
|
| 24 |
-
"tau-bench-2/airline": 0.54
|
| 25 |
-
}
|
| 26 |
-
}
|
| 27 |
-
]
|
| 28 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/tau-bench-2_retail.json
DELETED
|
@@ -1,28 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "anthropic/claude-opus-4-5",
|
| 5 |
-
"name": "claude-opus-4-5",
|
| 6 |
-
"developer": "Anthropic",
|
| 7 |
-
"scores": {
|
| 8 |
-
"tau-bench-2/retail": 0.78
|
| 9 |
-
}
|
| 10 |
-
},
|
| 11 |
-
{
|
| 12 |
-
"model_id": "google/gemini-3-pro-preview",
|
| 13 |
-
"name": "gemini-3-pro-preview",
|
| 14 |
-
"developer": "Google",
|
| 15 |
-
"scores": {
|
| 16 |
-
"tau-bench-2/retail": 0.7576
|
| 17 |
-
}
|
| 18 |
-
},
|
| 19 |
-
{
|
| 20 |
-
"model_id": "openai/gpt-5.2-2025-12-11",
|
| 21 |
-
"name": "gpt-5.2-2025-12-11",
|
| 22 |
-
"developer": "OpenAI",
|
| 23 |
-
"scores": {
|
| 24 |
-
"tau-bench-2/retail": 0.68
|
| 25 |
-
}
|
| 26 |
-
}
|
| 27 |
-
]
|
| 28 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/tau-bench-2_telecom.json
DELETED
|
@@ -1,28 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "anthropic/claude-opus-4-5",
|
| 5 |
-
"name": "claude-opus-4-5",
|
| 6 |
-
"developer": "Anthropic",
|
| 7 |
-
"scores": {
|
| 8 |
-
"tau-bench-2/telecom": 0.84
|
| 9 |
-
}
|
| 10 |
-
},
|
| 11 |
-
{
|
| 12 |
-
"model_id": "google/gemini-3-pro-preview",
|
| 13 |
-
"name": "gemini-3-pro-preview",
|
| 14 |
-
"developer": "Google",
|
| 15 |
-
"scores": {
|
| 16 |
-
"tau-bench-2/telecom": 0.73
|
| 17 |
-
}
|
| 18 |
-
},
|
| 19 |
-
{
|
| 20 |
-
"model_id": "openai/gpt-5.2-2025-12-11",
|
| 21 |
-
"name": "gpt-5.2-2025-12-11",
|
| 22 |
-
"developer": "OpenAI",
|
| 23 |
-
"scores": {
|
| 24 |
-
"tau-bench-2/telecom": 0.5354
|
| 25 |
-
}
|
| 26 |
-
}
|
| 27 |
-
]
|
| 28 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/terminal-bench-2.0.json
DELETED
|
@@ -1,300 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "alibaba/qwen-3-coder-480b",
|
| 5 |
-
"name": "Qwen 3 Coder 480B",
|
| 6 |
-
"developer": "Alibaba",
|
| 7 |
-
"scores": {
|
| 8 |
-
"terminal-bench-2.0": 23.9
|
| 9 |
-
}
|
| 10 |
-
},
|
| 11 |
-
{
|
| 12 |
-
"model_id": "anthropic/claude-haiku-4.5",
|
| 13 |
-
"name": "Claude Haiku 4.5",
|
| 14 |
-
"developer": "Anthropic",
|
| 15 |
-
"scores": {
|
| 16 |
-
"terminal-bench-2.0": 35.5
|
| 17 |
-
}
|
| 18 |
-
},
|
| 19 |
-
{
|
| 20 |
-
"model_id": "anthropic/claude-opus-4.1",
|
| 21 |
-
"name": "Claude Opus 4.1",
|
| 22 |
-
"developer": "Anthropic",
|
| 23 |
-
"scores": {
|
| 24 |
-
"terminal-bench-2.0": 35.1
|
| 25 |
-
}
|
| 26 |
-
},
|
| 27 |
-
{
|
| 28 |
-
"model_id": "anthropic/claude-opus-4.5",
|
| 29 |
-
"name": "Claude Opus 4.5",
|
| 30 |
-
"developer": "Anthropic",
|
| 31 |
-
"scores": {
|
| 32 |
-
"terminal-bench-2.0": 63.1
|
| 33 |
-
}
|
| 34 |
-
},
|
| 35 |
-
{
|
| 36 |
-
"model_id": "anthropic/claude-opus-4.6",
|
| 37 |
-
"name": "Claude Opus 4.6",
|
| 38 |
-
"developer": "Anthropic",
|
| 39 |
-
"scores": {
|
| 40 |
-
"terminal-bench-2.0": 74.7
|
| 41 |
-
}
|
| 42 |
-
},
|
| 43 |
-
{
|
| 44 |
-
"model_id": "anthropic/claude-sonnet-4.5",
|
| 45 |
-
"name": "Claude Sonnet 4.5",
|
| 46 |
-
"developer": "Anthropic",
|
| 47 |
-
"scores": {
|
| 48 |
-
"terminal-bench-2.0": 42.5
|
| 49 |
-
}
|
| 50 |
-
},
|
| 51 |
-
{
|
| 52 |
-
"model_id": "deepseek/deepseek-v3.2",
|
| 53 |
-
"name": "DeepSeek-V3.2",
|
| 54 |
-
"developer": "DeepSeek",
|
| 55 |
-
"scores": {
|
| 56 |
-
"terminal-bench-2.0": 39.6
|
| 57 |
-
}
|
| 58 |
-
},
|
| 59 |
-
{
|
| 60 |
-
"model_id": "google/gemini-2.5-flash",
|
| 61 |
-
"name": "Gemini 2.5 Flash",
|
| 62 |
-
"developer": "Google",
|
| 63 |
-
"scores": {
|
| 64 |
-
"terminal-bench-2.0": 16.9
|
| 65 |
-
}
|
| 66 |
-
},
|
| 67 |
-
{
|
| 68 |
-
"model_id": "google/gemini-2.5-pro",
|
| 69 |
-
"name": "Gemini 2.5 Pro",
|
| 70 |
-
"developer": "Google",
|
| 71 |
-
"scores": {
|
| 72 |
-
"terminal-bench-2.0": 19.6
|
| 73 |
-
}
|
| 74 |
-
},
|
| 75 |
-
{
|
| 76 |
-
"model_id": "google/gemini-3-flash",
|
| 77 |
-
"name": "Gemini 3 Flash",
|
| 78 |
-
"developer": "Google",
|
| 79 |
-
"scores": {
|
| 80 |
-
"terminal-bench-2.0": 47.4
|
| 81 |
-
}
|
| 82 |
-
},
|
| 83 |
-
{
|
| 84 |
-
"model_id": "google/gemini-3-pro",
|
| 85 |
-
"name": "Gemini 3 Pro",
|
| 86 |
-
"developer": "Google",
|
| 87 |
-
"scores": {
|
| 88 |
-
"terminal-bench-2.0": 56.9
|
| 89 |
-
}
|
| 90 |
-
},
|
| 91 |
-
{
|
| 92 |
-
"model_id": "google/gemini-3.1-pro",
|
| 93 |
-
"name": "Gemini 3.1 Pro",
|
| 94 |
-
"developer": "Google",
|
| 95 |
-
"scores": {
|
| 96 |
-
"terminal-bench-2.0": 74.8
|
| 97 |
-
}
|
| 98 |
-
},
|
| 99 |
-
{
|
| 100 |
-
"model_id": "minimax/minimax-m2",
|
| 101 |
-
"name": "MiniMax M2",
|
| 102 |
-
"developer": "MiniMax",
|
| 103 |
-
"scores": {
|
| 104 |
-
"terminal-bench-2.0": 30.0
|
| 105 |
-
}
|
| 106 |
-
},
|
| 107 |
-
{
|
| 108 |
-
"model_id": "minimax/minimax-m2.1",
|
| 109 |
-
"name": "MiniMax M2.1",
|
| 110 |
-
"developer": "MiniMax",
|
| 111 |
-
"scores": {
|
| 112 |
-
"terminal-bench-2.0": 29.2
|
| 113 |
-
}
|
| 114 |
-
},
|
| 115 |
-
{
|
| 116 |
-
"model_id": "minimax/minimax-m2.5",
|
| 117 |
-
"name": "Minimax m2.5",
|
| 118 |
-
"developer": "Minimax",
|
| 119 |
-
"scores": {
|
| 120 |
-
"terminal-bench-2.0": 42.2
|
| 121 |
-
}
|
| 122 |
-
},
|
| 123 |
-
{
|
| 124 |
-
"model_id": "moonshot-ai/kimi-k2-instruct",
|
| 125 |
-
"name": "Kimi K2 Instruct",
|
| 126 |
-
"developer": "Moonshot AI",
|
| 127 |
-
"scores": {
|
| 128 |
-
"terminal-bench-2.0": 27.8
|
| 129 |
-
}
|
| 130 |
-
},
|
| 131 |
-
{
|
| 132 |
-
"model_id": "moonshot-ai/kimi-k2-thinking",
|
| 133 |
-
"name": "Kimi K2 Thinking",
|
| 134 |
-
"developer": "Moonshot AI",
|
| 135 |
-
"scores": {
|
| 136 |
-
"terminal-bench-2.0": 35.7
|
| 137 |
-
}
|
| 138 |
-
},
|
| 139 |
-
{
|
| 140 |
-
"model_id": "moonshot-ai/kimi-k2.5",
|
| 141 |
-
"name": "Kimi K2.5",
|
| 142 |
-
"developer": "Kimi",
|
| 143 |
-
"scores": {
|
| 144 |
-
"terminal-bench-2.0": 43.2
|
| 145 |
-
}
|
| 146 |
-
},
|
| 147 |
-
{
|
| 148 |
-
"model_id": "multiple/multiple",
|
| 149 |
-
"name": "Multiple",
|
| 150 |
-
"developer": "Multiple",
|
| 151 |
-
"scores": {
|
| 152 |
-
"terminal-bench-2.0": 59.1
|
| 153 |
-
}
|
| 154 |
-
},
|
| 155 |
-
{
|
| 156 |
-
"model_id": "openai/gpt-5",
|
| 157 |
-
"name": "GPT-5",
|
| 158 |
-
"developer": "OpenAI",
|
| 159 |
-
"scores": {
|
| 160 |
-
"terminal-bench-2.0": 49.6
|
| 161 |
-
}
|
| 162 |
-
},
|
| 163 |
-
{
|
| 164 |
-
"model_id": "openai/gpt-5-codex",
|
| 165 |
-
"name": "GPT-5-Codex",
|
| 166 |
-
"developer": "OpenAI",
|
| 167 |
-
"scores": {
|
| 168 |
-
"terminal-bench-2.0": 44.3
|
| 169 |
-
}
|
| 170 |
-
},
|
| 171 |
-
{
|
| 172 |
-
"model_id": "openai/gpt-5-mini",
|
| 173 |
-
"name": "GPT-5-Mini",
|
| 174 |
-
"developer": "OpenAI",
|
| 175 |
-
"scores": {
|
| 176 |
-
"terminal-bench-2.0": 31.9
|
| 177 |
-
}
|
| 178 |
-
},
|
| 179 |
-
{
|
| 180 |
-
"model_id": "openai/gpt-5-nano",
|
| 181 |
-
"name": "GPT-5-Nano",
|
| 182 |
-
"developer": "OpenAI",
|
| 183 |
-
"scores": {
|
| 184 |
-
"terminal-bench-2.0": 7.0
|
| 185 |
-
}
|
| 186 |
-
},
|
| 187 |
-
{
|
| 188 |
-
"model_id": "openai/gpt-5.1",
|
| 189 |
-
"name": "GPT-5.1",
|
| 190 |
-
"developer": "OpenAI",
|
| 191 |
-
"scores": {
|
| 192 |
-
"terminal-bench-2.0": 47.6
|
| 193 |
-
}
|
| 194 |
-
},
|
| 195 |
-
{
|
| 196 |
-
"model_id": "openai/gpt-5.1-codex",
|
| 197 |
-
"name": "GPT-5.1-Codex",
|
| 198 |
-
"developer": "OpenAI",
|
| 199 |
-
"scores": {
|
| 200 |
-
"terminal-bench-2.0": 53.5
|
| 201 |
-
}
|
| 202 |
-
},
|
| 203 |
-
{
|
| 204 |
-
"model_id": "openai/gpt-5.1-codex-max",
|
| 205 |
-
"name": "GPT-5.1-Codex-Max",
|
| 206 |
-
"developer": "OpenAI",
|
| 207 |
-
"scores": {
|
| 208 |
-
"terminal-bench-2.0": 60.4
|
| 209 |
-
}
|
| 210 |
-
},
|
| 211 |
-
{
|
| 212 |
-
"model_id": "openai/gpt-5.1-codex-mini",
|
| 213 |
-
"name": "GPT-5.1-Codex-Mini",
|
| 214 |
-
"developer": "OpenAI",
|
| 215 |
-
"scores": {
|
| 216 |
-
"terminal-bench-2.0": 43.1
|
| 217 |
-
}
|
| 218 |
-
},
|
| 219 |
-
{
|
| 220 |
-
"model_id": "openai/gpt-5.2",
|
| 221 |
-
"name": "GPT-5.2",
|
| 222 |
-
"developer": "OpenAI",
|
| 223 |
-
"scores": {
|
| 224 |
-
"terminal-bench-2.0": 64.9
|
| 225 |
-
}
|
| 226 |
-
},
|
| 227 |
-
{
|
| 228 |
-
"model_id": "openai/gpt-5.2-codex",
|
| 229 |
-
"name": "GPT-5.2-Codex",
|
| 230 |
-
"developer": "OpenAI",
|
| 231 |
-
"scores": {
|
| 232 |
-
"terminal-bench-2.0": 66.5
|
| 233 |
-
}
|
| 234 |
-
},
|
| 235 |
-
{
|
| 236 |
-
"model_id": "openai/gpt-5.3-codex",
|
| 237 |
-
"name": "GPT-5.3-Codex",
|
| 238 |
-
"developer": "OpenAI",
|
| 239 |
-
"scores": {
|
| 240 |
-
"terminal-bench-2.0": 74.6
|
| 241 |
-
}
|
| 242 |
-
},
|
| 243 |
-
{
|
| 244 |
-
"model_id": "openai/gpt-oss-120b",
|
| 245 |
-
"name": "GPT-OSS-120B",
|
| 246 |
-
"developer": "OpenAI",
|
| 247 |
-
"scores": {
|
| 248 |
-
"terminal-bench-2.0": 18.7
|
| 249 |
-
}
|
| 250 |
-
},
|
| 251 |
-
{
|
| 252 |
-
"model_id": "openai/gpt-oss-20b",
|
| 253 |
-
"name": "GPT-OSS-20B",
|
| 254 |
-
"developer": "OpenAI",
|
| 255 |
-
"scores": {
|
| 256 |
-
"terminal-bench-2.0": 3.4
|
| 257 |
-
}
|
| 258 |
-
},
|
| 259 |
-
{
|
| 260 |
-
"model_id": "xai/grok-4",
|
| 261 |
-
"name": "Grok 4",
|
| 262 |
-
"developer": "xAI",
|
| 263 |
-
"scores": {
|
| 264 |
-
"terminal-bench-2.0": 23.1
|
| 265 |
-
}
|
| 266 |
-
},
|
| 267 |
-
{
|
| 268 |
-
"model_id": "xai/grok-code-fast-1",
|
| 269 |
-
"name": "Grok Code Fast 1",
|
| 270 |
-
"developer": "xAI",
|
| 271 |
-
"scores": {
|
| 272 |
-
"terminal-bench-2.0": 14.2
|
| 273 |
-
}
|
| 274 |
-
},
|
| 275 |
-
{
|
| 276 |
-
"model_id": "zhipu-ai/glm-4.6",
|
| 277 |
-
"name": "GLM 4.6",
|
| 278 |
-
"developer": "Z.ai",
|
| 279 |
-
"scores": {
|
| 280 |
-
"terminal-bench-2.0": 24.5
|
| 281 |
-
}
|
| 282 |
-
},
|
| 283 |
-
{
|
| 284 |
-
"model_id": "zhipu-ai/glm-4.7",
|
| 285 |
-
"name": "GLM 4.7",
|
| 286 |
-
"developer": "Z-AI",
|
| 287 |
-
"scores": {
|
| 288 |
-
"terminal-bench-2.0": 33.3
|
| 289 |
-
}
|
| 290 |
-
},
|
| 291 |
-
{
|
| 292 |
-
"model_id": "zhipu-ai/glm-5",
|
| 293 |
-
"name": "GLM 5",
|
| 294 |
-
"developer": "Z-AI",
|
| 295 |
-
"scores": {
|
| 296 |
-
"terminal-bench-2.0": 52.4
|
| 297 |
-
}
|
| 298 |
-
}
|
| 299 |
-
]
|
| 300 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
data/benchmarks/theory_of_mind.json
DELETED
|
@@ -1,12 +0,0 @@
|
|
| 1 |
-
{
|
| 2 |
-
"models": [
|
| 3 |
-
{
|
| 4 |
-
"model_id": "Qwen/Qwen2.5-3B-Instruct",
|
| 5 |
-
"name": "Qwen2.5-3B-Instruct",
|
| 6 |
-
"developer": "Qwen",
|
| 7 |
-
"scores": {
|
| 8 |
-
"accuracy on theory_of_mind for scorer model_graded_fact": 0.78
|
| 9 |
-
}
|
| 10 |
-
}
|
| 11 |
-
]
|
| 12 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
lib/benchmark-metadata.ts
CHANGED
|
@@ -1,98 +1,66 @@
|
|
| 1 |
import "server-only"
|
| 2 |
|
| 3 |
-
import { promises as fs, type Dirent } from "fs"
|
| 4 |
-
import path from "path"
|
| 5 |
-
|
| 6 |
import type { BenchmarkCard } from "@/lib/benchmark-schema"
|
| 7 |
-
import {
|
|
|
|
| 8 |
|
| 9 |
export { normalizeBenchmarkKey }
|
| 10 |
|
| 11 |
-
|
| 12 |
-
benchmark_cards?: Record<string, BenchmarkCard>
|
| 13 |
-
}
|
| 14 |
-
|
| 15 |
-
function getBenchmarkDataDirectory() {
|
| 16 |
-
return path.join(process.cwd(), "data", "benchmarks")
|
| 17 |
-
}
|
| 18 |
|
| 19 |
-
async function
|
| 20 |
-
const
|
| 21 |
const map = new Map<string, BenchmarkCard>()
|
| 22 |
|
| 23 |
-
|
| 24 |
-
|
| 25 |
-
|
| 26 |
-
|
| 27 |
-
return map
|
| 28 |
-
}
|
| 29 |
|
| 30 |
-
|
| 31 |
-
|
| 32 |
-
|
| 33 |
-
jsonFiles.map(async (entry) => {
|
| 34 |
-
try {
|
| 35 |
-
const raw = await fs.readFile(path.join(dir, entry.name), "utf8")
|
| 36 |
-
const parsed = JSON.parse(raw) as IndexedBenchmarkDetailFile
|
| 37 |
-
const embeddedCards = parsed.benchmark_cards
|
| 38 |
-
|
| 39 |
-
if (!embeddedCards || typeof embeddedCards !== "object") {
|
| 40 |
-
return
|
| 41 |
-
}
|
| 42 |
-
|
| 43 |
-
for (const [metricName, card] of Object.entries(embeddedCards)) {
|
| 44 |
-
if (!card?.benchmark_details?.name) {
|
| 45 |
-
continue
|
| 46 |
-
}
|
| 47 |
-
|
| 48 |
-
for (const key of candidateKeys(metricName)) {
|
| 49 |
-
if (!map.has(key)) map.set(key, card)
|
| 50 |
-
}
|
| 51 |
-
|
| 52 |
-
for (const key of candidateKeys(card.benchmark_details.name)) {
|
| 53 |
-
if (!map.has(key)) map.set(key, card)
|
| 54 |
-
}
|
| 55 |
-
}
|
| 56 |
-
} catch (err) {
|
| 57 |
-
console.warn(`benchmark-metadata: failed to load embedded cards from ${entry.name}:`, err)
|
| 58 |
}
|
| 59 |
-
}
|
| 60 |
-
|
| 61 |
|
| 62 |
return map
|
| 63 |
}
|
| 64 |
|
| 65 |
-
let cachedMapPromise: Promise<Map<string, BenchmarkCard>> | null = null
|
| 66 |
-
|
| 67 |
function getMap(): Promise<Map<string, BenchmarkCard>> {
|
| 68 |
-
if (
|
| 69 |
-
|
| 70 |
-
return cachedMapPromise
|
| 71 |
}
|
| 72 |
-
|
|
|
|
| 73 |
}
|
| 74 |
|
| 75 |
-
/** Look up a BenchmarkCard by any commonly-used benchmark name. Returns null if not found. */
|
| 76 |
export async function getBenchmarkCard(benchmarkName: string): Promise<BenchmarkCard | null> {
|
| 77 |
const map = await getMap()
|
|
|
|
| 78 |
for (const key of candidateKeys(benchmarkName)) {
|
| 79 |
const card = map.get(key)
|
| 80 |
-
if (card)
|
|
|
|
|
|
|
| 81 |
}
|
|
|
|
| 82 |
return null
|
| 83 |
}
|
| 84 |
|
| 85 |
-
/** Returns all loaded BenchmarkCards keyed by their normalised canonical name. */
|
| 86 |
export async function getAllBenchmarkCards(): Promise<Record<string, BenchmarkCard>> {
|
| 87 |
const map = await getMap()
|
| 88 |
-
// Deduplicate: only emit one entry per card (by canonical name)
|
| 89 |
const seen = new Set<BenchmarkCard>()
|
| 90 |
const result: Record<string, BenchmarkCard> = {}
|
| 91 |
-
|
| 92 |
-
|
| 93 |
-
|
| 94 |
-
|
| 95 |
}
|
|
|
|
|
|
|
|
|
|
| 96 |
}
|
|
|
|
| 97 |
return result
|
| 98 |
}
|
|
|
|
| 1 |
import "server-only"
|
| 2 |
|
|
|
|
|
|
|
|
|
|
| 3 |
import type { BenchmarkCard } from "@/lib/benchmark-schema"
|
| 4 |
+
import { candidateBenchmarkKeys as candidateKeys, normalizeBenchmarkKey } from "@/lib/benchmark-metadata-utils"
|
| 5 |
+
import { fetchBenchmarkMetadataMap } from "@/lib/hf-data"
|
| 6 |
|
| 7 |
export { normalizeBenchmarkKey }
|
| 8 |
|
| 9 |
+
let cachedMapPromise: Promise<Map<string, BenchmarkCard>> | null = null
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 10 |
|
| 11 |
+
async function readPipelineBenchmarkCards(): Promise<Map<string, BenchmarkCard>> {
|
| 12 |
+
const cards = await fetchBenchmarkMetadataMap()
|
| 13 |
const map = new Map<string, BenchmarkCard>()
|
| 14 |
|
| 15 |
+
for (const card of Object.values(cards)) {
|
| 16 |
+
if (!card?.benchmark_details?.name) {
|
| 17 |
+
continue
|
| 18 |
+
}
|
|
|
|
|
|
|
| 19 |
|
| 20 |
+
for (const key of candidateKeys(card.benchmark_details.name)) {
|
| 21 |
+
if (!map.has(key)) {
|
| 22 |
+
map.set(key, card)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 23 |
}
|
| 24 |
+
}
|
| 25 |
+
}
|
| 26 |
|
| 27 |
return map
|
| 28 |
}
|
| 29 |
|
|
|
|
|
|
|
| 30 |
function getMap(): Promise<Map<string, BenchmarkCard>> {
|
| 31 |
+
if (!cachedMapPromise) {
|
| 32 |
+
cachedMapPromise = readPipelineBenchmarkCards()
|
|
|
|
| 33 |
}
|
| 34 |
+
|
| 35 |
+
return cachedMapPromise
|
| 36 |
}
|
| 37 |
|
|
|
|
| 38 |
export async function getBenchmarkCard(benchmarkName: string): Promise<BenchmarkCard | null> {
|
| 39 |
const map = await getMap()
|
| 40 |
+
|
| 41 |
for (const key of candidateKeys(benchmarkName)) {
|
| 42 |
const card = map.get(key)
|
| 43 |
+
if (card) {
|
| 44 |
+
return card
|
| 45 |
+
}
|
| 46 |
}
|
| 47 |
+
|
| 48 |
return null
|
| 49 |
}
|
| 50 |
|
|
|
|
| 51 |
export async function getAllBenchmarkCards(): Promise<Record<string, BenchmarkCard>> {
|
| 52 |
const map = await getMap()
|
|
|
|
| 53 |
const seen = new Set<BenchmarkCard>()
|
| 54 |
const result: Record<string, BenchmarkCard> = {}
|
| 55 |
+
|
| 56 |
+
for (const card of map.values()) {
|
| 57 |
+
if (seen.has(card)) {
|
| 58 |
+
continue
|
| 59 |
}
|
| 60 |
+
|
| 61 |
+
seen.add(card)
|
| 62 |
+
result[normalizeBenchmarkKey(card.benchmark_details.name)] = card
|
| 63 |
}
|
| 64 |
+
|
| 65 |
return result
|
| 66 |
}
|
lib/model-data.ts
CHANGED
|
@@ -1039,6 +1039,179 @@ function aggregateBenchmarkSummaries(
|
|
| 1039 |
}
|
| 1040 |
}
|
| 1041 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1042 |
// ---------------------------------------------------------------------------
|
| 1043 |
// Public API
|
| 1044 |
// ---------------------------------------------------------------------------
|
|
@@ -1361,6 +1534,34 @@ export async function getEvalSummaryById(evalId: string) {
|
|
| 1361 |
return aggregateBenchmarkSummaries(validSummaries, aggregateKey)
|
| 1362 |
}
|
| 1363 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1364 |
// Direct eval lookup
|
| 1365 |
const detail = await fetchHFEvalDetail(evalId)
|
| 1366 |
if (detail) {
|
|
|
|
| 1039 |
}
|
| 1040 |
}
|
| 1041 |
|
| 1042 |
+
const SYNTHETIC_MATRIX_EVAL_PREFIX = "matrix__"
|
| 1043 |
+
|
| 1044 |
+
function buildSingleMetricSuiteMatrixSummary(
|
| 1045 |
+
details: HFEvalDetail[],
|
| 1046 |
+
suiteKey: string
|
| 1047 |
+
): BenchmarkEvalSummary | null {
|
| 1048 |
+
if (details.length < 2) {
|
| 1049 |
+
return null
|
| 1050 |
+
}
|
| 1051 |
+
|
| 1052 |
+
const suiteDisplayName = getBenchmarkDisplayName(suiteKey)
|
| 1053 |
+
const validDetails = [...details]
|
| 1054 |
+
.filter((detail) => (detail.metrics?.length ?? 0) === 1 && extractDetailSubtasks(detail).length === 0)
|
| 1055 |
+
.sort((left, right) =>
|
| 1056 |
+
(left.benchmark_leaf_name || left.eval_summary_id).localeCompare(right.benchmark_leaf_name || right.eval_summary_id)
|
| 1057 |
+
)
|
| 1058 |
+
|
| 1059 |
+
if (validDetails.length < 2) {
|
| 1060 |
+
return null
|
| 1061 |
+
}
|
| 1062 |
+
|
| 1063 |
+
const leaderboardMetrics: NonNullable<BenchmarkEvalSummary["leaderboard_metrics"]> = []
|
| 1064 |
+
const rowStates = new Map<
|
| 1065 |
+
string,
|
| 1066 |
+
NonNullable<BenchmarkEvalSummary["leaderboard_rows"]>[number] & { _timestampValue: number }
|
| 1067 |
+
>()
|
| 1068 |
+
|
| 1069 |
+
let metricConfig: BenchmarkEvalSummary["metric_config"] | null = null
|
| 1070 |
+
let benchmarkCard: BenchmarkCard | undefined
|
| 1071 |
+
const metricNames = new Set<string>()
|
| 1072 |
+
|
| 1073 |
+
for (const detail of validDetails) {
|
| 1074 |
+
const metric = detail.metrics?.[0]
|
| 1075 |
+
if (!metric) {
|
| 1076 |
+
continue
|
| 1077 |
+
}
|
| 1078 |
+
|
| 1079 |
+
if (!metricConfig) {
|
| 1080 |
+
metricConfig = toSummaryMetricConfig(metric)
|
| 1081 |
+
}
|
| 1082 |
+
|
| 1083 |
+
if (!benchmarkCard && detail.benchmark_card) {
|
| 1084 |
+
benchmarkCard = detail.benchmark_card
|
| 1085 |
+
}
|
| 1086 |
+
|
| 1087 |
+
const summaryMetric = toBenchmarkSummaryMetric(metric)
|
| 1088 |
+
metricNames.add(summaryMetric.metric_name)
|
| 1089 |
+
const subtaskKey = detail.benchmark_leaf_key || slugifyEvalId(detail.eval_summary_id)
|
| 1090 |
+
const subtaskName = detail.benchmark_leaf_name || detail.canonical_display_name || detail.eval_summary_id || subtaskKey
|
| 1091 |
+
const metricToken =
|
| 1092 |
+
summaryMetric.metric_summary_id ||
|
| 1093 |
+
summaryMetric.metric_key ||
|
| 1094 |
+
slugifyEvalId(summaryMetric.display_name)
|
| 1095 |
+
const columnKey = ["subtask", subtaskKey, metricToken].join(":")
|
| 1096 |
+
|
| 1097 |
+
leaderboardMetrics.push({
|
| 1098 |
+
column_key: columnKey,
|
| 1099 |
+
metric_summary_id: summaryMetric.metric_summary_id,
|
| 1100 |
+
metric_name: summaryMetric.metric_name,
|
| 1101 |
+
display_name: summaryMetric.display_name,
|
| 1102 |
+
canonical_display_name: summaryMetric.canonical_display_name,
|
| 1103 |
+
lower_is_better: summaryMetric.lower_is_better,
|
| 1104 |
+
unit: summaryMetric.unit,
|
| 1105 |
+
scope: "subtask",
|
| 1106 |
+
subtask_key: subtaskKey,
|
| 1107 |
+
subtask_name: subtaskName,
|
| 1108 |
+
})
|
| 1109 |
+
|
| 1110 |
+
const benchmarkKey = detail.benchmark ?? suiteKey
|
| 1111 |
+
const sourceName = detail.source_data?.dataset_name || benchmarkKey
|
| 1112 |
+
const sourceOrganization = detail.source_data?.hf_repo || sourceName
|
| 1113 |
+
const sourceMetadata: SourceMetadata = {
|
| 1114 |
+
source_type: "documentation",
|
| 1115 |
+
source_name: sourceName,
|
| 1116 |
+
source_organization_name: sourceOrganization,
|
| 1117 |
+
evaluator_relationship: "other",
|
| 1118 |
+
}
|
| 1119 |
+
const sourceData = detail.source_data ?? { dataset_name: benchmarkKey }
|
| 1120 |
+
|
| 1121 |
+
for (const modelResult of metric.model_results ?? []) {
|
| 1122 |
+
const modelId = modelResult.model_id || modelResult.model_name
|
| 1123 |
+
if (!modelId) {
|
| 1124 |
+
continue
|
| 1125 |
+
}
|
| 1126 |
+
|
| 1127 |
+
const nextTimestamp = normalizeEvalTimestamp(modelResult.retrieved_timestamp ?? "")
|
| 1128 |
+
const existing = rowStates.get(modelId)
|
| 1129 |
+
|
| 1130 |
+
if (!existing) {
|
| 1131 |
+
rowStates.set(modelId, {
|
| 1132 |
+
model_info: {
|
| 1133 |
+
name: modelResult.model_name ?? "",
|
| 1134 |
+
id: modelId,
|
| 1135 |
+
developer: modelResult.developer ?? "",
|
| 1136 |
+
},
|
| 1137 |
+
model_route_id: modelResult.model_route_id,
|
| 1138 |
+
evaluation_timestamp: modelResult.retrieved_timestamp ?? "",
|
| 1139 |
+
source_metadata: sourceMetadata,
|
| 1140 |
+
source_data: sourceData,
|
| 1141 |
+
values: { [columnKey]: modelResult.score ?? null },
|
| 1142 |
+
metrics_present: 0,
|
| 1143 |
+
_timestampValue: nextTimestamp,
|
| 1144 |
+
})
|
| 1145 |
+
continue
|
| 1146 |
+
}
|
| 1147 |
+
|
| 1148 |
+
existing.values[columnKey] = modelResult.score ?? null
|
| 1149 |
+
if (!existing.model_route_id && modelResult.model_route_id) {
|
| 1150 |
+
existing.model_route_id = modelResult.model_route_id
|
| 1151 |
+
}
|
| 1152 |
+
if (nextTimestamp >= existing._timestampValue) {
|
| 1153 |
+
existing.evaluation_timestamp = modelResult.retrieved_timestamp ?? existing.evaluation_timestamp
|
| 1154 |
+
existing.source_metadata = sourceMetadata
|
| 1155 |
+
existing.source_data = sourceData
|
| 1156 |
+
existing._timestampValue = nextTimestamp
|
| 1157 |
+
}
|
| 1158 |
+
}
|
| 1159 |
+
}
|
| 1160 |
+
|
| 1161 |
+
if (leaderboardMetrics.length < 2) {
|
| 1162 |
+
return null
|
| 1163 |
+
}
|
| 1164 |
+
|
| 1165 |
+
const sharedMetricName = metricNames.size === 1 ? Array.from(metricNames)[0] : undefined
|
| 1166 |
+
const suiteMetricConfig = metricConfig
|
| 1167 |
+
? {
|
| 1168 |
+
...metricConfig,
|
| 1169 |
+
evaluation_description: sharedMetricName ?? metricConfig.evaluation_description,
|
| 1170 |
+
}
|
| 1171 |
+
: {
|
| 1172 |
+
evaluation_description: sharedMetricName ?? "",
|
| 1173 |
+
lower_is_better: false,
|
| 1174 |
+
score_type: "continuous" as const,
|
| 1175 |
+
min_score: 0,
|
| 1176 |
+
max_score: 1,
|
| 1177 |
+
}
|
| 1178 |
+
|
| 1179 |
+
const leaderboardRows = Array.from(rowStates.values()).map(({ _timestampValue, ...row }) => ({
|
| 1180 |
+
...row,
|
| 1181 |
+
metrics_present: leaderboardMetrics.reduce(
|
| 1182 |
+
(count, metric) => count + (typeof row.values[metric.column_key] === "number" ? 1 : 0),
|
| 1183 |
+
0
|
| 1184 |
+
),
|
| 1185 |
+
}))
|
| 1186 |
+
|
| 1187 |
+
return {
|
| 1188 |
+
evaluation_name: suiteDisplayName,
|
| 1189 |
+
evaluation_id: `${SYNTHETIC_MATRIX_EVAL_PREFIX}${suiteKey}`,
|
| 1190 |
+
canonical_display_name: suiteDisplayName,
|
| 1191 |
+
composite_benchmark_key: suiteKey,
|
| 1192 |
+
composite_benchmark_name: suiteDisplayName,
|
| 1193 |
+
category: inferCategoryFromBenchmark(suiteDisplayName),
|
| 1194 |
+
metric_config: suiteMetricConfig,
|
| 1195 |
+
model_results: [],
|
| 1196 |
+
models_count: leaderboardRows.length,
|
| 1197 |
+
evaluator_names: [],
|
| 1198 |
+
source_types: [],
|
| 1199 |
+
latest_source_name: suiteDisplayName,
|
| 1200 |
+
third_party_ratio: 0,
|
| 1201 |
+
missing_generation_config_count: 0,
|
| 1202 |
+
best_model: null,
|
| 1203 |
+
worst_model: null,
|
| 1204 |
+
avg_score: 0,
|
| 1205 |
+
avg_score_norm: 0,
|
| 1206 |
+
benchmark_card: benchmarkCard,
|
| 1207 |
+
metrics_count: leaderboardMetrics.length,
|
| 1208 |
+
metric_names: leaderboardMetrics.map((metric) => `${metric.subtask_name} / ${metric.metric_name}`),
|
| 1209 |
+
source_data: { dataset_name: suiteDisplayName },
|
| 1210 |
+
leaderboard_metrics: leaderboardMetrics,
|
| 1211 |
+
leaderboard_rows: leaderboardRows,
|
| 1212 |
+
}
|
| 1213 |
+
}
|
| 1214 |
+
|
| 1215 |
// ---------------------------------------------------------------------------
|
| 1216 |
// Public API
|
| 1217 |
// ---------------------------------------------------------------------------
|
|
|
|
| 1534 |
return aggregateBenchmarkSummaries(validSummaries, aggregateKey)
|
| 1535 |
}
|
| 1536 |
|
| 1537 |
+
if (evalId.startsWith(SYNTHETIC_MATRIX_EVAL_PREFIX)) {
|
| 1538 |
+
const suiteKey = evalId.replace(new RegExp(`^${SYNTHETIC_MATRIX_EVAL_PREFIX}`), "")
|
| 1539 |
+
const normalizedSuiteKey = normalizeBenchmarkKeyForLookup(suiteKey)
|
| 1540 |
+
const { evals } = await fetchHFEvalListLite()
|
| 1541 |
+
const matchingEvals = evals.filter((entry) => {
|
| 1542 |
+
if (entry.is_summary_score) {
|
| 1543 |
+
return false
|
| 1544 |
+
}
|
| 1545 |
+
|
| 1546 |
+
const parentKey = normalizeBenchmarkKeyForLookup(
|
| 1547 |
+
entry.benchmark_parent_key || entry.benchmark_family_key || entry.benchmark
|
| 1548 |
+
)
|
| 1549 |
+
return parentKey === normalizedSuiteKey
|
| 1550 |
+
})
|
| 1551 |
+
|
| 1552 |
+
if (matchingEvals.length < 2) {
|
| 1553 |
+
return null
|
| 1554 |
+
}
|
| 1555 |
+
|
| 1556 |
+
const details = await Promise.all(
|
| 1557 |
+
matchingEvals.map(async (entry) => fetchHFEvalDetail(entry.eval_summary_id))
|
| 1558 |
+
)
|
| 1559 |
+
|
| 1560 |
+
const validDetails = details.filter((detail): detail is HFEvalDetail => detail !== null)
|
| 1561 |
+
const syntheticSummary = buildSingleMetricSuiteMatrixSummary(validDetails, suiteKey)
|
| 1562 |
+
return syntheticSummary ? attachBenchmarkCardToSummary(syntheticSummary) : null
|
| 1563 |
+
}
|
| 1564 |
+
|
| 1565 |
// Direct eval lookup
|
| 1566 |
const detail = await fetchHFEvalDetail(evalId)
|
| 1567 |
if (detail) {
|