/* app.js — Static dashboard logic (artifact-layer aware) */
const HF_BASE = "https://huggingface.co";
const HF_SPACES = "https://huggingface.co/spaces";
const MODELS_FILE = "models.json";
const LINKS_FILE = "links.json";
const $ = (id) => document.getElementById(id);
const $$ = (sel) => Array.from(document.querySelectorAll(sel));
/** Proper HTML escaping. */
function esc(str) {
return String(str ?? "")
.replaceAll("&", "&")
.replaceAll("<", "<")
.replaceAll(">", ">")
.replaceAll('"', """)
.replaceAll("'", "'");
}
let CATALOG_ROOT = { models: [] };
let CATALOG = [];
let CATALOG_NOTE = "Auto-generated from Hugging Face repositories.";
/* Tier-1 KPI metadata from models.json: key -> {label, unit, direction, modalities}.
Lets renderers show a metric's name/unit without hardcoding a second copy of the table
generate_models_json.py already built (see KPI_REGISTRY there). */
let KPI_REGISTRY = {};
let LINKS = { spaces: {}, collections: {} };
let KEY_TO_MODEL = {};
let ARCHS = ["All"];
let MODALITIES = ["All"];
let VARIANTS = ["All"];
const ALL_FORMATS = ["ONNX", "GGUF"];
const ALL_HW = ["CPU", "NPU", "DSP"];
function uniqueSorted(values) {
const out = [];
const seen = new Set();
for (const v of values) {
const x = String(v ?? "Unknown");
if (!seen.has(x)) { seen.add(x); out.push(x); }
}
return out.sort((a, b) => a.localeCompare(b));
}
async function loadJSON(path, fallback) {
try {
const res = await fetch(path, { cache: "no-store" });
if (!res.ok) throw new Error(`HTTP ${res.status} ${res.statusText}`);
return await res.json();
} catch {
return fallback;
}
}
function hfModelUrl(repoId) { return `${HF_BASE}/${repoId}`; }
function hfSpaceUrl(spaceId) { return `${HF_SPACES}/${spaceId}`; }
function safeList(x, fallback = []) { return Array.isArray(x) ? x : fallback; }
function defaultPurposeForFormat(fmt) {
return fmt === "ONNX" ? "PerformanceEstimate" : "llama.cpp Benchmarking";
}
function validatePurpose(fmt, purpose) {
if ((purpose === "PerformanceEstimate" || purpose === "ORT Benchmarking") && fmt !== "ONNX") {
return { ok: false, msg: "⚠️ PerformanceEstimate / ORT benchmarking is ONNX-only. Switch Format → ONNX." };
}
if (purpose === "llama.cpp Benchmarking" && fmt !== "GGUF") {
return { ok: false, msg: "⚠️ llama.cpp benchmarking is GGUF-only. Switch Format → GGUF." };
}
return { ok: true, msg: "" };
}
/* ---------- Badge builders (HTML) ---------- */
function formatBadgesHTML(model) {
const parts = [];
if (model.onnx_repo) parts.push(`ONNX`);
if (model.gguf_repo) parts.push(`GGUF`);
return `
${parts.join("") || `—`}
`;
}
function hwBadgesHTML(targets) {
const t = safeList(targets, []);
const parts = [];
if (t.includes("CPU")) parts.push(`CPU`);
if (t.includes("NPU")) parts.push(`NPU`);
if (t.includes("DSP")) parts.push(`DSP`);
return `
${parts.join("") || `—`}
`;
}
function repoButtonsSmallHTML(model) {
const parts = [];
if (model.onnx_repo) {
parts.push(`ONNX`);
}
if (model.gguf_repo) {
parts.push(`GGUF`);
}
return `
`;
}
/* ---------- Metrics formatting ---------- */
// peak_mem_mb actually holds memory bandwidth (MB/s); the UI always shows it as GB/s.
function memBwGBs(mb) {
return mb == null ? null : Math.round((mb / 1024) * 10) / 10;
}
/* KPI_REGISTRY-driven label/unit lookup for the Tier-1 fields that aren't already
hand-labeled below (tok_s/latency_ms_p50/peak_mem_mb keep their existing wording). */
function kpiLabel(key, fallback) {
return (KPI_REGISTRY[key] && KPI_REGISTRY[key].label) || fallback;
}
function kpiUnit(key, fallback) {
return (KPI_REGISTRY[key] && KPI_REGISTRY[key].unit) || fallback;
}
function summarizeUnit(name, unitObj) {
if (!unitObj) return [];
const lines = [];
if (unitObj.tok_s != null) lines.push(`${name} tok/s: ${unitObj.tok_s}`);
if (unitObj.prefill_tok_s != null) lines.push(`${name} ${kpiLabel("prefill_tok_s", "prefill")} ${kpiUnit("prefill_tok_s", "tok/s")}: ${unitObj.prefill_tok_s}`);
if (unitObj.ttft_ms != null) lines.push(`${name} ${kpiLabel("ttft_ms", "TTFT")} ${kpiUnit("ttft_ms", "ms")}: ${unitObj.ttft_ms}`);
if (unitObj.latency_ms_p50 != null) lines.push(`${name} p50 ms: ${unitObj.latency_ms_p50}`);
if (unitObj.peak_mem_mb != null) lines.push(`${name} mem BW GB/s: ${memBwGBs(unitObj.peak_mem_mb)}`);
return lines;
}
function metricSummaryV2(obj) {
if (!obj) return "—";
const cpu = obj.cpu;
const npu = obj.npu;
const dsp = obj.dsp;
const total = obj.total || {};
const setup = obj.setup;
const updated = obj.last_updated;
let lines = [];
lines = lines.concat(summarizeUnit("CPU", cpu));
lines = lines.concat(summarizeUnit("NPU", npu));
lines = lines.concat(summarizeUnit("DSP", dsp));
if (total && typeof total === "object") {
if (total.tok_s != null) lines.push(`TOTAL tok/s: ${total.tok_s}`);
if (total.prefill_tok_s != null) lines.push(`TOTAL ${kpiLabel("prefill_tok_s", "prefill")} ${kpiUnit("prefill_tok_s", "tok/s")}: ${total.prefill_tok_s}`);
if (total.ttft_ms != null) lines.push(`TOTAL ${kpiLabel("ttft_ms", "TTFT")} ${kpiUnit("ttft_ms", "ms")}: ${total.ttft_ms}`);
if (total.peak_mem_mb != null) lines.push(`TOTAL mem BW GB/s: ${memBwGBs(total.peak_mem_mb)}`);
if (total.latency_ms_p50 != null) lines.push(`TOTAL p50 ms: ${total.latency_ms_p50}`);
}
// Tier-2 generic stage timings (e.g. SigLIP's Vision Encoder / Text Encoder split).
for (const s of safeList(obj.stages, [])) {
if (s && s.ms != null) lines.push(`${s.label}: ${s.ms} ms`);
}
if (setup) lines.push(`Setup: ${setup}`);
if (updated) lines.push(`Updated: ${updated}`);
if (obj.accuracy != null) lines.push(`Accuracy: ${obj.accuracy}`);
if (obj.llm_metrics?.overall != null) lines.push(`LLM overall: ${obj.llm_metrics.overall}`);
if (obj.vlm_metrics?.overall != null) lines.push(`VLM overall: ${obj.vlm_metrics.overall}`);
return lines.length ? lines.join("\n") : "✓";
}
/* Shows every KPI the block actually reports (subject to the "KPIs to display" filter):
throughput/prefill/TTFT, any Tier-2 stage timing (e.g. SigLIP's Vision/Text Encoder
split), latency/total, and accuracy — one row each, in that order, plus a fixed mem-BW
row. A block that reports nothing meaningful just renders the mem-BW row (still "—" if
that's absent too), rather than three placeholder rows for metrics it never had. */
function metricBriefHTML(block) {
if (!block || typeof block !== "object") return `—`;
const total = (block.total && typeof block.total === "object") ? block.total : {};
const mem = memBwGBs(total.peak_mem_mb);
const rows = visibleKpiChips(block)
.map(c => `
${esc(c.label)}${esc(fmtChipFull(c))}
`);
rows.push(`
mem BW${mem != null ? esc(mem) + " GB/s" : "—"}
`);
return `
${rows.join("")}
`;
}
/* ---------- Catalog helpers ---------- */
function getCfg(model, cfgId) {
for (const c of safeList(model.hardware_configs, [])) {
if (c.id === cfgId) return c;
}
return null;
}
function safeFirstCfg(model) {
const cfgs = safeList(model.hardware_configs, []);
return cfgs.length ? cfgs[0] : null;
}
function getArtifactsFor(modelKey, fmt) {
const m = KEY_TO_MODEL[modelKey];
if (!m) return [];
return (fmt === "ONNX") ? safeList(m.onnx_artifacts, []) : safeList(m.gguf_artifacts, []);
}
/* True only if the artifact actually carries benchmark metadata (from its
benchmarks/ subfolder → an estimate or hil block). Checked against the
per-artifact map directly, NOT metricsForArtifact (whose legacy fallback
would make every artifact look benchmarked). A legacy config with no
artifacts map at all is treated as benchmarked so it isn't hidden. */
function artifactHasBenchmark(m, art) {
for (const cfg of safeList(m?.hardware_configs, [])) {
const metrics = cfg.metrics || {};
const artMap = metrics.artifacts;
if (artMap && typeof artMap === "object") {
const a = artMap[art];
if (a && (a.estimate || a.hil)) return true;
} else if (metrics.estimate || metrics.hil) {
return true;
}
}
return false;
}
/* Artifacts to actually surface in the results UI: only those with benchmark
data (e.g. 8B's fp16 folder has no benchmark, so it must not appear). */
function benchmarkedArtifactsFor(modelKey, fmt) {
const m = KEY_TO_MODEL[modelKey];
if (!m) return [];
return getArtifactsFor(modelKey, fmt).filter(a => artifactHasBenchmark(m, a));
}
/**
* Artifact-layer aware:
* - Preferred: cfg.metrics.artifacts[artifact].estimate/hil
* - Fallback: cfg.metrics.estimate/hil (legacy)
*/
function metricsForArtifact(cfgMetrics, artifact) {
const m = cfgMetrics || {};
const artMap = m.artifacts;
if (artifact && artMap && typeof artMap === "object" && artMap[artifact]) {
const a = artMap[artifact] || {};
return { estimate: a.estimate || null, hil: a.hil || null };
}
return { estimate: m.estimate || null, hil: m.hil || null };
}
function bestAvailableBlockForArtifact(cfgMetrics, artifact) {
const { estimate, hil } = metricsForArtifact(cfgMetrics, artifact);
return hil || estimate || null;
}
/* Every benchmark run recorded for an artifact (one per runtime/engine, tagged
with its kind: "hil" | "estimate"). The estimate/hil blocks above are the
representative single run per kind; runs[] keeps the full per-runtime detail
(e.g. ResNet50 int8 measured on both onnxruntime and mwmx). */
function artifactRuns(cfgMetrics, artifact) {
const artMap = (cfgMetrics || {}).artifacts;
if (artifact && artMap && typeof artMap === "object" && artMap[artifact]
&& Array.isArray(artMap[artifact].runs)) {
return artMap[artifact].runs;
}
return [];
}
/* Human-friendly task label: "image-classification" -> "Image classification". */
function prettyTask(t) {
const s = String(t ?? "").trim();
if (!s) return "";
return s.replace(/[-_]+/g, " ").replace(/^\w/, c => c.toUpperCase());
}
/* Accuracy for one artifact under a config, scanning all runs (HIL-first, then
estimate), falling back to the representative blocks. Returns a number or null. */
function accuracyForArtifact(cfg, art) {
const runs = artifactRuns(cfg?.metrics || {}, art);
let a = runs.filter(r => r.kind === "hil").map(accuracyFromBlock).find(v => v != null);
if (a == null) a = runs.filter(r => r.kind === "estimate").map(accuracyFromBlock).find(v => v != null);
if (a == null) {
const { estimate, hil } = metricsForArtifact(cfg?.metrics || {}, art);
a = accuracyFromBlock(hil);
if (a == null) a = accuracyFromBlock(estimate);
}
return a;
}
/* The full-precision reference accuracy (fp32, else fp16) for a model+format,
used to express how far a quantized precision drops from the float baseline. */
function referenceAccuracy(m, cfg, fmt) {
const arts = getArtifactsFor(m.key, fmt);
for (const ref of ["fp32", "fp16"]) {
if (arts.includes(ref)) {
const a = accuracyForArtifact(cfg, ref);
if (a != null) return { art: ref, acc: a };
}
}
return null;
}
/* ---------- Accuracy + runtime extraction ---------- */
function accuracyFromBlock(block) {
if (!block || typeof block !== "object") return null;
for (const k of ["accuracy", "accuracy_pct", "acc", "quality"]) {
if (block[k] != null && isFinite(Number(block[k]))) return Number(block[k]);
}
const tryList = (obj, keys) => {
if (!obj || typeof obj !== "object") return null;
for (const k of keys) {
if (obj[k] != null && isFinite(Number(obj[k]))) return Number(obj[k]);
}
return null;
};
let v = tryList(block.llm_metrics, ["overall", "mmlu", "gsm8k", "hellaswag", "truthfulqa", "mt_bench"]);
if (v != null) return v;
v = tryList(block.vlm_metrics, ["overall", "mmbench", "vqav2", "seedbench", "pope"]);
if (v != null) return v;
return null;
}
function runtimeTotalMsFromBlock(block) {
if (!block || typeof block !== "object") return null;
if (block.total && block.total.latency_ms_p50 != null && isFinite(Number(block.total.latency_ms_p50))) {
return Number(block.total.latency_ms_p50);
}
const sumUnits = ["cpu", "npu", "dsp"]
.map(u => block[u]?.latency_ms_p50)
.filter(v => v != null && isFinite(Number(v)))
.map(Number);
if (sumUnits.length) return sumUnits.reduce((a, b) => a + b, 0);
const tok = (block.total && block.total.tok_s != null) ? block.total.tok_s : block.tok_s;
if (tok != null && isFinite(Number(tok)) && Number(tok) > 0) {
return 1000 / Number(tok);
}
const fps = (block.total && block.total.fps != null) ? block.total.fps : block.fps;
if (fps != null && isFinite(Number(fps)) && Number(fps) > 0) {
return 1000 / Number(fps);
}
return null;
}
/* ---------- Throughput / latency (LLM report tok/s, vision models img/s) ----------
These read either the block's `total` roll-up or, failing that, the per-unit
(npu/cpu/dsp) values, so the UI can render a single figure with the right unit
regardless of whether a model is measured in tokens or frames per second. */
function _firstUnitVal(block, key) {
if (!block || typeof block !== "object") return null;
const tot = block.total || {};
if (tot[key] != null && isFinite(Number(tot[key]))) return Number(tot[key]);
for (const u of ["npu", "cpu", "dsp"]) {
const v = block[u]?.[key];
if (v != null && isFinite(Number(v))) return Number(v);
}
return null;
}
/* Throughput as {value, unit}: tokens/sec for language models, frames/sec
(img/s) for vision models. Prefers tok/s when both are somehow present. */
function throughputOf(block) {
const tok = _firstUnitVal(block, "tok_s");
if (tok != null) return { value: tok, unit: "tok/s" };
const fps = _firstUnitVal(block, "fps");
if (fps != null) return { value: fps, unit: "img/s" };
return { value: null, unit: "tok/s" };
}
function throughputValueOf(block) { return throughputOf(block).value; }
/* Median inference latency (ms). LLM benchmarks and vision benchmarks are both
normalized to latency_ms_p50 by the generator. */
function latencyMsOf(block) { return _firstUnitVal(block, "latency_ms_p50"); }
/* ---------- Generic KPI chips (every metric a block actually reports) ----------
Every model family reports a different mix of Tier-1 fields (tok_s/prefill/ttft/fps),
Tier-2 generic stage timings (block.stages — e.g. SigLIP's Vision/Text Encoder split),
and accuracy. This turns whatever a block actually has into a flat, orderable list so
cardHTML/metricBriefHTML can show "whatever is in the performance section" instead of
hardcoding one metric per model family — and so the KPI display filter (KPI_KINDS /
CATALOG_STATE.kpi) can hide/show each *kind* uniformly across every render surface. */
const KPI_KINDS = [
{ id: "throughput", label: "Throughput (tok/s, fps)" },
{ id: "prefill", label: "Prefill rate" },
{ id: "ttft", label: "TTFT" },
{ id: "stage", label: "Stage timings (e.g. encoder split)" },
{ id: "latency", label: "Latency / Total" },
{ id: "accuracy", label: "Accuracy" },
];
function kpiChipsOf(block) {
if (!block || typeof block !== "object") return [];
const total = block.total || {};
const chips = [];
const thr = throughputOf(block);
if (thr.value != null) chips.push({ kind: "throughput", label: "Throughput", value: thr.value, unit: thr.unit });
if (total.prefill_tok_s != null) {
chips.push({ kind: "prefill", label: kpiLabel("prefill_tok_s", "Prefill"), value: total.prefill_tok_s, unit: kpiUnit("prefill_tok_s", "tok/s") });
}
if (total.ttft_ms != null) {
chips.push({ kind: "ttft", label: kpiLabel("ttft_ms", "TTFT"), value: total.ttft_ms, unit: kpiUnit("ttft_ms", "ms") });
}
for (const s of safeList(block.stages, [])) {
if (s && s.ms != null) chips.push({ kind: "stage", label: s.label, value: s.ms, unit: "ms" });
}
const lat = latencyMsOf(block);
if (lat != null) {
chips.push({ kind: "latency", label: safeList(block.stages, []).length ? "Total" : "Latency", value: lat, unit: "ms" });
}
if (block.accuracy != null) chips.push({ kind: "accuracy", label: "Accuracy", value: block.accuracy * 100, unit: "%" });
if (block.top5_accuracy != null) chips.push({ kind: "accuracy", label: "Top-5 Accuracy", value: block.top5_accuracy * 100, unit: "%" });
return chips;
}
/* KPI chips filtered by the user's "KPIs to display" selection (CATALOG_STATE.kpi). */
function visibleKpiChips(block) {
const enabled = CATALOG_STATE.kpi;
return kpiChipsOf(block).filter(c => !enabled || enabled.has(c.kind));
}
/* The one chip to headline (matches the old tok/fps -> latency -> accuracy fallback
order), so existing LLM/CV cards keep showing throughput as the hero number. */
function primaryKpiChip(chips) {
return chips.find(c => c.kind === "throughput")
|| chips.find(c => c.kind === "latency")
|| chips.find(c => c.kind === "stage")
|| chips.find(c => c.kind === "prefill")
|| chips.find(c => c.kind === "ttft")
|| chips.find(c => c.kind === "accuracy")
|| null;
}
/* The single best *remaining* KPI after the headline — one fixed extra line, never
a variable-length list, so the card's stat box is the same size for every model
regardless of how many KPIs it reports. Prefers a stage breakdown (e.g. SigLIP's
Vision Encoder) over prefill/TTFT/accuracy/a second latency figure. */
function secondaryKpiChip(chips, primary) {
const rest = chips.filter(c => c !== primary);
return rest.find(c => c.kind === "stage")
|| rest.find(c => c.kind === "prefill")
|| rest.find(c => c.kind === "ttft")
|| rest.find(c => c.kind === "accuracy")
|| rest.find(c => c.kind === "latency")
|| null;
}
function fmtChipValue(c) {
if (!c || c.value == null || !isFinite(Number(c.value))) return "—";
return String(Math.round(Number(c.value) * 10) / 10);
}
/* "76.3%" for percentages (no space, matching the rest of the UI), "396.8 ms" otherwise. */
function fmtChipFull(c) {
const v = fmtChipValue(c);
if (v === "—") return v;
return c.unit === "%" ? `${v}%` : `${v} ${c.unit}`;
}
/* ---------- UI builders ---------- */
function buildRadioGroup(container, name, choices, value) {
if (!container) return;
container.innerHTML = choices.map(c => {
const id = `${name}_${c.replace(/\W+/g, "_")}`;
const checked = (c === value) ? "checked" : "";
return `
`;
}).join("");
}
function buildCheckGroup(container, name, choices, values) {
if (!container) return;
const set = new Set(values || []);
container.innerHTML = choices.map(c => {
const id = `${name}_${c.replace(/\W+/g, "_")}`;
const checked = set.has(c) ? "checked" : "";
return `
`;
}).join("");
}
function readRadio(name, fallback) {
const el = document.querySelector(`input[name="${CSS.escape(name)}"]:checked`);
return el ? el.value : fallback;
}
function readChecks(name) {
return Array.from(document.querySelectorAll(`input[name="${CSS.escape(name)}"]:checked`)).map(x => x.value);
}
/* A minimal detail panel for not-yet-published models: no benchmarks, no
artifacts, no download command to fabricate, and no repo link — the repo
for these is either absent or just a placeholder, so linking to it would
send someone to an empty page. Just what's known about the model.
Bypasses the full benchmark rendering below entirely, since building a
download command against zero artifacts would otherwise produce a
nonsensical path. Shared by the modal and the full page, each passing the
same `dense` flag their real-model rendering uses (see renderModalBodyHTML
/ renderFullModelPageHTML) so a coming-soon card sits in the same
rail-plus-content skeleton as a published one instead of falling back to
a one-off layout. */
function comingSoonDetailsHTML(m, dense) {
const specs = modelSpecPairs(m, null, "", "", null).filter(s => s.k !== "Input resolution" && s.k !== "Format / compute");
const specsKvHTML = specs.map(s => `
🚧 Coming soon. This model is in progress — benchmarks and download artifacts have not been published yet.
`;
if (dense) {
return `
${exampleImageHTML(m)}
${specsKvHTML}
${notice}
`;
}
return `
Model summary
${specsGridHTML}
${exampleImageHTML(m)}
${notice}`;
}
/* Fallback illustration (animated SVG) chosen by modality. */
const MODALITY_FALLBACK = {
LLM: "assets/fallback-llm.svg",
VLM: "assets/fallback-vlm.svg",
VLA: "assets/fallback-vla.svg",
CV: "assets/fallback-cv.svg",
ALM: "assets/fallback-alm.svg",
// No dedicated dual-encoder illustration yet — reuse the VLM one (closest: image + text I/O).
EMBED: "assets/fallback-vlm.svg",
};
/* CV has multiple distinct tasks with their own illustration — a per-object
bounding-box scene reads as object detection, not classification. */
const CV_TASK_FALLBACK = {
"object-detection": "assets/fallback-cv.svg",
"image-classification": "assets/fallback-cv-classification.svg",
"3d-object-detection": "assets/fallback-cv-3d.svg",
"object-detection-3d": "assets/fallback-cv-3d.svg",
"lane-detection": "assets/fallback-cv-lane.svg",
"image-segmentation": "assets/fallback-cv-segmentation.svg",
"semantic-segmentation": "assets/fallback-cv-segmentation.svg",
"video-classification": "assets/fallback-cv-video.svg",
"image-feature-extraction": "assets/fallback-cv-feature.svg",
};
/* VLA covers very different domains — a humanoid-robot arm reads wrong for a
driving model. Same idea as CV_TASK_FALLBACK: pick by the model's own task. */
const VLA_TASK_FALLBACK = {
"autonomous-driving": "assets/fallback-vla-driving.svg",
};
/* Detail-page example figure: the single modality/task-appropriate animated
SVG illustration for a model (unchanged — the richer "what it's doing"
scene shown in the modal/full page, distinct from the compact card set
below). */
function fallbackIllustrationSrc(m) {
const modality = m.modality || "LLM";
const task = safeList(m.tasks, [])[0] || "";
const bev = modality === "CV" && bevKey(m);
return (bev && `assets/fallback-${bev}.svg`)
|| (modality === "CV" && CV_TASK_FALLBACK[task])
|| (modality === "VLA" && VLA_TASK_FALLBACK[task])
|| MODALITY_FALLBACK[modality]
|| MODALITY_FALLBACK.LLM;
}
/* Same modality/task routing as fallbackIllustrationSrc, but returning a key
into CARD_ILLUSTRATION_VARIANTS rather than a single file. */
const CV_TASK_KEY = {
"image-classification": "cv-classification",
"3d-object-detection": "cv-3d",
"object-detection-3d": "cv-3d",
"lane-detection": "cv-lane",
"image-segmentation": "cv-segmentation",
"semantic-segmentation": "cv-segmentation",
"video-classification": "cv-video",
"image-feature-extraction": "cv-feature",
};
/* Bird's-eye-view models (BEVFormer, BEVFusion, BEVLaneDet) share a top-down
look regardless of their nominal task, so they're detected by family/key. */
function bevKey(m) {
const id = `${m.family || ""} ${m.key || ""}`.toLowerCase();
if (!/(^|[\s_-])bev(former|fusion|lane|[\s_-]|$)/.test(id)) return null;
if (id.includes("fusion")) return "cv-bev-fusion";
if (safeList(m.tasks, []).includes("lane-detection")) return "cv-bev-lane";
return "cv-bev";
}
function illustrationTypeKey(m) {
const modality = m.modality || "LLM";
const task = safeList(m.tasks, [])[0] || "";
if (modality === "CV" && bevKey(m)) return bevKey(m);
if (modality === "CV" && CV_TASK_KEY[task]) return CV_TASK_KEY[task];
if (modality === "CV") return "cv";
if (modality === "VLA" && task === "autonomous-driving") return "vla-driving";
if (modality === "VLA") return "vla";
if (modality === "VLM" || modality === "EMBED") return "vlm";
if (modality === "ALM") return "alm";
return "llm";
}
/* Catalog-card illustration set — several hand-designed variants per
modality/task, sized and cropped for the short banner strip (as opposed
to the single wide scene each modality gets in the detail-page figure).
A model always lands on the same variant (stable hash of its key) so the
grid doesn't shuffle on re-render, but sibling models of the same
modality don't all show the identical picture. */
const CARD_ILLUSTRATION_VARIANTS = {
llm: ["assets/cards/llm-1.svg", "assets/cards/llm-2.svg", "assets/cards/llm-3.svg"],
vlm: ["assets/cards/vlm-1.svg", "assets/cards/vlm-2.svg", "assets/cards/vlm-3.svg"],
vla: ["assets/cards/vla-1.svg", "assets/cards/vla-2.svg", "assets/cards/vla-3.svg"],
"vla-driving": ["assets/cards/vla-driving-1.svg", "assets/cards/vla-driving-2.svg", "assets/cards/vla-driving-3.svg"],
cv: ["assets/cards/cv-1.svg", "assets/cards/cv-2.svg", "assets/cards/cv-3.svg"],
"cv-classification": ["assets/cards/cv-classification-1.svg", "assets/cards/cv-classification-2.svg", "assets/cards/cv-classification-3.svg"],
"cv-3d": ["assets/cards/cv-3d-1.svg", "assets/cards/cv-3d-2.svg", "assets/cards/cv-3d-3.svg"],
"cv-lane": ["assets/cards/cv-lane-1.svg", "assets/cards/cv-lane-2.svg", "assets/cards/cv-lane-3.svg"],
"cv-segmentation": ["assets/cards/cv-segmentation-1.svg", "assets/cards/cv-segmentation-2.svg", "assets/cards/cv-segmentation-3.svg"],
"cv-video": ["assets/cards/cv-video-1.svg", "assets/cards/cv-video-2.svg", "assets/cards/cv-video-3.svg"],
"cv-feature": ["assets/cards/cv-feature-1.svg", "assets/cards/cv-feature-2.svg", "assets/cards/cv-feature-3.svg"],
"cv-bev": ["assets/cards/cv-bev-1.svg"],
"cv-bev-fusion": ["assets/cards/cv-bev-fusion-1.svg"],
"cv-bev-lane": ["assets/cards/cv-bev-lane-1.svg"],
alm: ["assets/cards/alm-1.svg", "assets/cards/alm-2.svg", "assets/cards/alm-3.svg"],
};
function stableHash(str) {
let h = 0;
for (let i = 0; i < str.length; i++) h = (Math.imul(h, 31) + str.charCodeAt(i)) | 0;
return Math.abs(h);
}
/* compact = true swaps in the assets/cards/compact/ set: the same scenes
redrawn slightly zoomed-out so their focal details (a chat bubble's tail,
a robot arm's gripper, a detection badge) survive the shorter Overview
carousel banner's object-fit:cover crop instead of bleeding off the edge. */
function cardIllustrationSrc(m, { compact = false } = {}) {
const variants = CARD_ILLUSTRATION_VARIANTS[illustrationTypeKey(m)] || CARD_ILLUSTRATION_VARIANTS.llm;
const src = variants[stableHash(String(m.key || m.display_name || "")) % variants.length];
return compact ? src.replace("assets/cards/", "assets/cards/compact/") : src;
}
/* Same hash-picked variant per model as cardIllustrationSrc, but walked in
list order so that whenever two neighbors would land on the identical
illustration (same family + same variant index — most likely when a
burst of same-modality models land back to back, e.g. several YOLO
releases), the second one is nudged to the next variant instead. Used by
both the Last Update carousel (true left-right adjacency) and the
Catalog grid (render order = reading order, so this at least guarantees
no repeat within a row; the grid's column count isn't known at render
time to also de-dupe vertically). */
function sequentialIllustrationSrcs(list, { compact = false } = {}) {
let prevFamily = null, prevIdx = null;
return list.map(m => {
const familyKey = illustrationTypeKey(m);
const variants = CARD_ILLUSTRATION_VARIANTS[familyKey] || CARD_ILLUSTRATION_VARIANTS.llm;
let idx = stableHash(String(m.key || m.display_name || "")) % variants.length;
if (familyKey === prevFamily && idx === prevIdx && variants.length > 1) {
idx = (idx + 1) % variants.length;
}
prevFamily = familyKey; prevIdx = idx;
const src = variants[idx];
return compact ? src.replace("assets/cards/", "assets/cards/compact/") : src;
});
}
/* "What it's doing" — a sample/preview image pulled from the model repo; falls
back to a modality-appropriate animated SVG illustration when the repo ships
none or it fails to load. Shown for all modalities. */
function exampleImageHTML(m) {
const url = m && m.sample_image ? String(m.sample_image) : "";
const modality = m.modality || "LLM";
const task = safeList(m.tasks, [])[0] || "";
const fallbackSrc = fallbackIllustrationSrc(m);
const taskLabel = task ? prettyTask(task) : modality;
/* If the repo supplies an image, show it over the animated fallback.
onerror removes the , revealing the animated SVG beneath. */
const fallbackImg = ``;
const repoImg = url
? ``
: "";
return `
${fallbackImg}${repoImg}
Example${task ? " · " + esc(taskLabel) : " · " + esc(modality)}`;
}
/* Runs for an artifact: real per-runtime runs when present, else a synthetic
one-row-per-kind fallback built from the aggregate hil/estimate blocks
(mirrors accuracyForArtifact's HIL-first fallback). */
function runsForArtifact(cfgMetrics, art) {
const direct = artifactRuns(cfgMetrics, art).filter(Boolean);
if (direct.length) return direct;
const { estimate, hil } = metricsForArtifact(cfgMetrics, art);
const synth = [];
if (hil) synth.push({ ...hil, kind: "hil" });
if (estimate) synth.push({ ...estimate, kind: "estimate" });
return synth;
}
/* Device utilization: how much of the chip a run actually drove (NPU instances,
AI cores, clock frequency, TOPS) vs. what the hardware config has available —
sourced straight from the benchmark YAML's hardware/configuration blocks. */
function utilizationOf(block) {
return (block && block.utilization) ? block.utilization : null;
}
// function downloadIncludePattern(fmt, artifact) {
// const a = String(artifact || "").trim();
// if (!a) return "*";
// if (fmt === "GGUF") return `*${a}*.gguf`;
// return `*${a}*`;
// }
/* ---------- Download / Deploy helpers ---------- */
/* Mode A — “Download repo (filtered)”.
Pulls only the artifact’s folder from the repo (host-side benchmarking
or further conversion). Keeps the existing behaviour: one hf-download line. */
function downloadIncludePattern(fmt, artifact) {
const a = String(artifact || "").trim();
if (!a) return "*";
// Same pattern for both formats: the artifact name is a top-level folder
// in every Renesas repo (fp16/, w4a16/, w4a8/, fp32/).
return `${a}/*`;
}
/* Mode B — “Deploy with prebuilt binaries”.
Resolves installer / model / runner filenames from naming conventions in
the per-repo README, with optional per-model overrides via
`model.deploy[artifact]` in models.json. Returns null when no prebuilt
binaries are published for the (artifact, hardware) pair, in which case
the UI will gracefully degrade. Currently GGUF + RCAR-X5H only. */
function deployInfoFor(model, fmt, artifact, cfg) {
if (!model || !artifact || fmt !== "GGUF") return null;
// No published binaries for ONNX or non-X5H targets yet.
const cfgId = (cfg?.id || "").toUpperCase();
if (cfgId && cfgId !== "RCAR-X5H") return null;
const ov = model.deploy?.[artifact] || {};
const isQuant = artifact.toLowerCase() !== "fp16";
const runnerKind = isQuant ? "llama-quant-runner" : "llama-runner";
const installerVer = ov.installer_version || "0.1.0";
const installer = ov.installer || `${runnerKind}-${installerVer}-Linux.sh`;
const installDir = ov.install_dir || runnerKind;
const runnerBin = ov.runner_bin || runnerKind;
const xosTag = ov.xos_tag || "xOS_v4.32";
const boardTag = ov.board_tag || "rcar-x5hv1";
const modelTag = isQuant ? artifact : "f16"; // README uses *-f16.gguf for FP16
const modelFile = ov.model_file || `${model.display_name || model.key}-${modelTag}.gguf`;
return {
installer, installDir, runnerBin, modelFile,
installerPath: `${artifact}/binaries/${boardTag}/${xosTag}/${installer}`,
modelPath: `${artifact}/${modelFile}`,
};
}
/* Render the Download section as two tabs: “Download repo” (filtered hf
download) and “Deploy with binaries” (4-step install + run on X5H board). */
function renderActions(modelKey, fmt, compute, purpose, artifact, cfgId, mode = "explore") {
const m = KEY_TO_MODEL[modelKey];
if (!m) return "No actions.";
const v = validatePurpose(fmt, purpose);
if (!v.ok) {
return `
${v.msg}
Actions are disabled until the selection is valid.
`;
}
/* Modal density's precision + NPU selectors — same row-per-axis pattern
as cfgBarHTML, minus hardware config/qualification (not shown in the
compact modal). */
function modalSelectorsHTML(V) {
return `
`;
}
/* A hybrid run's bar (npuPct + cpuPct both published) is a cumulative stack
— the NPU share at the base and the CPU fallback share on top, together
filling the same height a flat HIL/SIL bar would — instead of one flat
color that hides how much of the graph actually fell back to CPU. */
function barFillHTML(b) {
if (b.npuPct != null && b.cpuPct != null) {
return `
`;
}
return ``;
}
function barsHTML(V) {
if (!V.bars.length) {
return `
No ${V.unit === "lat" ? "latency" : "throughput"} published at this allocation.
`;
}
/* NPU/CPU offload split — only rendered when the benchmark YAML actually
published it (configuration.npu_offload_pct / cpu_offload_pct on a hybrid
run); a pure-NPU run carries neither field. Given its own two-tone bar
(NPU/CPU badge colors) rather than the generic single-value track used
below, since "what fraction ran where" is a different kind of fact than
"how much of the available resource did this run use". */
function resourcesSplitHTML(alloc) {
if (!alloc || (alloc.npuPct == null && alloc.cpuPct == null)) return "";
return `
NPU / CPU split${alloc.npuPct != null ? `${esc(String(alloc.npuPct))}%` : "—"} NPU${alloc.cpuPct != null ? `${esc(String(alloc.cpuPct))}%` : "—"} CPU
`;
}
function resourcesCardHTML(V) {
const alloc = V.alloc;
if (!alloc) return `
`;
}
/* Compact "route to the full page" links — sits beside the resources card
in the modal's right column (not a separate full-width row: at 248px
density there's no room to spare below a two-column area). No direct
"Open repo" link here — the repo is always reachable via the Download tab
(repo link + hf download command), so a second shortcut was redundant. */
function modalQuickLinksHTML(state, V, runsCount) {
const rows = [
{ tab: "runs", label: `All ${runsCount} benchmark runs →` },
// No model file for the selected precision -> nothing to download yet, so
// the link to the full page's Download & run tab is dropped rather than
// sending the user to a tab whose command can't actually be run.
...(artifactFileMissing(V.m, V.fmt, V.art) ? [] : [{ tab: "dl", label: "Download & run instructions →" }]),
].map(l => `${esc(l.label)}`).join("");
return `
${rows}
`;
}
function runsTableHTML(V) {
const rows = allRunsRowsFor(V.m, V.cfg, V.fmt);
if (!rows.length) return `
No benchmark runs published for this hardware config yet.
`;
const body = rows.map(({ art, r, allocId, allocLabel }) => {
const sel = art === V.art && allocId === V.allocId;
const thr = throughputOf(r);
const lat = latencyMsOf(r);
const acc = accuracyFromBlock(r);
const qual = r.kind === "hil" ? `HIL` : `SIL`;
return `
Rows matching the current selection are highlighted
Precision
NPU allocation
Runtime
Qual.
Throughput
Latency
Accuracy
Updated
${body}
HIL = measured on hardware. SIL = estimated by the PPA estimator. Blank cells mean the metric was not reported by that run — never zero.
`;
}
/* MWMX AI Compiler installer/package live behind Renesas's customer portal,
not on a public URL the doc reveals — link straight to the gated portal
page rather than the generic myRenesas landing page. Only relevant for the
MWMX/NNAC (ONNX) toolchain; GGUF's llama.cpp-style runner needs no
separate compiler download, so callers gate this on the run's engine. */
const MYRENESAS_AI_COMPILER_URL = "https://www.renesas.com/en/myrenesas/secure-portals/gen5-r-car-x5x-sw-ai";
function compilerLinkHTML(run) {
if (String(run?.engine || "").toLowerCase() !== "mwmx") return "";
return `
`;
}
/* A benchmark YAML can carry an explicit `reproduce:` block — exact commands
its author verified on real hardware (see generate_models_json.py
parse_reproduce()) — because the real deployment flow (single vs. multi NPU
cluster, an ORT-split subgraph, a GGUF runner binary, ...) varies per model
and isn't safe to guess. When the selected run has one, it replaces the
generic per-format placeholder flow below entirely; it keeps the same
two-column layout (toolchain meta + commands) so the two paths look like
one feature, not two different UIs. */
function reproduceStepsHTML(run, cfg, toolchain) {
const rp = run.reproduce;
const steps = safeList(rp.steps, []);
const body = steps.map((s, i) => {
const isNote = s.kind === "note";
return `
${i + 1} · ${esc(s.title || "Run")}
${isNote
? `
${esc(s.command)}
`
: `
${esc(s.command)}
`}
${s.expected ? `
Expected: ${esc(s.expected)}
` : ""}
`;
}).join("");
return `
Toolchain requirements
${toolchain.map(t => `
${esc(t.k)}${esc(t.v)}
`).join("")}
${compilerLinkHTML(run)}
${rp.reference ? `
Verified against ${esc(rp.reference)}.
` : ""}
Reproduce · ${esc(cfg?.label || cfg?.id || "")}
${body}
${rp.notes ? `
${esc(rp.notes)}
` : ""}
`;
}
/* Driven by the selected precision. ONNX gets the README's own-repo
layout (download cmd + onnxruntime/RcarNpuExecutionProvider snippet +
board setup + toolchain + a note on what the catalog can't show yet);
GGUF reuses the existing deploy-with-binaries renderActions() flow. Either
is overridden by an explicit reproduce: block when the selected run has one. */
function downloadRunHTML(V) {
const { m, cfg, fmt, art } = V;
const bestRun = V.hilRuns[0] || V.silRuns[0] || V.runs[0] || null;
const inputRes = V.runs.map(r => r.input_resolution).find(Boolean);
const toolchain = [
{ k: "Hardware config", v: cfg?.label || cfg?.id || "—" },
{ k: "Runtime engine", v: bestRun?.engine || "—" },
{ k: "Toolchain version", v: bestRun?.toolchain_version || "—" },
{ k: "Execution provider", v: bestRun?.execution_provider ? String(bestRun.execution_provider).toUpperCase() : "NPU" },
{ k: "Batch size", v: bestRun?.batch_size != null ? String(bestRun.batch_size) : "—" },
];
if (inputRes) toolchain.push({ k: "Input resolution", v: inputRes });
if (bestRun?.reproduce?.steps?.length) {
return reproduceStepsHTML(bestRun, cfg, toolchain);
}
if (fmt !== "ONNX") {
return renderActions(m.key, fmt, defaultRowCompute(m, cfg), defaultPurposeForFormat(fmt), art, cfg?.id, "tab");
}
const repoId = m.onnx_repo || "";
if (!repoId) return `
No ONNX repo published for this model yet.
`;
if (!art) return `
No benchmarked precision to build a download command from yet.
Artifact file sizes and checksums aren't captured by generate_models_json.py yet — it only reads filenames from the repo's file tree. Browse the exact files in the ONNX repo.
1 · Download the ${esc(art.toUpperCase())} artifact
${esc(cmdRepo)}
2 · Run inference on the board
${esc(py)}
Replace <artifact_file>.onnx with the exact filename from the repo — the catalog generator doesn't capture individual filenames yet.
`;
}
/* The page's own topbar already carries the Renesas wordmark + primary
nav (Overview/Catalog) — repeating it here would just be a second
header. The name and modality are both restated a breath away (hero
title, then chips), so a "Catalog / CV / " trail here would be
the third repetition on screen; a bare back arrow says "where am I"
just as well and keeps this row tight. */
/* Breadcrumb row (navigation) + title row (name, meta, chips) merged into
one compact head block — kept as two separate bands before, which
duplicated padding/borders and pushed the actual performance content
far down the page for no reason (breadcrumb and title never need to
scroll independently of each other). */
function pageHeadHTML(m, fmt, compute) {
const repoId = fmt === "GGUF" ? (m.gguf_repo || "") : (m.onnx_repo || "");
const metaBits = [m.architecture, safeList(m.tasks, [])[0] ? prettyTask(m.tasks[0]) : null, m.model_size_m != null ? `${m.model_size_m}M params` : null].filter(Boolean);
return `
`;
}
function footerHTML() {
return `
`;
}
/* ---------- Full page (#sec-model) ---------- */
function renderFullModelPageHTML(state) {
const m = KEY_TO_MODEL[state.key];
if (!m) return `
Model not found.
`;
if (isComingSoon(m)) {
return `${pageHeadHTML(m, state.fmt, defaultRowCompute(m, null))}${comingSoonDetailsHTML(m)}${footerHTML()}`;
}
const V = buildDetailView(state.key, state.cfgId, state.fmt, state.precision, state.allocationId, state.unit);
state.cfgId = V.cfg?.id || state.cfgId || "";
state.precision = V.art;
state.allocationId = V.allocId;
state.unit = V.unit;
const compute = defaultRowCompute(V.m, V.cfg);
const uid = `mdlp_${String(state.key).replace(/\W+/g, "_")}`;
// No model file for the selected precision -> the Download & run tab has
// nothing runnable to show, so it's dropped rather than left open on a
// command that would fail. Falls back to the runs tab if that's where the
// (now-hidden) dl tab was left selected.
const fileMissing = artifactFileMissing(m, state.fmt, V.art);
const tab = (!fileMissing && state.tab === "dl") ? "dl" : "runs";
return `
${pageHeadHTML(m, state.fmt, compute)}
${missingFileBannerHTML(m, state.fmt, V.art)}
${summaryRowHTML(m, V.cfg, state.fmt, compute, V.art)}
${cfgBarHTML(V, "page")}
`;
}).join("\n");
setupPinnedColumns(table, pinCount);
}
function setupPinnedColumns(table, count) {
if (!table) return;
const headCells = table.querySelectorAll(`thead th.pin`);
if (!headCells.length) return;
const lefts = [];
let acc = 0;
for (let i = 0; i < count; i++) {
const cell = table.querySelector(`thead th.pin[data-pin="${i}"]`);
if (!cell) break;
lefts[i] = acc;
acc += cell.getBoundingClientRect().width;
}
for (let i = 0; i < lefts.length; i++) {
table.querySelectorAll(`.pin[data-pin="${i}"]`).forEach(el => { el.style.left = `${lefts[i]}px`; });
}
}
/* ---------- Model modal ---------- */
function openModelModal(modelKey, cfgId, fmt) {
const m = KEY_TO_MODEL[modelKey];
if (!m) return;
const modal = $("modelModal");
modal.classList.add("open");
modal.setAttribute("aria-hidden", "false");
const cfg = cfgId ? getCfg(m, cfgId) : safeFirstCfg(m);
MODAL_STATE = { key: modelKey, cfgId: cfg?.id || cfgId || "", fmt, precision: null, allocationId: null, unit: null };
$("modalTitle").textContent = m.display_name || m.key;
const sub = $("modalSubtitle");
if (sub) sub.innerHTML = modalHeaderMetaHTML(m, fmt, defaultRowCompute(m, cfg));
renderModalNow();
}
function closeModelModal() {
const modal = $("modelModal");
modal.classList.remove("open");
modal.setAttribute("aria-hidden", "true");
const el = $("modalDetails");
if (el) el.innerHTML = "";
MODAL_STATE = null;
}
/* ====================================================================== */
/* OVERVIEW — constellation (size vs throughput) */
/* ====================================================================== */
const FAMILY_PALETTE = ["#C96442", "#6E8B6A", "#C9A24B", "#5B7A99", "#A6573F", "#8E6E9E", "#4C8C7D", "#B07A3C", "#9A6B6B", "#7C8B5A"];
const COLOR_CACHE = {};
function colorFor(label) {
if (COLOR_CACHE[label] != null) return COLOR_CACHE[label];
const c = FAMILY_PALETTE[Object.keys(COLOR_CACHE).length % FAMILY_PALETTE.length];
COLOR_CACHE[label] = c;
return c;
}
function allArtifactsOf(m) {
return uniqueSorted([...safeList(m.onnx_artifacts, []), ...safeList(m.gguf_artifacts, [])]);
}
/* A repo with no artifact folders at all has nothing published yet — treated
as a "coming soon" teaser rather than a model with missing benchmarks. This
flips automatically the day real artifacts are pushed, with no status flag
to remember to clear. */
function isComingSoon(m) {
return allArtifactsOf(m).length === 0;
}
/* File-vs-benchmark decorrelation — a benchmark YAML lands the moment a run
completes, which is routinely before the (much larger) weight file itself
is uploaded. `_file_status[artifact]` (generate_models_json.py,
artifact_has_payload()) is `false` only when that specific artifact has no
real payload; absent/undefined means the model predates this field, so it
is treated as available rather than flagged. Kept per-format because the
same precision name can be a real file in one repo and not the other
(Llama-3.1-8B-Instruct: GGUF w4a16 ships coefficients, ONNX w4a16 doesn't). */
function fileStatusFor(m, fmt) {
return (fmt === "ONNX" ? m?.onnx_file_status : m?.gguf_file_status) || {};
}
function artifactFileMissing(m, fmt, art) {
return !!art && fileStatusFor(m, fmt)[art] === false;
}
function modelHasMissingFile(m) {
const vals = [...Object.values(m?.onnx_file_status || {}), ...Object.values(m?.gguf_file_status || {})];
return vals.some(v => v === false);
}
/* True if at least one artifact (either format) actually has a downloadable
file — i.e. the model isn't "coming soon" (no artifacts at all) and isn't
stuck with every artifact's file still unpublished (see fileStatusFor). */
function modelHasAvailableFile(m) {
if (isComingSoon(m)) return false;
const onnxOk = safeList(m.onnx_artifacts, []).some(a => !artifactFileMissing(m, "ONNX", a));
const ggufOk = safeList(m.gguf_artifacts, []).some(a => !artifactFileMissing(m, "GGUF", a));
return onnxOk || ggufOk;
}
/* Shared banner for the modal and the full model page — same warning style as
the "Coming soon" notice, scoped to just the currently-selected precision
instead of the whole model. */
function missingFileBannerHTML(m, fmt, art) {
if (!artifactFileMissing(m, fmt, art)) return "";
return `
⏳ ${esc(String(art).toUpperCase())} model file not yet uploaded. The benchmark numbers below for this precision were published ahead of the model weights — download will not work until the file is added to the repo.
`;
}
function fmtOfArtifact(m, art) {
if (safeList(m.gguf_artifacts, []).includes(art)) return "GGUF";
if (safeList(m.onnx_artifacts, []).includes(art)) return "ONNX";
return "—";
}
function pickBlock(cfgMetrics, art, source) {
const { estimate, hil } = metricsForArtifact(cfgMetrics || {}, art);
if (source === "HIL") return hil ? { block: hil, source: "HIL" } : null;
if (source === "Estimate") return estimate ? { block: estimate, source: "Estimate" } : null;
if (hil) return { block: hil, source: "HIL" };
if (estimate) return { block: estimate, source: "Estimate" };
return null;
}
function tokSOf(block) {
if (!block) return null;
let t = (block.total && block.total.tok_s != null) ? block.total.tok_s : null;
if (t == null) t = block.npu?.tok_s ?? block.cpu?.tok_s ?? block.dsp?.tok_s ?? null;
return (t != null && isFinite(Number(t))) ? Number(t) : null;
}
/* ---------------------------------------------------------------------- */
/* Measurement regimes = the constellation tabs */
/* ----------------------------------------------------------------------
The catalog deliberately mixes model *types* (CNN, dual encoder,
transformer decoder, LLM/VLM/ALM/VLA), and each type answers a different
question with a different unit. Plotting them on one pair of axes is not
just crowded, it is wrong: 952 img/s (MobileNetV2, one full image per
inference) and 41 tok/s (Llama-3.2-1B, one *token* per inference step)
are not the same quantity, and putting them on a shared linear axis both
implies a comparison that doesn't exist and squeezes every decoder into
the left margin.
So the chart is split by what "one inference" means, which is what fixes
the unit — and therefore the axis and even the chart form:
gen one generated token -> tok/s -> scatter
cnn one full image, fixed graph -> img/s, ms -> scatter
enc one pass per encoder tower -> ms per stage -> bar chart
soon undefined (no KPI/artifacts) -> none -> cards
A model lands in a regime by modality, with a data-shape fallback for
modalities this table doesn't know yet, so a new modality shows up in a
sensible tab instead of vanishing. */
const CSTL_REGIMES = [
{
id: "gen",
label: "Token generators",
unit: "tok/s",
chart: "scatter",
modalities: ["LLM", "VLM", "ALM"],
why: `One inference = one generated token. Decoders share the same memory-bound decode loop,
so tok/s and size are directly comparable here.`,
x: ["tok", "size", "prefill", "ttft"],
y: ["size", "tok", "ttft", "mem"],
},
{
id: "cnn",
label: "Vision CNNs",
unit: "img/s · ms",
chart: "scatter",
modalities: ["CV"],
why: `One inference = one fixed-shape image. Compute-bound and quantized, so img/s and
latency sit orders of magnitude above any token rate — hence the separate scale.`,
x: ["fps", "lat", "size"],
y: ["size", "lat", "fps"],
/* Single-pass, batch-1, fixed-shape graph → throughput is the reciprocal of
latency, and the CV benchmark files state that convention themselves
("fps: null # throughput: 1000 / latency" in RetinaNet's int8 run). It
also holds in the data: MobileNetV2 952.4 vs 1000/1.05, EfficientNetV2-B0
480 vs 1000/2.08, ResNet50 303 vs 1000/3.23 — within ~2%. So a CV variant
that timed a run but left fps null still gets a point, drawn dashed and
labelled as derived. Never enabled for the decode regime, where latency
and tok/s measure different things. */
deriveFps: true,
},
{
id: "enc",
label: "Encoders & embeddings",
unit: "ms / stage",
chart: "stages",
modalities: ["EMBED"],
why: `No inference loop — two towers, different rates. A blended throughput would be fiction,
so each tower's stage latency is charted separately.`,
},
{
id: "soon",
label: "Roadmap",
unit: "no KPI yet",
chart: "cards",
modalities: ["VLA"],
why: `Nothing measurable published yet. No KPI pipeline or artifacts yet — listed as cards
so they stay visible.`,
},
];
const CSTL_REGIME_BY_ID = Object.fromEntries(CSTL_REGIMES.map(r => [r.id, r]));
/* Every benchmark block a model publishes, across configs/artifacts/kinds. */
function allBlocksOf(m) {
const out = [];
for (const cfg of safeList(m.hardware_configs, [])) {
for (const art of allArtifactsOf(m)) {
const { estimate, hil } = metricsForArtifact(cfg.metrics || {}, art);
if (estimate) out.push(estimate);
if (hil) out.push(hil);
}
}
return out;
}
function publishesMetric(m, key) {
return allBlocksOf(m).some(b => _firstUnitVal(b, key) != null);
}
/* Which tab a model belongs to. Modality first (it is the declared model type),
then the shape of the published numbers — so a modality not listed above, or a
VLA that one day reports a token rate, still lands somewhere sensible. */
function regimeOf(m) {
if (isComingSoon(m)) return "soon";
const mod = String(m.modality || "").trim().toUpperCase();
if (mod === "VLA") return publishesMetric(m, "tok_s") ? "gen" : "soon";
const byMod = CSTL_REGIMES.find(r => safeList(r.modalities, []).includes(mod));
if (byMod) return byMod.id;
if (publishesMetric(m, "tok_s")) return "gen";
if (publishesMetric(m, "fps")) return "cnn";
if (allBlocksOf(m).some(b => safeList(b.stages, []).length)) return "enc";
return "soon";
}
function fmtAxisNum(v) {
return Math.abs(v) >= 100 ? String(Math.round(v)) : String(Math.round(v * 10) / 10);
}
/* Measured values keep their tenth of a millisecond (396.8 ms, not 397 ms) — axis
ticks are the only place rounding to whole units is fine. */
function fmtMs(v) { return v == null ? "—" : String(Math.round(Number(v) * 10) / 10); }
/* Axis registry. `get` returns null when a variant never published that metric —
the caller reports those as excluded rather than dropping them silently. */
const AXES = {
tok: { get: p => p.tok, label: "Decode throughput (tok/s)", tick: fmtAxisNum },
prefill: { get: p => p.prefill, label: "Prefill rate (tok/s)", tick: fmtAxisNum },
ttft: { get: p => p.ttft, label: "Time to first token (ms)", tick: fmtAxisNum, lowerBetter: true },
fps: { get: p => p.fps, label: "Throughput (img/s)", tick: fmtAxisNum },
lat: { get: p => p.lat, label: "Latency p50 (ms)", tick: fmtAxisNum, lowerBetter: true },
mem: { get: p => p.memBw, label: "Memory bandwidth (GB/s)", tick: fmtAxisNum },
size: { get: p => p.sizeM, label: "Model size (params)", tick: v => (v >= 1000 ? `${Math.round(v / 100) / 10}B` : `${Math.round(v)}M`) },
};
/* One point per (model × hardware config × artifact) inside one regime, carrying
every metric the block reports so each regime can pick its own axes. */
function regimePoints(regimeId, source) {
const pts = [];
const regime = CSTL_REGIME_BY_ID[regimeId];
for (const m of CATALOG) {
if (regimeOf(m) !== regimeId) continue;
for (const cfg of safeList(m.hardware_configs, [])) {
for (const art of allArtifactsOf(m)) {
const got = pickBlock(cfg.metrics || {}, art, source);
if (!got) continue;
const b = got.block;
/* Size is per point, not per model: an artifact that declares its own
parameter count in /.metadata.yaml wins over the repo-level
figure. Falls back to the model value, which is the common case. */
const sizeM = artifactSizeM(m, art);
const lat = latencyMsOf(b);
let fps = _firstUnitVal(b, "fps");
let fpsDerived = false;
if (fps == null && regime?.deriveFps && lat != null && lat > 0) {
fps = 1000 / lat;
fpsDerived = true;
}
pts.push({
key: m.key, cfgId: cfg.id, fmt: fmtOfArtifact(m, art),
name: m.display_name || m.key,
arch: m.architecture || m.display_name || m.key,
family: familyOf(m),
quant: art,
precision: artifactPrecision(m, art),
target: safeList(cfg.targets, safeList(m.targets, []))[0] || "—",
hwLabel: cfg.label || cfg.id || "—",
sizeM,
tok: _firstUnitVal(b, "tok_s"),
fps, fpsDerived, lat,
prefill: _firstUnitVal(b, "prefill_tok_s"),
ttft: _firstUnitVal(b, "ttft_ms"),
memBw: memBwGBs(_firstUnitVal(b, "peak_mem_mb")),
acc: b.accuracy != null ? Number(b.accuracy) : null,
stages: safeList(b.stages, []).filter(s => s && s.ms != null),
source: got.source, block: b,
});
}
}
}
return pts;
}
function colorKey(p, colorBy) {
return colorBy === "quant" ? p.precision
: colorBy === "format" ? p.fmt
: colorBy === "target" ? p.target
: colorBy === "family" ? p.family
: p.arch;
}
/* Parameter count for one artifact of one model, in millions.
`artifact_info[].size_m` comes from that artifact's own .metadata.yaml and
overrides the repo-level `model_size_m` — the generator collected those files
but never read them, so per-artifact sizes used to be invisible here. */
function artifactSizeM(m, art) {
const info = (m.artifact_info || {})[art];
if (info && info.size_m != null) return Number(info.size_m);
return (m.model_size_m != null) ? Number(m.model_size_m) : null;
}
/* Some .metadata.yaml files record precision as a {weights, activations}
object (e.g. "w4a16" written out longhand as {weights: "int4", activations:
"int16"}) instead of the short code — String()'ing that object is where the
catalog's "[object Object]" quant badge came from. Collapses it back to the
standard wNaM shorthand, and folds the "f16"/"f32" spelling some repos use
into the "fp16"/"fp32" names used everywhere else. */
function normalizeQuantName(raw) {
if (raw && typeof raw === "object") {
const bits = v => { const m = String(v ?? "").match(/\d+/); return m ? m[0] : null; };
const w = bits(raw.weights), a = bits(raw.activations);
return (w && a) ? `w${w}a${a}` : "mixed";
}
const s = String(raw ?? "").trim().toLowerCase();
return (s === "f16") ? "fp16" : (s === "f32") ? "fp32" : s;
}
/* The actual quantization/precision (fp32, int8, w4a16, …) of an artifact —
NOT the artifact folder name, which is usually the same string but isn't
always: e.g. ResNet18-OpticalFlow-ONNX's artifacts are named "xavier-export"
/ "a100-export" (two reference-target exports, both int8 per artifact_info),
so listing the raw folder name as a "quantization" is simply wrong. Falls
back to the artifact name for repos with no artifact_info override, where
the folder name already *is* the precision. */
function artifactPrecision(m, art) {
const info = (m.artifact_info || {})[art];
return normalizeQuantName((info && info.precision) ? info.precision : art);
}
function sizeLabel(sizeM) {
if (sizeM == null) return "size n/a";
return sizeM >= 1000 ? `${Math.round(sizeM / 100) / 10}B` : `${Math.round(sizeM)}M`;
}
/* Tooltip / excluded-list body: whatever the block actually reports, unfiltered
by the Catalog's KPI display toggles (those belong to the Catalog surface). */
function pointKpiText(p) {
const chips = kpiChipsOf(p.block).map(c => `${c.label} ${fmtChipFull(c)}`);
if (p.fpsDerived) chips.unshift(`Throughput ${fmtAxisNum(p.fps)} img/s (derived: 1000 / ${fmtMs(p.lat)} ms)`);
return chips.length ? chips : ["no metrics recorded"];
}
/* A block with accuracy but no timing at all is not a failed performance run —
it is the FP32 reference the quantized variants are scored against (the CV
repos' fp32 runs say so: "reference accuracy measured on physical X5H
silicon", with fps and latency explicitly null). Charting it as a missing
point would report a deliberate baseline as a gap, so these are pulled out of
the point set and listed as references instead. */
function isAccuracyReference(p) {
return p.tok == null && p.fps == null && p.lat == null && p.acc != null;
}
/* The reference accuracy for a model, so a quantized point can show what its
accuracy is measured against. */
function accuracyReferenceOf(key, refs) {
return safeList(refs, []).find(r => r.key === key) || null;
}
let CSTL_HIT = [];
/* Accuracy-reference runs for the open regime, kept so a point's tooltip can
name the baseline its quantization is scored against. */
let CSTL_REFS = [];
const CSTL_FONT = "system-ui, -apple-system, Segoe UI, Roboto, Arial";
const CSTL_INK = "rgba(31,30,28,0.72)";
const CSTL_INK_SOFT = "rgba(31,30,28,0.50)";
const CSTL_GRID = "rgba(31,30,28,0.08)";
/* Which regime tab is open, plus the axis pair chosen per regime (kept per tab so
switching back restores what you were looking at). */
const CSTL_STATE = { regime: CSTL_REGIMES[0].id, axes: {} };
function cstlAxesFor(r) {
if (!CSTL_STATE.axes[r.id]) {
CSTL_STATE.axes[r.id] = { x: safeList(r.x, ["tok"])[0], y: safeList(r.y, ["size"])[0] };
}
return CSTL_STATE.axes[r.id];
}
/* Circles (scatter) hit by distance, bars (stage chart) by rectangle. */
function hitTest(list, evt, canvas) {
const rect = canvas.getBoundingClientRect();
const x = evt.clientX - rect.left, y = evt.clientY - rect.top;
let best = null, bestD = Infinity;
for (const p of list) {
if (p.rect) {
if (x >= p.rect[0] && x <= p.rect[2] && y >= p.rect[1] && y <= p.rect[3]) return p;
continue;
}
const dx = x - p.x, dy = y - p.y, d = dx * dx + dy * dy;
if (d <= p.r * p.r && d < bestD) { best = p; bestD = d; }
}
return best;
}
function prepCanvas(canvas) {
const ctx = canvas.getContext("2d");
const w = canvas.clientWidth || canvas.parentElement?.clientWidth || 600;
const h = canvas.clientHeight || 460;
const dpr = window.devicePixelRatio || 1;
canvas.width = Math.max(1, Math.floor(w * dpr));
canvas.height = Math.max(1, Math.floor(h * dpr));
ctx.setTransform(dpr, 0, 0, dpr, 0, 0);
ctx.clearRect(0, 0, w, h);
ctx.font = `12px ${CSTL_FONT}`;
ctx.textAlign = "left";
CSTL_HIT = [];
return { ctx, w, h };
}
function emptyCanvasMsg(ctx, w, h, msg) {
ctx.fillStyle = CSTL_INK;
ctx.textAlign = "center";
ctx.fillText(msg, w / 2, h / 2);
ctx.textAlign = "left";
}
/* Ellipsis-truncate to fit maxWidth under the ctx's current font — used for the
stage chart's model-name column, since architecture names vary wildly in
length (e.g. "SigLIP-SO400M-patch14-384") and must never overlap the bars. */
function truncateToWidth(ctx, text, maxWidth) {
if (ctx.measureText(text).width <= maxWidth) return text;
let lo = 0, hi = text.length;
while (lo < hi) {
const mid = (lo + hi + 1) >> 1;
if (ctx.measureText(text.slice(0, mid) + "…").width <= maxWidth) lo = mid; else hi = mid - 1;
}
return lo > 0 ? text.slice(0, lo) + "…" : "…";
}
function barPath(ctx, x, y, w, h, r) {
const rr = Math.min(r, h / 2, w / 2);
ctx.beginPath();
ctx.moveTo(x + rr, y);
ctx.lineTo(x + w - rr, y);
ctx.quadraticCurveTo(x + w, y, x + w, y + rr);
ctx.lineTo(x + w, y + h - rr);
ctx.quadraticCurveTo(x + w, y + h, x + w - rr, y + h);
ctx.lineTo(x + rr, y + h);
ctx.quadraticCurveTo(x, y + h, x, y + h - rr);
ctx.lineTo(x, y + rr);
ctx.quadraticCurveTo(x, y, x + rr, y);
ctx.closePath();
}
/* Point labels so the chart reads without hovering. For each point: try four
offsets, prefer the first that clears both the other labels and every bubble,
fall back to the first that at least clears the other labels, and skip the
point if even that fails. Labels are drawn with a white halo so the fallback
case stays legible where the catalog clusters (small quantized models all
land in the same corner). */
function drawPointLabels(ctx, items, box) {
const overlaps = (a, b) => !(a[2] < b[0] || a[0] > b[2] || a[3] < b[1] || a[1] > b[3]);
const bubbles = items.map(it => [it.x - it.r, it.y - it.r, it.x + it.r, it.y + it.r]);
const placed = [];
ctx.font = `600 11px ${CSTL_FONT}`;
ctx.lineJoin = "round";
for (const it of items) {
const tw = ctx.measureText(it.text).width;
const cands = [
[it.x + it.r + 5, it.y + 4],
[it.x - it.r - 5 - tw, it.y + 4],
[it.x - tw / 2, it.y - it.r - 6],
[it.x - tw / 2, it.y + it.r + 14],
].map(([lx, ly]) => ({ lx, ly, rect: [lx - 2, ly - 11, lx + tw + 2, ly + 3] }))
.filter(c => c.rect[0] >= box.l && c.rect[2] <= box.r && c.rect[1] >= box.t && c.rect[3] <= box.b)
.filter(c => !placed.some(q => overlaps(c.rect, q)));
const pick = cands.find(c => !bubbles.some(b => overlaps(c.rect, b))) || cands[0];
if (!pick) continue;
placed.push(pick.rect);
ctx.strokeStyle = "rgba(255,255,255,0.92)";
ctx.lineWidth = 3;
ctx.strokeText(it.text, pick.lx, pick.ly);
ctx.fillStyle = "rgba(31,30,28,0.72)";
ctx.fillText(it.text, pick.lx, pick.ly);
}
ctx.lineWidth = 1;
ctx.font = `12px ${CSTL_FONT}`;
}
/* Scatter for the regimes whose models publish two commensurable numbers.
Returns the split between what could be placed and what could not (and why),
so the caller can show the misses instead of dropping them. */
function drawScatter(canvas, points, opts) {
const { ctx, w, h } = prepCanvas(canvas);
const ax = AXES[opts.xKey] || AXES.tok;
const ay = AXES[opts.yKey] || AXES.size;
const plotted = [], excluded = [];
for (const p of points) {
const miss = [];
if (ax.get(p) == null) miss.push(ax.label);
if (ay.get(p) == null) miss.push(ay.label);
if (miss.length) excluded.push({ p, miss }); else plotted.push(p);
}
if (!plotted.length) {
emptyCanvasMsg(ctx, w, h, points.length
? "No variant reports both of these axes — see the list below."
: "No benchmark data for this selection.");
return { plotted, excluded };
}
const pad = { l: 70, r: 22, t: 24, b: 52 };
const innerW = Math.max(1, w - pad.l - pad.r);
const innerH = Math.max(1, h - pad.t - pad.b);
const xs = plotted.map(ax.get), ys = plotted.map(ay.get);
let xMin = Math.min(0, ...xs), xMax = Math.max(...xs);
let yMin = Math.min(0, ...ys), yMax = Math.max(...ys);
if (xMax - xMin < 1e-9) xMax = xMin + 1;
if (yMax - yMin < 1e-9) yMax = yMin + 1;
xMax += (xMax - xMin) * 0.10;
yMax += (yMax - yMin) * 0.14;
const xOf = v => pad.l + ((v - xMin) / (xMax - xMin)) * innerW;
const yOf = v => pad.t + innerH - ((v - yMin) / (yMax - yMin)) * innerH;
const ticks = 4;
ctx.lineWidth = 1;
ctx.strokeStyle = CSTL_GRID;
for (let i = 0; i <= ticks; i++) {
const ty = pad.t + innerH - (innerH * i / ticks);
ctx.beginPath(); ctx.moveTo(pad.l, ty); ctx.lineTo(pad.l + innerW, ty); ctx.stroke();
ctx.fillStyle = CSTL_INK;
ctx.fillText(ay.tick(yMin + (yMax - yMin) * (i / ticks)), 8, ty + 4);
}
for (let i = 0; i <= ticks; i++) {
const tx = pad.l + (innerW * i / ticks);
ctx.beginPath(); ctx.moveTo(tx, pad.t); ctx.lineTo(tx, pad.t + innerH); ctx.stroke();
ctx.fillStyle = CSTL_INK;
ctx.textAlign = i === ticks ? "right" : "center";
ctx.fillText(ax.tick(xMin + (xMax - xMin) * (i / ticks)), tx, h - 30);
ctx.textAlign = "left";
}
ctx.fillStyle = CSTL_INK_SOFT;
ctx.fillText(`${ax.label} ${ax.lowerBetter ? "← lower is better" : "→ higher is better"}`, pad.l, h - 10);
ctx.fillText(`↑ ${ay.label}${ay.lowerBetter ? " (lower is better)" : ""}`, 8, 14);
const maxSz = Math.max(1, ...plotted.map(p => p.sizeM || 0));
const rOf = s => (s == null ? 8 : 6 + 16 * Math.sqrt(s / maxSz));
// Big bubbles first so a small fast variant is never buried under a large one.
const order = plotted.slice().sort((a, b) => (b.sizeM || 0) - (a.sizeM || 0));
const labels = [];
order.forEach(p => {
const x = xOf(ax.get(p)), y = yOf(ay.get(p)), r = rOf(p.sizeM);
const col = colorFor(colorKey(p, opts.colorBy));
// A dashed ring marks a point whose plotted throughput was derived from its
// own latency rather than reported directly.
const derivedOnAxis = p.fpsDerived && (opts.xKey === "fps" || opts.yKey === "fps");
ctx.setLineDash(derivedOnAxis ? [4, 3] : []);
ctx.beginPath(); ctx.arc(x, y, r, 0, Math.PI * 2);
if (p.source === "HIL") { ctx.fillStyle = col + "C0"; ctx.fill(); ctx.strokeStyle = col; ctx.lineWidth = 1.5; ctx.stroke(); }
else { ctx.fillStyle = col + "33"; ctx.fill(); ctx.strokeStyle = col; ctx.lineWidth = 1.8; ctx.stroke(); }
ctx.setLineDash([]);
CSTL_HIT.push({ x, y, r: Math.max(r, 10), data: p });
labels.push({ x, y, r, text: `${p.arch} ${p.quant}` });
});
if (labels.length <= 18) {
drawPointLabels(ctx, labels, { l: 4, r: w - 4, t: pad.t - 12, b: pad.t + innerH + 10 });
}
return { plotted, excluded };
}
/* Stage chart for the encoder regime: one row per artifact/source, one bar per
encoder tower. Bars are scaled to the largest *stage*, and the run total is
printed in the row label rather than drawn as a bar — it is an order of
magnitude larger than the towers and is not their sum. */
function drawStageChart(canvas, points) {
const { ctx, w, h } = prepCanvas(canvas);
// Grouped by model first, so several dual-encoder models sort into visibly
// separate blocks instead of interleaving by quant/source.
const rows = points.filter(p => p.stages.length)
.sort((a, b) => a.arch.localeCompare(b.arch) || a.quant.localeCompare(b.quant) || a.source.localeCompare(b.source));
const excluded = points.filter(p => !p.stages.length).map(p => ({ p, miss: ["per-stage timings"] }));
const stageNames = [];
rows.forEach(p => p.stages.forEach(s => { if (!stageNames.includes(s.label)) stageNames.push(s.label); }));
if (!rows.length) {
emptyCanvasMsg(ctx, w, h, "No per-stage timings for this selection.");
return { plotted: rows, excluded, stageNames };
}
const maxMs = Math.max(1, ...rows.flatMap(p => p.stages.map(s => Number(s.ms) || 0)));
const pad = { l: 186, r: 96, t: 24, b: 46 };
const innerW = Math.max(1, w - pad.l - pad.r);
const innerH = Math.max(1, h - pad.t - pad.b);
const rowH = innerH / rows.length;
const xOf = ms => pad.l + (ms / maxMs) * innerW;
const ticks = 4;
ctx.lineWidth = 1;
ctx.strokeStyle = CSTL_GRID;
for (let i = 0; i <= ticks; i++) {
const tx = pad.l + (innerW * i / ticks);
ctx.beginPath(); ctx.moveTo(tx, pad.t); ctx.lineTo(tx, pad.t + innerH); ctx.stroke();
ctx.fillStyle = CSTL_INK;
ctx.textAlign = i === ticks ? "right" : "center";
ctx.fillText(fmtAxisNum(maxMs * i / ticks), tx, h - 26);
ctx.textAlign = "left";
}
ctx.fillStyle = CSTL_INK_SOFT;
ctx.fillText("Per-stage NPU time (ms) ← lower is better", pad.l, h - 8);
rows.forEach((p, i) => {
const top = pad.t + i * rowH;
const newModel = i === 0 || rows[i - 1].arch !== p.arch;
if (i) {
// A heavier line at a model boundary reads as a group break; the default
// hairline just separates two variants of the same model.
ctx.strokeStyle = newModel ? "rgba(31,30,28,0.16)" : CSTL_GRID;
ctx.lineWidth = newModel ? 1.4 : 1;
ctx.beginPath(); ctx.moveTo(8, top); ctx.lineTo(w - 8, top); ctx.stroke();
ctx.lineWidth = 1;
}
const labelMaxW = pad.l - 22;
// Model name — the identity that used to be missing — with a color dot
// matching the same model's swatch elsewhere (cards/scatter), so several
// models scan at a glance even before reading the text.
ctx.font = `700 12px ${CSTL_FONT}`;
const nameText = truncateToWidth(ctx, p.arch, labelMaxW - 12);
ctx.fillStyle = colorFor(p.arch);
ctx.beginPath(); ctx.arc(11, top + rowH / 2 - 15, 3.5, 0, Math.PI * 2); ctx.fill();
ctx.fillStyle = CSTL_INK;
ctx.fillText(nameText, 20, top + rowH / 2 - 11);
ctx.font = `600 11.5px ${CSTL_FONT}`;
ctx.fillStyle = CSTL_INK_SOFT;
ctx.fillText(truncateToWidth(ctx, `${p.quant} · ${p.source}`, labelMaxW), 8, top + rowH / 2 + 3);
ctx.font = `11px ${CSTL_FONT}`;
ctx.fillStyle = CSTL_INK_SOFT;
ctx.fillText(p.lat != null ? `Σ run ${fmtMs(p.lat)} ms` : `${p.fmt} · ${p.target}`, 8, top + rowH / 2 + 17);
ctx.font = `12px ${CSTL_FONT}`;
const bars = p.stages;
const barH = Math.max(10, Math.min(24, (rowH - 18) / bars.length - 6));
const stackH = bars.length * (barH + 6) - 6;
bars.forEach((s, j) => {
const y = top + (rowH - stackH) / 2 + j * (barH + 6);
const bw = Math.max(3, xOf(Number(s.ms)) - pad.l);
const col = colorFor(s.label);
barPath(ctx, pad.l, y, bw, barH, 3);
ctx.fillStyle = col + "D9"; ctx.fill();
ctx.strokeStyle = col; ctx.lineWidth = 1; ctx.stroke();
const nameW = ctx.measureText(s.label).width;
const inside = bw > nameW + 18 && barH >= 14;
if (inside) {
ctx.fillStyle = "#FFFFFF";
ctx.fillText(s.label, pad.l + 9, y + barH / 2 + 4);
}
ctx.fillStyle = CSTL_INK;
ctx.fillText(`${fmtMs(s.ms)} ms${inside ? "" : " · " + s.label}`, pad.l + bw + 7, y + barH / 2 + 4);
CSTL_HIT.push({ rect: [pad.l, y, pad.l + bw, y + barH], data: p });
});
});
return { plotted: rows, excluded, stageNames };
}
function setLegend(items, extraHTML) {
const el = $("cstlLegend");
if (!el) return;
el.innerHTML = items.map(it =>
`${esc(it.label)}`
).join("") + (extraHTML || "");
}
/* Roadmap regime: no metric exists, so there is nothing to plot — the models are
listed with the reason, which is the point of the tab. */
function roadmapReasons(m) {
const mod = String(m.modality || "").trim().toUpperCase();
const out = [];
if (mod === "VLA") {
out.push("Emits an action chunk per step — the comparable KPI is closed-loop control rate (Hz) and task success, which the benchmark pipeline does not produce yet.");
}
if (isComingSoon(m)) out.push("No artifact folder published yet, so there is nothing to measure or download.");
if (!out.length) out.push("No plottable metric published yet.");
return out;
}
function roadmapCardHTML(m) {
const col = colorFor(m.architecture || "Unknown");
const d = designerOf(m);
const cfg = safeFirstCfg(m);
const defFmt = m.gguf_repo ? "GGUF" : (m.onnx_repo ? "ONNX" : "GGUF");
return `
`;
}
function renderRoadmapCards(models) {
const el = $("cstlCards");
if (!el) return;
el.innerHTML = models.length
? models.map(roadmapCardHTML).join("")
: `
Nothing pending — every model in the catalog reports a plottable metric.
`;
}
/* One row per variant that isn't a point on this chart, in two groups with very
different meanings: a genuine metadata gap (the metric is missing) versus an
accuracy reference run (deliberately has no timing). */
function cstlRowHTML(p, missHTML) {
return `