/* app.js — Static dashboard logic (artifact-layer aware) */ const HF_BASE = "https://huggingface.co"; const HF_SPACES = "https://huggingface.co/spaces"; const MODELS_FILE = "models.json"; const LINKS_FILE = "links.json"; const $ = (id) => document.getElementById(id); const $$ = (sel) => Array.from(document.querySelectorAll(sel)); /** Proper HTML escaping. */ function esc(str) { return String(str ?? "") .replaceAll("&", "&") .replaceAll("<", "<") .replaceAll(">", ">") .replaceAll('"', """) .replaceAll("'", "'"); } let CATALOG_ROOT = { models: [] }; let CATALOG = []; let CATALOG_NOTE = "Auto-generated from Hugging Face repositories."; /* Tier-1 KPI metadata from models.json: key -> {label, unit, direction, modalities}. Lets renderers show a metric's name/unit without hardcoding a second copy of the table generate_models_json.py already built (see KPI_REGISTRY there). */ let KPI_REGISTRY = {}; let LINKS = { spaces: {}, collections: {} }; let KEY_TO_MODEL = {}; let ARCHS = ["All"]; let MODALITIES = ["All"]; let VARIANTS = ["All"]; const ALL_FORMATS = ["ONNX", "GGUF"]; const ALL_HW = ["CPU", "NPU", "DSP"]; function uniqueSorted(values) { const out = []; const seen = new Set(); for (const v of values) { const x = String(v ?? "Unknown"); if (!seen.has(x)) { seen.add(x); out.push(x); } } return out.sort((a, b) => a.localeCompare(b)); } async function loadJSON(path, fallback) { try { const res = await fetch(path, { cache: "no-store" }); if (!res.ok) throw new Error(`HTTP ${res.status} ${res.statusText}`); return await res.json(); } catch { return fallback; } } function hfModelUrl(repoId) { return `${HF_BASE}/${repoId}`; } function hfSpaceUrl(spaceId) { return `${HF_SPACES}/${spaceId}`; } function safeList(x, fallback = []) { return Array.isArray(x) ? x : fallback; } function defaultPurposeForFormat(fmt) { return fmt === "ONNX" ? "PerformanceEstimate" : "llama.cpp Benchmarking"; } function validatePurpose(fmt, purpose) { if ((purpose === "PerformanceEstimate" || purpose === "ORT Benchmarking") && fmt !== "ONNX") { return { ok: false, msg: "⚠️ PerformanceEstimate / ORT benchmarking is ONNX-only. Switch Format → ONNX." }; } if (purpose === "llama.cpp Benchmarking" && fmt !== "GGUF") { return { ok: false, msg: "⚠️ llama.cpp benchmarking is GGUF-only. Switch Format → GGUF." }; } return { ok: true, msg: "" }; } /* ---------- Badge builders (HTML) ---------- */ function formatBadgesHTML(model) { const parts = []; if (model.onnx_repo) parts.push(`ONNX`); if (model.gguf_repo) parts.push(`GGUF`); return `
${parts.join("") || `—`}
`; } function hwBadgesHTML(targets) { const t = safeList(targets, []); const parts = []; if (t.includes("CPU")) parts.push(`CPU`); if (t.includes("NPU")) parts.push(`NPU`); if (t.includes("DSP")) parts.push(`DSP`); return `
${parts.join("") || `—`}
`; } function repoButtonsSmallHTML(model) { const parts = []; if (model.onnx_repo) { parts.push(`ONNX`); } if (model.gguf_repo) { parts.push(`GGUF`); } return `
${parts.join("") || `—`}
`; } /* Precision/Quant badges (horizontal wrap) */ function artifactsBadgesHTML(model) { const a1 = safeList(model.onnx_artifacts, []); const a2 = safeList(model.gguf_artifacts, []); const arts = uniqueSorted([...a1, ...a2].map(a => artifactPrecision(model, a))); if (!arts.length) return `
—
`; return `
${ arts.map(a => `${esc(a)}`).join("") }
`; } /* ---------- Metrics formatting ---------- */ // peak_mem_mb actually holds memory bandwidth (MB/s); the UI always shows it as GB/s. function memBwGBs(mb) { return mb == null ? null : Math.round((mb / 1024) * 10) / 10; } /* KPI_REGISTRY-driven label/unit lookup for the Tier-1 fields that aren't already hand-labeled below (tok_s/latency_ms_p50/peak_mem_mb keep their existing wording). */ function kpiLabel(key, fallback) { return (KPI_REGISTRY[key] && KPI_REGISTRY[key].label) || fallback; } function kpiUnit(key, fallback) { return (KPI_REGISTRY[key] && KPI_REGISTRY[key].unit) || fallback; } function summarizeUnit(name, unitObj) { if (!unitObj) return []; const lines = []; if (unitObj.tok_s != null) lines.push(`${name} tok/s: ${unitObj.tok_s}`); if (unitObj.prefill_tok_s != null) lines.push(`${name} ${kpiLabel("prefill_tok_s", "prefill")} ${kpiUnit("prefill_tok_s", "tok/s")}: ${unitObj.prefill_tok_s}`); if (unitObj.ttft_ms != null) lines.push(`${name} ${kpiLabel("ttft_ms", "TTFT")} ${kpiUnit("ttft_ms", "ms")}: ${unitObj.ttft_ms}`); if (unitObj.latency_ms_p50 != null) lines.push(`${name} p50 ms: ${unitObj.latency_ms_p50}`); if (unitObj.peak_mem_mb != null) lines.push(`${name} mem BW GB/s: ${memBwGBs(unitObj.peak_mem_mb)}`); return lines; } function metricSummaryV2(obj) { if (!obj) return "—"; const cpu = obj.cpu; const npu = obj.npu; const dsp = obj.dsp; const total = obj.total || {}; const setup = obj.setup; const updated = obj.last_updated; let lines = []; lines = lines.concat(summarizeUnit("CPU", cpu)); lines = lines.concat(summarizeUnit("NPU", npu)); lines = lines.concat(summarizeUnit("DSP", dsp)); if (total && typeof total === "object") { if (total.tok_s != null) lines.push(`TOTAL tok/s: ${total.tok_s}`); if (total.prefill_tok_s != null) lines.push(`TOTAL ${kpiLabel("prefill_tok_s", "prefill")} ${kpiUnit("prefill_tok_s", "tok/s")}: ${total.prefill_tok_s}`); if (total.ttft_ms != null) lines.push(`TOTAL ${kpiLabel("ttft_ms", "TTFT")} ${kpiUnit("ttft_ms", "ms")}: ${total.ttft_ms}`); if (total.peak_mem_mb != null) lines.push(`TOTAL mem BW GB/s: ${memBwGBs(total.peak_mem_mb)}`); if (total.latency_ms_p50 != null) lines.push(`TOTAL p50 ms: ${total.latency_ms_p50}`); } // Tier-2 generic stage timings (e.g. SigLIP's Vision Encoder / Text Encoder split). for (const s of safeList(obj.stages, [])) { if (s && s.ms != null) lines.push(`${s.label}: ${s.ms} ms`); } if (setup) lines.push(`Setup: ${setup}`); if (updated) lines.push(`Updated: ${updated}`); if (obj.accuracy != null) lines.push(`Accuracy: ${obj.accuracy}`); if (obj.llm_metrics?.overall != null) lines.push(`LLM overall: ${obj.llm_metrics.overall}`); if (obj.vlm_metrics?.overall != null) lines.push(`VLM overall: ${obj.vlm_metrics.overall}`); return lines.length ? lines.join("\n") : "✓"; } /* Shows every KPI the block actually reports (subject to the "KPIs to display" filter): throughput/prefill/TTFT, any Tier-2 stage timing (e.g. SigLIP's Vision/Text Encoder split), latency/total, and accuracy — one row each, in that order, plus a fixed mem-BW row. A block that reports nothing meaningful just renders the mem-BW row (still "—" if that's absent too), rather than three placeholder rows for metrics it never had. */ function metricBriefHTML(block) { if (!block || typeof block !== "object") return `—`; const total = (block.total && typeof block.total === "object") ? block.total : {}; const mem = memBwGBs(total.peak_mem_mb); const rows = visibleKpiChips(block) .map(c => `
${esc(c.label)}${esc(fmtChipFull(c))}
`); rows.push(`
mem BW${mem != null ? esc(mem) + " GB/s" : "—"}
`); return `
${rows.join("")}
`; } /* ---------- Catalog helpers ---------- */ function getCfg(model, cfgId) { for (const c of safeList(model.hardware_configs, [])) { if (c.id === cfgId) return c; } return null; } function safeFirstCfg(model) { const cfgs = safeList(model.hardware_configs, []); return cfgs.length ? cfgs[0] : null; } function getArtifactsFor(modelKey, fmt) { const m = KEY_TO_MODEL[modelKey]; if (!m) return []; return (fmt === "ONNX") ? safeList(m.onnx_artifacts, []) : safeList(m.gguf_artifacts, []); } /* True only if the artifact actually carries benchmark metadata (from its benchmarks/ subfolder → an estimate or hil block). Checked against the per-artifact map directly, NOT metricsForArtifact (whose legacy fallback would make every artifact look benchmarked). A legacy config with no artifacts map at all is treated as benchmarked so it isn't hidden. */ function artifactHasBenchmark(m, art) { for (const cfg of safeList(m?.hardware_configs, [])) { const metrics = cfg.metrics || {}; const artMap = metrics.artifacts; if (artMap && typeof artMap === "object") { const a = artMap[art]; if (a && (a.estimate || a.hil)) return true; } else if (metrics.estimate || metrics.hil) { return true; } } return false; } /* Artifacts to actually surface in the results UI: only those with benchmark data (e.g. 8B's fp16 folder has no benchmark, so it must not appear). */ function benchmarkedArtifactsFor(modelKey, fmt) { const m = KEY_TO_MODEL[modelKey]; if (!m) return []; return getArtifactsFor(modelKey, fmt).filter(a => artifactHasBenchmark(m, a)); } /** * Artifact-layer aware: * - Preferred: cfg.metrics.artifacts[artifact].estimate/hil * - Fallback: cfg.metrics.estimate/hil (legacy) */ function metricsForArtifact(cfgMetrics, artifact) { const m = cfgMetrics || {}; const artMap = m.artifacts; if (artifact && artMap && typeof artMap === "object" && artMap[artifact]) { const a = artMap[artifact] || {}; return { estimate: a.estimate || null, hil: a.hil || null }; } return { estimate: m.estimate || null, hil: m.hil || null }; } function bestAvailableBlockForArtifact(cfgMetrics, artifact) { const { estimate, hil } = metricsForArtifact(cfgMetrics, artifact); return hil || estimate || null; } /* Every benchmark run recorded for an artifact (one per runtime/engine, tagged with its kind: "hil" | "estimate"). The estimate/hil blocks above are the representative single run per kind; runs[] keeps the full per-runtime detail (e.g. ResNet50 int8 measured on both onnxruntime and mwmx). */ function artifactRuns(cfgMetrics, artifact) { const artMap = (cfgMetrics || {}).artifacts; if (artifact && artMap && typeof artMap === "object" && artMap[artifact] && Array.isArray(artMap[artifact].runs)) { return artMap[artifact].runs; } return []; } /* Human-friendly task label: "image-classification" -> "Image classification". */ function prettyTask(t) { const s = String(t ?? "").trim(); if (!s) return ""; return s.replace(/[-_]+/g, " ").replace(/^\w/, c => c.toUpperCase()); } /* Accuracy for one artifact under a config, scanning all runs (HIL-first, then estimate), falling back to the representative blocks. Returns a number or null. */ function accuracyForArtifact(cfg, art) { const runs = artifactRuns(cfg?.metrics || {}, art); let a = runs.filter(r => r.kind === "hil").map(accuracyFromBlock).find(v => v != null); if (a == null) a = runs.filter(r => r.kind === "estimate").map(accuracyFromBlock).find(v => v != null); if (a == null) { const { estimate, hil } = metricsForArtifact(cfg?.metrics || {}, art); a = accuracyFromBlock(hil); if (a == null) a = accuracyFromBlock(estimate); } return a; } /* The full-precision reference accuracy (fp32, else fp16) for a model+format, used to express how far a quantized precision drops from the float baseline. */ function referenceAccuracy(m, cfg, fmt) { const arts = getArtifactsFor(m.key, fmt); for (const ref of ["fp32", "fp16"]) { if (arts.includes(ref)) { const a = accuracyForArtifact(cfg, ref); if (a != null) return { art: ref, acc: a }; } } return null; } /* ---------- Accuracy + runtime extraction ---------- */ function accuracyFromBlock(block) { if (!block || typeof block !== "object") return null; for (const k of ["accuracy", "accuracy_pct", "acc", "quality"]) { if (block[k] != null && isFinite(Number(block[k]))) return Number(block[k]); } const tryList = (obj, keys) => { if (!obj || typeof obj !== "object") return null; for (const k of keys) { if (obj[k] != null && isFinite(Number(obj[k]))) return Number(obj[k]); } return null; }; let v = tryList(block.llm_metrics, ["overall", "mmlu", "gsm8k", "hellaswag", "truthfulqa", "mt_bench"]); if (v != null) return v; v = tryList(block.vlm_metrics, ["overall", "mmbench", "vqav2", "seedbench", "pope"]); if (v != null) return v; return null; } function runtimeTotalMsFromBlock(block) { if (!block || typeof block !== "object") return null; if (block.total && block.total.latency_ms_p50 != null && isFinite(Number(block.total.latency_ms_p50))) { return Number(block.total.latency_ms_p50); } const sumUnits = ["cpu", "npu", "dsp"] .map(u => block[u]?.latency_ms_p50) .filter(v => v != null && isFinite(Number(v))) .map(Number); if (sumUnits.length) return sumUnits.reduce((a, b) => a + b, 0); const tok = (block.total && block.total.tok_s != null) ? block.total.tok_s : block.tok_s; if (tok != null && isFinite(Number(tok)) && Number(tok) > 0) { return 1000 / Number(tok); } const fps = (block.total && block.total.fps != null) ? block.total.fps : block.fps; if (fps != null && isFinite(Number(fps)) && Number(fps) > 0) { return 1000 / Number(fps); } return null; } /* ---------- Throughput / latency (LLM report tok/s, vision models img/s) ---------- These read either the block's `total` roll-up or, failing that, the per-unit (npu/cpu/dsp) values, so the UI can render a single figure with the right unit regardless of whether a model is measured in tokens or frames per second. */ function _firstUnitVal(block, key) { if (!block || typeof block !== "object") return null; const tot = block.total || {}; if (tot[key] != null && isFinite(Number(tot[key]))) return Number(tot[key]); for (const u of ["npu", "cpu", "dsp"]) { const v = block[u]?.[key]; if (v != null && isFinite(Number(v))) return Number(v); } return null; } /* Throughput as {value, unit}: tokens/sec for language models, frames/sec (img/s) for vision models. Prefers tok/s when both are somehow present. */ function throughputOf(block) { const tok = _firstUnitVal(block, "tok_s"); if (tok != null) return { value: tok, unit: "tok/s" }; const fps = _firstUnitVal(block, "fps"); if (fps != null) return { value: fps, unit: "img/s" }; return { value: null, unit: "tok/s" }; } function throughputValueOf(block) { return throughputOf(block).value; } /* Median inference latency (ms). LLM benchmarks and vision benchmarks are both normalized to latency_ms_p50 by the generator. */ function latencyMsOf(block) { return _firstUnitVal(block, "latency_ms_p50"); } /* ---------- Generic KPI chips (every metric a block actually reports) ---------- Every model family reports a different mix of Tier-1 fields (tok_s/prefill/ttft/fps), Tier-2 generic stage timings (block.stages — e.g. SigLIP's Vision/Text Encoder split), and accuracy. This turns whatever a block actually has into a flat, orderable list so cardHTML/metricBriefHTML can show "whatever is in the performance section" instead of hardcoding one metric per model family — and so the KPI display filter (KPI_KINDS / CATALOG_STATE.kpi) can hide/show each *kind* uniformly across every render surface. */ const KPI_KINDS = [ { id: "throughput", label: "Throughput (tok/s, fps)" }, { id: "prefill", label: "Prefill rate" }, { id: "ttft", label: "TTFT" }, { id: "stage", label: "Stage timings (e.g. encoder split)" }, { id: "latency", label: "Latency / Total" }, { id: "accuracy", label: "Accuracy" }, ]; function kpiChipsOf(block) { if (!block || typeof block !== "object") return []; const total = block.total || {}; const chips = []; const thr = throughputOf(block); if (thr.value != null) chips.push({ kind: "throughput", label: "Throughput", value: thr.value, unit: thr.unit }); if (total.prefill_tok_s != null) { chips.push({ kind: "prefill", label: kpiLabel("prefill_tok_s", "Prefill"), value: total.prefill_tok_s, unit: kpiUnit("prefill_tok_s", "tok/s") }); } if (total.ttft_ms != null) { chips.push({ kind: "ttft", label: kpiLabel("ttft_ms", "TTFT"), value: total.ttft_ms, unit: kpiUnit("ttft_ms", "ms") }); } for (const s of safeList(block.stages, [])) { if (s && s.ms != null) chips.push({ kind: "stage", label: s.label, value: s.ms, unit: "ms" }); } const lat = latencyMsOf(block); if (lat != null) { chips.push({ kind: "latency", label: safeList(block.stages, []).length ? "Total" : "Latency", value: lat, unit: "ms" }); } if (block.accuracy != null) chips.push({ kind: "accuracy", label: "Accuracy", value: block.accuracy * 100, unit: "%" }); if (block.top5_accuracy != null) chips.push({ kind: "accuracy", label: "Top-5 Accuracy", value: block.top5_accuracy * 100, unit: "%" }); return chips; } /* KPI chips filtered by the user's "KPIs to display" selection (CATALOG_STATE.kpi). */ function visibleKpiChips(block) { const enabled = CATALOG_STATE.kpi; return kpiChipsOf(block).filter(c => !enabled || enabled.has(c.kind)); } /* The one chip to headline (matches the old tok/fps -> latency -> accuracy fallback order), so existing LLM/CV cards keep showing throughput as the hero number. */ function primaryKpiChip(chips) { return chips.find(c => c.kind === "throughput") || chips.find(c => c.kind === "latency") || chips.find(c => c.kind === "stage") || chips.find(c => c.kind === "prefill") || chips.find(c => c.kind === "ttft") || chips.find(c => c.kind === "accuracy") || null; } /* The single best *remaining* KPI after the headline — one fixed extra line, never a variable-length list, so the card's stat box is the same size for every model regardless of how many KPIs it reports. Prefers a stage breakdown (e.g. SigLIP's Vision Encoder) over prefill/TTFT/accuracy/a second latency figure. */ function secondaryKpiChip(chips, primary) { const rest = chips.filter(c => c !== primary); return rest.find(c => c.kind === "stage") || rest.find(c => c.kind === "prefill") || rest.find(c => c.kind === "ttft") || rest.find(c => c.kind === "accuracy") || rest.find(c => c.kind === "latency") || null; } function fmtChipValue(c) { if (!c || c.value == null || !isFinite(Number(c.value))) return "—"; return String(Math.round(Number(c.value) * 10) / 10); } /* "76.3%" for percentages (no space, matching the rest of the UI), "396.8 ms" otherwise. */ function fmtChipFull(c) { const v = fmtChipValue(c); if (v === "—") return v; return c.unit === "%" ? `${v}%` : `${v} ${c.unit}`; } /* ---------- UI builders ---------- */ function buildRadioGroup(container, name, choices, value) { if (!container) return; container.innerHTML = choices.map(c => { const id = `${name}_${c.replace(/\W+/g, "_")}`; const checked = (c === value) ? "checked" : ""; return ` `; }).join(""); } function buildCheckGroup(container, name, choices, values) { if (!container) return; const set = new Set(values || []); container.innerHTML = choices.map(c => { const id = `${name}_${c.replace(/\W+/g, "_")}`; const checked = set.has(c) ? "checked" : ""; return ` `; }).join(""); } function readRadio(name, fallback) { const el = document.querySelector(`input[name="${CSS.escape(name)}"]:checked`); return el ? el.value : fallback; } function readChecks(name) { return Array.from(document.querySelectorAll(`input[name="${CSS.escape(name)}"]:checked`)).map(x => x.value); } /* A minimal detail panel for not-yet-published models: no benchmarks, no artifacts, no download command to fabricate, and no repo link — the repo for these is either absent or just a placeholder, so linking to it would send someone to an empty page. Just what's known about the model. Bypasses the full benchmark rendering below entirely, since building a download command against zero artifacts would otherwise produce a nonsensical path. Shared by the modal and the full page, each passing the same `dense` flag their real-model rendering uses (see renderModalBodyHTML / renderFullModelPageHTML) so a coming-soon card sits in the same rail-plus-content skeleton as a published one instead of falling back to a one-off layout. */ function comingSoonDetailsHTML(m, dense) { const specs = modelSpecPairs(m, null, "", "", null).filter(s => s.k !== "Input resolution" && s.k !== "Format / compute"); const specsKvHTML = specs.map(s => `
${esc(s.k)}${esc(s.v)}
`).join(""); const specsGridHTML = specs.map(s => `
${esc(s.k)}
${esc(s.v)}
`).join(""); const notice = `
🚧 Coming soon. This model is in progress — benchmarks and download artifacts have not been published yet.
`; if (dense) { return `
${exampleImageHTML(m)}
${specsKvHTML}
${notice}
`; } return `
Model summary
${specsGridHTML}
${exampleImageHTML(m)}
${notice}`; } /* Fallback illustration (animated SVG) chosen by modality. */ const MODALITY_FALLBACK = { LLM: "assets/fallback-llm.svg", VLM: "assets/fallback-vlm.svg", VLA: "assets/fallback-vla.svg", CV: "assets/fallback-cv.svg", ALM: "assets/fallback-alm.svg", // No dedicated dual-encoder illustration yet — reuse the VLM one (closest: image + text I/O). EMBED: "assets/fallback-vlm.svg", }; /* CV has multiple distinct tasks with their own illustration — a per-object bounding-box scene reads as object detection, not classification. */ const CV_TASK_FALLBACK = { "object-detection": "assets/fallback-cv.svg", "image-classification": "assets/fallback-cv-classification.svg", "3d-object-detection": "assets/fallback-cv-3d.svg", "object-detection-3d": "assets/fallback-cv-3d.svg", "lane-detection": "assets/fallback-cv-lane.svg", "image-segmentation": "assets/fallback-cv-segmentation.svg", "semantic-segmentation": "assets/fallback-cv-segmentation.svg", "video-classification": "assets/fallback-cv-video.svg", "image-feature-extraction": "assets/fallback-cv-feature.svg", }; /* VLA covers very different domains — a humanoid-robot arm reads wrong for a driving model. Same idea as CV_TASK_FALLBACK: pick by the model's own task. */ const VLA_TASK_FALLBACK = { "autonomous-driving": "assets/fallback-vla-driving.svg", }; /* Detail-page example figure: the single modality/task-appropriate animated SVG illustration for a model (unchanged — the richer "what it's doing" scene shown in the modal/full page, distinct from the compact card set below). */ function fallbackIllustrationSrc(m) { const modality = m.modality || "LLM"; const task = safeList(m.tasks, [])[0] || ""; const bev = modality === "CV" && bevKey(m); return (bev && `assets/fallback-${bev}.svg`) || (modality === "CV" && CV_TASK_FALLBACK[task]) || (modality === "VLA" && VLA_TASK_FALLBACK[task]) || MODALITY_FALLBACK[modality] || MODALITY_FALLBACK.LLM; } /* Same modality/task routing as fallbackIllustrationSrc, but returning a key into CARD_ILLUSTRATION_VARIANTS rather than a single file. */ const CV_TASK_KEY = { "image-classification": "cv-classification", "3d-object-detection": "cv-3d", "object-detection-3d": "cv-3d", "lane-detection": "cv-lane", "image-segmentation": "cv-segmentation", "semantic-segmentation": "cv-segmentation", "video-classification": "cv-video", "image-feature-extraction": "cv-feature", }; /* Bird's-eye-view models (BEVFormer, BEVFusion, BEVLaneDet) share a top-down look regardless of their nominal task, so they're detected by family/key. */ function bevKey(m) { const id = `${m.family || ""} ${m.key || ""}`.toLowerCase(); if (!/(^|[\s_-])bev(former|fusion|lane|[\s_-]|$)/.test(id)) return null; if (id.includes("fusion")) return "cv-bev-fusion"; if (safeList(m.tasks, []).includes("lane-detection")) return "cv-bev-lane"; return "cv-bev"; } function illustrationTypeKey(m) { const modality = m.modality || "LLM"; const task = safeList(m.tasks, [])[0] || ""; if (modality === "CV" && bevKey(m)) return bevKey(m); if (modality === "CV" && CV_TASK_KEY[task]) return CV_TASK_KEY[task]; if (modality === "CV") return "cv"; if (modality === "VLA" && task === "autonomous-driving") return "vla-driving"; if (modality === "VLA") return "vla"; if (modality === "VLM" || modality === "EMBED") return "vlm"; if (modality === "ALM") return "alm"; return "llm"; } /* Catalog-card illustration set — several hand-designed variants per modality/task, sized and cropped for the short banner strip (as opposed to the single wide scene each modality gets in the detail-page figure). A model always lands on the same variant (stable hash of its key) so the grid doesn't shuffle on re-render, but sibling models of the same modality don't all show the identical picture. */ const CARD_ILLUSTRATION_VARIANTS = { llm: ["assets/cards/llm-1.svg", "assets/cards/llm-2.svg", "assets/cards/llm-3.svg"], vlm: ["assets/cards/vlm-1.svg", "assets/cards/vlm-2.svg", "assets/cards/vlm-3.svg"], vla: ["assets/cards/vla-1.svg", "assets/cards/vla-2.svg", "assets/cards/vla-3.svg"], "vla-driving": ["assets/cards/vla-driving-1.svg", "assets/cards/vla-driving-2.svg", "assets/cards/vla-driving-3.svg"], cv: ["assets/cards/cv-1.svg", "assets/cards/cv-2.svg", "assets/cards/cv-3.svg"], "cv-classification": ["assets/cards/cv-classification-1.svg", "assets/cards/cv-classification-2.svg", "assets/cards/cv-classification-3.svg"], "cv-3d": ["assets/cards/cv-3d-1.svg", "assets/cards/cv-3d-2.svg", "assets/cards/cv-3d-3.svg"], "cv-lane": ["assets/cards/cv-lane-1.svg", "assets/cards/cv-lane-2.svg", "assets/cards/cv-lane-3.svg"], "cv-segmentation": ["assets/cards/cv-segmentation-1.svg", "assets/cards/cv-segmentation-2.svg", "assets/cards/cv-segmentation-3.svg"], "cv-video": ["assets/cards/cv-video-1.svg", "assets/cards/cv-video-2.svg", "assets/cards/cv-video-3.svg"], "cv-feature": ["assets/cards/cv-feature-1.svg", "assets/cards/cv-feature-2.svg", "assets/cards/cv-feature-3.svg"], "cv-bev": ["assets/cards/cv-bev-1.svg"], "cv-bev-fusion": ["assets/cards/cv-bev-fusion-1.svg"], "cv-bev-lane": ["assets/cards/cv-bev-lane-1.svg"], alm: ["assets/cards/alm-1.svg", "assets/cards/alm-2.svg", "assets/cards/alm-3.svg"], }; function stableHash(str) { let h = 0; for (let i = 0; i < str.length; i++) h = (Math.imul(h, 31) + str.charCodeAt(i)) | 0; return Math.abs(h); } /* compact = true swaps in the assets/cards/compact/ set: the same scenes redrawn slightly zoomed-out so their focal details (a chat bubble's tail, a robot arm's gripper, a detection badge) survive the shorter Overview carousel banner's object-fit:cover crop instead of bleeding off the edge. */ function cardIllustrationSrc(m, { compact = false } = {}) { const variants = CARD_ILLUSTRATION_VARIANTS[illustrationTypeKey(m)] || CARD_ILLUSTRATION_VARIANTS.llm; const src = variants[stableHash(String(m.key || m.display_name || "")) % variants.length]; return compact ? src.replace("assets/cards/", "assets/cards/compact/") : src; } /* Same hash-picked variant per model as cardIllustrationSrc, but walked in list order so that whenever two neighbors would land on the identical illustration (same family + same variant index — most likely when a burst of same-modality models land back to back, e.g. several YOLO releases), the second one is nudged to the next variant instead. Used by both the Last Update carousel (true left-right adjacency) and the Catalog grid (render order = reading order, so this at least guarantees no repeat within a row; the grid's column count isn't known at render time to also de-dupe vertically). */ function sequentialIllustrationSrcs(list, { compact = false } = {}) { let prevFamily = null, prevIdx = null; return list.map(m => { const familyKey = illustrationTypeKey(m); const variants = CARD_ILLUSTRATION_VARIANTS[familyKey] || CARD_ILLUSTRATION_VARIANTS.llm; let idx = stableHash(String(m.key || m.display_name || "")) % variants.length; if (familyKey === prevFamily && idx === prevIdx && variants.length > 1) { idx = (idx + 1) % variants.length; } prevFamily = familyKey; prevIdx = idx; const src = variants[idx]; return compact ? src.replace("assets/cards/", "assets/cards/compact/") : src; }); } /* "What it's doing" — a sample/preview image pulled from the model repo; falls back to a modality-appropriate animated SVG illustration when the repo ships none or it fails to load. Shown for all modalities. */ function exampleImageHTML(m) { const url = m && m.sample_image ? String(m.sample_image) : ""; const modality = m.modality || "LLM"; const task = safeList(m.tasks, [])[0] || ""; const fallbackSrc = fallbackIllustrationSrc(m); const taskLabel = task ? prettyTask(task) : modality; /* If the repo supplies an image, show it over the animated fallback. onerror removes the , revealing the animated SVG beneath. */ const fallbackImg = `${esc(modality)} illustration`; const repoImg = url ? `${esc(taskLabel)} example` : ""; return `
${fallbackImg}${repoImg}
Example${task ? " · " + esc(taskLabel) : " · " + esc(modality)}
`; } /* Runs for an artifact: real per-runtime runs when present, else a synthetic one-row-per-kind fallback built from the aggregate hil/estimate blocks (mirrors accuracyForArtifact's HIL-first fallback). */ function runsForArtifact(cfgMetrics, art) { const direct = artifactRuns(cfgMetrics, art).filter(Boolean); if (direct.length) return direct; const { estimate, hil } = metricsForArtifact(cfgMetrics, art); const synth = []; if (hil) synth.push({ ...hil, kind: "hil" }); if (estimate) synth.push({ ...estimate, kind: "estimate" }); return synth; } /* Device utilization: how much of the chip a run actually drove (NPU instances, AI cores, clock frequency, TOPS) vs. what the hardware config has available — sourced straight from the benchmark YAML's hardware/configuration blocks. */ function utilizationOf(block) { return (block && block.utilization) ? block.utilization : null; } // function downloadIncludePattern(fmt, artifact) { // const a = String(artifact || "").trim(); // if (!a) return "*"; // if (fmt === "GGUF") return `*${a}*.gguf`; // return `*${a}*`; // } /* ---------- Download / Deploy helpers ---------- */ /* Mode A — “Download repo (filtered)”. Pulls only the artifact’s folder from the repo (host-side benchmarking or further conversion). Keeps the existing behaviour: one hf-download line. */ function downloadIncludePattern(fmt, artifact) { const a = String(artifact || "").trim(); if (!a) return "*"; // Same pattern for both formats: the artifact name is a top-level folder // in every Renesas repo (fp16/, w4a16/, w4a8/, fp32/). return `${a}/*`; } /* Mode B — “Deploy with prebuilt binaries”. Resolves installer / model / runner filenames from naming conventions in the per-repo README, with optional per-model overrides via `model.deploy[artifact]` in models.json. Returns null when no prebuilt binaries are published for the (artifact, hardware) pair, in which case the UI will gracefully degrade. Currently GGUF + RCAR-X5H only. */ function deployInfoFor(model, fmt, artifact, cfg) { if (!model || !artifact || fmt !== "GGUF") return null; // No published binaries for ONNX or non-X5H targets yet. const cfgId = (cfg?.id || "").toUpperCase(); if (cfgId && cfgId !== "RCAR-X5H") return null; const ov = model.deploy?.[artifact] || {}; const isQuant = artifact.toLowerCase() !== "fp16"; const runnerKind = isQuant ? "llama-quant-runner" : "llama-runner"; const installerVer = ov.installer_version || "0.1.0"; const installer = ov.installer || `${runnerKind}-${installerVer}-Linux.sh`; const installDir = ov.install_dir || runnerKind; const runnerBin = ov.runner_bin || runnerKind; const xosTag = ov.xos_tag || "xOS_v4.32"; const boardTag = ov.board_tag || "rcar-x5hv1"; const modelTag = isQuant ? artifact : "f16"; // README uses *-f16.gguf for FP16 const modelFile = ov.model_file || `${model.display_name || model.key}-${modelTag}.gguf`; return { installer, installDir, runnerBin, modelFile, installerPath: `${artifact}/binaries/${boardTag}/${xosTag}/${installer}`, modelPath: `${artifact}/${modelFile}`, }; } /* Render the Download section as two tabs: “Download repo” (filtered hf download) and “Deploy with binaries” (4-step install + run on X5H board). */ function renderActions(modelKey, fmt, compute, purpose, artifact, cfgId, mode = "explore") { const m = KEY_TO_MODEL[modelKey]; if (!m) return "No actions."; const v = validatePurpose(fmt, purpose); if (!v.ok) { return `
${v.msg}
Actions are disabled until the selection is valid.
`; } const repoId = (fmt === "ONNX") ? (m.onnx_repo || "") : (m.gguf_repo || ""); const include = downloadIncludePattern(fmt, artifact); const cfg = cfgId ? getCfg(m, cfgId) : safeFirstCfg(m); const deploy = deployInfoFor(m, fmt, artifact, cfg); const repoUrl = repoId ? hfModelUrl(repoId) : ""; const headCls = mode === "modal" ? "no-border" : ""; const showH3 = mode !== "tab"; if (!repoId) { return `${showH3 ? `

Download

` : ""}
Repo not set for the selected format (${esc(fmt)}).
`; } /* Mode A — single filtered hf download */ const cmdRepo = `hf download ${repoId} --repo-type=model --include "${include}"`; /* Mode B — installer + model + on-board install + run */ const deployBlock = deploy ? `
  1. 1 Download installer (host)
    hf download ${esc(repoId)} --repo-type=model --include "${esc(deploy.installerPath)}"
  2. 2 Download GGUF model (host)
    hf download ${esc(repoId)} --repo-type=model --include "${esc(deploy.modelPath)}"
  3. 3 Copy to X5H board, install & stage model
    bash ./${esc(deploy.installer)} --prefix=./ --exclude-subdir --skip-license
    mv ${esc(deploy.modelFile)} ${esc(deploy.installDir)}/
  4. 4 Run (on board)
    cd ${esc(deploy.installDir)}
    bash ./setup_npu.sh
    ./${esc(deploy.runnerBin)} "<PROMPT>"

Expected on-board layout:

${esc(deploy.installDir)}/
├── ${esc(deploy.modelFile)}
├── ${esc(deploy.runnerBin)}
├── setup_npu.sh
├── firmwares/
├── kernel_modules/
└── scripts/
` : `
No prebuilt binaries published for ${esc(artifact || "—")} on this hardware config yet. Use Download repo instead.
`; /* Sibling-selector mode cards (no JS) */ const uid = `dl_${(modelKey || "x").replace(/\W+/g, "_")}_${(artifact || "x").replace(/\W+/g, "_")}_${fmt}_${mode}`; return ` ${showH3 ? `

Download

` : ""}
${esc(cmdRepo)}
${deployBlock}
`; } /* ====================================================================== */ /* MODEL DETAIL — full page (#sec-model) + modal density (openModelModal) */ /* Config bar = three axes: hardware config × precision (artifact) × NPU */ /* allocation (derived below, not modeled in the data). Qualification */ /* (HIL/SIL) is never a selector — both are always computed and shown */ /* side by side. Page and modal keep independent state and share every */ /* builder below (PAGE_STATE / MODAL_STATE). */ /* ====================================================================== */ let PAGE_STATE = null; // { key, cfgId, fmt, precision, allocationId, unit, tab } let MODAL_STATE = null; // { key, cfgId, fmt, precision, allocationId, unit } function fmtNumStr(v, d) { return v == null ? null : Number(v).toFixed(d); } function fmtAccStr(a) { if (a == null) return "—"; return a <= 1.5 ? `${Math.round(a * 1000) / 10}%` : `${Math.round(a * 10) / 10}`; } function pctOf(used, total) { return (used != null && total != null && total > 0) ? Math.max(2, Math.round((used / total) * 100)) : 0; } function allocKeyOf(u) { if (!u) return "default"; return `${u.cores_used ?? "x"}|${u.freq_used_mhz ?? "x"}|${u.tops_used ?? "x"}`; } /* Groups a precision's runs[] into NPU-allocation buckets — the config bar's third axis — by utilization.cores_used + freq_used_mhz + tops_used. E.g. MobileNetV2 int8's mwmx runs land at 1 core / 7 TOPS (1.05 ms, 0.704 ms) and at 12 cores / 84 TOPS (0.751 ms): two allocations, not two duplicate "mwmx" bars. Runs with no utilization block collapse into one "default" bucket instead of one per run. */ function allocationsFor(runs) { const groups = new Map(); for (const r of runs) { const u = r.utilization || null; const key = allocKeyOf(u); if (!groups.has(key)) { groups.set(key, { id: key, cores: u?.cores_used ?? null, freq: u?.freq_used_mhz ?? null, tops: u?.tops_used ?? null, insts: u?.instances_used ?? null, coresTotal: u?.cores_total ?? null, freqTotal: u?.freq_total_mhz ?? null, topsTotal: u?.tops_total ?? null, instsTotal: u?.instances_total ?? null, npuPct: u?.npu_offload_pct ?? null, cpuPct: u?.cpu_offload_pct ?? null, }); } } const list = Array.from(groups.values()); list.forEach(g => { // Only spell out the clock speed when two allocations otherwise share the // same core count + TOPS and would read as duplicates without it. const siblings = list.filter(o => o.cores === g.cores && o.tops === g.tops); const needFreq = siblings.length > 1 && new Set(siblings.map(o => o.freq)).size > 1; const parts = []; if (g.cores != null) parts.push(`${g.cores} AI core${g.cores === 1 ? "" : "s"}`); if (needFreq && g.freq != null) parts.push(`${g.freq} MHz`); if (g.tops != null) parts.push(`${g.tops} TOPS`); g.label = parts.join(" · ") || "Unspecified allocation"; }); list.sort((a, b) => (a.cores ?? 0) - (b.cores ?? 0) || (a.tops ?? 0) - (b.tops ?? 0) || (a.freq ?? 0) - (b.freq ?? 0)); return list; } /* Default allocation = the one the best HIL run (highest throughput, else lowest latency) was measured at — matches the existing DEFAULT_FMT / DEFAULT_COMPUTE convention of opening on data that actually exists. */ function pickDefaultAllocationId(allocations, hilRuns) { if (!allocations.length) return null; let best = null, bestScore = -Infinity; for (const r of hilRuns) { const thr = throughputValueOf(r); const lat = latencyMsOf(r); const score = thr != null ? thr : (lat != null ? -lat : null); if (score != null && score > bestScore) { bestScore = score; best = r; } } const ref = best || hilRuns[0]; const key = ref ? allocKeyOf(ref.utilization) : null; return (key && allocations.some(a => a.id === key)) ? key : allocations[0].id; } /* Best value per metric across a run list — one number per KPI card. */ function bestOf(runs, metricFn, higherBetter) { let best = null; for (const r of runs) { const v = metricFn(r); if (v == null) continue; if (!best || (higherBetter ? v > best.value : v < best.value)) best = { value: v, run: r }; } return best; } /* The bar-chart fix: one best run per engine+kind, from the HIL runs of the selected allocation plus *all* SIL runs of this (config, precision) — SIL keeps its own allocation, since the estimator may have run at a different one than the HIL measurement being compared against. */ function bestPerEngineKind(hilRuns, silRuns, metricFn, higherBetter) { const pool = hilRuns.map(r => ({ r, kind: "hil" })).concat(silRuns.map(r => ({ r, kind: "estimate" }))); const byKey = new Map(); for (const { r, kind } of pool) { const v = metricFn(r); if (v == null) continue; const key = `${r.engine || "—"}|${kind}`; const cur = byKey.get(key); if (!cur || (higherBetter ? v > cur.value : v < cur.value)) { byKey.set(key, { engine: r.engine || "—", kind, value: v, run: r }); } } const list = Array.from(byKey.values()); list.sort((a, b) => higherBetter ? b.value - a.value : a.value - b.value); return list; } /* Every benchmarked (precision, run) row for the runs table, each carrying the allocation label derived from *its own* precision's grouping (an allocation id is only meaningful within the precision it was grouped from). */ function allRunsRowsFor(m, cfg, fmt) { const rows = []; for (const art of benchmarkedArtifactsFor(m.key, fmt)) { const runs = cfg ? runsForArtifact(cfg.metrics || {}, art) : []; const allocs = allocationsFor(runs); for (const r of runs) { const allocId = allocKeyOf(r.utilization); const allocLabel = (allocs.find(a => a.id === allocId) || {}).label || "—"; rows.push({ art, r, allocId, allocLabel }); } } return rows; } /* Everything below the config bar recomputes from (cfgId, precision, allocationId, unit) — the single source of truth both the full page and the modal read from. Returns null only when the model itself is missing. */ function buildDetailView(key, cfgId, fmt, precision, allocationId, unit) { const m = KEY_TO_MODEL[key]; if (!m) return null; const hwConfigs = safeList(m.hardware_configs, []); const cfg = (cfgId && getCfg(m, cfgId)) || safeFirstCfg(m) || null; const arts = benchmarkedArtifactsFor(m.key, fmt); const art = (precision && arts.includes(precision)) ? precision : (arts[0] || null); const runs = (art && cfg) ? runsForArtifact(cfg.metrics || {}, art) : []; const allocations = allocationsFor(runs); const allHilRuns = runs.filter(r => r.kind === "hil"); const allocId = (allocationId && allocations.some(a => a.id === allocationId)) ? allocationId : pickDefaultAllocationId(allocations, allHilRuns); const alloc = allocations.find(a => a.id === allocId) || null; const hilRuns = alloc ? allHilRuns.filter(r => allocKeyOf(r.utilization) === alloc.id) : []; const silRuns = runs.filter(r => r.kind === "estimate"); // allocation-independent — always all of them const unitSafe = (unit === "lat") ? "lat" : "thr"; /* Dual-encoder models (SigLIP, CLIP-style) report per-tower timings (run.stages — e.g. Vision Encoder / Text Encoder) instead of a single throughput/latency figure, so the fixed throughput/latency/accuracy KPI triple below is always empty for them. Surface each stage as its own extra KPI card and, when no real throughput/latency bars exist, use the stages as the bar chart too — otherwise this view renders nothing at all for them even though the data is right there in every run. */ const stageNames = []; for (const r of runs) { for (const s of safeList(r.stages, [])) { if (s && s.label && !stageNames.includes(s.label)) stageNames.push(s.label); } } const stageMetricFn = (name) => (r) => { const s = safeList(r.stages, []).find(x => x && x.label === name); return (s && s.ms != null && isFinite(Number(s.ms))) ? Number(s.ms) : null; }; /* Single-pass, batch-1, fixed-shape CV graphs make throughput the reciprocal of latency — the same convention the constellation chart (CSTL_REGIMES.cnn.deriveFps) already applies, so a CV run that only published latency (e.g. the ppa-estimator SIL run) still gets a throughput figure instead of silently dropping off this view. Never applied to the decode regime, where latency and tok/s measure different things. */ const canDeriveFps = ["CV", "DIFFUSION"].includes(String(m.modality || "").toUpperCase()); const derivedFps = (r) => { const lat = latencyMsOf(r); return (lat != null && lat > 0) ? 1000 / lat : null; }; // Real reported throughput always outranks a derived one — derivation only // fills a gap when *no* run in the set reports throughput directly, never // overrides a run that did. function bestThrOf(runs) { const real = bestOf(runs, throughputValueOf, true); if (real || !canDeriveFps) return real; return bestOf(runs, derivedFps, true); } /* ---- KPI strip: throughput · latency/TTFT · accuracy · NPU compute used ---- */ const thrHil = bestThrOf(hilRuns); const thrSil = bestThrOf(silRuns); const thrUnit = canDeriveFps ? "img/s" : ((thrHil && throughputOf(thrHil.run).unit) || (thrSil && throughputOf(thrSil.run).unit) || (stageNames.length ? "ms" : "tok/s")); const thrLabel = kpiLabel(thrUnit === "img/s" ? "fps" : "tok_s", "Throughput"); const thrHilDerived = !!(thrHil && throughputValueOf(thrHil.run) == null); const thrSilDerived = !!(thrSil && throughputValueOf(thrSil.run) == null); // TTFT stands in for latency only when this model's own runs actually // publish it (LLM/VLM/VLA/ALM) — never hardcoded per modality. const hasTtft = runs.some(r => _firstUnitVal(r, "ttft_ms") != null); const secKey = hasTtft ? "ttft_ms" : "latency_ms_p50"; const secMetric = hasTtft ? (r => _firstUnitVal(r, "ttft_ms")) : latencyMsOf; const secHil = bestOf(hilRuns, secMetric, false); const secSil = bestOf(silRuns, secMetric, false); const secLabel = kpiLabel(secKey, hasTtft ? "TTFT" : "Latency p50"); const secUnit = kpiUnit(secKey, "ms"); const accHil = bestOf(hilRuns, accuracyFromBlock, true); const accSil = bestOf(silRuns, accuracyFromBlock, true); const accRef = cfg ? referenceAccuracy(m, cfg, fmt) : null; const isRefArt = !!(accRef && accRef.art === art); let accDeltaPts = null; if (!isRefArt && accHil && accRef && accHil.value <= 1.5 && accRef.acc <= 1.5) { accDeltaPts = (accHil.value - accRef.acc) * 100; } const stageKpis = stageNames.map(name => { const metricFn = stageMetricFn(name); const hilS = bestOf(hilRuns, metricFn, false); const silS = bestOf(silRuns, metricFn, false); return { key: `stage:${name}`, label: name, unit: "ms", showSil: true, hil: hilS != null ? fmtNumStr(hilS.value, 1) : null, sil: silS != null ? fmtNumStr(silS.value, 1) : null, empty: hilS == null, note: hilS ? `Best run · ${hilS.run.engine || "—"} · batch ${hilS.run.batch_size ?? 1} · lower is better` : `No ${name.toLowerCase()} timing published at this allocation`, }; }); const kpis = [ { key: "throughput", label: thrLabel, unit: thrUnit, showSil: true, hil: thrHil != null ? fmtNumStr(thrHil.value, 1) : null, sil: thrSil != null ? `${fmtNumStr(thrSil.value, 1)} ${thrUnit}${thrSilDerived ? "*" : ""}` : null, empty: thrHil == null, note: thrHil ? `Best run · ${thrHil.run.engine || "—"} · batch ${thrHil.run.batch_size ?? 1}${thrHilDerived ? " · derived from latency (1000 / p50)" : ""}` : `No ${thrLabel.toLowerCase()} published at this allocation`, }, { key: "second", label: secLabel, unit: secUnit, showSil: true, hil: secHil != null ? fmtNumStr(secHil.value, 2) : null, sil: secSil != null ? `${fmtNumStr(secSil.value, 2)} ${secUnit}` : null, empty: secHil == null, note: secHil ? `Best run · ${secHil.run.engine || "—"} · lower is better` : `No ${secLabel.toLowerCase()} published at this allocation`, }, { key: "accuracy", label: "Accuracy", unit: "", showSil: true, hil: accHil != null ? fmtAccStr(accHil.value) : null, sil: accSil != null ? fmtAccStr(accSil.value) : null, empty: accHil == null, warn: accDeltaPts != null && accDeltaPts < -2, note: isRefArt ? `${(accRef && accRef.art || "FP32").toUpperCase()} reference` : accDeltaPts == null ? (accHil == null ? "No accuracy run at this allocation" : "No FP32/FP16 reference published to compare against") : `${accDeltaPts < -0.05 ? "▼" : accDeltaPts > 0.05 ? "▲" : "±"} ${fmtNumStr(Math.abs(accDeltaPts), 1)} pts vs ${accRef.art.toUpperCase()} (${fmtAccStr(accRef.acc)}) · ${accHil.run.engine || "—"}`, }, // "NPU compute used" used to live here as its own KPI card, but it was // just the resourcesCardHTML "Compute" row (alloc.tops/topsTotal) plus // its cores/freq/instances restated in the note — the same allocation // data shown twice in one glance. Dropped; resourcesCardHTML is now the // single place hardware/compute utilization is shown. ].concat(stageKpis); /* ---- Bar chart: one best bar per engine + kind ---- Real throughput always wins within an engine+kind; a derived value only fills in for an engine+kind that reported no real throughput at all (never overrides one that did — see bestThrOf above). */ function bestThrPerEngineKind() { const real = bestPerEngineKind(hilRuns, silRuns, throughputValueOf, true); if (!canDeriveFps) return real; const covered = new Set(real.map(b => `${b.engine}|${b.kind}`)); const derivedOnly = bestPerEngineKind( hilRuns.filter(r => !covered.has(`${r.engine || "—"}|hil`)), silRuns.filter(r => !covered.has(`${r.engine || "—"}|estimate`)), derivedFps, true ).map(b => ({ ...b, derived: true })); return real.concat(derivedOnly).sort((a, b) => b.value - a.value); } let barList = unitSafe === "lat" ? bestPerEngineKind(hilRuns, silRuns, latencyMsOf, false) : bestThrPerEngineKind(); let chartUnit = unitSafe === "lat" ? "ms" : thrUnit; /* Kept to a single short clause — best-run/allocation/HIL·SIL/derived-value context is already visible per bar (kind label, "*" marker + its own hover tooltip below), so repeating it here was pure duplication. */ let chartNote = unitSafe === "lat" ? `${secLabel === "TTFT" ? "Latency" : secLabel} (ms) — lower is better.` : `${thrLabel} (${thrUnit}) — higher is better.`; // Neither throughput nor latency exists for a dual-encoder model (its bars // would otherwise always read "No throughput published") — chart one bar // per encoder tower instead, same per-stage data as the stageKpis cards. let usedStageBars = false; if (!barList.length && stageNames.length) { barList = stageNames.map(name => { const metricFn = stageMetricFn(name); const best = bestOf(hilRuns, metricFn, false) || bestOf(silRuns, metricFn, false); return best ? { engine: name, kind: best.run.kind === "hil" ? "hil" : "estimate", value: best.value, run: best.run } : null; }).filter(Boolean); chartUnit = "ms"; chartNote = "Per-stage NPU time (ms) — lower is better."; usedStageBars = true; } const maxV = barList.length ? Math.max(...barList.map(b => b.value)) : 1; const bars = barList.map(b => ({ ...b, valStr: (unitSafe === "lat" && !usedStageBars) ? fmtNumStr(b.value, 2) : fmtNumStr(b.value, 1), pct: Math.max(8, Math.round((b.value / (maxV || 1)) * 100)), // Present on a hybrid run's own utilization block — lets the bar itself // render as an NPU/CPU stack instead of one flat HIL/SIL color. npuPct: b.run?.utilization?.npu_offload_pct ?? null, cpuPct: b.run?.utilization?.cpu_offload_pct ?? null, })); /* ---- Hardware resources used, straight from the selected allocation ---- Compute (TOPS) is promoted to the card's headline number (see resourcesCardHTML) instead of sitting in this list as just another row — it's the one figure "NPU compute used" used to headline before that KPI card was folded into this one. */ const resources = alloc ? [ { l: "NPU instances", v: `${alloc.insts ?? "—"}${alloc.instsTotal != null ? ` / ${alloc.instsTotal}` : ""}`, pct: pctOf(alloc.insts, alloc.instsTotal) }, { l: "AI cores", v: `${alloc.cores ?? "—"}${alloc.coresTotal != null ? ` / ${alloc.coresTotal}` : ""}`, pct: pctOf(alloc.cores, alloc.coresTotal) }, { l: "Frequency", v: `${alloc.freq ?? "—"}${alloc.freqTotal != null ? ` / ${alloc.freqTotal} MHz` : ""}`, pct: pctOf(alloc.freq, alloc.freqTotal) }, ] : []; /* ---- Accuracy vs reference card ---- */ const accIsPct = accHil != null && accHil.value <= 1.5; const accCard = { valueStr: accHil != null ? fmtAccStr(accHil.value) : "—", pct: 0, deltaHTML: "", note: "" }; if (accHil != null) accCard.pct = Math.max(0, Math.min(100, accIsPct ? Math.round(accHil.value * 100) : Math.round(accHil.value))); if (accHil == null) { accCard.note = "No accuracy run at this allocation."; } else if (isRefArt) { accCard.note = `Reference precision${accRef ? ` (${accRef.art.toUpperCase()})` : ""} — quantized precisions are scored against this.`; } else if (accRef && accDeltaPts != null) { const cls = accDeltaPts < -0.05 ? "acc-down" : accDeltaPts > 0.05 ? "acc-up" : ""; const sign = accDeltaPts < -0.05 ? "▼" : accDeltaPts > 0.05 ? "▲" : "±"; accCard.deltaHTML = `
${sign} ${esc(fmtNumStr(Math.abs(accDeltaPts), 1))} pts vs ${esc(accRef.art.toUpperCase())}
`; accCard.note = `Best run · ${accHil.run.engine || "—"} · ${accRef.art.toUpperCase()} reference ${fmtAccStr(accRef.acc)}`; } else { accCard.note = `Best run · ${accHil.run.engine || "—"} · no FP32/FP16 reference published to compare against.`; } /* ---- Selector chips ---- */ const precisionChips = arts.map(a => { const aRuns = cfg ? runsForArtifact(cfg.metrics || {}, a) : []; const isRef = accRef && accRef.art === a; return { id: a, label: a.toUpperCase(), sub: isRef ? "reference" : `${aRuns.length || 1} run${(aRuns.length || 1) === 1 ? "" : "s"}`, selected: a === art }; }); const allocationChips = allocations.map(a => ({ id: a.id, label: a.label, selected: a.id === allocId, title: `${a.cores ?? "—"}${a.coresTotal != null ? `/${a.coresTotal}` : ""} cores · ${a.freq ?? "—"} MHz · ${a.tops ?? "—"}${a.topsTotal != null ? `/${a.topsTotal}` : ""} TOPS`, })); const hwChips = hwConfigs.map(c => ({ id: c.id, label: c.label || c.id, selected: !!(cfg && c.id === cfg.id) })); return { m, cfg, hwConfigs, fmt, art, arts, precisionChips, allocationChips, hwChips, allocations, alloc, allocId, unit: unitSafe, runs, hilRuns, silRuns, kpis, bars, chartUnit, chartNote, resources, accCard, }; } /* ---------- Shared markup builders (page + modal) ---------- */ function chipRowHTML(items, field, scope, extraCls = "") { if (!items.length) return ""; return items.map(it => ` `).join(""); } /* "NPU + CPU" (see defaultRowCompute) is a hybrid run, not the CPU fallback this ternary would otherwise mis-color it as — give it its own badge class. */ function computeBadgeCls(compute) { return compute === "NPU" ? "npu" : compute === "DSP" ? "dsp" : compute === "NPU + CPU" ? "hybrid" : "cpu"; } function heroChipsHTML(m, fmt, compute) { const chips = [`${esc(m.modality || "LLM")}`, `${esc(fmt)}`]; if (compute) chips.push(`${esc(compute)}`); if (m.license) chips.push(`${esc(String(m.license).toUpperCase())}`); return chips.join(""); } /* Modal header line under the model name: architecture text + the same modality/format/compute badges as the full page's hero (license omitted — the header is a title bar, not the summary). */ function modalHeaderMetaHTML(m, fmt, compute) { const parts = []; if (m.architecture) parts.push(`${esc(m.architecture)}`); parts.push(`${esc(m.modality || "LLM")}`); parts.push(`${esc(fmt)}`); if (compute) parts.push(`${esc(compute)}`); return parts.join(""); } function modelSpecPairs(m, cfg, fmt, compute, art) { const runs = (art && cfg) ? runsForArtifact(cfg.metrics || {}, art) : []; const inputRes = runs.map(r => r.input_resolution).find(Boolean) || null; return [ { k: "Architecture", v: m.architecture || "Unknown" }, { k: "Base model", v: m.base_model || "—" }, { k: "Task", v: safeList(m.tasks, []).map(prettyTask).join(", ") || "—" }, { k: "Modality", v: m.modality || "LLM" }, { k: "Model size", v: m.model_size_m != null ? `${m.model_size_m}M params` : "—" }, { k: "Input resolution", v: inputRes || "—" }, { k: "Variant", v: m.variant || "Default" }, { k: "Format / compute", v: `${fmt}${compute ? " · " + compute : ""}` }, { k: "License", v: m.license || "—" }, ]; } /* Each axis is its own full-width row (label + wrapping chip strip) rather than one shared flex line — a model with many precisions/allocations (e.g. EfficientNet's 9 size variants) wraps within its own row instead of shoving the next group onto a stray line or stranding Qualification in dead space. */ function cfgBarHTML(V, scope) { const cfgCount = allRunsRowsFor(V.m, V.cfg, V.fmt).length; return `
Configuration
${cfgCount} configuration${cfgCount === 1 ? "" : "s"} available
Hardware config
${V.hwChips.length ? chipRowHTML(V.hwChips, "cfgId", scope) : `No hardware config published`}
Precision
${V.precisionChips.length ? chipRowHTML(V.precisionChips, "precision", scope) : `No benchmarked precision`}
NPU allocation
${V.allocationChips.length ? chipRowHTML(V.allocationChips, "allocationId", scope) : `No allocation data`}
`; } /* Modal density's precision + NPU selectors — same row-per-axis pattern as cfgBarHTML, minus hardware config/qualification (not shown in the compact modal). */ function modalSelectorsHTML(V) { return `
Precision
${V.precisionChips.length ? chipRowHTML(V.precisionChips, "precision", "modal") : `No benchmarked precision`}
NPU
${V.allocationChips.length ? chipRowHTML(V.allocationChips, "allocationId", "modal") : `No allocation data`}
`; } function kpiCardHTML(k, compact) { return `
${esc(k.label)}
${!compact ? `
HIL · MEASURED
` : ""}
${k.hil != null ? esc(k.hil) : "—"}${k.unit ? ` ${esc(k.unit)}` : ""}
${k.showSil ? `
${!compact ? `
SIL · EST.
` : ""}
${k.sil != null ? esc(k.sil) : (compact ? "" : "—")}
` : ""}
${esc(k.note)}
`; } function kpiStripHTML(V, compact) { return `
${V.kpis.map(k => kpiCardHTML(k, compact)).join("")}
`; } /* A hybrid run's bar (npuPct + cpuPct both published) is a cumulative stack — the NPU share at the base and the CPU fallback share on top, together filling the same height a flat HIL/SIL bar would — instead of one flat color that hides how much of the graph actually fell back to CPU. */ function barFillHTML(b) { if (b.npuPct != null && b.cpuPct != null) { return `
`; } return `
`; } function barsHTML(V) { if (!V.bars.length) { return `
No ${V.unit === "lat" ? "latency" : "throughput"} published at this allocation.
`; } const cols = V.bars.map(b => `
${esc(b.valStr)}${b.derived ? "*" : ""}
${barFillHTML(b)}
`).join(""); const labels = V.bars.map(b => `
${esc(b.engine)}
${b.kind === "hil" ? "HIL · measured" : "SIL · estimated"}${b.derived ? " *" : ""}
`).join(""); const anySplit = V.bars.some(b => b.npuPct != null && b.cpuPct != null); const legend = anySplit ? `
NPU CPU
` : ""; return `
${cols}
${labels}
${legend}`; } function chartCardHTML(V, scope) { const units = [{ id: "thr", label: V.chartUnit }, { id: "lat", label: "ms" }]; const toggle = units.map(u => ``).join(""); return `
By runtime
${toggle}
${esc(V.chartNote)}
${barsHTML(V)}
`; } /* Headline number + note, styled like the old standalone "NPU compute used" KPI card (same .mdl-kpi-num treatment) — kept so folding that card into this one didn't also flatten its visual weight into just another list row. */ function resourcesHighlightHTML(alloc) { if (!alloc || alloc.tops == null) return ""; const note = [ alloc.cores != null ? `${esc(String(alloc.cores))}${alloc.coresTotal != null ? ` of ${esc(String(alloc.coresTotal))}` : ""} AI cores` : null, alloc.freq != null ? `${esc(String(alloc.freq))} MHz` : null, alloc.insts != null ? `${esc(String(alloc.insts))}${alloc.instsTotal != null ? `/${esc(String(alloc.instsTotal))}` : ""} NPU instances` : null, ].filter(Boolean).join(" · "); return `
${esc(String(alloc.tops))} ${alloc.topsTotal != null ? `of ${esc(String(alloc.topsTotal))} TOPS` : "TOPS"}
${note ? `
${note}
` : ""}
`; } /* NPU/CPU offload split — only rendered when the benchmark YAML actually published it (configuration.npu_offload_pct / cpu_offload_pct on a hybrid run); a pure-NPU run carries neither field. Given its own two-tone bar (NPU/CPU badge colors) rather than the generic single-value track used below, since "what fraction ran where" is a different kind of fact than "how much of the available resource did this run use". */ function resourcesSplitHTML(alloc) { if (!alloc || (alloc.npuPct == null && alloc.cpuPct == null)) return ""; return `
NPU / CPU split ${alloc.npuPct != null ? `${esc(String(alloc.npuPct))}%` : "—"} NPU ${alloc.cpuPct != null ? `${esc(String(alloc.cpuPct))}%` : "—"} CPU
`; } function resourcesCardHTML(V) { const alloc = V.alloc; if (!alloc) return `
Hardware resources used
No allocation selected.
`; return `
Hardware resources used
${resourcesHighlightHTML(alloc)} ${resourcesSplitHTML(alloc)}
${V.resources.map(r => `
${esc(r.l)}${esc(r.v)}
`).join("")}
`; } function accuracyCardHTML(V) { const c = V.accCard; return `
Accuracy vs reference
${esc(c.valueStr)}
${c.deltaHTML}
${esc(c.note)}
`; } function perfRowHTML(V, scope, compact, extraSideHTML = "") { if (compact) { return `
${chartCardHTML(V, scope)}
${resourcesCardHTML(V)}${extraSideHTML}
`; } return `
${chartCardHTML(V, scope)}${resourcesCardHTML(V)}${accuracyCardHTML(V)}
`; } /* Compact "route to the full page" links — sits beside the resources card in the modal's right column (not a separate full-width row: at 248px density there's no room to spare below a two-column area). No direct "Open repo" link here — the repo is always reachable via the Download tab (repo link + hf download command), so a second shortcut was redundant. */ function modalQuickLinksHTML(state, V, runsCount) { const rows = [ { tab: "runs", label: `All ${runsCount} benchmark runs →` }, // No model file for the selected precision -> nothing to download yet, so // the link to the full page's Download & run tab is dropped rather than // sending the user to a tab whose command can't actually be run. ...(artifactFileMissing(V.m, V.fmt, V.art) ? [] : [{ tab: "dl", label: "Download & run instructions →" }]), ].map(l => `${esc(l.label)}`).join(""); return ``; } function runsTableHTML(V) { const rows = allRunsRowsFor(V.m, V.cfg, V.fmt); if (!rows.length) return `
No benchmark runs published for this hardware config yet.
`; const body = rows.map(({ art, r, allocId, allocLabel }) => { const sel = art === V.art && allocId === V.allocId; const thr = throughputOf(r); const lat = latencyMsOf(r); const acc = accuracyFromBlock(r); const qual = r.kind === "hil" ? `HIL` : `SIL`; return ` ${esc(art)} ${esc(allocLabel)} ${esc(r.engine || "—")} ${qual} ${thr.value != null ? esc(fmtNumStr(thr.value, 1)) + " " + esc(thr.unit) : "—"} ${lat != null ? esc(fmtNumStr(lat, 2)) + " ms" : "—"} ${esc(fmtAccStr(acc))} ${esc(r.last_updated || "—")} `; }).join(""); return `
All benchmark runs · ${rows.length} published
Rows matching the current selection are highlighted
${body}
PrecisionNPU allocationRuntimeQual.ThroughputLatencyAccuracyUpdated
HIL = measured on hardware. SIL = estimated by the PPA estimator. Blank cells mean the metric was not reported by that run — never zero.
`; } /* MWMX AI Compiler installer/package live behind Renesas's customer portal, not on a public URL the doc reveals — link straight to the gated portal page rather than the generic myRenesas landing page. Only relevant for the MWMX/NNAC (ONNX) toolchain; GGUF's llama.cpp-style runner needs no separate compiler download, so callers gate this on the run's engine. */ const MYRENESAS_AI_COMPILER_URL = "https://www.renesas.com/en/myrenesas/secure-portals/gen5-r-car-x5x-sw-ai"; function compilerLinkHTML(run) { if (String(run?.engine || "").toLowerCase() !== "mwmx") return ""; return `
Get the MWMX AI Compiler (installer + NNAC portable package, myRenesas account required) from myRenesas › Secure Software Portal.
`; } /* A benchmark YAML can carry an explicit `reproduce:` block — exact commands its author verified on real hardware (see generate_models_json.py parse_reproduce()) — because the real deployment flow (single vs. multi NPU cluster, an ORT-split subgraph, a GGUF runner binary, ...) varies per model and isn't safe to guess. When the selected run has one, it replaces the generic per-format placeholder flow below entirely; it keeps the same two-column layout (toolchain meta + commands) so the two paths look like one feature, not two different UIs. */ function reproduceStepsHTML(run, cfg, toolchain) { const rp = run.reproduce; const steps = safeList(rp.steps, []); const body = steps.map((s, i) => { const isNote = s.kind === "note"; return `
${i + 1} · ${esc(s.title || "Run")}
${isNote ? `
${esc(s.command)}
` : `
${esc(s.command)}
`} ${s.expected ? `
Expected: ${esc(s.expected)}
` : ""}
`; }).join(""); return `
Toolchain requirements
${toolchain.map(t => `
${esc(t.k)}${esc(t.v)}
`).join("")}
${compilerLinkHTML(run)} ${rp.reference ? `
Verified against ${esc(rp.reference)}.
` : ""}
Reproduce · ${esc(cfg?.label || cfg?.id || "")}
${body}
${rp.notes ? `
${esc(rp.notes)}
` : ""}
`; } /* Driven by the selected precision. ONNX gets the README's own-repo layout (download cmd + onnxruntime/RcarNpuExecutionProvider snippet + board setup + toolchain + a note on what the catalog can't show yet); GGUF reuses the existing deploy-with-binaries renderActions() flow. Either is overridden by an explicit reproduce: block when the selected run has one. */ function downloadRunHTML(V) { const { m, cfg, fmt, art } = V; const bestRun = V.hilRuns[0] || V.silRuns[0] || V.runs[0] || null; const inputRes = V.runs.map(r => r.input_resolution).find(Boolean); const toolchain = [ { k: "Hardware config", v: cfg?.label || cfg?.id || "—" }, { k: "Runtime engine", v: bestRun?.engine || "—" }, { k: "Toolchain version", v: bestRun?.toolchain_version || "—" }, { k: "Execution provider", v: bestRun?.execution_provider ? String(bestRun.execution_provider).toUpperCase() : "NPU" }, { k: "Batch size", v: bestRun?.batch_size != null ? String(bestRun.batch_size) : "—" }, ]; if (inputRes) toolchain.push({ k: "Input resolution", v: inputRes }); if (bestRun?.reproduce?.steps?.length) { return reproduceStepsHTML(bestRun, cfg, toolchain); } if (fmt !== "ONNX") { return renderActions(m.key, fmt, defaultRowCompute(m, cfg), defaultPurposeForFormat(fmt), art, cfg?.id, "tab"); } const repoId = m.onnx_repo || ""; if (!repoId) return `
No ONNX repo published for this model yet.
`; if (!art) return `
No benchmarked precision to build a download command from yet.
`; const include = downloadIncludePattern(fmt, art); const cmdRepo = `hf download ${repoId} --repo-type=model --include "${include}"`; const isCV = ["CV", "DIFFUSION"].includes(String(m.modality || "").toUpperCase()); const dims = (inputRes && /^\d+x\d+$/i.test(inputRes)) ? inputRes.split(/x/i).join(", ") : "224, 224"; const py = isCV ? `import onnxruntime as ort, numpy as np\n` + `sess = ort.InferenceSession("${art}/.onnx", providers=["RcarNpuExecutionProvider"])\n` + `x = np.zeros((1, 3, ${dims}), dtype=np.float32)\n` + `print(sess.run(None, {sess.get_inputs()[0].name: x})[0].argmax())` : `import onnxruntime as ort, numpy as np\n` + `sess = ort.InferenceSession("${art}/.onnx", providers=["RcarNpuExecutionProvider"])\n` + `input_ids = np.array([[1]], dtype=np.int64) # replace with a real tokenized prompt\n` + `print(sess.run(None, {sess.get_inputs()[0].name: input_ids}))`; const boardSteps = [ { n: "1", t: "Copy the artifact to the board", c: `scp -r ${art}/ root@x5h:/opt/models/${m.key}/` }, { n: "2", t: "Bring up the NPU", c: `cd /opt/models/${m.key} && bash ./setup_npu.sh` }, { n: "3", t: "Benchmark it yourself", c: `python3 run_infer.py --model ${art}/.onnx --provider RcarNpuExecutionProvider` }, ]; return `
Toolchain requirements
${toolchain.map(t => `
${esc(t.k)}${esc(t.v)}
`).join("")}
${compilerLinkHTML(bestRun)}
Artifact file sizes and checksums aren't captured by generate_models_json.py yet — it only reads filenames from the repo's file tree. Browse the exact files in the ONNX repo.
1 · Download the ${esc(art.toUpperCase())} artifact
${esc(cmdRepo)}
2 · Run inference on the board
${esc(py)}
Replace <artifact_file>.onnx with the exact filename from the repo — the catalog generator doesn't capture individual filenames yet.
Board setup · ${esc(cfg?.label || cfg?.id || "R-Car X5H")}
    ${boardSteps.map(s => `
  1. ${s.n} ${esc(s.t)}
    ${esc(s.c)}
  2. `).join("")}
`; } /* The page's own topbar already carries the Renesas wordmark + primary nav (Overview/Catalog) — repeating it here would just be a second header. The name and modality are both restated a breath away (hero title, then chips), so a "Catalog / CV / " trail here would be the third repetition on screen; a bare back arrow says "where am I" just as well and keeps this row tight. */ /* Breadcrumb row (navigation) + title row (name, meta, chips) merged into one compact head block — kept as two separate bands before, which duplicated padding/borders and pushed the actual performance content far down the page for no reason (breadcrumb and title never need to scroll independently of each other). */ function pageHeadHTML(m, fmt, compute) { const repoId = fmt === "GGUF" ? (m.gguf_repo || "") : (m.onnx_repo || ""); const metaBits = [m.architecture, safeList(m.tasks, [])[0] ? prettyTask(m.tasks[0]) : null, m.model_size_m != null ? `${m.model_size_m}M params` : null].filter(Boolean); return `
${esc(m.display_name || m.key)}
Benchmarks updated ${esc(m.last_modified ? String(m.last_modified).slice(0, 10) : "—")}
${repoId ? `Open ${esc(fmt)} repo ↗` : ""}
${esc(metaBits.join(" · "))} ${heroChipsHTML(m, fmt, compute)}
`; } function summaryRowHTML(m, cfg, fmt, compute, art) { const specs = modelSpecPairs(m, cfg, fmt, compute, art); return `
Model summary
${specs.map(s => `
${esc(s.k)}
${esc(s.v)}
`).join("")}
${exampleImageHTML(m)}
`; } function footerHTML() { return ` `; } /* ---------- Full page (#sec-model) ---------- */ function renderFullModelPageHTML(state) { const m = KEY_TO_MODEL[state.key]; if (!m) return `
Model not found.
`; if (isComingSoon(m)) { return `${pageHeadHTML(m, state.fmt, defaultRowCompute(m, null))}${comingSoonDetailsHTML(m)}${footerHTML()}`; } const V = buildDetailView(state.key, state.cfgId, state.fmt, state.precision, state.allocationId, state.unit); state.cfgId = V.cfg?.id || state.cfgId || ""; state.precision = V.art; state.allocationId = V.allocId; state.unit = V.unit; const compute = defaultRowCompute(V.m, V.cfg); const uid = `mdlp_${String(state.key).replace(/\W+/g, "_")}`; // No model file for the selected precision -> the Download & run tab has // nothing runnable to show, so it's dropped rather than left open on a // command that would fail. Falls back to the runs tab if that's where the // (now-hidden) dl tab was left selected. const fileMissing = artifactFileMissing(m, state.fmt, V.art); const tab = (!fileMissing && state.tab === "dl") ? "dl" : "runs"; return ` ${pageHeadHTML(m, state.fmt, compute)} ${missingFileBannerHTML(m, state.fmt, V.art)} ${summaryRowHTML(m, V.cfg, state.fmt, compute, V.art)} ${cfgBarHTML(V, "page")}
Performance · ${esc(state.fmt)} · ${esc(compute)} · ${esc((V.art || "—").toUpperCase())}${V.alloc ? " · " + esc(V.alloc.label) : ""}
Every figure below reflects the selection above
${kpiStripHTML(V, false)} ${perfRowHTML(V, "page", false)} ${fileMissing ? "" : ``} ${fileMissing ? "" : `
`}
${runsTableHTML(V)}
${fileMissing ? "" : `
${downloadRunHTML(V)}
`} ${footerHTML()}`; } function renderPageNow() { if (!PAGE_STATE) return; const container = $("modelPage"); if (!container) return; if (container.querySelector(".det-radio--dl:checked")) PAGE_STATE.tab = "dl"; else if (container.querySelector(".det-radio--runs:checked")) PAGE_STATE.tab = "runs"; container.innerHTML = renderFullModelPageHTML(PAGE_STATE); } function openModelPage(modelKey, cfgId, fmt, opts = {}) { const m = KEY_TO_MODEL[modelKey]; if (!m) return; const cfg = cfgId ? getCfg(m, cfgId) : safeFirstCfg(m); PAGE_STATE = { key: modelKey, cfgId: cfg?.id || cfgId || "", fmt: fmt || defaultRowFmt(m), precision: null, allocationId: null, unit: null, tab: opts.tab === "dl" ? "dl" : "runs", }; const container = $("modelPage"); if (container) container.innerHTML = renderFullModelPageHTML(PAGE_STATE); setActiveSection("sec-model"); const hash = `#/model/${encodeURIComponent(modelKey)}`; if (location.hash !== hash) history.replaceState ? history.replaceState(null, "", hash) : (location.hash = hash); window.scrollTo(0, 0); } function routeFromHash() { const match = /^#\/model\/([^/?#]+)/.exec(location.hash || ""); if (!match) return; const key = decodeURIComponent(match[1]); if (KEY_TO_MODEL[key]) openModelPage(key, "", defaultRowFmt(KEY_TO_MODEL[key])); } /* ---------- Modal density ---------- */ function renderModalBodyHTML(state) { const m = KEY_TO_MODEL[state.key]; if (!m) return `
Model not found.
`; if (isComingSoon(m)) return comingSoonDetailsHTML(m, true); const V = buildDetailView(state.key, state.cfgId, state.fmt, state.precision, state.allocationId, state.unit); state.cfgId = V.cfg?.id || state.cfgId || ""; state.precision = V.art; state.allocationId = V.allocId; state.unit = V.unit; const compute = defaultRowCompute(V.m, V.cfg); const specs = modelSpecPairs(m, V.cfg, state.fmt, compute, V.art); const runsCount = allRunsRowsFor(m, V.cfg, state.fmt).length; return `
${exampleImageHTML(m)}
${specs.map(s => `
${esc(s.k)}${esc(s.v)}
`).join("")}
${modalSelectorsHTML(V)} ${missingFileBannerHTML(m, state.fmt, V.art)} ${kpiStripHTML(V, true)} ${perfRowHTML(V, "modal", true, modalQuickLinksHTML(state, V, runsCount))}
`; } function renderModalNow() { if (!MODAL_STATE) return; const el = $("modalDetails"); if (el) el.innerHTML = renderModalBodyHTML(MODAL_STATE); } /* ---------- Navigation ---------- */ function setActiveSection(targetId) { // Leaving the model page: drop its #/model/ hash so the address bar // matches what's on screen instead of staying pinned to the last model // (routeFromHash would otherwise reopen it on the next reload/back nav). if (targetId !== "sec-model" && /^#\/model\//.test(location.hash || "")) { const url = location.pathname + location.search; history.replaceState ? history.replaceState(null, "", url) : (location.hash = ""); } $$(".section").forEach(sec => sec.classList.remove("visible")); document.getElementById(targetId)?.classList.add("visible"); $$(".nav-item").forEach(btn => btn.classList.remove("active")); document.querySelector(`.nav-item[data-target="${CSS.escape(targetId)}"]`)?.classList.add("active"); if (targetId === "sec-overview") requestAnimationFrame(redrawConstellation); if (targetId === "sec-catalog" && CATALOG_STATE.view === "table") { requestAnimationFrame(() => setupPinnedColumns($("modelsTable"), 1)); } } /* ---------- All Models (table) ---------- */ function defaultRowFmt(model) { return model.onnx_repo ? "ONNX" : (model.gguf_repo ? "GGUF" : "ONNX"); } function defaultRowCompute(model, cfg) { const targets = safeList(cfg?.targets, safeList(model.targets, ["CPU"])); // A hybrid run genuinely executes on both — label it as such rather than // collapsing to whichever of NPU/CPU happens to win the single-badge priority // below, which would hide that the other unit is doing real work too. if (targets.includes("NPU") && targets.includes("CPU")) return "NPU + CPU"; return targets.includes("CPU") ? "CPU" : (targets.includes("NPU") ? "NPU" : "DSP"); } function defaultRowArtifact(model, fmt) { const arts = (fmt === "ONNX") ? safeList(model.onnx_artifacts, []) : safeList(model.gguf_artifacts, []); return arts[0] || null; } function buildTableRows(models) { const rows = []; for (const m of models) { const cfg = safeFirstCfg(m) || {}; const fmt = defaultRowFmt(m); const defArt = defaultRowArtifact(m, fmt); const { hil } = metricsForArtifact(cfg.metrics || {}, defArt); const detailsBtn = ``; rows.push({ _key: m.key || "", _cfg: cfg.id || "", _fmt: fmt, "Name": `${esc(m.display_name || m.key || "")}${isComingSoon(m) ? `Coming Soon` : modelHasMissingFile(m) ? `⏳ File pending` : ""}`, "Modality": esc(m.modality || "LLM"), "Task": esc(safeList(m.tasks, [])[0] || "—"), "Repo": repoButtonsSmallHTML(m), "Precision/Quant": artifactsBadgesHTML(m), "HIL": metricBriefHTML(hil), "": detailsBtn }); } return rows; } function renderTable(rows) { const table = $("modelsTable"); if (!table) return; const thead = table.querySelector("thead"); const tbody = table.querySelector("tbody"); if (!rows.length) { thead.innerHTML = ""; tbody.innerHTML = `No rows.`; return; } const cols = Object.keys(rows[0]).filter(k => !k.startsWith("_")); const pinCount = 1; // pin only the identifier (Name) column thead.innerHTML = `${cols.map((c, i) => { const isPinned = i < pinCount; const isDetails = (c === ""); const isName = (c === "Name"); const cls = `${isPinned ? "pin" : ""}${isDetails ? " details-col" : ""}${isName ? " cell-name" : ""}`.trim(); const pinAttr = isPinned ? ` data-pin="${i}"` : ""; const header = isDetails ? "" : esc(c); return `${header}`; }).join("")}`; const htmlCols = new Set(["Name", "Repo", "Precision/Quant", "HIL", ""]); tbody.innerHTML = rows.map((r, ri) => { return `${ cols.map((c, i) => { const isPinned = i < pinCount; const isDetails = (c === ""); const isName = (c === "Name"); const cls = `${isPinned ? "pin" : ""}${isDetails ? " details-col" : ""}${isName ? " cell-name" : ""}`.trim(); const pinAttr = isPinned ? ` data-pin="${i}"` : ""; // data-label drives the stacked-card layout's field labels on narrow screens const labelAttr = ` data-label="${esc(c)}"`; const val = r[c]; if (htmlCols.has(c)) return `${val ?? ""}`; return `${esc(val)}`; }).join("") }`; }).join("\n"); setupPinnedColumns(table, pinCount); } function setupPinnedColumns(table, count) { if (!table) return; const headCells = table.querySelectorAll(`thead th.pin`); if (!headCells.length) return; const lefts = []; let acc = 0; for (let i = 0; i < count; i++) { const cell = table.querySelector(`thead th.pin[data-pin="${i}"]`); if (!cell) break; lefts[i] = acc; acc += cell.getBoundingClientRect().width; } for (let i = 0; i < lefts.length; i++) { table.querySelectorAll(`.pin[data-pin="${i}"]`).forEach(el => { el.style.left = `${lefts[i]}px`; }); } } /* ---------- Model modal ---------- */ function openModelModal(modelKey, cfgId, fmt) { const m = KEY_TO_MODEL[modelKey]; if (!m) return; const modal = $("modelModal"); modal.classList.add("open"); modal.setAttribute("aria-hidden", "false"); const cfg = cfgId ? getCfg(m, cfgId) : safeFirstCfg(m); MODAL_STATE = { key: modelKey, cfgId: cfg?.id || cfgId || "", fmt, precision: null, allocationId: null, unit: null }; $("modalTitle").textContent = m.display_name || m.key; const sub = $("modalSubtitle"); if (sub) sub.innerHTML = modalHeaderMetaHTML(m, fmt, defaultRowCompute(m, cfg)); renderModalNow(); } function closeModelModal() { const modal = $("modelModal"); modal.classList.remove("open"); modal.setAttribute("aria-hidden", "true"); const el = $("modalDetails"); if (el) el.innerHTML = ""; MODAL_STATE = null; } /* ====================================================================== */ /* OVERVIEW — constellation (size vs throughput) */ /* ====================================================================== */ const FAMILY_PALETTE = ["#C96442", "#6E8B6A", "#C9A24B", "#5B7A99", "#A6573F", "#8E6E9E", "#4C8C7D", "#B07A3C", "#9A6B6B", "#7C8B5A"]; const COLOR_CACHE = {}; function colorFor(label) { if (COLOR_CACHE[label] != null) return COLOR_CACHE[label]; const c = FAMILY_PALETTE[Object.keys(COLOR_CACHE).length % FAMILY_PALETTE.length]; COLOR_CACHE[label] = c; return c; } function allArtifactsOf(m) { return uniqueSorted([...safeList(m.onnx_artifacts, []), ...safeList(m.gguf_artifacts, [])]); } /* A repo with no artifact folders at all has nothing published yet — treated as a "coming soon" teaser rather than a model with missing benchmarks. This flips automatically the day real artifacts are pushed, with no status flag to remember to clear. */ function isComingSoon(m) { return allArtifactsOf(m).length === 0; } /* File-vs-benchmark decorrelation — a benchmark YAML lands the moment a run completes, which is routinely before the (much larger) weight file itself is uploaded. `_file_status[artifact]` (generate_models_json.py, artifact_has_payload()) is `false` only when that specific artifact has no real payload; absent/undefined means the model predates this field, so it is treated as available rather than flagged. Kept per-format because the same precision name can be a real file in one repo and not the other (Llama-3.1-8B-Instruct: GGUF w4a16 ships coefficients, ONNX w4a16 doesn't). */ function fileStatusFor(m, fmt) { return (fmt === "ONNX" ? m?.onnx_file_status : m?.gguf_file_status) || {}; } function artifactFileMissing(m, fmt, art) { return !!art && fileStatusFor(m, fmt)[art] === false; } function modelHasMissingFile(m) { const vals = [...Object.values(m?.onnx_file_status || {}), ...Object.values(m?.gguf_file_status || {})]; return vals.some(v => v === false); } /* True if at least one artifact (either format) actually has a downloadable file — i.e. the model isn't "coming soon" (no artifacts at all) and isn't stuck with every artifact's file still unpublished (see fileStatusFor). */ function modelHasAvailableFile(m) { if (isComingSoon(m)) return false; const onnxOk = safeList(m.onnx_artifacts, []).some(a => !artifactFileMissing(m, "ONNX", a)); const ggufOk = safeList(m.gguf_artifacts, []).some(a => !artifactFileMissing(m, "GGUF", a)); return onnxOk || ggufOk; } /* Shared banner for the modal and the full model page — same warning style as the "Coming soon" notice, scoped to just the currently-selected precision instead of the whole model. */ function missingFileBannerHTML(m, fmt, art) { if (!artifactFileMissing(m, fmt, art)) return ""; return `
⏳ ${esc(String(art).toUpperCase())} model file not yet uploaded. The benchmark numbers below for this precision were published ahead of the model weights — download will not work until the file is added to the repo.
`; } function fmtOfArtifact(m, art) { if (safeList(m.gguf_artifacts, []).includes(art)) return "GGUF"; if (safeList(m.onnx_artifacts, []).includes(art)) return "ONNX"; return "—"; } function pickBlock(cfgMetrics, art, source) { const { estimate, hil } = metricsForArtifact(cfgMetrics || {}, art); if (source === "HIL") return hil ? { block: hil, source: "HIL" } : null; if (source === "Estimate") return estimate ? { block: estimate, source: "Estimate" } : null; if (hil) return { block: hil, source: "HIL" }; if (estimate) return { block: estimate, source: "Estimate" }; return null; } function tokSOf(block) { if (!block) return null; let t = (block.total && block.total.tok_s != null) ? block.total.tok_s : null; if (t == null) t = block.npu?.tok_s ?? block.cpu?.tok_s ?? block.dsp?.tok_s ?? null; return (t != null && isFinite(Number(t))) ? Number(t) : null; } /* ---------------------------------------------------------------------- */ /* Measurement regimes = the constellation tabs */ /* ---------------------------------------------------------------------- The catalog deliberately mixes model *types* (CNN, dual encoder, transformer decoder, LLM/VLM/ALM/VLA), and each type answers a different question with a different unit. Plotting them on one pair of axes is not just crowded, it is wrong: 952 img/s (MobileNetV2, one full image per inference) and 41 tok/s (Llama-3.2-1B, one *token* per inference step) are not the same quantity, and putting them on a shared linear axis both implies a comparison that doesn't exist and squeezes every decoder into the left margin. So the chart is split by what "one inference" means, which is what fixes the unit — and therefore the axis and even the chart form: gen one generated token -> tok/s -> scatter cnn one full image, fixed graph -> img/s, ms -> scatter enc one pass per encoder tower -> ms per stage -> bar chart soon undefined (no KPI/artifacts) -> none -> cards A model lands in a regime by modality, with a data-shape fallback for modalities this table doesn't know yet, so a new modality shows up in a sensible tab instead of vanishing. */ const CSTL_REGIMES = [ { id: "gen", label: "Token generators", unit: "tok/s", chart: "scatter", modalities: ["LLM", "VLM", "ALM"], why: `One inference = one generated token. Decoders share the same memory-bound decode loop, so tok/s and size are directly comparable here.`, x: ["tok", "size", "prefill", "ttft"], y: ["size", "tok", "ttft", "mem"], }, { id: "cnn", label: "Vision CNNs", unit: "img/s · ms", chart: "scatter", modalities: ["CV"], why: `One inference = one fixed-shape image. Compute-bound and quantized, so img/s and latency sit orders of magnitude above any token rate — hence the separate scale.`, x: ["fps", "lat", "size"], y: ["size", "lat", "fps"], /* Single-pass, batch-1, fixed-shape graph → throughput is the reciprocal of latency, and the CV benchmark files state that convention themselves ("fps: null # throughput: 1000 / latency" in RetinaNet's int8 run). It also holds in the data: MobileNetV2 952.4 vs 1000/1.05, EfficientNetV2-B0 480 vs 1000/2.08, ResNet50 303 vs 1000/3.23 — within ~2%. So a CV variant that timed a run but left fps null still gets a point, drawn dashed and labelled as derived. Never enabled for the decode regime, where latency and tok/s measure different things. */ deriveFps: true, }, { id: "enc", label: "Encoders & embeddings", unit: "ms / stage", chart: "stages", modalities: ["EMBED"], why: `No inference loop — two towers, different rates. A blended throughput would be fiction, so each tower's stage latency is charted separately.`, }, { id: "soon", label: "Roadmap", unit: "no KPI yet", chart: "cards", modalities: ["VLA"], why: `Nothing measurable published yet. No KPI pipeline or artifacts yet — listed as cards so they stay visible.`, }, ]; const CSTL_REGIME_BY_ID = Object.fromEntries(CSTL_REGIMES.map(r => [r.id, r])); /* Every benchmark block a model publishes, across configs/artifacts/kinds. */ function allBlocksOf(m) { const out = []; for (const cfg of safeList(m.hardware_configs, [])) { for (const art of allArtifactsOf(m)) { const { estimate, hil } = metricsForArtifact(cfg.metrics || {}, art); if (estimate) out.push(estimate); if (hil) out.push(hil); } } return out; } function publishesMetric(m, key) { return allBlocksOf(m).some(b => _firstUnitVal(b, key) != null); } /* Which tab a model belongs to. Modality first (it is the declared model type), then the shape of the published numbers — so a modality not listed above, or a VLA that one day reports a token rate, still lands somewhere sensible. */ function regimeOf(m) { if (isComingSoon(m)) return "soon"; const mod = String(m.modality || "").trim().toUpperCase(); if (mod === "VLA") return publishesMetric(m, "tok_s") ? "gen" : "soon"; const byMod = CSTL_REGIMES.find(r => safeList(r.modalities, []).includes(mod)); if (byMod) return byMod.id; if (publishesMetric(m, "tok_s")) return "gen"; if (publishesMetric(m, "fps")) return "cnn"; if (allBlocksOf(m).some(b => safeList(b.stages, []).length)) return "enc"; return "soon"; } function fmtAxisNum(v) { return Math.abs(v) >= 100 ? String(Math.round(v)) : String(Math.round(v * 10) / 10); } /* Measured values keep their tenth of a millisecond (396.8 ms, not 397 ms) — axis ticks are the only place rounding to whole units is fine. */ function fmtMs(v) { return v == null ? "—" : String(Math.round(Number(v) * 10) / 10); } /* Axis registry. `get` returns null when a variant never published that metric — the caller reports those as excluded rather than dropping them silently. */ const AXES = { tok: { get: p => p.tok, label: "Decode throughput (tok/s)", tick: fmtAxisNum }, prefill: { get: p => p.prefill, label: "Prefill rate (tok/s)", tick: fmtAxisNum }, ttft: { get: p => p.ttft, label: "Time to first token (ms)", tick: fmtAxisNum, lowerBetter: true }, fps: { get: p => p.fps, label: "Throughput (img/s)", tick: fmtAxisNum }, lat: { get: p => p.lat, label: "Latency p50 (ms)", tick: fmtAxisNum, lowerBetter: true }, mem: { get: p => p.memBw, label: "Memory bandwidth (GB/s)", tick: fmtAxisNum }, size: { get: p => p.sizeM, label: "Model size (params)", tick: v => (v >= 1000 ? `${Math.round(v / 100) / 10}B` : `${Math.round(v)}M`) }, }; /* One point per (model × hardware config × artifact) inside one regime, carrying every metric the block reports so each regime can pick its own axes. */ function regimePoints(regimeId, source) { const pts = []; const regime = CSTL_REGIME_BY_ID[regimeId]; for (const m of CATALOG) { if (regimeOf(m) !== regimeId) continue; for (const cfg of safeList(m.hardware_configs, [])) { for (const art of allArtifactsOf(m)) { const got = pickBlock(cfg.metrics || {}, art, source); if (!got) continue; const b = got.block; /* Size is per point, not per model: an artifact that declares its own parameter count in /.metadata.yaml wins over the repo-level figure. Falls back to the model value, which is the common case. */ const sizeM = artifactSizeM(m, art); const lat = latencyMsOf(b); let fps = _firstUnitVal(b, "fps"); let fpsDerived = false; if (fps == null && regime?.deriveFps && lat != null && lat > 0) { fps = 1000 / lat; fpsDerived = true; } pts.push({ key: m.key, cfgId: cfg.id, fmt: fmtOfArtifact(m, art), name: m.display_name || m.key, arch: m.architecture || m.display_name || m.key, family: familyOf(m), quant: art, precision: artifactPrecision(m, art), target: safeList(cfg.targets, safeList(m.targets, []))[0] || "—", hwLabel: cfg.label || cfg.id || "—", sizeM, tok: _firstUnitVal(b, "tok_s"), fps, fpsDerived, lat, prefill: _firstUnitVal(b, "prefill_tok_s"), ttft: _firstUnitVal(b, "ttft_ms"), memBw: memBwGBs(_firstUnitVal(b, "peak_mem_mb")), acc: b.accuracy != null ? Number(b.accuracy) : null, stages: safeList(b.stages, []).filter(s => s && s.ms != null), source: got.source, block: b, }); } } } return pts; } function colorKey(p, colorBy) { return colorBy === "quant" ? p.precision : colorBy === "format" ? p.fmt : colorBy === "target" ? p.target : colorBy === "family" ? p.family : p.arch; } /* Parameter count for one artifact of one model, in millions. `artifact_info[].size_m` comes from that artifact's own .metadata.yaml and overrides the repo-level `model_size_m` — the generator collected those files but never read them, so per-artifact sizes used to be invisible here. */ function artifactSizeM(m, art) { const info = (m.artifact_info || {})[art]; if (info && info.size_m != null) return Number(info.size_m); return (m.model_size_m != null) ? Number(m.model_size_m) : null; } /* Some .metadata.yaml files record precision as a {weights, activations} object (e.g. "w4a16" written out longhand as {weights: "int4", activations: "int16"}) instead of the short code — String()'ing that object is where the catalog's "[object Object]" quant badge came from. Collapses it back to the standard wNaM shorthand, and folds the "f16"/"f32" spelling some repos use into the "fp16"/"fp32" names used everywhere else. */ function normalizeQuantName(raw) { if (raw && typeof raw === "object") { const bits = v => { const m = String(v ?? "").match(/\d+/); return m ? m[0] : null; }; const w = bits(raw.weights), a = bits(raw.activations); return (w && a) ? `w${w}a${a}` : "mixed"; } const s = String(raw ?? "").trim().toLowerCase(); return (s === "f16") ? "fp16" : (s === "f32") ? "fp32" : s; } /* The actual quantization/precision (fp32, int8, w4a16, …) of an artifact — NOT the artifact folder name, which is usually the same string but isn't always: e.g. ResNet18-OpticalFlow-ONNX's artifacts are named "xavier-export" / "a100-export" (two reference-target exports, both int8 per artifact_info), so listing the raw folder name as a "quantization" is simply wrong. Falls back to the artifact name for repos with no artifact_info override, where the folder name already *is* the precision. */ function artifactPrecision(m, art) { const info = (m.artifact_info || {})[art]; return normalizeQuantName((info && info.precision) ? info.precision : art); } function sizeLabel(sizeM) { if (sizeM == null) return "size n/a"; return sizeM >= 1000 ? `${Math.round(sizeM / 100) / 10}B` : `${Math.round(sizeM)}M`; } /* Tooltip / excluded-list body: whatever the block actually reports, unfiltered by the Catalog's KPI display toggles (those belong to the Catalog surface). */ function pointKpiText(p) { const chips = kpiChipsOf(p.block).map(c => `${c.label} ${fmtChipFull(c)}`); if (p.fpsDerived) chips.unshift(`Throughput ${fmtAxisNum(p.fps)} img/s (derived: 1000 / ${fmtMs(p.lat)} ms)`); return chips.length ? chips : ["no metrics recorded"]; } /* A block with accuracy but no timing at all is not a failed performance run — it is the FP32 reference the quantized variants are scored against (the CV repos' fp32 runs say so: "reference accuracy measured on physical X5H silicon", with fps and latency explicitly null). Charting it as a missing point would report a deliberate baseline as a gap, so these are pulled out of the point set and listed as references instead. */ function isAccuracyReference(p) { return p.tok == null && p.fps == null && p.lat == null && p.acc != null; } /* The reference accuracy for a model, so a quantized point can show what its accuracy is measured against. */ function accuracyReferenceOf(key, refs) { return safeList(refs, []).find(r => r.key === key) || null; } let CSTL_HIT = []; /* Accuracy-reference runs for the open regime, kept so a point's tooltip can name the baseline its quantization is scored against. */ let CSTL_REFS = []; const CSTL_FONT = "system-ui, -apple-system, Segoe UI, Roboto, Arial"; const CSTL_INK = "rgba(31,30,28,0.72)"; const CSTL_INK_SOFT = "rgba(31,30,28,0.50)"; const CSTL_GRID = "rgba(31,30,28,0.08)"; /* Which regime tab is open, plus the axis pair chosen per regime (kept per tab so switching back restores what you were looking at). */ const CSTL_STATE = { regime: CSTL_REGIMES[0].id, axes: {} }; function cstlAxesFor(r) { if (!CSTL_STATE.axes[r.id]) { CSTL_STATE.axes[r.id] = { x: safeList(r.x, ["tok"])[0], y: safeList(r.y, ["size"])[0] }; } return CSTL_STATE.axes[r.id]; } /* Circles (scatter) hit by distance, bars (stage chart) by rectangle. */ function hitTest(list, evt, canvas) { const rect = canvas.getBoundingClientRect(); const x = evt.clientX - rect.left, y = evt.clientY - rect.top; let best = null, bestD = Infinity; for (const p of list) { if (p.rect) { if (x >= p.rect[0] && x <= p.rect[2] && y >= p.rect[1] && y <= p.rect[3]) return p; continue; } const dx = x - p.x, dy = y - p.y, d = dx * dx + dy * dy; if (d <= p.r * p.r && d < bestD) { best = p; bestD = d; } } return best; } function prepCanvas(canvas) { const ctx = canvas.getContext("2d"); const w = canvas.clientWidth || canvas.parentElement?.clientWidth || 600; const h = canvas.clientHeight || 460; const dpr = window.devicePixelRatio || 1; canvas.width = Math.max(1, Math.floor(w * dpr)); canvas.height = Math.max(1, Math.floor(h * dpr)); ctx.setTransform(dpr, 0, 0, dpr, 0, 0); ctx.clearRect(0, 0, w, h); ctx.font = `12px ${CSTL_FONT}`; ctx.textAlign = "left"; CSTL_HIT = []; return { ctx, w, h }; } function emptyCanvasMsg(ctx, w, h, msg) { ctx.fillStyle = CSTL_INK; ctx.textAlign = "center"; ctx.fillText(msg, w / 2, h / 2); ctx.textAlign = "left"; } /* Ellipsis-truncate to fit maxWidth under the ctx's current font — used for the stage chart's model-name column, since architecture names vary wildly in length (e.g. "SigLIP-SO400M-patch14-384") and must never overlap the bars. */ function truncateToWidth(ctx, text, maxWidth) { if (ctx.measureText(text).width <= maxWidth) return text; let lo = 0, hi = text.length; while (lo < hi) { const mid = (lo + hi + 1) >> 1; if (ctx.measureText(text.slice(0, mid) + "…").width <= maxWidth) lo = mid; else hi = mid - 1; } return lo > 0 ? text.slice(0, lo) + "…" : "…"; } function barPath(ctx, x, y, w, h, r) { const rr = Math.min(r, h / 2, w / 2); ctx.beginPath(); ctx.moveTo(x + rr, y); ctx.lineTo(x + w - rr, y); ctx.quadraticCurveTo(x + w, y, x + w, y + rr); ctx.lineTo(x + w, y + h - rr); ctx.quadraticCurveTo(x + w, y + h, x + w - rr, y + h); ctx.lineTo(x + rr, y + h); ctx.quadraticCurveTo(x, y + h, x, y + h - rr); ctx.lineTo(x, y + rr); ctx.quadraticCurveTo(x, y, x + rr, y); ctx.closePath(); } /* Point labels so the chart reads without hovering. For each point: try four offsets, prefer the first that clears both the other labels and every bubble, fall back to the first that at least clears the other labels, and skip the point if even that fails. Labels are drawn with a white halo so the fallback case stays legible where the catalog clusters (small quantized models all land in the same corner). */ function drawPointLabels(ctx, items, box) { const overlaps = (a, b) => !(a[2] < b[0] || a[0] > b[2] || a[3] < b[1] || a[1] > b[3]); const bubbles = items.map(it => [it.x - it.r, it.y - it.r, it.x + it.r, it.y + it.r]); const placed = []; ctx.font = `600 11px ${CSTL_FONT}`; ctx.lineJoin = "round"; for (const it of items) { const tw = ctx.measureText(it.text).width; const cands = [ [it.x + it.r + 5, it.y + 4], [it.x - it.r - 5 - tw, it.y + 4], [it.x - tw / 2, it.y - it.r - 6], [it.x - tw / 2, it.y + it.r + 14], ].map(([lx, ly]) => ({ lx, ly, rect: [lx - 2, ly - 11, lx + tw + 2, ly + 3] })) .filter(c => c.rect[0] >= box.l && c.rect[2] <= box.r && c.rect[1] >= box.t && c.rect[3] <= box.b) .filter(c => !placed.some(q => overlaps(c.rect, q))); const pick = cands.find(c => !bubbles.some(b => overlaps(c.rect, b))) || cands[0]; if (!pick) continue; placed.push(pick.rect); ctx.strokeStyle = "rgba(255,255,255,0.92)"; ctx.lineWidth = 3; ctx.strokeText(it.text, pick.lx, pick.ly); ctx.fillStyle = "rgba(31,30,28,0.72)"; ctx.fillText(it.text, pick.lx, pick.ly); } ctx.lineWidth = 1; ctx.font = `12px ${CSTL_FONT}`; } /* Scatter for the regimes whose models publish two commensurable numbers. Returns the split between what could be placed and what could not (and why), so the caller can show the misses instead of dropping them. */ function drawScatter(canvas, points, opts) { const { ctx, w, h } = prepCanvas(canvas); const ax = AXES[opts.xKey] || AXES.tok; const ay = AXES[opts.yKey] || AXES.size; const plotted = [], excluded = []; for (const p of points) { const miss = []; if (ax.get(p) == null) miss.push(ax.label); if (ay.get(p) == null) miss.push(ay.label); if (miss.length) excluded.push({ p, miss }); else plotted.push(p); } if (!plotted.length) { emptyCanvasMsg(ctx, w, h, points.length ? "No variant reports both of these axes — see the list below." : "No benchmark data for this selection."); return { plotted, excluded }; } const pad = { l: 70, r: 22, t: 24, b: 52 }; const innerW = Math.max(1, w - pad.l - pad.r); const innerH = Math.max(1, h - pad.t - pad.b); const xs = plotted.map(ax.get), ys = plotted.map(ay.get); let xMin = Math.min(0, ...xs), xMax = Math.max(...xs); let yMin = Math.min(0, ...ys), yMax = Math.max(...ys); if (xMax - xMin < 1e-9) xMax = xMin + 1; if (yMax - yMin < 1e-9) yMax = yMin + 1; xMax += (xMax - xMin) * 0.10; yMax += (yMax - yMin) * 0.14; const xOf = v => pad.l + ((v - xMin) / (xMax - xMin)) * innerW; const yOf = v => pad.t + innerH - ((v - yMin) / (yMax - yMin)) * innerH; const ticks = 4; ctx.lineWidth = 1; ctx.strokeStyle = CSTL_GRID; for (let i = 0; i <= ticks; i++) { const ty = pad.t + innerH - (innerH * i / ticks); ctx.beginPath(); ctx.moveTo(pad.l, ty); ctx.lineTo(pad.l + innerW, ty); ctx.stroke(); ctx.fillStyle = CSTL_INK; ctx.fillText(ay.tick(yMin + (yMax - yMin) * (i / ticks)), 8, ty + 4); } for (let i = 0; i <= ticks; i++) { const tx = pad.l + (innerW * i / ticks); ctx.beginPath(); ctx.moveTo(tx, pad.t); ctx.lineTo(tx, pad.t + innerH); ctx.stroke(); ctx.fillStyle = CSTL_INK; ctx.textAlign = i === ticks ? "right" : "center"; ctx.fillText(ax.tick(xMin + (xMax - xMin) * (i / ticks)), tx, h - 30); ctx.textAlign = "left"; } ctx.fillStyle = CSTL_INK_SOFT; ctx.fillText(`${ax.label} ${ax.lowerBetter ? "← lower is better" : "→ higher is better"}`, pad.l, h - 10); ctx.fillText(`↑ ${ay.label}${ay.lowerBetter ? " (lower is better)" : ""}`, 8, 14); const maxSz = Math.max(1, ...plotted.map(p => p.sizeM || 0)); const rOf = s => (s == null ? 8 : 6 + 16 * Math.sqrt(s / maxSz)); // Big bubbles first so a small fast variant is never buried under a large one. const order = plotted.slice().sort((a, b) => (b.sizeM || 0) - (a.sizeM || 0)); const labels = []; order.forEach(p => { const x = xOf(ax.get(p)), y = yOf(ay.get(p)), r = rOf(p.sizeM); const col = colorFor(colorKey(p, opts.colorBy)); // A dashed ring marks a point whose plotted throughput was derived from its // own latency rather than reported directly. const derivedOnAxis = p.fpsDerived && (opts.xKey === "fps" || opts.yKey === "fps"); ctx.setLineDash(derivedOnAxis ? [4, 3] : []); ctx.beginPath(); ctx.arc(x, y, r, 0, Math.PI * 2); if (p.source === "HIL") { ctx.fillStyle = col + "C0"; ctx.fill(); ctx.strokeStyle = col; ctx.lineWidth = 1.5; ctx.stroke(); } else { ctx.fillStyle = col + "33"; ctx.fill(); ctx.strokeStyle = col; ctx.lineWidth = 1.8; ctx.stroke(); } ctx.setLineDash([]); CSTL_HIT.push({ x, y, r: Math.max(r, 10), data: p }); labels.push({ x, y, r, text: `${p.arch} ${p.quant}` }); }); if (labels.length <= 18) { drawPointLabels(ctx, labels, { l: 4, r: w - 4, t: pad.t - 12, b: pad.t + innerH + 10 }); } return { plotted, excluded }; } /* Stage chart for the encoder regime: one row per artifact/source, one bar per encoder tower. Bars are scaled to the largest *stage*, and the run total is printed in the row label rather than drawn as a bar — it is an order of magnitude larger than the towers and is not their sum. */ function drawStageChart(canvas, points) { const { ctx, w, h } = prepCanvas(canvas); // Grouped by model first, so several dual-encoder models sort into visibly // separate blocks instead of interleaving by quant/source. const rows = points.filter(p => p.stages.length) .sort((a, b) => a.arch.localeCompare(b.arch) || a.quant.localeCompare(b.quant) || a.source.localeCompare(b.source)); const excluded = points.filter(p => !p.stages.length).map(p => ({ p, miss: ["per-stage timings"] })); const stageNames = []; rows.forEach(p => p.stages.forEach(s => { if (!stageNames.includes(s.label)) stageNames.push(s.label); })); if (!rows.length) { emptyCanvasMsg(ctx, w, h, "No per-stage timings for this selection."); return { plotted: rows, excluded, stageNames }; } const maxMs = Math.max(1, ...rows.flatMap(p => p.stages.map(s => Number(s.ms) || 0))); const pad = { l: 186, r: 96, t: 24, b: 46 }; const innerW = Math.max(1, w - pad.l - pad.r); const innerH = Math.max(1, h - pad.t - pad.b); const rowH = innerH / rows.length; const xOf = ms => pad.l + (ms / maxMs) * innerW; const ticks = 4; ctx.lineWidth = 1; ctx.strokeStyle = CSTL_GRID; for (let i = 0; i <= ticks; i++) { const tx = pad.l + (innerW * i / ticks); ctx.beginPath(); ctx.moveTo(tx, pad.t); ctx.lineTo(tx, pad.t + innerH); ctx.stroke(); ctx.fillStyle = CSTL_INK; ctx.textAlign = i === ticks ? "right" : "center"; ctx.fillText(fmtAxisNum(maxMs * i / ticks), tx, h - 26); ctx.textAlign = "left"; } ctx.fillStyle = CSTL_INK_SOFT; ctx.fillText("Per-stage NPU time (ms) ← lower is better", pad.l, h - 8); rows.forEach((p, i) => { const top = pad.t + i * rowH; const newModel = i === 0 || rows[i - 1].arch !== p.arch; if (i) { // A heavier line at a model boundary reads as a group break; the default // hairline just separates two variants of the same model. ctx.strokeStyle = newModel ? "rgba(31,30,28,0.16)" : CSTL_GRID; ctx.lineWidth = newModel ? 1.4 : 1; ctx.beginPath(); ctx.moveTo(8, top); ctx.lineTo(w - 8, top); ctx.stroke(); ctx.lineWidth = 1; } const labelMaxW = pad.l - 22; // Model name — the identity that used to be missing — with a color dot // matching the same model's swatch elsewhere (cards/scatter), so several // models scan at a glance even before reading the text. ctx.font = `700 12px ${CSTL_FONT}`; const nameText = truncateToWidth(ctx, p.arch, labelMaxW - 12); ctx.fillStyle = colorFor(p.arch); ctx.beginPath(); ctx.arc(11, top + rowH / 2 - 15, 3.5, 0, Math.PI * 2); ctx.fill(); ctx.fillStyle = CSTL_INK; ctx.fillText(nameText, 20, top + rowH / 2 - 11); ctx.font = `600 11.5px ${CSTL_FONT}`; ctx.fillStyle = CSTL_INK_SOFT; ctx.fillText(truncateToWidth(ctx, `${p.quant} · ${p.source}`, labelMaxW), 8, top + rowH / 2 + 3); ctx.font = `11px ${CSTL_FONT}`; ctx.fillStyle = CSTL_INK_SOFT; ctx.fillText(p.lat != null ? `Σ run ${fmtMs(p.lat)} ms` : `${p.fmt} · ${p.target}`, 8, top + rowH / 2 + 17); ctx.font = `12px ${CSTL_FONT}`; const bars = p.stages; const barH = Math.max(10, Math.min(24, (rowH - 18) / bars.length - 6)); const stackH = bars.length * (barH + 6) - 6; bars.forEach((s, j) => { const y = top + (rowH - stackH) / 2 + j * (barH + 6); const bw = Math.max(3, xOf(Number(s.ms)) - pad.l); const col = colorFor(s.label); barPath(ctx, pad.l, y, bw, barH, 3); ctx.fillStyle = col + "D9"; ctx.fill(); ctx.strokeStyle = col; ctx.lineWidth = 1; ctx.stroke(); const nameW = ctx.measureText(s.label).width; const inside = bw > nameW + 18 && barH >= 14; if (inside) { ctx.fillStyle = "#FFFFFF"; ctx.fillText(s.label, pad.l + 9, y + barH / 2 + 4); } ctx.fillStyle = CSTL_INK; ctx.fillText(`${fmtMs(s.ms)} ms${inside ? "" : " · " + s.label}`, pad.l + bw + 7, y + barH / 2 + 4); CSTL_HIT.push({ rect: [pad.l, y, pad.l + bw, y + barH], data: p }); }); }); return { plotted: rows, excluded, stageNames }; } function setLegend(items, extraHTML) { const el = $("cstlLegend"); if (!el) return; el.innerHTML = items.map(it => `${esc(it.label)}` ).join("") + (extraHTML || ""); } /* Roadmap regime: no metric exists, so there is nothing to plot — the models are listed with the reason, which is the point of the tab. */ function roadmapReasons(m) { const mod = String(m.modality || "").trim().toUpperCase(); const out = []; if (mod === "VLA") { out.push("Emits an action chunk per step — the comparable KPI is closed-loop control rate (Hz) and task success, which the benchmark pipeline does not produce yet."); } if (isComingSoon(m)) out.push("No artifact folder published yet, so there is nothing to measure or download."); if (!out.length) out.push("No plottable metric published yet."); return out; } function roadmapCardHTML(m) { const col = colorFor(m.architecture || "Unknown"); const d = designerOf(m); const cfg = safeFirstCfg(m); const defFmt = m.gguf_repo ? "GGUF" : (m.onnx_repo ? "ONNX" : "GGUF"); return `
${esc(monogramText(m))} ${esc(m.display_name || m.key)} ${esc(m.architecture || "")}
${esc(m.modality || "—")} ${isComingSoon(m) ? `Coming soon` : ""}
${roadmapReasons(m).map(r => `${esc(r)}`).join("")}
${esc(d.name)}${esc(sizeLabel(m.model_size_m))}
`; } function renderRoadmapCards(models) { const el = $("cstlCards"); if (!el) return; el.innerHTML = models.length ? models.map(roadmapCardHTML).join("") : `
Nothing pending — every model in the catalog reports a plottable metric.
`; } /* One row per variant that isn't a point on this chart, in two groups with very different meanings: a genuine metadata gap (the metric is missing) versus an accuracy reference run (deliberately has no timing). */ function cstlRowHTML(p, missHTML) { return `
${esc(p.arch)} ${esc(p.quant)} ${esc(p.source)} ${missHTML} ${esc(pointKpiText(p).join(" · "))}
`; } function renderExcluded(list, refs) { const el = $("cstlExcluded"); if (!el) return; const refList = safeList(refs, []); if (!list.length && !refList.length) { el.hidden = true; el.innerHTML = ""; return; } el.hidden = false; const groups = []; if (list.length) { groups.push(`
Not plottable on these axes (${list.length}) — metric missing from the repo metadata
${list.map(({ p, miss }) => cstlRowHTML(p, `no ${esc(miss.join(" · no "))}`)).join("")}
`); } if (refList.length) { groups.push(`
Accuracy reference runs (${refList.length}) — no timing by design
${refList.map(p => cstlRowHTML(p, `baseline for the quantized variants`)).join("")}
`); } el.innerHTML = groups.join(""); } function syncRegimeUI(r) { $$(".cstl-tab").forEach(btn => { const on = btn.getAttribute("data-regime") === r.id; btn.classList.toggle("active", on); btn.setAttribute("aria-selected", on ? "true" : "false"); btn.tabIndex = on ? 0 : -1; }); $("cstlPanel")?.setAttribute("aria-labelledby", `cstlTab_${r.id}`); const why = $("cstlWhy"); if (why) why.innerHTML = r.why || ""; const showXY = r.chart === "scatter"; const showSource = r.chart !== "cards"; const fields = { cstlFieldX: showXY, cstlFieldY: showXY, cstlFieldColor: showXY, cstlFieldSource: showSource }; for (const [id, on] of Object.entries(fields)) { const el = $(id); if (el) el.hidden = !on; } const anyControl = showXY || showSource; const ctl = $("cstlControls"); if (ctl) ctl.hidden = !anyControl; const tgl = $("cstlFilterToggle"); if (tgl) tgl.hidden = !anyControl; if (showXY) { const sel = cstlAxesFor(r); fillAxisSelect($("cstlX"), safeList(r.x, []), sel.x); fillAxisSelect($("cstlY"), safeList(r.y, []), sel.y); } } /* Rebuilds the `).join(""); } sel.value = value; } function buildRegimeTabs() { const el = $("cstlTabs"); if (!el) return; const counts = {}; CSTL_REGIMES.forEach(r => { counts[r.id] = 0; }); CATALOG.forEach(m => { const id = regimeOf(m); if (counts[id] != null) counts[id]++; }); el.innerHTML = CSTL_REGIMES.map(r => { const on = r.id === CSTL_STATE.regime; return ` `; }).join(""); } function setRegime(id) { if (!CSTL_REGIME_BY_ID[id] || id === CSTL_STATE.regime) return; CSTL_STATE.regime = id; redrawConstellation(); } function redrawConstellation() { const canvas = $("cstlCanvas"); if (!canvas) return; const r = CSTL_REGIME_BY_ID[CSTL_STATE.regime] || CSTL_REGIMES[0]; /* Capture the axis pickers before syncRegimeUI rebuilds their