diff --git a/site/assets/app.js b/site/assets/app.js
index a51f554d..dc53590e 100644
--- a/site/assets/app.js
+++ b/site/assets/app.js
@@ -374,6 +374,11 @@ const I18N = {
"Shown / corpus": "显示数/目录总数",
"source document": "份来源文档",
"Document publication date": "文档发布日期",
+ "Model release": "模型发布日期",
+ "Document published": "文档发布日期",
+ "Model with a reported score": "已有报告成绩的模型",
+ "Filter by model, e.g. GPT-6 Sol": "按模型筛选,例如 GPT-6 Sol",
+ "Searching all benchmarks (score cutoff paused)": "搜索全部 benchmark(暂不按分数筛选)",
"Reporting organization color key": "报告机构颜色图例",
"Distinct cited documents, including model reports and registry pages": "按引用的独立文档去重,包括模型报告和登记页面",
"Each source document counts once per benchmark record.": "每份来源文档对同一条 benchmark 记录只计一次。",
@@ -1228,6 +1233,7 @@ const state = {
// arrived, so "not yet loaded" and "failed to load" must not look alike.
benchmarkIndexLoaded: false,
benchmarkQuery: "",
+ benchmarkModel: "",
leaderboardShowAll: false,
leaderboardTopExpanded: false,
todayResultsKey: "",
@@ -1466,6 +1472,7 @@ function readUrl() {
state.lheight = ["cards", "documents"].includes(params.get("lheight")) ? "documents" : "models";
state.benchmarkVisibleLimit = BENCHMARK_SEARCH_LIMIT;
state.benchmarkQuery = (params.get("bq") || "").trim();
+ state.benchmarkModel = (params.get("bmodel") || "").trim();
state.lfrontier = params.get("lfrontier") || "";
state.lfrontierExplicit = Boolean(state.lfrontier);
// Existing benchmark permalinks follow the score history to its new tab.
@@ -1549,6 +1556,7 @@ function writeUrl(mode = "replace") {
if (!utility && state.view === "saturation") {
params.set("lscore", state.lscore);
if (state.benchmarkQuery) params.set("bq", state.benchmarkQuery);
+ if (state.benchmarkModel) params.set("bmodel", state.benchmarkModel);
// A benchmark auto-picked as the default is not the reader's choice, so it
// stays out of the URL until they select one themselves.
if (state.lfrontierExplicit && state.lfrontier) {
@@ -4704,8 +4712,22 @@ function saturationRows() {
const matches = benchmarkQueryIds();
// Search is a lookup across the whole catalog. It temporarily bypasses the
// browsing cutoff without changing the shared slider value.
- const rows = scoreBrowseRows(matches ? 100 : state.lscore);
- return matches ? rows.filter((row) => matches.has(row.id)) : rows;
+ const rows = scoreBrowseRows(matches || state.benchmarkModel ? 100 : state.lscore);
+ const named = matches ? rows.filter((row) => matches.has(row.id)) : rows;
+ if (!state.benchmarkModel) return named;
+ const needle = foldName(state.benchmarkModel);
+ if (!needle) return [];
+ const index = new Map((state.benchmarkIndex || []).map((record) => [record.slug, record]));
+ return named.filter((row) => (index.get(row.id)?.scored_models || [])
+ .some((model) => foldName(model.name).includes(needle)))
+ .sort((a, b) => (latestMatchingModel(index.get(b.id), needle)?.date || "")
+ .localeCompare(latestMatchingModel(index.get(a.id), needle)?.date || "") || a.name.localeCompare(b.name));
+}
+
+function latestMatchingModel(record, needle) {
+ return (record?.scored_models || [])
+ .filter((model) => foldName(model.name).includes(needle))
+ .sort((a, b) => (b.date || "").localeCompare(a.date || ""))[0];
}
function scoreRankingRows() {
@@ -4722,6 +4744,9 @@ function scoreSummaryLabel(summary) {
}
function scoreBrowseResultRow(row) {
+ const record = (state.benchmarkIndex || []).find((item) => item.slug === row.id);
+ const needle = foldName(state.benchmarkModel);
+ const matchedModel = needle ? latestMatchingModel(record, needle) : null;
const button = element("button", {
className: "benchmark-result score-browse-result",
attrs: { type: "button", "aria-pressed": row.id === state.lfrontier ? "true" : "false" },
@@ -4730,6 +4755,7 @@ function scoreBrowseResultRow(row) {
element("span", { className: "benchmark-result-facts", text: [scoreSourceLabel(row.source),
row.date ? `${t(benchmarkDateLabel(row))} ${formatDate(row.date, { dateStyle: "medium" })}` : t("Date unknown"),
matchesScoreCutoff(row.summary, 100) ? "" : t("No score reported"),
+ matchedModel ? `${matchedModel.name}${matchedModel.date ? ` · ${t(matchedModel.date_precision === "model_announcement" ? "Model release" : "Document published")} ${formatDate(matchedModel.date, { dateStyle: "medium" })}` : ""}` : "",
].filter(Boolean).join(" · ") }),
]);
button.addEventListener("click", () => {
@@ -4814,6 +4840,7 @@ function renderBenchmarkSearch() {
const status = byId("benchmark-search-status");
if (!container || !status || !state.data) return;
byId("benchmark-search-input").value = state.benchmarkQuery;
+ byId("benchmark-model-input").value = state.benchmarkModel;
const rows = saturationRows();
const shown = rows.slice(0, state.benchmarkVisibleLimit);
replaceChildren(container, shown.map(scoreBrowseResultRow));
@@ -4824,11 +4851,11 @@ function renderBenchmarkSearch() {
status.textContent = [t("{shown} of {total} matches")
.replace("{shown}", shown.length.toLocaleString())
.replace("{total}", rows.length.toLocaleString()),
- state.benchmarkQuery ? t("Searching all benchmarks (filters paused)") : "", coverage].filter(Boolean).join(" · ");
+ state.benchmarkQuery || state.benchmarkModel ? t("Searching all benchmarks (score cutoff paused)") : "", coverage].filter(Boolean).join(" · ");
if (!rows.length) {
container.append(element("p", { className: "empty-state", text: loading
? t("Loading benchmark details…") : t("No benchmarks match these filters.") }));
- if (!state.benchmarkQuery && state.lscore < 100) {
+ if (!state.benchmarkQuery && !state.benchmarkModel && state.lscore < 100) {
const all = element("button", { className: "clear-button", text: t("All"), attrs: { type: "button" } });
all.addEventListener("click", () => setScoreFilter(100));
container.append(all);
@@ -5266,6 +5293,13 @@ function initBenchmarkSearch() {
writeUrl();
});
input.addEventListener("input", onInput);
+ const modelInput = byId("benchmark-model-input");
+ modelInput.addEventListener("input", debounce(() => {
+ state.benchmarkModel = modelInput.value.trim();
+ state.benchmarkVisibleLimit = BENCHMARK_SEARCH_LIMIT;
+ renderSaturation();
+ writeUrl();
+ }));
loadBenchmarkIndex().then((records) => {
// Loading and failure remain explicit; no source-specific fallback corpus.
state.benchmarkIndex = records;
diff --git a/site/assets/styles.css b/site/assets/styles.css
index 1e627311..3214405e 100644
--- a/site/assets/styles.css
+++ b/site/assets/styles.css
@@ -2795,6 +2795,14 @@ td a {
font-size: 0.78rem;
}
+.benchmark-model-label {
+ display: block;
+ margin: 0.85rem 0 0.4rem;
+ font-family: var(--utility);
+ font-size: 0.78rem;
+ font-weight: 700;
+}
+
.benchmark-navigator p:not(.eyebrow):not(.benchmark-search-status) {
color: var(--muted);
font-size: 0.8rem;
diff --git a/site/index.html b/site/index.html
index 8ccd5def..de883091 100644
--- a/site/index.html
+++ b/site/index.html
@@ -768,6 +768,13 @@
class="benchmark-search-input"
data-i18n-placeholder="Search benchmarks, tasks, domains…"
placeholder="Search benchmarks, tasks, domains…" />
+
+
diff --git a/src/benchmark_radar/catalog.py b/src/benchmark_radar/catalog.py
index 762dc18c..923ef77b 100644
--- a/src/benchmark_radar/catalog.py
+++ b/src/benchmark_radar/catalog.py
@@ -675,6 +675,7 @@ def write_catalog(
def build_benchmark_index(
records: list[dict[str, Any]],
series_by_key: dict[str, dict[str, Any]] | None = None,
+ observations: list[dict[str, Any]] | None = None,
) -> list[dict[str, Any]]:
"""The small search payload: one entry per source record, never per merge.
@@ -684,6 +685,21 @@ def build_benchmark_index(
it is invisible. One row per record keeps a bad grouping a display bug.
"""
series_by_key = series_by_key or {}
+ models_by_key: dict[str, dict[str, dict[str, Any]]] = {}
+ for observation in observations or []:
+ name = observation.get("model_name")
+ if not name or not isinstance(observation.get("value"), (int, float)):
+ continue
+ models = models_by_key.setdefault(observation["key"], {})
+ # One model name per source benchmark. Keep the most recent recorded
+ # date and its precision; a model release is not an evaluation update.
+ current = models.get(name)
+ if current is None or (observation.get("reported_date") or "") > (current["date"] or ""):
+ models[name] = {
+ "name": name,
+ "date": observation.get("reported_date"),
+ "date_precision": observation.get("date_precision"),
+ }
index: list[dict[str, Any]] = []
for record in records:
openness = record.get("openness") or {}
@@ -735,6 +751,10 @@ def build_benchmark_index(
"openness": openness.get("status", "unknown"),
"modality": record.get("modality"),
"score_count": series.get("observation_count", 0),
+ "scored_models": sorted(
+ models_by_key.get(record["key"], {}).values(),
+ key=lambda model: model["name"].lower(),
+ ),
"score_summary": (
series.get("score_summary") if series.get("observation_count", 0) else None
),
diff --git a/src/benchmark_radar/cli.py b/src/benchmark_radar/cli.py
index 93e3e0ef..955908fc 100644
--- a/src/benchmark_radar/cli.py
+++ b/src/benchmark_radar/cli.py
@@ -449,7 +449,7 @@ def main() -> None:
# One index over every source, one row per source record. Two sources
# describing the same benchmark stay two rows until identity.yml says
# otherwise under human review.
- index = build_benchmark_index(resolved_records, series_by_key)
+ index = build_benchmark_index(resolved_records, series_by_key, all_observations)
index_path = write_benchmark_index(
index,
Path("site/data/benchmark-index.json"),
diff --git a/tests/score_filters_harness.mjs b/tests/score_filters_harness.mjs
index f37b8ff6..ba326e08 100644
--- a/tests/score_filters_harness.mjs
+++ b/tests/score_filters_harness.mjs
@@ -49,10 +49,23 @@ state.benchmarkIndex = [
{slug:'missing',name:'Missing',source:'llm_stats',score_summary:summary(null,0)},
];
state.lscore = 70;
-const scoreNames = ['scoreRecord','matchesScoreFilter','scoreBrowseRows','saturationRows','scoreRankingRows','benchmarkQueryIds','searchBenchmarkIndex','foldName','frontierDefaultEntry','catalogDisplayFactor'];
+const scoreNames = ['scoreRecord','matchesScoreFilter','scoreBrowseRows','saturationRows','latestMatchingModel','scoreRankingRows','benchmarkQueryIds','searchBenchmarkIndex','foldName','frontierDefaultEntry','catalogDisplayFactor'];
state.benchmarkQuery = '';
const scores = new Function('state', 'scorePopulation', 'matchesScoreCutoff', 'SKYLINE_START_DATE', `${scoreNames.map(fn).join('\n')}\nreturn {${scoreNames.join(',')}};`)(state, scorePopulation, matchesScoreCutoff, SKYLINE_START_DATE);
assert.deepEqual(scores.scoreBrowseRows().map(r=>r.id), ['external','curated','missing']);
+state.benchmarkIndex[0].scored_models = [{name:'GPT-5.6 Sol',date:'2026-06-01',date_precision:'document_publication'}];
+state.benchmarkIndex[1].scored_models = [{name:'GPT-5.6 Sol (high)',date:'2026-09-01',date_precision:'model_announcement'}];
+state.benchmarkModel = 'gpt-5.6-sol';
+assert.deepEqual(scores.saturationRows().map(r=>r.id), ['external','curated'], 'model filtering searches every source beyond the score cutoff and orders by recorded date');
+state.benchmarkModel = 'gpt-6-sol';
+assert.deepEqual(scores.saturationRows(), [], 'an unrecorded model does not imply a benchmark result');
+for (const query of ['-', '---', '()']) {
+ state.benchmarkModel = query;
+ assert.deepEqual(scores.saturationRows(), [], 'punctuation-only queries have no model match');
+}
+assert.equal(state.lscore, 70, 'unmatched model lookup preserves the cutoff');
+state.benchmarkModel = '';
+assert.deepEqual(scores.saturationRows().map(r=>r.id), ['external','curated','missing'], 'clearing the model restores cutoff browsing');
assert.equal(scores.frontierDefaultEntry(state.data.model_card_leaderboard).id,'external');
assert.equal(scores.matchesScoreFilter(summary(69.999)),true);
assert.equal(scores.matchesScoreFilter(summary(70)),false);
diff --git a/tests/test_catalog.py b/tests/test_catalog.py
index 3379e9bd..69421279 100644
--- a/tests/test_catalog.py
+++ b/tests/test_catalog.py
@@ -71,6 +71,22 @@ def test_search_index_carries_semantic_source_fields_without_inference(normalize
assert isinstance(row["languages"], list)
+def test_index_model_filter_evidence_is_scoped_to_its_source_record(normalized: dict) -> None:
+ from benchmark_radar.catalog import build_benchmark_index
+
+ observations = normalized["score_observations"]
+ index = build_benchmark_index(normalized["source_records"], observations=observations)
+ by_key = {record["key"]: record for record in index}
+ for key, record in by_key.items():
+ source_rows = [row for row in observations if row["key"] == key]
+ assert {model["name"] for model in record["scored_models"]} == {
+ row["model_name"] for row in source_rows
+ }
+ assert all(
+ model["date_precision"] == "model_announcement" for model in record["scored_models"]
+ )
+
+
def test_obs_id_is_unique(normalized: dict) -> None:
"""Without this a rerun silently duplicates every score row."""
obs_ids = [row["obs_id"] for row in normalized["score_observations"]]
@@ -509,7 +525,14 @@ def test_index_has_one_row_per_source_record(normalized: dict) -> None:
from benchmark_radar.catalog_opencompass import normalize_opencompass
records = normalized["source_records"] + normalize_opencompass()["source_records"]
- index = build_benchmark_index(records, {row["key"]: row for row in normalized["score_series"]})
+ series_by_key = {row["key"]: row for row in normalized["score_series"]}
+ unscored_key = next(
+ record["key"] for record in records if record["source"] == "opencompass_hub"
+ )
+ series_by_key[unscored_key] = {"observation_count": 0, "score_summary": {"max": 99}}
+ index = build_benchmark_index(
+ records, series_by_key, observations=normalized["score_observations"]
+ )
assert len(index) == 1148
assert len({row["key"] for row in index}) == 1148
assert len({row["slug"] for row in index}) == 1148
@@ -526,6 +549,8 @@ def test_index_has_one_row_per_source_record(normalized: dict) -> None:
assert unscored["first_score_source_reference"] is None
assert unscored["first_score_record"] is None
assert unscored["score_summary"] is None
+ assert unscored["scored_models"] == []
+ assert any(row["scored_models"] for row in index)
scored = [row for row in index if row["first_score_record"]]
assert len(scored) > 600, "the index must preserve dates for the whole scored catalog"
for row in scored:
diff --git a/tests/test_site.py b/tests/test_site.py
index 94e65889..27c4f967 100644
--- a/tests/test_site.py
+++ b/tests/test_site.py
@@ -432,7 +432,7 @@ def test_each_view_serializes_only_the_filters_it_reads():
1
].split('if (!utility && state.view === "saturation")', 1)
assert 'params.set("lfrontier"' not in leaderboard
- for key in ("lscore", "bq", "lfrontier"):
+ for key in ("lscore", "bq", "bmodel", "lfrontier"):
assert f'params.set("{key}"' in saturation
for key in ("lq", "ldomain", "lorg", "lera", "lscore"):
assert f'params.set("{key}"' in leaderboard
@@ -658,12 +658,12 @@ def section(start: str, end: str) -> str:
results.legacyBenchmarks.push(window.location.pathname + window.location.search);
}}
-install("/saturation/?lscore=40&bq=bench-95&lfrontier=bench-95&lq=agent&lheight=documents");
+install("/saturation/?lscore=40&bq=bench-95&bmodel=GPT-6+Sol&lfrontier=bench-95&lq=agent&lheight=documents");
readUrl();
writeUrl("replace");
results.saturation = {{
url: window.location.pathname + window.location.search,
- view: state.view, query: state.benchmarkQuery, cutoff: state.lscore,
+ view: state.view, query: state.benchmarkQuery, model: state.benchmarkModel, cutoff: state.lscore,
selected: state.lfrontier, explicit: state.lfrontierExplicit,
}};
state.view = "leaderboard";
@@ -671,7 +671,10 @@ def section(start: str, end: str) -> str:
results.sharedCutoff = window.location.pathname + window.location.search;
state.view = "saturation";
state.benchmarkQuery = "";
+state.benchmarkModel = "";
writeUrl("replace");
+readUrl();
+results.clearedModel = state.benchmarkModel;
results.clearedSearch = window.location.pathname + window.location.search;
install("/#rubric=2");
@@ -733,14 +736,16 @@ def section(start: str, end: str) -> str:
* 2
)
assert routes["saturation"] == {
- "url": "/saturation/?lscore=40&bq=bench-95&lfrontier=bench-95",
+ "url": "/saturation/?lscore=40&bq=bench-95&bmodel=GPT-6+Sol&lfrontier=bench-95",
"view": "saturation",
"query": "bench-95",
+ "model": "GPT-6 Sol",
"cutoff": 40,
"selected": "bench-95",
"explicit": True,
}
assert routes["sharedCutoff"] == "/leaderboard/?lscore=40&lheight=documents&lq=agent"
+ assert routes["clearedModel"] == ""
assert routes["clearedSearch"] == "/saturation/?lscore=40&lfrontier=bench-95"
assert routes["legacyRubric"] == "/rubric/?version=2"
assert routes["legacyReturns"] is False
@@ -3271,3 +3276,50 @@ def test_blog_styles_cannot_reach_the_dashboard():
assert selectors
assert all(selector.startswith(".blog-page") for selector in selectors)
assert "blog-page" not in Path("site/index.html").read_text(encoding="utf-8")
+
+
+def test_model_filter_controls_and_status_have_chinese_translations():
+ import json
+ import shutil
+ import subprocess
+
+ import pytest
+
+ node = shutil.which("node")
+ if not node:
+ pytest.skip("node is not installed")
+ script = Path("site/assets/app.js").read_text(encoding="utf-8")
+ html = Path("site/index.html").read_text(encoding="utf-8")
+ keys = [
+ "Model with a reported score",
+ "Filter by model, e.g. GPT-6 Sol",
+ "Model release",
+ "Document published",
+ "Searching all benchmarks (score cutoff paused)",
+ ]
+ assert f'data-i18n="{keys[0]}"' in html
+ assert f'data-i18n-placeholder="{keys[1]}"' in html
+ start = script.index("const I18N = {")
+ dictionary = script[start : script.index("\n};", start) + 3]
+ start = script.index("function t(")
+ translate = script[start : script.index("\n}\n", start) + 2]
+ program = (
+ dictionary
+ + translate
+ + "\nlet lang = 'zh'; function getLang() { return lang; }\n"
+ + f"const keys = {json.dumps(keys)};\n"
+ + "const zh = keys.map(key => t(key)); lang = 'en';\n"
+ + "console.log(JSON.stringify({zh, en: keys.map(key => t(key))}));"
+ )
+ result = subprocess.run(
+ [node, "-e", program], capture_output=True, text=True, timeout=60, check=True
+ )
+ translated = json.loads(result.stdout)
+ assert translated["en"] == keys
+ assert translated["zh"] == [
+ "已有报告成绩的模型",
+ "按模型筛选,例如 GPT-6 Sol",
+ "模型发布日期",
+ "文档发布日期",
+ "搜索全部 benchmark(暂不按分数筛选)",
+ ]