diff --git a/site/assets/app.js b/site/assets/app.js index a51f554d..dc53590e 100644 --- a/site/assets/app.js +++ b/site/assets/app.js @@ -374,6 +374,11 @@ const I18N = { "Shown / corpus": "显示数/目录总数", "source document": "份来源文档", "Document publication date": "文档发布日期", + "Model release": "模型发布日期", + "Document published": "文档发布日期", + "Model with a reported score": "已有报告成绩的模型", + "Filter by model, e.g. GPT-6 Sol": "按模型筛选,例如 GPT-6 Sol", + "Searching all benchmarks (score cutoff paused)": "搜索全部 benchmark(暂不按分数筛选)", "Reporting organization color key": "报告机构颜色图例", "Distinct cited documents, including model reports and registry pages": "按引用的独立文档去重,包括模型报告和登记页面", "Each source document counts once per benchmark record.": "每份来源文档对同一条 benchmark 记录只计一次。", @@ -1228,6 +1233,7 @@ const state = { // arrived, so "not yet loaded" and "failed to load" must not look alike. benchmarkIndexLoaded: false, benchmarkQuery: "", + benchmarkModel: "", leaderboardShowAll: false, leaderboardTopExpanded: false, todayResultsKey: "", @@ -1466,6 +1472,7 @@ function readUrl() { state.lheight = ["cards", "documents"].includes(params.get("lheight")) ? "documents" : "models"; state.benchmarkVisibleLimit = BENCHMARK_SEARCH_LIMIT; state.benchmarkQuery = (params.get("bq") || "").trim(); + state.benchmarkModel = (params.get("bmodel") || "").trim(); state.lfrontier = params.get("lfrontier") || ""; state.lfrontierExplicit = Boolean(state.lfrontier); // Existing benchmark permalinks follow the score history to its new tab. @@ -1549,6 +1556,7 @@ function writeUrl(mode = "replace") { if (!utility && state.view === "saturation") { params.set("lscore", state.lscore); if (state.benchmarkQuery) params.set("bq", state.benchmarkQuery); + if (state.benchmarkModel) params.set("bmodel", state.benchmarkModel); // A benchmark auto-picked as the default is not the reader's choice, so it // stays out of the URL until they select one themselves. if (state.lfrontierExplicit && state.lfrontier) { @@ -4704,8 +4712,22 @@ function saturationRows() { const matches = benchmarkQueryIds(); // Search is a lookup across the whole catalog. It temporarily bypasses the // browsing cutoff without changing the shared slider value. - const rows = scoreBrowseRows(matches ? 100 : state.lscore); - return matches ? rows.filter((row) => matches.has(row.id)) : rows; + const rows = scoreBrowseRows(matches || state.benchmarkModel ? 100 : state.lscore); + const named = matches ? rows.filter((row) => matches.has(row.id)) : rows; + if (!state.benchmarkModel) return named; + const needle = foldName(state.benchmarkModel); + if (!needle) return []; + const index = new Map((state.benchmarkIndex || []).map((record) => [record.slug, record])); + return named.filter((row) => (index.get(row.id)?.scored_models || []) + .some((model) => foldName(model.name).includes(needle))) + .sort((a, b) => (latestMatchingModel(index.get(b.id), needle)?.date || "") + .localeCompare(latestMatchingModel(index.get(a.id), needle)?.date || "") || a.name.localeCompare(b.name)); +} + +function latestMatchingModel(record, needle) { + return (record?.scored_models || []) + .filter((model) => foldName(model.name).includes(needle)) + .sort((a, b) => (b.date || "").localeCompare(a.date || ""))[0]; } function scoreRankingRows() { @@ -4722,6 +4744,9 @@ function scoreSummaryLabel(summary) { } function scoreBrowseResultRow(row) { + const record = (state.benchmarkIndex || []).find((item) => item.slug === row.id); + const needle = foldName(state.benchmarkModel); + const matchedModel = needle ? latestMatchingModel(record, needle) : null; const button = element("button", { className: "benchmark-result score-browse-result", attrs: { type: "button", "aria-pressed": row.id === state.lfrontier ? "true" : "false" }, @@ -4730,6 +4755,7 @@ function scoreBrowseResultRow(row) { element("span", { className: "benchmark-result-facts", text: [scoreSourceLabel(row.source), row.date ? `${t(benchmarkDateLabel(row))} ${formatDate(row.date, { dateStyle: "medium" })}` : t("Date unknown"), matchesScoreCutoff(row.summary, 100) ? "" : t("No score reported"), + matchedModel ? `${matchedModel.name}${matchedModel.date ? ` · ${t(matchedModel.date_precision === "model_announcement" ? "Model release" : "Document published")} ${formatDate(matchedModel.date, { dateStyle: "medium" })}` : ""}` : "", ].filter(Boolean).join(" · ") }), ]); button.addEventListener("click", () => { @@ -4814,6 +4840,7 @@ function renderBenchmarkSearch() { const status = byId("benchmark-search-status"); if (!container || !status || !state.data) return; byId("benchmark-search-input").value = state.benchmarkQuery; + byId("benchmark-model-input").value = state.benchmarkModel; const rows = saturationRows(); const shown = rows.slice(0, state.benchmarkVisibleLimit); replaceChildren(container, shown.map(scoreBrowseResultRow)); @@ -4824,11 +4851,11 @@ function renderBenchmarkSearch() { status.textContent = [t("{shown} of {total} matches") .replace("{shown}", shown.length.toLocaleString()) .replace("{total}", rows.length.toLocaleString()), - state.benchmarkQuery ? t("Searching all benchmarks (filters paused)") : "", coverage].filter(Boolean).join(" · "); + state.benchmarkQuery || state.benchmarkModel ? t("Searching all benchmarks (score cutoff paused)") : "", coverage].filter(Boolean).join(" · "); if (!rows.length) { container.append(element("p", { className: "empty-state", text: loading ? t("Loading benchmark details…") : t("No benchmarks match these filters.") })); - if (!state.benchmarkQuery && state.lscore < 100) { + if (!state.benchmarkQuery && !state.benchmarkModel && state.lscore < 100) { const all = element("button", { className: "clear-button", text: t("All"), attrs: { type: "button" } }); all.addEventListener("click", () => setScoreFilter(100)); container.append(all); @@ -5266,6 +5293,13 @@ function initBenchmarkSearch() { writeUrl(); }); input.addEventListener("input", onInput); + const modelInput = byId("benchmark-model-input"); + modelInput.addEventListener("input", debounce(() => { + state.benchmarkModel = modelInput.value.trim(); + state.benchmarkVisibleLimit = BENCHMARK_SEARCH_LIMIT; + renderSaturation(); + writeUrl(); + })); loadBenchmarkIndex().then((records) => { // Loading and failure remain explicit; no source-specific fallback corpus. state.benchmarkIndex = records; diff --git a/site/assets/styles.css b/site/assets/styles.css index 1e627311..3214405e 100644 --- a/site/assets/styles.css +++ b/site/assets/styles.css @@ -2795,6 +2795,14 @@ td a { font-size: 0.78rem; } +.benchmark-model-label { + display: block; + margin: 0.85rem 0 0.4rem; + font-family: var(--utility); + font-size: 0.78rem; + font-weight: 700; +} + .benchmark-navigator p:not(.eyebrow):not(.benchmark-search-status) { color: var(--muted); font-size: 0.8rem; diff --git a/site/index.html b/site/index.html index 8ccd5def..de883091 100644 --- a/site/index.html +++ b/site/index.html @@ -768,6 +768,13 @@

class="benchmark-search-input" data-i18n-placeholder="Search benchmarks, tasks, domains…" placeholder="Search benchmarks, tasks, domains…" /> + +

diff --git a/src/benchmark_radar/catalog.py b/src/benchmark_radar/catalog.py index 762dc18c..923ef77b 100644 --- a/src/benchmark_radar/catalog.py +++ b/src/benchmark_radar/catalog.py @@ -675,6 +675,7 @@ def write_catalog( def build_benchmark_index( records: list[dict[str, Any]], series_by_key: dict[str, dict[str, Any]] | None = None, + observations: list[dict[str, Any]] | None = None, ) -> list[dict[str, Any]]: """The small search payload: one entry per source record, never per merge. @@ -684,6 +685,21 @@ def build_benchmark_index( it is invisible. One row per record keeps a bad grouping a display bug. """ series_by_key = series_by_key or {} + models_by_key: dict[str, dict[str, dict[str, Any]]] = {} + for observation in observations or []: + name = observation.get("model_name") + if not name or not isinstance(observation.get("value"), (int, float)): + continue + models = models_by_key.setdefault(observation["key"], {}) + # One model name per source benchmark. Keep the most recent recorded + # date and its precision; a model release is not an evaluation update. + current = models.get(name) + if current is None or (observation.get("reported_date") or "") > (current["date"] or ""): + models[name] = { + "name": name, + "date": observation.get("reported_date"), + "date_precision": observation.get("date_precision"), + } index: list[dict[str, Any]] = [] for record in records: openness = record.get("openness") or {} @@ -735,6 +751,10 @@ def build_benchmark_index( "openness": openness.get("status", "unknown"), "modality": record.get("modality"), "score_count": series.get("observation_count", 0), + "scored_models": sorted( + models_by_key.get(record["key"], {}).values(), + key=lambda model: model["name"].lower(), + ), "score_summary": ( series.get("score_summary") if series.get("observation_count", 0) else None ), diff --git a/src/benchmark_radar/cli.py b/src/benchmark_radar/cli.py index 93e3e0ef..955908fc 100644 --- a/src/benchmark_radar/cli.py +++ b/src/benchmark_radar/cli.py @@ -449,7 +449,7 @@ def main() -> None: # One index over every source, one row per source record. Two sources # describing the same benchmark stay two rows until identity.yml says # otherwise under human review. - index = build_benchmark_index(resolved_records, series_by_key) + index = build_benchmark_index(resolved_records, series_by_key, all_observations) index_path = write_benchmark_index( index, Path("site/data/benchmark-index.json"), diff --git a/tests/score_filters_harness.mjs b/tests/score_filters_harness.mjs index f37b8ff6..ba326e08 100644 --- a/tests/score_filters_harness.mjs +++ b/tests/score_filters_harness.mjs @@ -49,10 +49,23 @@ state.benchmarkIndex = [ {slug:'missing',name:'Missing',source:'llm_stats',score_summary:summary(null,0)}, ]; state.lscore = 70; -const scoreNames = ['scoreRecord','matchesScoreFilter','scoreBrowseRows','saturationRows','scoreRankingRows','benchmarkQueryIds','searchBenchmarkIndex','foldName','frontierDefaultEntry','catalogDisplayFactor']; +const scoreNames = ['scoreRecord','matchesScoreFilter','scoreBrowseRows','saturationRows','latestMatchingModel','scoreRankingRows','benchmarkQueryIds','searchBenchmarkIndex','foldName','frontierDefaultEntry','catalogDisplayFactor']; state.benchmarkQuery = ''; const scores = new Function('state', 'scorePopulation', 'matchesScoreCutoff', 'SKYLINE_START_DATE', `${scoreNames.map(fn).join('\n')}\nreturn {${scoreNames.join(',')}};`)(state, scorePopulation, matchesScoreCutoff, SKYLINE_START_DATE); assert.deepEqual(scores.scoreBrowseRows().map(r=>r.id), ['external','curated','missing']); +state.benchmarkIndex[0].scored_models = [{name:'GPT-5.6 Sol',date:'2026-06-01',date_precision:'document_publication'}]; +state.benchmarkIndex[1].scored_models = [{name:'GPT-5.6 Sol (high)',date:'2026-09-01',date_precision:'model_announcement'}]; +state.benchmarkModel = 'gpt-5.6-sol'; +assert.deepEqual(scores.saturationRows().map(r=>r.id), ['external','curated'], 'model filtering searches every source beyond the score cutoff and orders by recorded date'); +state.benchmarkModel = 'gpt-6-sol'; +assert.deepEqual(scores.saturationRows(), [], 'an unrecorded model does not imply a benchmark result'); +for (const query of ['-', '---', '()']) { + state.benchmarkModel = query; + assert.deepEqual(scores.saturationRows(), [], 'punctuation-only queries have no model match'); +} +assert.equal(state.lscore, 70, 'unmatched model lookup preserves the cutoff'); +state.benchmarkModel = ''; +assert.deepEqual(scores.saturationRows().map(r=>r.id), ['external','curated','missing'], 'clearing the model restores cutoff browsing'); assert.equal(scores.frontierDefaultEntry(state.data.model_card_leaderboard).id,'external'); assert.equal(scores.matchesScoreFilter(summary(69.999)),true); assert.equal(scores.matchesScoreFilter(summary(70)),false); diff --git a/tests/test_catalog.py b/tests/test_catalog.py index 3379e9bd..69421279 100644 --- a/tests/test_catalog.py +++ b/tests/test_catalog.py @@ -71,6 +71,22 @@ def test_search_index_carries_semantic_source_fields_without_inference(normalize assert isinstance(row["languages"], list) +def test_index_model_filter_evidence_is_scoped_to_its_source_record(normalized: dict) -> None: + from benchmark_radar.catalog import build_benchmark_index + + observations = normalized["score_observations"] + index = build_benchmark_index(normalized["source_records"], observations=observations) + by_key = {record["key"]: record for record in index} + for key, record in by_key.items(): + source_rows = [row for row in observations if row["key"] == key] + assert {model["name"] for model in record["scored_models"]} == { + row["model_name"] for row in source_rows + } + assert all( + model["date_precision"] == "model_announcement" for model in record["scored_models"] + ) + + def test_obs_id_is_unique(normalized: dict) -> None: """Without this a rerun silently duplicates every score row.""" obs_ids = [row["obs_id"] for row in normalized["score_observations"]] @@ -509,7 +525,14 @@ def test_index_has_one_row_per_source_record(normalized: dict) -> None: from benchmark_radar.catalog_opencompass import normalize_opencompass records = normalized["source_records"] + normalize_opencompass()["source_records"] - index = build_benchmark_index(records, {row["key"]: row for row in normalized["score_series"]}) + series_by_key = {row["key"]: row for row in normalized["score_series"]} + unscored_key = next( + record["key"] for record in records if record["source"] == "opencompass_hub" + ) + series_by_key[unscored_key] = {"observation_count": 0, "score_summary": {"max": 99}} + index = build_benchmark_index( + records, series_by_key, observations=normalized["score_observations"] + ) assert len(index) == 1148 assert len({row["key"] for row in index}) == 1148 assert len({row["slug"] for row in index}) == 1148 @@ -526,6 +549,8 @@ def test_index_has_one_row_per_source_record(normalized: dict) -> None: assert unscored["first_score_source_reference"] is None assert unscored["first_score_record"] is None assert unscored["score_summary"] is None + assert unscored["scored_models"] == [] + assert any(row["scored_models"] for row in index) scored = [row for row in index if row["first_score_record"]] assert len(scored) > 600, "the index must preserve dates for the whole scored catalog" for row in scored: diff --git a/tests/test_site.py b/tests/test_site.py index 94e65889..27c4f967 100644 --- a/tests/test_site.py +++ b/tests/test_site.py @@ -432,7 +432,7 @@ def test_each_view_serializes_only_the_filters_it_reads(): 1 ].split('if (!utility && state.view === "saturation")', 1) assert 'params.set("lfrontier"' not in leaderboard - for key in ("lscore", "bq", "lfrontier"): + for key in ("lscore", "bq", "bmodel", "lfrontier"): assert f'params.set("{key}"' in saturation for key in ("lq", "ldomain", "lorg", "lera", "lscore"): assert f'params.set("{key}"' in leaderboard @@ -658,12 +658,12 @@ def section(start: str, end: str) -> str: results.legacyBenchmarks.push(window.location.pathname + window.location.search); }} -install("/saturation/?lscore=40&bq=bench-95&lfrontier=bench-95&lq=agent&lheight=documents"); +install("/saturation/?lscore=40&bq=bench-95&bmodel=GPT-6+Sol&lfrontier=bench-95&lq=agent&lheight=documents"); readUrl(); writeUrl("replace"); results.saturation = {{ url: window.location.pathname + window.location.search, - view: state.view, query: state.benchmarkQuery, cutoff: state.lscore, + view: state.view, query: state.benchmarkQuery, model: state.benchmarkModel, cutoff: state.lscore, selected: state.lfrontier, explicit: state.lfrontierExplicit, }}; state.view = "leaderboard"; @@ -671,7 +671,10 @@ def section(start: str, end: str) -> str: results.sharedCutoff = window.location.pathname + window.location.search; state.view = "saturation"; state.benchmarkQuery = ""; +state.benchmarkModel = ""; writeUrl("replace"); +readUrl(); +results.clearedModel = state.benchmarkModel; results.clearedSearch = window.location.pathname + window.location.search; install("/#rubric=2"); @@ -733,14 +736,16 @@ def section(start: str, end: str) -> str: * 2 ) assert routes["saturation"] == { - "url": "/saturation/?lscore=40&bq=bench-95&lfrontier=bench-95", + "url": "/saturation/?lscore=40&bq=bench-95&bmodel=GPT-6+Sol&lfrontier=bench-95", "view": "saturation", "query": "bench-95", + "model": "GPT-6 Sol", "cutoff": 40, "selected": "bench-95", "explicit": True, } assert routes["sharedCutoff"] == "/leaderboard/?lscore=40&lheight=documents&lq=agent" + assert routes["clearedModel"] == "" assert routes["clearedSearch"] == "/saturation/?lscore=40&lfrontier=bench-95" assert routes["legacyRubric"] == "/rubric/?version=2" assert routes["legacyReturns"] is False @@ -3271,3 +3276,50 @@ def test_blog_styles_cannot_reach_the_dashboard(): assert selectors assert all(selector.startswith(".blog-page") for selector in selectors) assert "blog-page" not in Path("site/index.html").read_text(encoding="utf-8") + + +def test_model_filter_controls_and_status_have_chinese_translations(): + import json + import shutil + import subprocess + + import pytest + + node = shutil.which("node") + if not node: + pytest.skip("node is not installed") + script = Path("site/assets/app.js").read_text(encoding="utf-8") + html = Path("site/index.html").read_text(encoding="utf-8") + keys = [ + "Model with a reported score", + "Filter by model, e.g. GPT-6 Sol", + "Model release", + "Document published", + "Searching all benchmarks (score cutoff paused)", + ] + assert f'data-i18n="{keys[0]}"' in html + assert f'data-i18n-placeholder="{keys[1]}"' in html + start = script.index("const I18N = {") + dictionary = script[start : script.index("\n};", start) + 3] + start = script.index("function t(") + translate = script[start : script.index("\n}\n", start) + 2] + program = ( + dictionary + + translate + + "\nlet lang = 'zh'; function getLang() { return lang; }\n" + + f"const keys = {json.dumps(keys)};\n" + + "const zh = keys.map(key => t(key)); lang = 'en';\n" + + "console.log(JSON.stringify({zh, en: keys.map(key => t(key))}));" + ) + result = subprocess.run( + [node, "-e", program], capture_output=True, text=True, timeout=60, check=True + ) + translated = json.loads(result.stdout) + assert translated["en"] == keys + assert translated["zh"] == [ + "已有报告成绩的模型", + "按模型筛选,例如 GPT-6 Sol", + "模型发布日期", + "文档发布日期", + "搜索全部 benchmark(暂不按分数筛选)", + ]