Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
42 changes: 38 additions & 4 deletions site/assets/app.js
Original file line number Diff line number Diff line change
Expand Up @@ -374,6 +374,11 @@ const I18N = {
"Shown / corpus": "显示数/目录总数",
"source document": "份来源文档",
"Document publication date": "文档发布日期",
"Model release": "模型发布日期",
"Document published": "文档发布日期",
"Model with a reported score": "已有报告成绩的模型",
"Filter by model, e.g. GPT-6 Sol": "按模型筛选,例如 GPT-6 Sol",
"Searching all benchmarks (score cutoff paused)": "搜索全部 benchmark(暂不按分数筛选)",
"Reporting organization color key": "报告机构颜色图例",
"Distinct cited documents, including model reports and registry pages": "按引用的独立文档去重,包括模型报告和登记页面",
"Each source document counts once per benchmark record.": "每份来源文档对同一条 benchmark 记录只计一次。",
Expand Down Expand Up @@ -1228,6 +1233,7 @@ const state = {
// arrived, so "not yet loaded" and "failed to load" must not look alike.
benchmarkIndexLoaded: false,
benchmarkQuery: "",
benchmarkModel: "",
leaderboardShowAll: false,
leaderboardTopExpanded: false,
todayResultsKey: "",
Expand Down Expand Up @@ -1466,6 +1472,7 @@ function readUrl() {
state.lheight = ["cards", "documents"].includes(params.get("lheight")) ? "documents" : "models";
state.benchmarkVisibleLimit = BENCHMARK_SEARCH_LIMIT;
state.benchmarkQuery = (params.get("bq") || "").trim();
state.benchmarkModel = (params.get("bmodel") || "").trim();
state.lfrontier = params.get("lfrontier") || "";
state.lfrontierExplicit = Boolean(state.lfrontier);
// Existing benchmark permalinks follow the score history to its new tab.
Expand Down Expand Up @@ -1549,6 +1556,7 @@ function writeUrl(mode = "replace") {
if (!utility && state.view === "saturation") {
params.set("lscore", state.lscore);
if (state.benchmarkQuery) params.set("bq", state.benchmarkQuery);
if (state.benchmarkModel) params.set("bmodel", state.benchmarkModel);
Comment thread
maxffarrell marked this conversation as resolved.
// A benchmark auto-picked as the default is not the reader's choice, so it
// stays out of the URL until they select one themselves.
if (state.lfrontierExplicit && state.lfrontier) {
Expand Down Expand Up @@ -4704,8 +4712,22 @@ function saturationRows() {
const matches = benchmarkQueryIds();
// Search is a lookup across the whole catalog. It temporarily bypasses the
// browsing cutoff without changing the shared slider value.
const rows = scoreBrowseRows(matches ? 100 : state.lscore);
return matches ? rows.filter((row) => matches.has(row.id)) : rows;
const rows = scoreBrowseRows(matches || state.benchmarkModel ? 100 : state.lscore);
const named = matches ? rows.filter((row) => matches.has(row.id)) : rows;
if (!state.benchmarkModel) return named;
const needle = foldName(state.benchmarkModel);
if (!needle) return [];
const index = new Map((state.benchmarkIndex || []).map((record) => [record.slug, record]));
return named.filter((row) => (index.get(row.id)?.scored_models || [])
.some((model) => foldName(model.name).includes(needle)))
.sort((a, b) => (latestMatchingModel(index.get(b.id), needle)?.date || "")
.localeCompare(latestMatchingModel(index.get(a.id), needle)?.date || "") || a.name.localeCompare(b.name));
}

function latestMatchingModel(record, needle) {
return (record?.scored_models || [])
.filter((model) => foldName(model.name).includes(needle))
.sort((a, b) => (b.date || "").localeCompare(a.date || ""))[0];
}

function scoreRankingRows() {
Expand All @@ -4722,6 +4744,9 @@ function scoreSummaryLabel(summary) {
}

function scoreBrowseResultRow(row) {
const record = (state.benchmarkIndex || []).find((item) => item.slug === row.id);
const needle = foldName(state.benchmarkModel);
const matchedModel = needle ? latestMatchingModel(record, needle) : null;
const button = element("button", {
className: "benchmark-result score-browse-result",
attrs: { type: "button", "aria-pressed": row.id === state.lfrontier ? "true" : "false" },
Expand All @@ -4730,6 +4755,7 @@ function scoreBrowseResultRow(row) {
element("span", { className: "benchmark-result-facts", text: [scoreSourceLabel(row.source),
row.date ? `${t(benchmarkDateLabel(row))} ${formatDate(row.date, { dateStyle: "medium" })}` : t("Date unknown"),
matchesScoreCutoff(row.summary, 100) ? "" : t("No score reported"),
matchedModel ? `${matchedModel.name}${matchedModel.date ? ` · ${t(matchedModel.date_precision === "model_announcement" ? "Model release" : "Document published")} ${formatDate(matchedModel.date, { dateStyle: "medium" })}` : ""}` : "",
Comment thread
maxffarrell marked this conversation as resolved.
].filter(Boolean).join(" · ") }),
]);
button.addEventListener("click", () => {
Expand Down Expand Up @@ -4814,6 +4840,7 @@ function renderBenchmarkSearch() {
const status = byId("benchmark-search-status");
if (!container || !status || !state.data) return;
byId("benchmark-search-input").value = state.benchmarkQuery;
byId("benchmark-model-input").value = state.benchmarkModel;
const rows = saturationRows();
const shown = rows.slice(0, state.benchmarkVisibleLimit);
replaceChildren(container, shown.map(scoreBrowseResultRow));
Expand All @@ -4824,11 +4851,11 @@ function renderBenchmarkSearch() {
status.textContent = [t("{shown} of {total} matches")
.replace("{shown}", shown.length.toLocaleString())
.replace("{total}", rows.length.toLocaleString()),
state.benchmarkQuery ? t("Searching all benchmarks (filters paused)") : "", coverage].filter(Boolean).join(" · ");
state.benchmarkQuery || state.benchmarkModel ? t("Searching all benchmarks (score cutoff paused)") : "", coverage].filter(Boolean).join(" · ");
Comment thread
maxffarrell marked this conversation as resolved.
if (!rows.length) {
container.append(element("p", { className: "empty-state", text: loading
? t("Loading benchmark details…") : t("No benchmarks match these filters.") }));
if (!state.benchmarkQuery && state.lscore < 100) {
if (!state.benchmarkQuery && !state.benchmarkModel && state.lscore < 100) {
const all = element("button", { className: "clear-button", text: t("All"), attrs: { type: "button" } });
all.addEventListener("click", () => setScoreFilter(100));
container.append(all);
Expand Down Expand Up @@ -5266,6 +5293,13 @@ function initBenchmarkSearch() {
writeUrl();
});
input.addEventListener("input", onInput);
const modelInput = byId("benchmark-model-input");
modelInput.addEventListener("input", debounce(() => {
state.benchmarkModel = modelInput.value.trim();
state.benchmarkVisibleLimit = BENCHMARK_SEARCH_LIMIT;
renderSaturation();
writeUrl();
}));
loadBenchmarkIndex().then((records) => {
// Loading and failure remain explicit; no source-specific fallback corpus.
state.benchmarkIndex = records;
Expand Down
8 changes: 8 additions & 0 deletions site/assets/styles.css
Original file line number Diff line number Diff line change
Expand Up @@ -2795,6 +2795,14 @@ td a {
font-size: 0.78rem;
}

.benchmark-model-label {
display: block;
margin: 0.85rem 0 0.4rem;
font-family: var(--utility);
font-size: 0.78rem;
font-weight: 700;
}

.benchmark-navigator p:not(.eyebrow):not(.benchmark-search-status) {
color: var(--muted);
font-size: 0.8rem;
Expand Down
7 changes: 7 additions & 0 deletions site/index.html
Original file line number Diff line number Diff line change
Expand Up @@ -768,6 +768,13 @@ <h2 class="benchmark-search-label" id="benchmark-search-heading">
class="benchmark-search-input"
data-i18n-placeholder="Search benchmarks, tasks, domains…"
placeholder="Search benchmarks, tasks, domains…" />
<label class="benchmark-model-label" for="benchmark-model-input"
data-i18n="Model with a reported score">Model with a reported score</label>
<input id="benchmark-model-input" type="search" autocomplete="off"
class="benchmark-search-input benchmark-model-input"
data-i18n-placeholder="Filter by model, e.g. GPT-6 Sol"
placeholder="Filter by model, e.g. GPT-6 Sol"
aria-controls="benchmark-search-results frontier-benchmark" />
<p id="benchmark-search-status" class="benchmark-search-status"></p>
<div id="benchmark-search-results"></div>
<button id="benchmark-search-more" class="clear-button" type="button" data-i18n="Show more" hidden>Show more</button>
Expand Down
20 changes: 20 additions & 0 deletions src/benchmark_radar/catalog.py
Original file line number Diff line number Diff line change
Expand Up @@ -675,6 +675,7 @@ def write_catalog(
def build_benchmark_index(
records: list[dict[str, Any]],
series_by_key: dict[str, dict[str, Any]] | None = None,
observations: list[dict[str, Any]] | None = None,
) -> list[dict[str, Any]]:
"""The small search payload: one entry per source record, never per merge.

Expand All @@ -684,6 +685,21 @@ def build_benchmark_index(
it is invisible. One row per record keeps a bad grouping a display bug.
"""
series_by_key = series_by_key or {}
models_by_key: dict[str, dict[str, dict[str, Any]]] = {}
for observation in observations or []:
name = observation.get("model_name")
if not name or not isinstance(observation.get("value"), (int, float)):
continue
models = models_by_key.setdefault(observation["key"], {})
# One model name per source benchmark. Keep the most recent recorded
# date and its precision; a model release is not an evaluation update.
current = models.get(name)
if current is None or (observation.get("reported_date") or "") > (current["date"] or ""):
models[name] = {
"name": name,
"date": observation.get("reported_date"),
"date_precision": observation.get("date_precision"),
}
index: list[dict[str, Any]] = []
for record in records:
openness = record.get("openness") or {}
Expand Down Expand Up @@ -735,6 +751,10 @@ def build_benchmark_index(
"openness": openness.get("status", "unknown"),
"modality": record.get("modality"),
"score_count": series.get("observation_count", 0),
"scored_models": sorted(
models_by_key.get(record["key"], {}).values(),
key=lambda model: model["name"].lower(),
),
"score_summary": (
series.get("score_summary") if series.get("observation_count", 0) else None
),
Expand Down
2 changes: 1 addition & 1 deletion src/benchmark_radar/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -449,7 +449,7 @@ def main() -> None:
# One index over every source, one row per source record. Two sources
# describing the same benchmark stay two rows until identity.yml says
# otherwise under human review.
index = build_benchmark_index(resolved_records, series_by_key)
index = build_benchmark_index(resolved_records, series_by_key, all_observations)
index_path = write_benchmark_index(
index,
Path("site/data/benchmark-index.json"),
Expand Down
15 changes: 14 additions & 1 deletion tests/score_filters_harness.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -49,10 +49,23 @@ state.benchmarkIndex = [
{slug:'missing',name:'Missing',source:'llm_stats',score_summary:summary(null,0)},
];
state.lscore = 70;
const scoreNames = ['scoreRecord','matchesScoreFilter','scoreBrowseRows','saturationRows','scoreRankingRows','benchmarkQueryIds','searchBenchmarkIndex','foldName','frontierDefaultEntry','catalogDisplayFactor'];
const scoreNames = ['scoreRecord','matchesScoreFilter','scoreBrowseRows','saturationRows','latestMatchingModel','scoreRankingRows','benchmarkQueryIds','searchBenchmarkIndex','foldName','frontierDefaultEntry','catalogDisplayFactor'];
state.benchmarkQuery = '';
const scores = new Function('state', 'scorePopulation', 'matchesScoreCutoff', 'SKYLINE_START_DATE', `${scoreNames.map(fn).join('\n')}\nreturn {${scoreNames.join(',')}};`)(state, scorePopulation, matchesScoreCutoff, SKYLINE_START_DATE);
assert.deepEqual(scores.scoreBrowseRows().map(r=>r.id), ['external','curated','missing']);
state.benchmarkIndex[0].scored_models = [{name:'GPT-5.6 Sol',date:'2026-06-01',date_precision:'document_publication'}];
state.benchmarkIndex[1].scored_models = [{name:'GPT-5.6 Sol (high)',date:'2026-09-01',date_precision:'model_announcement'}];
state.benchmarkModel = 'gpt-5.6-sol';
assert.deepEqual(scores.saturationRows().map(r=>r.id), ['external','curated'], 'model filtering searches every source beyond the score cutoff and orders by recorded date');
state.benchmarkModel = 'gpt-6-sol';
assert.deepEqual(scores.saturationRows(), [], 'an unrecorded model does not imply a benchmark result');
for (const query of ['-', '---', '()']) {
state.benchmarkModel = query;
assert.deepEqual(scores.saturationRows(), [], 'punctuation-only queries have no model match');
}
assert.equal(state.lscore, 70, 'unmatched model lookup preserves the cutoff');
state.benchmarkModel = '';
assert.deepEqual(scores.saturationRows().map(r=>r.id), ['external','curated','missing'], 'clearing the model restores cutoff browsing');
assert.equal(scores.frontierDefaultEntry(state.data.model_card_leaderboard).id,'external');
assert.equal(scores.matchesScoreFilter(summary(69.999)),true);
assert.equal(scores.matchesScoreFilter(summary(70)),false);
Expand Down
27 changes: 26 additions & 1 deletion tests/test_catalog.py
Original file line number Diff line number Diff line change
Expand Up @@ -71,6 +71,22 @@ def test_search_index_carries_semantic_source_fields_without_inference(normalize
assert isinstance(row["languages"], list)


def test_index_model_filter_evidence_is_scoped_to_its_source_record(normalized: dict) -> None:
from benchmark_radar.catalog import build_benchmark_index

observations = normalized["score_observations"]
index = build_benchmark_index(normalized["source_records"], observations=observations)
by_key = {record["key"]: record for record in index}
for key, record in by_key.items():
source_rows = [row for row in observations if row["key"] == key]
assert {model["name"] for model in record["scored_models"]} == {
row["model_name"] for row in source_rows
}
assert all(
model["date_precision"] == "model_announcement" for model in record["scored_models"]
)


def test_obs_id_is_unique(normalized: dict) -> None:
"""Without this a rerun silently duplicates every score row."""
obs_ids = [row["obs_id"] for row in normalized["score_observations"]]
Expand Down Expand Up @@ -509,7 +525,14 @@ def test_index_has_one_row_per_source_record(normalized: dict) -> None:
from benchmark_radar.catalog_opencompass import normalize_opencompass

records = normalized["source_records"] + normalize_opencompass()["source_records"]
index = build_benchmark_index(records, {row["key"]: row for row in normalized["score_series"]})
series_by_key = {row["key"]: row for row in normalized["score_series"]}
unscored_key = next(
record["key"] for record in records if record["source"] == "opencompass_hub"
)
series_by_key[unscored_key] = {"observation_count": 0, "score_summary": {"max": 99}}
index = build_benchmark_index(
records, series_by_key, observations=normalized["score_observations"]
)
assert len(index) == 1148
assert len({row["key"] for row in index}) == 1148
assert len({row["slug"] for row in index}) == 1148
Expand All @@ -526,6 +549,8 @@ def test_index_has_one_row_per_source_record(normalized: dict) -> None:
assert unscored["first_score_source_reference"] is None
assert unscored["first_score_record"] is None
assert unscored["score_summary"] is None
assert unscored["scored_models"] == []
assert any(row["scored_models"] for row in index)
scored = [row for row in index if row["first_score_record"]]
assert len(scored) > 600, "the index must preserve dates for the whole scored catalog"
for row in scored:
Expand Down
60 changes: 56 additions & 4 deletions tests/test_site.py
Original file line number Diff line number Diff line change
Expand Up @@ -432,7 +432,7 @@ def test_each_view_serializes_only_the_filters_it_reads():
1
].split('if (!utility && state.view === "saturation")', 1)
assert 'params.set("lfrontier"' not in leaderboard
for key in ("lscore", "bq", "lfrontier"):
for key in ("lscore", "bq", "bmodel", "lfrontier"):
assert f'params.set("{key}"' in saturation
for key in ("lq", "ldomain", "lorg", "lera", "lscore"):
assert f'params.set("{key}"' in leaderboard
Expand Down Expand Up @@ -658,20 +658,23 @@ def section(start: str, end: str) -> str:
results.legacyBenchmarks.push(window.location.pathname + window.location.search);
}}

install("/saturation/?lscore=40&bq=bench-95&lfrontier=bench-95&lq=agent&lheight=documents");
install("/saturation/?lscore=40&bq=bench-95&bmodel=GPT-6+Sol&lfrontier=bench-95&lq=agent&lheight=documents");
readUrl();
writeUrl("replace");
results.saturation = {{
url: window.location.pathname + window.location.search,
view: state.view, query: state.benchmarkQuery, cutoff: state.lscore,
view: state.view, query: state.benchmarkQuery, model: state.benchmarkModel, cutoff: state.lscore,
selected: state.lfrontier, explicit: state.lfrontierExplicit,
}};
state.view = "leaderboard";
writeUrl("push");
results.sharedCutoff = window.location.pathname + window.location.search;
state.view = "saturation";
state.benchmarkQuery = "";
state.benchmarkModel = "";
writeUrl("replace");
readUrl();
results.clearedModel = state.benchmarkModel;
results.clearedSearch = window.location.pathname + window.location.search;

install("/#rubric=2");
Expand Down Expand Up @@ -733,14 +736,16 @@ def section(start: str, end: str) -> str:
* 2
)
assert routes["saturation"] == {
"url": "/saturation/?lscore=40&bq=bench-95&lfrontier=bench-95",
"url": "/saturation/?lscore=40&bq=bench-95&bmodel=GPT-6+Sol&lfrontier=bench-95",
"view": "saturation",
"query": "bench-95",
"model": "GPT-6 Sol",
"cutoff": 40,
"selected": "bench-95",
"explicit": True,
}
assert routes["sharedCutoff"] == "/leaderboard/?lscore=40&lheight=documents&lq=agent"
assert routes["clearedModel"] == ""
assert routes["clearedSearch"] == "/saturation/?lscore=40&lfrontier=bench-95"
assert routes["legacyRubric"] == "/rubric/?version=2"
assert routes["legacyReturns"] is False
Expand Down Expand Up @@ -3271,3 +3276,50 @@ def test_blog_styles_cannot_reach_the_dashboard():
assert selectors
assert all(selector.startswith(".blog-page") for selector in selectors)
assert "blog-page" not in Path("site/index.html").read_text(encoding="utf-8")


def test_model_filter_controls_and_status_have_chinese_translations():
import json
import shutil
import subprocess

import pytest

node = shutil.which("node")
if not node:
pytest.skip("node is not installed")
script = Path("site/assets/app.js").read_text(encoding="utf-8")
html = Path("site/index.html").read_text(encoding="utf-8")
keys = [
"Model with a reported score",
"Filter by model, e.g. GPT-6 Sol",
"Model release",
"Document published",
"Searching all benchmarks (score cutoff paused)",
]
assert f'data-i18n="{keys[0]}"' in html
assert f'data-i18n-placeholder="{keys[1]}"' in html
start = script.index("const I18N = {")
dictionary = script[start : script.index("\n};", start) + 3]
start = script.index("function t(")
translate = script[start : script.index("\n}\n", start) + 2]
program = (
dictionary
+ translate
+ "\nlet lang = 'zh'; function getLang() { return lang; }\n"
+ f"const keys = {json.dumps(keys)};\n"
+ "const zh = keys.map(key => t(key)); lang = 'en';\n"
+ "console.log(JSON.stringify({zh, en: keys.map(key => t(key))}));"
)
result = subprocess.run(
[node, "-e", program], capture_output=True, text=True, timeout=60, check=True
)
translated = json.loads(result.stdout)
assert translated["en"] == keys
assert translated["zh"] == [
"已有报告成绩的模型",
"按模型筛选,例如 GPT-6 Sol",
"模型发布日期",
"文档发布日期",
"搜索全部 benchmark(暂不按分数筛选)",
]
Loading