From 14f594c7eb550a2d16666587c653abb05ac1bdfb Mon Sep 17 00:00:00 2001 From: Note Flow AI Date: Wed, 16 Sep 2026 01:16:01 +0800 Subject: [PATCH] Use explicit lab filenames on static hosts; release 0.11.1 --- CHANGELOG.md | 5 +++++ README.md | 4 ++-- README.zh-CN.md | 4 ++-- package-lock.json | 4 ++-- package.json | 2 +- pyproject.toml | 2 +- scripts/build_site.py | 28 ++++++++++++++++++++++++++++ scripts/check_site.cjs | 23 +++++++++++++++++------ site/index.html | 4 ++-- site/labs/research.html | 2 +- site/labs/skill-impact.html | 2 +- src/evalarc/__init__.py | 2 +- tests/test_site.py | 18 ++++++++++++++++++ 13 files changed, 81 insertions(+), 19 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 30ef232..d77d4a6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,10 @@ # Changelog +## 0.11.1 — 2026-09-15 + +- Use explicit HTML filenames for published lab navigation. Hugging Face redirects bare directory paths to Hub routes rather than serving each directory index. Cover homepage entries and research/skill-lab return links. +- Reseal only the published lab presentation files; recorded trials, raw JSON and original research archives stay byte-identical. Test navigation with a static server that rejects implicit directory indexes. + ## 0.11.0 — 2026-09-15 - Add offline AgentCore Evaluate import, golden-case/rubric comparison and recomputable trace review with responsive standalone reports. diff --git a/README.md b/README.md index c2f8110..43a029e 100644 --- a/README.md +++ b/README.md @@ -117,11 +117,11 @@ also requires all checks to pass. [Verification and limits](docs/verification.md ## Trace Workbench — 0.11.0 -Import saved AgentCore Evaluate responses, versioned golden cases and Skills Anywhere delivery receipts. Inspect zero scores, skipped judges, missing results and missed skills separately. Compare matching datasets/rubrics and verify preserved input bytes offline. [Try the authored controls](https://noteflowai.github.io/evalarc/trace-workbench/) · [Actual local MCP delivery](https://noteflowai.github.io/evalarc/trace-mcp/) · [Input contract](docs/trace-workbench.md). No live AWS evaluation is claimed. +Import saved AgentCore Evaluate responses, versioned golden cases and Skills Anywhere delivery receipts. Inspect zero scores, skipped judges, missing results and missed skills separately. Compare matching datasets/rubrics and verify preserved input bytes offline. [Try the authored controls](https://noteflowai.github.io/evalarc/trace-workbench/index.html) · [Actual local MCP delivery](https://noteflowai.github.io/evalarc/trace-mcp/index.html) · [Input contract](docs/trace-workbench.md). No live AWS evaluation is claimed. ## New in 0.9.0: research you can inspect -[Explore all 27 real GPU skill trials](https://noteflowai.github.io/evalarc/skill-impact/) and [the research pilots](docs/research-pilots.md). Robot Reel's [captured-scene editor](https://noteflowai.github.io/robot-reel/scene-lab/) and [official LIBERO-Plus replay](https://noteflowai.github.io/robot-reel/libero-plus/) connect real source records with portable skill delivery and independent grading. Every failed attempt stays visible; no skill efficacy, full-benchmark or real-hardware result is implied. +[Explore all 27 real GPU skill trials](https://noteflowai.github.io/evalarc/skill-impact/index.html) and [the research pilots](docs/research-pilots.md). Robot Reel's [captured-scene editor](https://noteflowai.github.io/robot-reel/scene-lab/) and [official LIBERO-Plus replay](https://noteflowai.github.io/robot-reel/libero-plus/) connect real source records with portable skill delivery and independent grading. Every failed attempt stays visible; no skill efficacy, full-benchmark or real-hardware result is implied. ## Run an audit diff --git a/README.zh-CN.md b/README.zh-CN.md index 4a9f591..4165595 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -100,11 +100,11 @@ Docker 评测:参考策略 3/3 轮完全通过,重复写入策略虽然平 ## 0.11.0:运行记录评估工作台 -导入已保存的 AgentCore Evaluate 结果、版本化黄金案例和 Skills Anywhere 加载回执,分别查看有效零分、评估跳过、缺少结果及漏调用技能。支持相同测试集与评分规则下的对比,以及原始输入的离线复核。[交互示例](https://noteflowai.github.io/evalarc/trace-workbench/) · [真实本地 MCP 加载](https://noteflowai.github.io/evalarc/trace-mcp/) · [数据契约](docs/trace-workbench.md)。示例明确区分合成评分与真实加载记录,未运行云端评估。 +导入已保存的 AgentCore Evaluate 结果、版本化黄金案例和 Skills Anywhere 加载回执,分别查看有效零分、评估跳过、缺少结果及漏调用技能。支持相同测试集与评分规则下的对比,以及原始输入的离线复核。[交互示例](https://noteflowai.github.io/evalarc/trace-workbench/index.html) · [真实本地 MCP 加载](https://noteflowai.github.io/evalarc/trace-mcp/index.html) · [数据契约](docs/trace-workbench.md)。示例明确区分合成评分与真实加载记录,未运行云端评估。 ## 0.9.0:有原始证据的研究场景 -[查看 27 次真实 GPU 技能评测](https://noteflowai.github.io/evalarc/skill-impact/),并阅读[完整方法与限制](docs/research-pilots.md)。新增[实景 Blender 编辑](https://noteflowai.github.io/robot-reel/scene-lab/)与[官方 LIBERO-Plus 子集回放](https://noteflowai.github.io/robot-reel/libero-plus/),把原始记录、技能交付与独立验收连接起来。失败尝试全部保留;不宣称技能提分、完整基准成绩或真机效果。 +[查看 27 次真实 GPU 技能评测](https://noteflowai.github.io/evalarc/skill-impact/index.html),并阅读[完整方法与限制](docs/research-pilots.md)。新增[实景 Blender 编辑](https://noteflowai.github.io/robot-reel/scene-lab/)与[官方 LIBERO-Plus 子集回放](https://noteflowai.github.io/robot-reel/libero-plus/),把原始记录、技能交付与独立验收连接起来。失败尝试全部保留;不宣称技能提分、完整基准成绩或真机效果。 ## 直接运行 diff --git a/package-lock.json b/package-lock.json index 2d5dadc..f9dd47a 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "evalarc-evidence-site", - "version": "0.11.0", + "version": "0.11.1", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "evalarc-evidence-site", - "version": "0.11.0", + "version": "0.11.1", "devDependencies": { "playwright": "1.63.0" } diff --git a/package.json b/package.json index b697ebe..7b76f1e 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "evalarc-evidence-site", - "version": "0.11.0", + "version": "0.11.1", "private": true, "description": "Browser checks for the static EvalArc evidence explorer", "scripts": { diff --git a/pyproject.toml b/pyproject.toml index a384d7e..9b2aedd 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "evalarc" -version = "0.11.0" +version = "0.11.1" description = "Auditable task environments and evaluations for coding and tool-using agents." readme = "README.md" requires-python = ">=3.11" diff --git a/scripts/build_site.py b/scripts/build_site.py index 0f992a0..94b2e16 100644 --- a/scripts/build_site.py +++ b/scripts/build_site.py @@ -225,6 +225,30 @@ def sha256(path: Path) -> str: return hashlib.sha256(path.read_bytes()).hexdigest() +def explicit_lab_navigation(folder: Path, commit: str) -> None: + """Update copied presentation links, preserving every original experiment byte.""" + page = folder / "index.html" + original = page.read_bytes() + updated = original.replace(b'href="../"', b'href="../index.html"').replace( + b'href="../skill-impact/"', b'href="../skill-impact/index.html"' + ) + if updated == original: + return + page.write_bytes(updated) + manifest_path = folder / "manifest.json" + manifest = json.loads(manifest_path.read_bytes()) + manifest["presentation"] = { + "source_commit": commit, + "original_index_sha256": hashlib.sha256(original).hexdigest(), + "change": "Explicit index.html navigation for static hosting", + } + manifest["files"]["index.html"] = { + "sha256": hashlib.sha256(updated).hexdigest(), + "bytes": len(updated), + } + manifest_path.write_text(json.dumps(manifest, indent=2) + "\n") + + def verify(folder: Path) -> dict: record = json.loads((folder / MANIFEST).read_text()) if record["source_repository"] != SOURCE or not re.fullmatch( @@ -394,6 +418,10 @@ def build(destination: Path) -> dict: verify_records(ROOT / "examples/research") shutil.copytree(ROOT / "examples/skill-impact", destination / "skill-impact") shutil.copytree(ROOT / "examples/research", destination / "research") + for lab in ("skill-impact", "research"): + explicit_lab_navigation(destination / lab, commit) + verify_lab(destination / "skill-impact") + verify_records(destination / "research") from evalarc.trace_review import import_trace trace_examples = ROOT / "examples/trace-workbench" diff --git a/scripts/check_site.cjs b/scripts/check_site.cjs index a472777..57e0b48 100644 --- a/scripts/check_site.cjs +++ b/scripts/check_site.cjs @@ -13,7 +13,12 @@ async function main() { const mime = {".html":"text/html", ".js":"text/javascript", ".json":"application/json", ".css":"text/css", ".svg":"image/svg+xml", ".zip":"application/zip"}; server = http.createServer((req, res) => { const route = decodeURIComponent(new URL(req.url, "http://localhost").pathname); - const filename = path.resolve(root, "." + (route.endsWith("/") ? route + "index.html" : route)); + // HF static hosting does not resolve subdirectory indexes. Keep that + // production constraint in the local browser fixture. + if (route !== "/" && route.endsWith("/")) { + res.writeHead(404).end(); return; + } + const filename = path.resolve(root, "." + (route === "/" ? "/index.html" : route)); if (!filename.startsWith(root + path.sep) || !fs.existsSync(filename) || !fs.statSync(filename).isFile()) { res.writeHead(404).end(); return; } @@ -226,7 +231,7 @@ async function main() { await app.getByRole("heading", {name:"support-routing", exact:true}).waitFor(); assert.match(await app.locator("body").innerText(), /retry-after-commit/); } - await app.locator("body").evaluate((element, url) => { location.href = new URL("skill-impact/", url).href; }, appUrl); + await app.locator("body").evaluate((element, url) => { location.href = new URL("skill-impact/index.html", url).href; }, appUrl); await app.locator("#workspace").waitFor({state:"visible"}); assert.equal(await app.locator(".trial").count(), 9); assert.match(await app.locator("#profile-totals").innerText(), /2\/9 fully resolved/); @@ -248,10 +253,14 @@ async function main() { fs.mkdirSync(process.env.RESEARCH_SCREENSHOTS,{recursive:true}); await page.screenshot({path:path.join(process.env.RESEARCH_SCREENSHOTS,"skill-impact.png"),fullPage:true}); } - await app.locator("body").evaluate((element, url) => { location.href = new URL("research/", url).href; }, appUrl); + await app.locator("body").evaluate((element, url) => { location.href = new URL("research/index.html", url).href; }, appUrl); await app.getByRole("heading",{name:"Skill composition: 12 attempts, 3 accepted"}).waitFor(); assert.equal(await app.locator("tbody tr").count(),18); assert.equal(await app.locator("body").evaluate(() => document.documentElement.scrollWidth > innerWidth), false); + await app.getByRole("link",{name:"27-trial Skill Impact Lab"}).click(); + await app.locator("#profile").waitFor(); + await app.getByRole("link",{name:"Playground"}).click(); + await app.locator("#workspace").waitFor({state:"visible"}); const coveragePage = await browser.newPage({viewport:{width,height:1000}}); try { await coveragePage.goto(new URL("robot/index.html#fault-3", suiteBase).href); @@ -298,7 +307,7 @@ async function main() { assert.equal(await page.locator("body").evaluate(() => document.documentElement.scrollWidth > innerWidth), false); assert.deepEqual(errors, []); failures.add("skill-impact/lab.json"); - await page.goto(new URL("skill-impact/",base).href); + await page.goto(new URL("skill-impact/index.html",base).href); await page.locator("#retry").waitFor({state:"visible"}); failures.clear(); await page.locator("#retry").click(); @@ -361,7 +370,8 @@ async function main() { const errors = []; page.on("pageerror", error => errors.push(error.message)); try { - await page.goto(new URL("trace-workbench/",base).href); + await page.goto(base); + await page.getByRole("link",{name:"Explore five authored review controls"}).click(); await page.locator("#filter-status").filter({hasText:"5 cases shown"}).waitFor(); assert.equal(await page.locator("article[data-gate=accepted]").count(),1); assert.equal(await page.locator("article[data-gate=rejected]").count(),2); @@ -387,7 +397,8 @@ async function main() { assert.equal(await download.failure(),null); const original = fs.readFileSync(path.join(root,"trace-workbench/input.json")); assert.deepEqual(fs.readFileSync(await download.path()),original); - await page.goto(new URL("trace-mcp/",base).href); + await page.goto(base); + await page.getByRole("link",{name:"Inspect an actual MCP delivery"}).click(); await page.locator("#filter-status").filter({hasText:"1 cases shown"}).waitFor(); assert.match(await page.locator("body").innerText(),/Actual local stdio MCP/); assert.match(await page.locator("article").innerText(),/MATCHED/); diff --git a/site/index.html b/site/index.html index 16cf984..6033a01 100644 --- a/site/index.html +++ b/site/index.html @@ -37,8 +37,8 @@

Look past
the score.

Generate Python or JavaScript candidates and audit independent controls against the same task contracts. Try both runtimes ↗. The recorded showcases below retain their original versions and fingerprints.

-

NEW / RECORDED MODEL EVIDENCE

A skill loaded. Did the task pass?

27 real Qwen3-8B trials compare no skill, direct loading and MCP delivery on attributed robot recordings. Inspect every tool receipt, candidate, independent score and ATIF trajectory. Three engineering profiles, including negative results.

L40S recordings; one public development task. No skill accuracy gain or general model ranking is claimed.

-

NEW / BRING YOUR AGENT RECORDS

Zero, skipped, or missing?

A zero can be a valid judgment. A skipped evaluator needs context. A required skill may never have loaded. Inspect each against a versioned golden case, with recording identity and explicit acceptance rules.

The five controls are synthetic. The separate MCP record contains a real local load and no evaluator scores. Review your own saved AgentCore Evaluate export offline with evalarc trace-import.

+

NEW / RECORDED MODEL EVIDENCE

A skill loaded. Did the task pass?

27 real Qwen3-8B trials compare no skill, direct loading and MCP delivery on attributed robot recordings. Inspect every tool receipt, candidate, independent score and ATIF trajectory. Three engineering profiles, including negative results.

L40S recordings; one public development task. No skill accuracy gain or general model ranking is claimed.

+

NEW / BRING YOUR AGENT RECORDS

Zero, skipped, or missing?

A zero can be a valid judgment. A skipped evaluator needs context. A required skill may never have loaded. Inspect each against a versioned golden case, with recording identity and explicit acceptance rules.

The five controls are synthetic. The separate MCP record contains a real local load and no evaluator scores. Review your own saved AgentCore Evaluate export offline with evalarc trace-import.

__TASK_PACK_COUNT__task packs
__FAULT_COUNT__declared faults caught
diff --git a/site/labs/research.html b/site/labs/research.html index d4e6e91..ee0d8d5 100644 --- a/site/labs/research.html +++ b/site/labs/research.html @@ -2,7 +2,7 @@ Research records · EvalArc -← EvalArc · 27-trial Skill Impact Lab +← EvalArc · 27-trial Skill Impact Lab

Keep the failed attempts.
Check the artifact.

Recorded public development experiments, 14 September 2026. Qwen3-8B/Qwen3-4B on an NVIDIA L40S. No private customer data or personal agent history.

These pilots test specific integration paths and inspectable outcomes. They do not establish skill or memory efficacy, a general leakage detector, or native commercial-agent interoperability.

diff --git a/site/labs/skill-impact.html b/site/labs/skill-impact.html index 2d3ea8b..a7d0dc1 100644 --- a/site/labs/skill-impact.html +++ b/site/labs/skill-impact.html @@ -12,7 +12,7 @@ -
← PlaygroundRobot Reel × Skills Anywhere × EvalArcMethods & limits
+
← PlaygroundRobot Reel × Skills Anywhere × EvalArcMethods & limits

Real model calls / NVIDIA L40S / Public development evidence

A skill loaded.
Did the task pass?

diff --git a/src/evalarc/__init__.py b/src/evalarc/__init__.py index 6332104..5a866cf 100644 --- a/src/evalarc/__init__.py +++ b/src/evalarc/__init__.py @@ -1,3 +1,3 @@ """Auditable evaluations for AI agents.""" -__version__ = "0.11.0" +__version__ = "0.11.1" diff --git a/tests/test_site.py b/tests/test_site.py index bd0e4d0..353e7f2 100644 --- a/tests/test_site.py +++ b/tests/test_site.py @@ -45,6 +45,24 @@ def test_bundle_rejects_changed_evidence_and_extra_files(tmp_path): assert "__AUDIT_COVERAGE__" not in (folder / "index.html").read_text() assert (folder / "index.html").read_text().count("single-case dependencies") == 3 assert (folder / "robot/index.html").is_file() + from scripts.verify_research import verify_lab, verify_records + + for name, verifier in (("skill-impact", verify_lab), ("research", verify_records)): + verifier(folder / name) + original = builder.ROOT / "examples" / name + for source in original.rglob("*"): + if source.is_file() and source.relative_to(original).as_posix() not in ( + "index.html", + "manifest.json", + ): + assert (folder / name / source.relative_to(original)).read_bytes() == ( + source.read_bytes() + ) + page = (folder / name / "index.html").read_text() + assert 'href="../index.html"' in page + assert 'href="../"' not in page + for name in ("skill-impact", "research", "trace-workbench", "trace-mcp"): + assert f'href="{name}/index.html"' in (folder / "index.html").read_text() assert "single-case dependency" in (folder / "robot/index.html").read_text() audit = folder / "support" / "audit.json" original = audit.read_bytes()