From 8e2e83809bf39d57842ae3d01ac289a5d3175875 Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Wed, 26 Aug 2026 23:30:37 +0200 Subject: [PATCH 01/16] feat(bench): add T3 Code application corpus --- benchmarks/README.md | 21 +-- benchmarks/corpus/t3code.json | 275 ++++++++++++++++++++++++++++++++ benchmarks/tests/corpus.test.ts | 2 +- 3 files changed, 287 insertions(+), 11 deletions(-) create mode 100644 benchmarks/corpus/t3code.json diff --git a/benchmarks/README.md b/benchmarks/README.md index 2f037bd..3fba33e 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -59,11 +59,12 @@ targets and a duplicate occurrence of the same chunk each contribute at most onc no gold target resolves. Manifest decoding rejects empty ground-truth arrays, and corpus preparation rejects authored targets that do not resolve to a chunk. -| Corpus | Revision | Language | Size band | Indexed scope | -| --------- | ------------------------------------------ | ---------- | --------- | ----------------------------- | -| FastAPI | `95f8322ee1dcda7ceace7b1c4f6c9915b36d748f` | Python | medium | `fastapi/**/*.py` | -| Effect v4 | `9263ba30c4535b655cf69a14f44a43cb9a93921e` | TypeScript | large | `packages/effect/src/**/*.ts` | -| fd | `41532d114e2ba565fb5367d606c111b29b96450c` | Rust | small | `src/**/*.rs` | +| Corpus | Revision | Language | Size band | Indexed scope | +| --------- | ------------------------------------------ | ---------- | --------- | -------------------------------------------------- | +| FastAPI | `95f8322ee1dcda7ceace7b1c4f6c9915b36d748f` | Python | medium | `fastapi/**/*.py` | +| Effect v4 | `9263ba30c4535b655cf69a14f44a43cb9a93921e` | TypeScript | large | `packages/effect/src/**/*.ts` | +| fd | `41532d114e2ba565fb5367d606c111b29b96450c` | Rust | small | `src/**/*.rs` | +| T3 Code | `badae6a5cc8325dcd5a145bea6f7b8ac692818a1` | TypeScript | medium | server, web state/browser, and shared source files | Effect's vendored Scalar and Swagger browser bundles are excluded explicitly. They are minified JavaScript stored in giant TypeScript string literals, exceed Tree-sitter's input limit, and are not @@ -95,9 +96,9 @@ vp run bench:retrieval:full | Profile | Repositories | Models | Validation | Static fusion | Router fusion | Diagnostics | | ---------- | ------------ | -------- | ----------------------------- | ------------- | ------------- | ---------------------- | | `smoke` | fd | MiniLM | grouped 5-fold | DBSF | DBSF | current router | -| `develop` | all three | MiniLM | grouped 3-fold | DBSF | DBSF | current router | -| `validate` | all three | MiniLM | grouped 5-fold and repository | DBSF | DBSF | current router | -| `full` | all three | selected | grouped 5-fold and repository | all three | all three | all active diagnostics | +| `develop` | all four | MiniLM | grouped 3-fold | DBSF | DBSF | current router | +| `validate` | all four | MiniLM | grouped 5-fold and repository | DBSF | DBSF | current router | +| `full` | all four | selected | grouped 5-fold and repository | all three | all three | all active diagnostics | `bench:retrieval` aliases `bench:retrieval:validate`. Every profile measures the same physical rankings and retrieval variants; profiles only control matrix size, holdout coverage, and expensive @@ -153,8 +154,8 @@ $env:PIX_BENCH_MODELS = "Xenova/bge-small-en-v1.5" vp run bench:retrieval:full ``` -The built-in repository IDs are `fastapi`, `effect-v4`, and `fd`; additional IDs come from added -manifests. Every profile runs exactly one model, +The built-in repository IDs are `fastapi`, `effect-v4`, `fd`, and `t3code`; additional IDs come from +added manifests. Every profile runs exactly one model, defaulting to MiniLM. Select another with `PIX_BENCH_MODELS`. Supported values are the three models in `MODEL_REGISTRY`: diff --git a/benchmarks/corpus/t3code.json b/benchmarks/corpus/t3code.json new file mode 100644 index 0000000..bf97f9f --- /dev/null +++ b/benchmarks/corpus/t3code.json @@ -0,0 +1,275 @@ +{ + "schemaVersion": 2, + "id": "t3code", + "repository": "https://github.com/pingdotgg/t3code.git", + "revision": "badae6a5cc8325dcd5a145bea6f7b8ac692818a1", + "language": "typescript", + "size": "medium", + "includeRoots": [ + "apps/server/src/vcs/", + "apps/server/src/workspace/", + "apps/server/src/textGeneration/", + "apps/web/src/state/", + "apps/web/src/browser/", + "packages/shared/src/" + ], + "excludePaths": ["apps/server/src/vcs/testing/"], + "extensions": [".ts", ".tsx"], + "questions": [ + { + "id": "t3code-001", + "queries": { + "identifier": "WorkspaceSearchIndex", + "searchPhrase": "workspace file path and content search index", + "naturalQuestion": "Where does the server build and query the workspace search index?", + "agentTask": "Find the service that indexes workspace paths and file contents for search." + }, + "category": "search", + "difficulty": "easy", + "groundTruth": [ + { + "file": "apps/server/src/workspace/WorkspaceSearchIndex.ts", + "symbol": "WorkspaceSearchIndex" + } + ] + }, + { + "id": "t3code-002", + "queries": { + "identifier": "WorkspaceFileSystem", + "searchPhrase": "safe workspace file read write path escape", + "naturalQuestion": "Which service handles workspace file operations and rejects paths outside the workspace?", + "agentTask": "Find the workspace filesystem boundary that validates paths before reading or writing files." + }, + "category": "filesystem", + "difficulty": "medium", + "groundTruth": [ + { + "file": "apps/server/src/workspace/WorkspaceFileSystem.ts", + "symbol": "WorkspaceFileSystem" + } + ] + }, + { + "id": "t3code-003", + "queries": { + "identifier": "VcsProvisioningService", + "searchPhrase": "provision version control driver project", + "naturalQuestion": "Where is version-control support provisioned for a project?", + "agentTask": "Find the service that prepares a VCS driver for a project workspace." + }, + "category": "version-control", + "difficulty": "medium", + "groundTruth": [ + { + "file": "apps/server/src/vcs/VcsProvisioningService.ts", + "symbol": "VcsProvisioningService" + } + ] + }, + { + "id": "t3code-004", + "queries": { + "identifier": "VcsStatusBroadcaster", + "searchPhrase": "broadcast git status remote refresh changes", + "naturalQuestion": "How does the server publish changing repository status to connected clients?", + "agentTask": "Find the service that watches VCS state and broadcasts status updates." + }, + "category": "events", + "difficulty": "hard", + "groundTruth": [ + { + "file": "apps/server/src/vcs/VcsStatusBroadcaster.ts", + "symbol": "VcsStatusBroadcaster" + } + ] + }, + { + "id": "t3code-005", + "queries": { + "identifier": "buildCommitMessagePrompt", + "searchPhrase": "generate commit message prompt from repository changes", + "naturalQuestion": "Where is the prompt for generating a commit message assembled?", + "agentTask": "Find the function that builds the commit-message generation prompt." + }, + "category": "text-generation", + "difficulty": "easy", + "groundTruth": [ + { + "file": "apps/server/src/textGeneration/TextGenerationPrompts.ts", + "symbol": "buildCommitMessagePrompt" + } + ] + }, + { + "id": "t3code-006", + "queries": { + "identifier": "makeOpenCodeTextGeneration", + "searchPhrase": "OpenCode text generation session prompt response", + "naturalQuestion": "Which adapter asks OpenCode to generate repository text?", + "agentTask": "Find the OpenCode-backed text-generation implementation." + }, + "category": "provider-adapter", + "difficulty": "medium", + "groundTruth": [ + { + "file": "apps/server/src/textGeneration/OpenCodeTextGeneration.ts", + "symbol": "makeOpenCodeTextGeneration" + } + ] + }, + { + "id": "t3code-007", + "queries": { + "identifier": "useThreadSearch", + "searchPhrase": "search agent threads query hook", + "naturalQuestion": "Which client hook searches existing agent threads?", + "agentTask": "Find the web-state query hook for thread search." + }, + "category": "state-query", + "difficulty": "easy", + "groundTruth": [ + { + "file": "apps/web/src/state/queries.ts", + "symbol": "useThreadSearch" + } + ] + }, + { + "id": "t3code-008", + "queries": { + "identifier": "useProjectContentSearch", + "searchPhrase": "project source content search client query", + "naturalQuestion": "Where does the web client request content search within a project?", + "agentTask": "Find the state hook that runs project-content searches." + }, + "category": "state-query", + "difficulty": "medium", + "groundTruth": [ + { + "file": "apps/web/src/state/queries.ts", + "symbol": "useProjectContentSearch" + } + ] + }, + { + "id": "t3code-009", + "queries": { + "identifier": "readThreadDetail", + "searchPhrase": "read scoped thread detail from entity state", + "naturalQuestion": "How can non-React code read the current details for a scoped thread?", + "agentTask": "Find the imperative state accessor for thread details." + }, + "category": "state-management", + "difficulty": "medium", + "groundTruth": [ + { + "file": "apps/web/src/state/entities.ts", + "symbol": "readThreadDetail" + } + ] + }, + { + "id": "t3code-010", + "queries": { + "identifier": "authEnvironment", + "searchPhrase": "authentication environment atoms connection runtime", + "naturalQuestion": "Where are authentication RPC atoms connected to the web runtime?", + "agentTask": "Find the web-state binding for the authentication environment." + }, + "category": "authentication", + "difficulty": "hard", + "groundTruth": [ + { + "file": "apps/web/src/state/auth.ts", + "symbol": "authEnvironment" + } + ] + }, + { + "id": "t3code-011", + "queries": { + "identifier": "planWebviewCrashRecovery", + "searchPhrase": "webview crash retry backoff recovery plan", + "naturalQuestion": "How does the browser preview decide whether and when to recover after a webview crash?", + "agentTask": "Find the function that plans browser webview crash recovery attempts." + }, + "category": "resilience", + "difficulty": "medium", + "groundTruth": [ + { + "file": "apps/web/src/browser/webviewCrashRecovery.ts", + "symbol": "planWebviewCrashRecovery" + } + ] + }, + { + "id": "t3code-012", + "queries": { + "identifier": "buildConnectAuthorizeRequestUrl", + "searchPhrase": "connect OAuth authorization request URL", + "naturalQuestion": "Where is the hosted authorization URL for connecting a local client created?", + "agentTask": "Find the shared function that builds a Connect authorization request URL." + }, + "category": "authentication", + "difficulty": "medium", + "groundTruth": [ + { + "file": "packages/shared/src/connectAuth.ts", + "symbol": "buildConnectAuthorizeRequestUrl" + } + ] + }, + { + "id": "t3code-013", + "queries": { + "identifier": "resolveAutoFeatureBranchName", + "searchPhrase": "derive automatic git feature branch name", + "naturalQuestion": "Which function turns a proposed branch label into the automatic feature branch name?", + "agentTask": "Find the shared branch-name resolver used for automatic worktree branches." + }, + "category": "version-control", + "difficulty": "easy", + "groundTruth": [ + { + "file": "packages/shared/src/git.ts", + "symbol": "resolveAutoFeatureBranchName" + } + ] + }, + { + "id": "t3code-014", + "queries": { + "identifier": "verifyDpopProof", + "searchPhrase": "verify DPoP proof request token binding", + "naturalQuestion": "Where does the application verify a DPoP proof against an HTTP request and access token?", + "agentTask": "Find the shared DPoP proof verification function." + }, + "category": "security", + "difficulty": "hard", + "groundTruth": [ + { + "file": "packages/shared/src/dpop.ts", + "symbol": "verifyDpopProof" + } + ] + }, + { + "id": "t3code-015", + "queries": { + "identifier": "resolveBackgroundActivitySettings", + "searchPhrase": "background agent activity preset settings override", + "naturalQuestion": "How are background activity presets and custom overrides combined?", + "agentTask": "Find the function that resolves effective background activity settings." + }, + "category": "configuration", + "difficulty": "hard", + "groundTruth": [ + { + "file": "packages/shared/src/backgroundActivitySettings.ts", + "symbol": "resolveBackgroundActivitySettings" + } + ] + } + ] +} diff --git a/benchmarks/tests/corpus.test.ts b/benchmarks/tests/corpus.test.ts index 9864cc5..8c1d919 100644 --- a/benchmarks/tests/corpus.test.ts +++ b/benchmarks/tests/corpus.test.ts @@ -48,7 +48,7 @@ it("rejects benchmark questions without exact ground truth", () => { it.effect("resolves every authored gold symbol in each pinned corpus", () => Effect.gen(function* () { const manifests = yield* loadCorpusManifests() - expect(manifests).toHaveLength(3) + expect(manifests).toHaveLength(4) expect(manifests.every((manifest) => manifest.questions.length === 15)).toBe(true) for (const manifest of manifests) { From 6fc9de5a34b9691b70d3965b80be866570a2ff0d Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Wed, 26 Aug 2026 23:49:33 +0200 Subject: [PATCH 02/16] feat(bench): add Python and Rust app corpora --- benchmarks/corpus/alacritty.json | 193 +++++++++++++++++++++++++++++ benchmarks/corpus/beets.json | 201 +++++++++++++++++++++++++++++++ benchmarks/tests/corpus.test.ts | 2 +- 3 files changed, 395 insertions(+), 1 deletion(-) create mode 100644 benchmarks/corpus/alacritty.json create mode 100644 benchmarks/corpus/beets.json diff --git a/benchmarks/corpus/alacritty.json b/benchmarks/corpus/alacritty.json new file mode 100644 index 0000000..521b311 --- /dev/null +++ b/benchmarks/corpus/alacritty.json @@ -0,0 +1,193 @@ +{ + "schemaVersion": 2, + "id": "alacritty", + "repository": "https://github.com/alacritty/alacritty.git", + "revision": "94e7c8874e526b1e67b349d9ba30ddf81669119e", + "language": "rust", + "size": "medium", + "includeRoots": ["alacritty/src/", "alacritty_terminal/src/"], + "excludePaths": ["alacritty_terminal/src/grid/tests.rs"], + "extensions": [".rs"], + "questions": [ + { + "id": "alacritty-001", + "queries": { + "identifier": "alacritty", + "searchPhrase": "terminal application startup configuration event loop", + "naturalQuestion": "Where does Alacritty initialize the terminal application and start its event loop?", + "agentTask": "Find the main application startup function after command-line parsing." + }, + "category": "startup", + "difficulty": "easy", + "groundTruth": [{ "file": "alacritty/src/main.rs", "symbol": "alacritty" }] + }, + { + "id": "alacritty-002", + "queries": { + "identifier": "WindowContext", + "searchPhrase": "per-window terminal display event state", + "naturalQuestion": "Which type owns the terminal, display, and event state for one window?", + "agentTask": "Find the per-window context that coordinates terminal UI state." + }, + "category": "window-management", + "difficulty": "medium", + "groundTruth": [{ "file": "alacritty/src/window_context.rs", "symbol": "WindowContext" }] + }, + { + "id": "alacritty-003", + "queries": { + "identifier": "Display", + "searchPhrase": "render terminal window OpenGL display", + "naturalQuestion": "Where is the terminal window's rendering state managed?", + "agentTask": "Find the display type that owns the window and renderer." + }, + "category": "rendering", + "difficulty": "easy", + "groundTruth": [{ "file": "alacritty/src/display/mod.rs", "symbol": "Display" }] + }, + { + "id": "alacritty-004", + "queries": { + "identifier": "RenderableContent", + "searchPhrase": "convert terminal state into renderable cells cursor", + "naturalQuestion": "How does the display collect visible terminal cells and cursor state for rendering?", + "agentTask": "Find the iterator that prepares terminal content for the renderer." + }, + "category": "rendering", + "difficulty": "hard", + "groundTruth": [{ "file": "alacritty/src/display/content.rs", "symbol": "RenderableContent" }] + }, + { + "id": "alacritty-005", + "queries": { + "identifier": "VisualBell", + "searchPhrase": "visual bell animation intensity timing", + "naturalQuestion": "Where is the visual bell animation state and intensity calculated?", + "agentTask": "Find the component that controls the visual bell animation." + }, + "category": "animation", + "difficulty": "medium", + "groundTruth": [{ "file": "alacritty/src/display/bell.rs", "symbol": "VisualBell" }] + }, + { + "id": "alacritty-006", + "queries": { + "identifier": "build_sequence", + "searchPhrase": "keyboard event terminal escape sequence modifiers mode", + "naturalQuestion": "Which function translates a keyboard event into bytes for the terminal?", + "agentTask": "Find the keyboard encoder that builds terminal escape sequences." + }, + "category": "input", + "difficulty": "hard", + "groundTruth": [{ "file": "alacritty/src/input/keyboard.rs", "symbol": "build_sequence" }] + }, + { + "id": "alacritty-007", + "queries": { + "identifier": "Clipboard", + "searchPhrase": "terminal clipboard selection load store", + "naturalQuestion": "Where does Alacritty read and write clipboard or selection contents?", + "agentTask": "Find the clipboard abstraction used by terminal actions." + }, + "category": "clipboard", + "difficulty": "easy", + "groundTruth": [{ "file": "alacritty/src/clipboard.rs", "symbol": "Clipboard" }] + }, + { + "id": "alacritty-008", + "queries": { + "identifier": "Scheduler", + "searchPhrase": "schedule timer events repeat deadlines", + "naturalQuestion": "How are delayed and repeating terminal events scheduled?", + "agentTask": "Find the timer scheduler that emits events at configured deadlines." + }, + "category": "events", + "difficulty": "medium", + "groundTruth": [{ "file": "alacritty/src/scheduler.rs", "symbol": "Scheduler" }] + }, + { + "id": "alacritty-009", + "queries": { + "identifier": "UiConfig", + "searchPhrase": "combined terminal window UI configuration", + "naturalQuestion": "Which type holds the effective UI and terminal configuration?", + "agentTask": "Find the root configuration type consumed by the graphical application." + }, + "category": "configuration", + "difficulty": "easy", + "groundTruth": [{ "file": "alacritty/src/config/ui_config.rs", "symbol": "UiConfig" }] + }, + { + "id": "alacritty-010", + "queries": { + "identifier": "EventLoop", + "searchPhrase": "PTY read write terminal parser event loop", + "naturalQuestion": "Where are PTY reads, terminal parsing, and writes coordinated?", + "agentTask": "Find the terminal event loop that connects the PTY to terminal state." + }, + "category": "process-io", + "difficulty": "hard", + "groundTruth": [{ "file": "alacritty_terminal/src/event_loop.rs", "symbol": "EventLoop" }] + }, + { + "id": "alacritty-011", + "queries": { + "identifier": "Term", + "searchPhrase": "terminal emulator grid modes cursor state", + "naturalQuestion": "Which type owns the terminal emulator's grid, modes, and cursor state?", + "agentTask": "Find the central terminal state machine." + }, + "category": "terminal-state", + "difficulty": "medium", + "groundTruth": [{ "file": "alacritty_terminal/src/term/mod.rs", "symbol": "Term" }] + }, + { + "id": "alacritty-012", + "queries": { + "identifier": "RegexSearch", + "searchPhrase": "incremental regex search terminal grid", + "naturalQuestion": "Where is regular-expression search over terminal contents implemented?", + "agentTask": "Find the terminal regex search engine." + }, + "category": "search", + "difficulty": "medium", + "groundTruth": [{ "file": "alacritty_terminal/src/term/search.rs", "symbol": "RegexSearch" }] + }, + { + "id": "alacritty-013", + "queries": { + "identifier": "Selection", + "searchPhrase": "terminal text selection anchors range lines", + "naturalQuestion": "How does the terminal track and convert a text selection into a range?", + "agentTask": "Find the terminal selection model used for selected text." + }, + "category": "selection", + "difficulty": "hard", + "groundTruth": [{ "file": "alacritty_terminal/src/selection.rs", "symbol": "Selection" }] + }, + { + "id": "alacritty-014", + "queries": { + "identifier": "ViModeCursor", + "searchPhrase": "vi mode cursor motion terminal navigation", + "naturalQuestion": "Where is the cursor position for vi-mode navigation represented?", + "agentTask": "Find the vi-mode cursor that applies terminal motions." + }, + "category": "navigation", + "difficulty": "medium", + "groundTruth": [{ "file": "alacritty_terminal/src/vi_mode.rs", "symbol": "ViModeCursor" }] + }, + { + "id": "alacritty-015", + "queries": { + "identifier": "setup_env", + "searchPhrase": "configure TERM COLORTERM terminal environment", + "naturalQuestion": "Where are environment variables prepared before the shell starts?", + "agentTask": "Find the function that configures the child terminal environment." + }, + "category": "process-environment", + "difficulty": "easy", + "groundTruth": [{ "file": "alacritty_terminal/src/tty/mod.rs", "symbol": "setup_env" }] + } + ] +} diff --git a/benchmarks/corpus/beets.json b/benchmarks/corpus/beets.json new file mode 100644 index 0000000..4661e26 --- /dev/null +++ b/benchmarks/corpus/beets.json @@ -0,0 +1,201 @@ +{ + "schemaVersion": 2, + "id": "beets", + "repository": "https://github.com/beetbox/beets.git", + "revision": "b7952299941543d4507ac7931edb223acd684b3d", + "language": "python", + "size": "medium", + "includeRoots": [ + "beets/importer/", + "beets/library/", + "beets/autotag/", + "beets/dbcore/", + "beets/ui/", + "beets/plugins.py", + "beets/metadata_plugins.py" + ], + "excludePaths": [], + "extensions": [".py"], + "questions": [ + { + "id": "beets-001", + "queries": { + "identifier": "ImportSession", + "searchPhrase": "music import session configuration pipeline", + "naturalQuestion": "Which object coordinates configuration and pipeline stages for a music import?", + "agentTask": "Find the session object that owns one beets import run." + }, + "category": "import-workflow", + "difficulty": "easy", + "groundTruth": [{ "file": "beets/importer/session.py", "symbol": "ImportSession" }] + }, + { + "id": "beets-002", + "queries": { + "identifier": "ImportTask", + "searchPhrase": "album import task files metadata candidates choice", + "naturalQuestion": "Where is the state for importing one album and its candidate choice stored?", + "agentTask": "Find the task model for an album import." + }, + "category": "import-workflow", + "difficulty": "medium", + "groundTruth": [{ "file": "beets/importer/tasks.py", "symbol": "ImportTask" }] + }, + { + "id": "beets-003", + "queries": { + "identifier": "ImportTaskFactory", + "searchPhrase": "scan paths create album singleton import tasks", + "naturalQuestion": "Which component turns paths into album or singleton import tasks?", + "agentTask": "Find the factory that discovers files and creates import tasks." + }, + "category": "filesystem", + "difficulty": "hard", + "groundTruth": [{ "file": "beets/importer/tasks.py", "symbol": "ImportTaskFactory" }] + }, + { + "id": "beets-004", + "queries": { + "identifier": "read_tasks", + "searchPhrase": "read import paths produce tasks pipeline stage", + "naturalQuestion": "Where does the importer read configured paths and start producing tasks?", + "agentTask": "Find the first importer stage that yields tasks from input paths." + }, + "category": "pipeline", + "difficulty": "medium", + "groundTruth": [{ "file": "beets/importer/stages.py", "symbol": "read_tasks" }] + }, + { + "id": "beets-005", + "queries": { + "identifier": "lookup_candidates", + "searchPhrase": "lookup metadata candidates imported album", + "naturalQuestion": "Which importer stage looks up metadata matches for an album?", + "agentTask": "Find the stage that attaches autotag candidates to an import task." + }, + "category": "metadata-matching", + "difficulty": "easy", + "groundTruth": [{ "file": "beets/importer/stages.py", "symbol": "lookup_candidates" }] + }, + { + "id": "beets-006", + "queries": { + "identifier": "manipulate_files", + "searchPhrase": "move copy link imported music files", + "naturalQuestion": "Where does the importer apply copy, move, link, and write operations to accepted files?", + "agentTask": "Find the importer stage that changes files after a choice is accepted." + }, + "category": "filesystem", + "difficulty": "hard", + "groundTruth": [{ "file": "beets/importer/stages.py", "symbol": "manipulate_files" }] + }, + { + "id": "beets-007", + "queries": { + "identifier": "Library", + "searchPhrase": "music library database albums items paths", + "naturalQuestion": "Which object provides the main database API for albums and items?", + "agentTask": "Find the central beets library abstraction." + }, + "category": "library", + "difficulty": "easy", + "groundTruth": [{ "file": "beets/library/library.py", "symbol": "Library" }] + }, + { + "id": "beets-008", + "queries": { + "identifier": "Item", + "searchPhrase": "single music file metadata database model", + "naturalQuestion": "Where is one track and its file metadata represented?", + "agentTask": "Find the database model for a single music item." + }, + "category": "data-model", + "difficulty": "easy", + "groundTruth": [{ "file": "beets/library/models.py", "symbol": "Item" }] + }, + { + "id": "beets-009", + "queries": { + "identifier": "Album", + "searchPhrase": "album metadata items art database model", + "naturalQuestion": "Which model groups tracks and album-level metadata?", + "agentTask": "Find the database model for an album." + }, + "category": "data-model", + "difficulty": "easy", + "groundTruth": [{ "file": "beets/library/models.py", "symbol": "Album" }] + }, + { + "id": "beets-010", + "queries": { + "identifier": "tag_album", + "searchPhrase": "rank metadata candidates complete album", + "naturalQuestion": "How does beets find and rank metadata candidates for an album?", + "agentTask": "Find the autotagger entry point for matching a complete album." + }, + "category": "metadata-matching", + "difficulty": "hard", + "groundTruth": [{ "file": "beets/autotag/match.py", "symbol": "tag_album" }] + }, + { + "id": "beets-011", + "queries": { + "identifier": "tag_item", + "searchPhrase": "rank metadata candidates single track", + "naturalQuestion": "Where does beets match a single track against metadata candidates?", + "agentTask": "Find the autotagger entry point for matching one item." + }, + "category": "metadata-matching", + "difficulty": "medium", + "groundTruth": [{ "file": "beets/autotag/match.py", "symbol": "tag_item" }] + }, + { + "id": "beets-012", + "queries": { + "identifier": "parse_query_string", + "searchPhrase": "parse user library query text sort fields", + "naturalQuestion": "Which function turns command-line query text into database queries and sorting?", + "agentTask": "Find the parser for user-facing library query strings." + }, + "category": "query-parsing", + "difficulty": "hard", + "groundTruth": [{ "file": "beets/library/queries.py", "symbol": "parse_query_string" }] + }, + { + "id": "beets-013", + "queries": { + "identifier": "Database", + "searchPhrase": "SQLite model transactions queries migrations", + "naturalQuestion": "Where are SQLite access, model queries, and transactions coordinated?", + "agentTask": "Find the base database implementation used by the music library." + }, + "category": "persistence", + "difficulty": "hard", + "groundTruth": [{ "file": "beets/dbcore/db.py", "symbol": "Database" }] + }, + { + "id": "beets-014", + "queries": { + "identifier": "BeetsPlugin", + "searchPhrase": "plugin commands listeners template fields lifecycle", + "naturalQuestion": "Which base class lets extensions register commands, listeners, and template fields?", + "agentTask": "Find the base class for beets plugins." + }, + "category": "plugins", + "difficulty": "medium", + "groundTruth": [{ "file": "beets/plugins.py", "symbol": "BeetsPlugin" }] + }, + { + "id": "beets-015", + "queries": { + "identifier": "candidates", + "searchPhrase": "aggregate album metadata candidates plugins", + "naturalQuestion": "Where are album candidates collected from all configured metadata plugins?", + "agentTask": "Find the function that yields album candidates from metadata sources." + }, + "category": "plugins", + "difficulty": "medium", + "groundTruth": [{ "file": "beets/metadata_plugins.py", "symbol": "candidates" }] + } + ] +} diff --git a/benchmarks/tests/corpus.test.ts b/benchmarks/tests/corpus.test.ts index 8c1d919..8ba1b5b 100644 --- a/benchmarks/tests/corpus.test.ts +++ b/benchmarks/tests/corpus.test.ts @@ -48,7 +48,7 @@ it("rejects benchmark questions without exact ground truth", () => { it.effect("resolves every authored gold symbol in each pinned corpus", () => Effect.gen(function* () { const manifests = yield* loadCorpusManifests() - expect(manifests).toHaveLength(4) + expect(manifests).toHaveLength(6) expect(manifests.every((manifest) => manifest.questions.length === 15)).toBe(true) for (const manifest of manifests) { From c662b4a2143b67d11c636f7cab1d7c678ba9db0c Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Wed, 26 Aug 2026 23:52:32 +0200 Subject: [PATCH 03/16] feat(bench): add retrieval matrix workflow --- benchmarks/README.md | 35 ++- benchmarks/matrix/full.json | 49 +++++ benchmarks/retrieval/evaluation/types.ts | 6 +- benchmarks/retrieval/matrix.ts | 256 ++++++++++++++++++++++ benchmarks/retrieval/runner.ts | 31 ++- benchmarks/tests/matrix-execution.test.ts | 54 +++++ benchmarks/tests/matrix.test.ts | 165 ++++++++++++++ benchmarks/tests/report.test.ts | 3 +- benchmarks/tests/retrieval.test.ts | 2 +- package.json | 1 + 10 files changed, 585 insertions(+), 17 deletions(-) create mode 100644 benchmarks/matrix/full.json create mode 100644 benchmarks/retrieval/matrix.ts create mode 100644 benchmarks/tests/matrix-execution.test.ts create mode 100644 benchmarks/tests/matrix.test.ts diff --git a/benchmarks/README.md b/benchmarks/README.md index 3fba33e..b2f2d50 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -62,8 +62,10 @@ rejects authored targets that do not resolve to a chunk. | Corpus | Revision | Language | Size band | Indexed scope | | --------- | ------------------------------------------ | ---------- | --------- | -------------------------------------------------- | | FastAPI | `95f8322ee1dcda7ceace7b1c4f6c9915b36d748f` | Python | medium | `fastapi/**/*.py` | +| beets | `b7952299941543d4507ac7931edb223acd684b3d` | Python | medium | importer, library, autotag, database, and UI code | | Effect v4 | `9263ba30c4535b655cf69a14f44a43cb9a93921e` | TypeScript | large | `packages/effect/src/**/*.ts` | | fd | `41532d114e2ba565fb5367d606c111b29b96450c` | Rust | small | `src/**/*.rs` | +| Alacritty | `94e7c8874e526b1e67b349d9ba30ddf81669119e` | Rust | medium | application and terminal source files | | T3 Code | `badae6a5cc8325dcd5a145bea6f7b8ac692818a1` | TypeScript | medium | server, web state/browser, and shared source files | Effect's vendored Scalar and Swagger browser bundles are excluded explicitly. They are minified @@ -93,16 +95,25 @@ vp run bench:retrieval:validate vp run bench:retrieval:full ``` +Run every profile, model, and optimization-profile invocation from the matrix manifest, then merge +the artifacts: + +```bash +vp run bench:retrieval:matrix +``` + +This command runs 30 model-backed benchmarks. Use it for release evidence, not the development loop. + | Profile | Repositories | Models | Validation | Static fusion | Router fusion | Diagnostics | | ---------- | ------------ | -------- | ----------------------------- | ------------- | ------------- | ---------------------- | | `smoke` | fd | MiniLM | grouped 5-fold | DBSF | DBSF | current router | -| `develop` | all four | MiniLM | grouped 3-fold | DBSF | DBSF | current router | -| `validate` | all four | MiniLM | grouped 5-fold and repository | DBSF | DBSF | current router | -| `full` | all four | selected | grouped 5-fold and repository | all three | all three | all active diagnostics | +| `develop` | all six | MiniLM | grouped 3-fold | DBSF | DBSF | current router | +| `validate` | all six | MiniLM | grouped 5-fold and repository | DBSF | DBSF | current router | +| `full` | all six | selected | grouped 5-fold and repository | all three | all three | all active diagnostics | `bench:retrieval` aliases `bench:retrieval:validate`. Every profile measures the same physical rankings and retrieval variants; profiles only control matrix size, holdout coverage, and expensive -diagnostics. The selected profile is recorded in schema-26 artifacts without changing retrieval +diagnostics. The selected profile is recorded in schema-32 artifacts without changing retrieval semantics. The full profile includes all three fusion methods; short profiles intentionally omit RRF to keep development runs fast. @@ -154,8 +165,8 @@ $env:PIX_BENCH_MODELS = "Xenova/bge-small-en-v1.5" vp run bench:retrieval:full ``` -The built-in repository IDs are `fastapi`, `effect-v4`, `fd`, and `t3code`; additional IDs come from -added manifests. Every profile runs exactly one model, +The built-in repository IDs are `alacritty`, `beets`, `effect-v4`, `fastapi`, `fd`, and `t3code`; +additional IDs come from added manifests. Every profile runs exactly one model, defaulting to MiniLM. Select another with `PIX_BENCH_MODELS`. Supported values are the three models in `MODEL_REGISTRY`: @@ -163,6 +174,18 @@ defaulting to MiniLM. Select another with `PIX_BENCH_MODELS`. Supported values a - `Xenova/bge-small-en-v1.5` - `jinaai/jina-embeddings-v2-base-code` +### Matrix manifest + +`benchmarks/matrix/full.json` defines the complete benchmark matrix as explicit run axes. It expands +to 12,960 coordinates across all models, repositories, fusion methods, optimization profiles, router +objectives, grouped folds, and repository holdouts. + +`expandBenchmarkMatrixManifest` expands the manifest in a stable order. Pass the resulting plan and +schema-32 artifacts to `mergeBenchmarkMatrix`. The merge rejects duplicate, missing, and unexpected +coordinates. Each merged coordinate retains the complete router result, search diagnostics, source +timestamp, and source timing record. `artifactSerializationDurationMs` records the preflight JSON +serialization time for each source artifact. + The router search defaults to `halving-funnel`. It proxy-scores 512 broad scouts, keeps 32 survivors, proxy-scores 16 radius-2 Sobol points per survivor, and fully evaluates 256 finalists plus the static base seeds once. Set `PIX_BENCH_ROUTER_STRATEGY=proxy-promotion` to run the slower coordinate-beam diff --git a/benchmarks/matrix/full.json b/benchmarks/matrix/full.json new file mode 100644 index 0000000..8724409 --- /dev/null +++ b/benchmarks/matrix/full.json @@ -0,0 +1,49 @@ +{ + "schemaVersion": 1, + "runs": [ + { + "benchmarkProfile": "develop", + "optimizationProfiles": [ + "search-priority", + "balanced", + "code-navigation", + "basic-exploration", + "natural-language" + ], + "models": [ + "Xenova/all-MiniLM-L6-v2", + "Xenova/bge-small-en-v1.5", + "jinaai/jina-embeddings-v2-base-code" + ], + "repositories": ["alacritty", "beets", "effect-v4", "fastapi", "fd", "t3code"], + "fusions": ["dbsf"], + "objectives": ["direct", "direct-recall-first", "reranker-top20", "reranker-top50"], + "validations": [{ "strategy": "grouped-3-fold", "folds": ["1", "2", "3"] }] + }, + { + "benchmarkProfile": "full", + "optimizationProfiles": [ + "search-priority", + "balanced", + "code-navigation", + "basic-exploration", + "natural-language" + ], + "models": [ + "Xenova/all-MiniLM-L6-v2", + "Xenova/bge-small-en-v1.5", + "jinaai/jina-embeddings-v2-base-code" + ], + "repositories": ["alacritty", "beets", "effect-v4", "fastapi", "fd", "t3code"], + "fusions": ["rrf", "relative-score", "dbsf"], + "objectives": ["direct", "direct-recall-first", "reranker-top20", "reranker-top50"], + "validations": [ + { "strategy": "grouped-5-fold", "folds": ["1", "2", "3", "4", "5"] }, + { + "strategy": "leave-one-repository-out", + "folds": ["alacritty", "beets", "effect-v4", "fastapi", "fd", "t3code"] + } + ] + } + ] +} diff --git a/benchmarks/retrieval/evaluation/types.ts b/benchmarks/retrieval/evaluation/types.ts index 0fcbefa..341fb07 100644 --- a/benchmarks/retrieval/evaluation/types.ts +++ b/benchmarks/retrieval/evaluation/types.ts @@ -474,7 +474,7 @@ export interface RecommendedEvidenceRouter { readonly searchDiagnostics: RouterSearchDiagnostics } -/** Compute-time breakdown for one benchmark invocation, excluding artifact file serialization. */ +/** Compute-time breakdown for one benchmark invocation; totalDurationMs excludes serialization. */ export interface BenchmarkTimings { readonly totalDurationMs: number readonly corpusPreparationDurationMs: number @@ -487,11 +487,13 @@ export interface BenchmarkTimings { readonly candidateQueueStartupDurationMs: number /** Time spent shutting down the shared native candidate queue. */ readonly candidateQueueShutdownDurationMs: number + /** Time spent on the first JSON serialization pass used to write the artifact. */ + readonly artifactSerializationDurationMs: number } /** Reproducible machine-readable output of one complete benchmark run. */ export interface BenchmarkArtifact { - readonly schemaVersion: 31 + readonly schemaVersion: 32 /** Profile controlling benchmark coverage without changing retrieval behavior. */ readonly benchmarkProfile: BenchmarkProfile /** Global-scout sequence used to seed the router beam search. */ diff --git a/benchmarks/retrieval/matrix.ts b/benchmarks/retrieval/matrix.ts new file mode 100644 index 0000000..667df2f --- /dev/null +++ b/benchmarks/retrieval/matrix.ts @@ -0,0 +1,256 @@ +import { Schema } from "effect" + +import type { FusionMethod } from "../../src/domain/retrieval.js" +import type { OptimizationProfile } from "./evaluation/optimization-profiles.js" +import type { + BenchmarkArtifact, + BenchmarkProfile, + BenchmarkTimings, + EvidenceRouterSearchResult, + RouterObjective, + RouterSearchDiagnostics, + ValidationStrategy, +} from "./evaluation/types.js" + +const BenchmarkProfileSchema = Schema.Literals(["smoke", "develop", "validate", "full"]) +const ValidationStrategySchema = Schema.Literals([ + "grouped-3-fold", + "grouped-5-fold", + "leave-one-repository-out", +]) +const FusionMethodSchema = Schema.Literals(["rrf", "relative-score", "dbsf"]) +const RouterObjectiveSchema = Schema.Literals([ + "direct", + "direct-recall-first", + "reranker-top20", + "reranker-top50", +]) + +/** One expected result coordinate in a mergeable retrieval benchmark matrix. */ +export const BenchmarkMatrixCoordinateSchema = Schema.Struct({ + benchmarkProfile: BenchmarkProfileSchema, + optimizationProfile: Schema.String, + model: Schema.String, + repository: Schema.String, + fusion: FusionMethodSchema, + objective: RouterObjectiveSchema, + validationStrategy: ValidationStrategySchema, + fold: Schema.String, +}) + +/** Explicit expected coverage for a set of independently generated benchmark artifacts. */ +export const BenchmarkMatrixPlanSchema = Schema.Struct({ + schemaVersion: Schema.Literal(1), + coordinates: Schema.NonEmptyArray(BenchmarkMatrixCoordinateSchema), +}) + +const BenchmarkMatrixValidationAxisSchema = Schema.Struct({ + strategy: ValidationStrategySchema, + folds: Schema.NonEmptyArray(Schema.String), +}) + +const BenchmarkMatrixRunSchema = Schema.Struct({ + benchmarkProfile: BenchmarkProfileSchema, + optimizationProfiles: Schema.NonEmptyArray(Schema.String), + models: Schema.NonEmptyArray(Schema.String), + repositories: Schema.NonEmptyArray(Schema.String), + fusions: Schema.NonEmptyArray(FusionMethodSchema), + objectives: Schema.NonEmptyArray(RouterObjectiveSchema), + validations: Schema.NonEmptyArray(BenchmarkMatrixValidationAxisSchema), +}) + +/** Versioned matrix manifest whose run axes expand to exact expected coordinates. */ +export const BenchmarkMatrixManifestSchema = Schema.Struct({ + schemaVersion: Schema.Literal(1), + runs: Schema.NonEmptyArray(BenchmarkMatrixRunSchema), +}) + +/** Decoded benchmark matrix coordinate. */ +export type BenchmarkMatrixCoordinate = typeof BenchmarkMatrixCoordinateSchema.Type + +/** Decoded benchmark matrix coverage plan. */ +export type BenchmarkMatrixPlan = typeof BenchmarkMatrixPlanSchema.Type + +/** Decoded benchmark matrix run manifest. */ +export type BenchmarkMatrixManifest = typeof BenchmarkMatrixManifestSchema.Type + +/** One benchmark process invocation derived from a matrix run. */ +export interface BenchmarkMatrixInvocation { + readonly benchmarkProfile: BenchmarkProfile + readonly optimizationProfile: string + readonly model: string + readonly repositories: readonly string[] +} + +/** Minimum router result identity required to merge benchmark artifacts. */ +export interface MatrixSearchResult { + readonly model: string + readonly fusion: FusionMethod + readonly objective: RouterObjective + readonly strategy: ValidationStrategy + readonly fold: string + readonly searchDiagnostics: RouterSearchDiagnostics +} + +/** Artifact fields consumed by matrix validation and merging. */ +export interface MatrixSourceArtifact< + Result extends MatrixSearchResult = EvidenceRouterSearchResult, +> { + readonly benchmarkProfile: BenchmarkProfile + readonly optimizationProfile: Pick + readonly generatedAt: string + readonly timings: BenchmarkTimings + readonly repositories: BenchmarkArtifact["repositories"] + readonly evidenceRouterSearch: readonly Result[] +} + +/** One matrix coordinate with its complete router result and source timing record. */ +export interface MergedBenchmarkMatrixEntry< + Result extends MatrixSearchResult = EvidenceRouterSearchResult, +> { + readonly coordinate: BenchmarkMatrixCoordinate + readonly sourceGeneratedAt: string + readonly sourceTimings: BenchmarkTimings + readonly result: Result +} + +/** Deterministic merge output after all expected coordinates pass coverage validation. */ +export interface MergedBenchmarkMatrix< + Result extends MatrixSearchResult = EvidenceRouterSearchResult, +> { + readonly schemaVersion: 1 + readonly coordinates: readonly MergedBenchmarkMatrixEntry[] +} + +/** Return the stable identity used for duplicate, missing, and unexpected-coordinate checks. */ +export const benchmarkMatrixCoordinateKey = (coordinate: BenchmarkMatrixCoordinate): string => + JSON.stringify([ + coordinate.benchmarkProfile, + coordinate.optimizationProfile, + coordinate.model, + coordinate.repository, + coordinate.fusion, + coordinate.objective, + coordinate.validationStrategy, + coordinate.fold, + ]) + +/** Expand a compact run-axis manifest into deterministic result coordinates. */ +export const expandBenchmarkMatrixManifest = ( + manifest: BenchmarkMatrixManifest, +): BenchmarkMatrixPlan => { + const coordinates = manifest.runs.flatMap((run) => + run.optimizationProfiles.flatMap((optimizationProfile) => + run.models.flatMap((model) => + run.repositories.flatMap((repository) => + run.fusions.flatMap((fusion) => + run.objectives.flatMap((objective) => + run.validations.flatMap((validation) => + validation.folds.map((fold) => ({ + benchmarkProfile: run.benchmarkProfile, + optimizationProfile, + model, + repository, + fusion, + objective, + validationStrategy: validation.strategy, + fold, + })), + ), + ), + ), + ), + ), + ), + ) + const first = coordinates[0] + if (first === undefined) throw new Error("Benchmark matrix manifest expanded to no coordinates") + return { schemaVersion: 1, coordinates: [first, ...coordinates.slice(1)] } +} + +/** Return the benchmark process invocations needed to produce every manifest coordinate. */ +export const benchmarkMatrixInvocations = ( + manifest: BenchmarkMatrixManifest, +): readonly BenchmarkMatrixInvocation[] => + manifest.runs.flatMap((run) => + run.optimizationProfiles.flatMap((optimizationProfile) => + run.models.map((model) => ({ + benchmarkProfile: run.benchmarkProfile, + optimizationProfile, + model, + repositories: run.repositories, + })), + ), + ) + +const coordinateFor = ( + artifact: MatrixSourceArtifact, + result: Result, + repository: string, +): BenchmarkMatrixCoordinate => ({ + benchmarkProfile: artifact.benchmarkProfile, + optimizationProfile: artifact.optimizationProfile.name, + model: result.model, + repository, + fusion: result.fusion, + objective: result.objective, + validationStrategy: result.strategy, + fold: result.fold, +}) + +/** Expand every artifact result across the repositories that participated in its search. */ +export const benchmarkMatrixCoordinates = ( + artifact: MatrixSourceArtifact, +): readonly BenchmarkMatrixCoordinate[] => + artifact.evidenceRouterSearch.flatMap((result) => + artifact.repositories.map((repository) => coordinateFor(artifact, result, repository.id)), + ) + +const uniquePlanCoordinates = ( + plan: BenchmarkMatrixPlan, +): ReadonlyMap => { + const coordinates = new Map() + for (const coordinate of plan.coordinates) { + const key = benchmarkMatrixCoordinateKey(coordinate) + if (coordinates.has(key)) throw new Error(`Duplicate benchmark matrix plan coordinate: ${key}`) + coordinates.set(key, coordinate) + } + return coordinates +} + +/** Merge artifact rows according to an explicit plan and reject incomplete or overlapping coverage. */ +export const mergeBenchmarkMatrix = ( + plan: BenchmarkMatrixPlan, + artifacts: readonly MatrixSourceArtifact[], +): MergedBenchmarkMatrix => { + const expected = uniquePlanCoordinates(plan) + const discovered = new Map>() + + for (const artifact of artifacts) { + for (const result of artifact.evidenceRouterSearch) { + for (const repository of artifact.repositories) { + const coordinate = coordinateFor(artifact, result, repository.id) + const key = benchmarkMatrixCoordinateKey(coordinate) + if (discovered.has(key)) throw new Error(`Duplicate benchmark artifact coordinate: ${key}`) + if (!expected.has(key)) throw new Error(`Unexpected benchmark artifact coordinate: ${key}`) + discovered.set(key, { + coordinate, + sourceGeneratedAt: artifact.generatedAt, + sourceTimings: artifact.timings, + result, + }) + } + } + } + + const missing = [...expected.keys()].filter((key) => !discovered.has(key)) + if (missing.length > 0) + throw new Error(`Missing benchmark artifact coordinates: ${missing.join(", ")}`) + + return { + schemaVersion: 1, + coordinates: [...discovered] + .sort(([left], [right]) => left.localeCompare(right)) + .map(([, row]) => row), + } +} diff --git a/benchmarks/retrieval/runner.ts b/benchmarks/retrieval/runner.ts index 529c3bd..6aba1eb 100644 --- a/benchmarks/retrieval/runner.ts +++ b/benchmarks/retrieval/runner.ts @@ -118,7 +118,9 @@ const selectOptimizationProfile = (): Effect.Effect : Effect.succeed(selected) } -const writeArtifact = (artifact: BenchmarkArtifact): Effect.Effect => +const writeArtifact = ( + artifact: BenchmarkArtifact, +): Effect.Effect<{ readonly artifact: BenchmarkArtifact; readonly outputPath: string }, Error> => Effect.gen(function* () { const outputDirectory = path.resolve("benchmarks/results") yield* Effect.tryPromise({ @@ -128,15 +130,30 @@ const writeArtifact = (artifact: BenchmarkArtifact): Effect.Effect writeFile(outputPath, `${JSON.stringify(artifact, null, 2)}\n`, "utf8"), + try: () => + writeFile( + outputPath, + `${JSON.stringify(artifactWithSerializationTiming, null, 2)}\n`, + "utf8", + ), catch: (cause) => new Error(`Could not write benchmark artifact ${outputPath}`, { cause }), }) yield* Effect.tryPromise({ - try: () => writeFile(reportPath, renderMarkdownReport(artifact), "utf8"), + try: () => + writeFile(reportPath, renderMarkdownReport(artifactWithSerializationTiming), "utf8"), catch: (cause) => new Error(`Could not write benchmark report ${reportPath}`, { cause }), }) - return outputPath + return { artifact: artifactWithSerializationTiming, outputPath } }) /** Run all selected repositories, embedding models, channel variants, and context budgets. */ @@ -218,7 +235,7 @@ export const runRetrievalBenchmark = ( ) const artifact: BenchmarkArtifact = { - schemaVersion: 31, + schemaVersion: 32, benchmarkProfile: profile, scoutSequence, seedHypotheses, @@ -252,6 +269,7 @@ export const runRetrievalBenchmark = ( evidenceRouterSearchDurationMs: search.evidenceRouterSearchDurationMs, candidateQueueStartupDurationMs: search.candidateQueueStartupDurationMs, candidateQueueShutdownDurationMs: search.candidateQueueShutdownDurationMs, + artifactSerializationDurationMs: 0, }, chunkConfig: { chunkTokens, @@ -281,6 +299,5 @@ export const runRetrievalBenchmark = ( recommendedEvidenceRouters: search.recommendedEvidenceRouters, promotionEvidence: search.promotionEvidence, } - const outputPath = yield* writeArtifact(artifact) - return { artifact, outputPath } + return yield* writeArtifact(artifact) }) diff --git a/benchmarks/tests/matrix-execution.test.ts b/benchmarks/tests/matrix-execution.test.ts new file mode 100644 index 0000000..6766069 --- /dev/null +++ b/benchmarks/tests/matrix-execution.test.ts @@ -0,0 +1,54 @@ +import { mkdir, readFile, writeFile } from "node:fs/promises" +import path from "node:path" + +import { expect, it } from "@effect/vitest" +import { Effect, Schema } from "effect" + +import type { BenchmarkArtifact } from "../retrieval/evaluation/types.js" +import { + benchmarkMatrixInvocations, + BenchmarkMatrixManifestSchema, + expandBenchmarkMatrixManifest, + mergeBenchmarkMatrix, +} from "../retrieval/matrix.js" +import { runRetrievalBenchmark } from "../retrieval/runner.js" + +const readManifest = Effect.tryPromise({ + try: async () => { + const input: unknown = JSON.parse(await readFile("benchmarks/matrix/full.json", "utf8")) + return Schema.decodeUnknownSync(BenchmarkMatrixManifestSchema)(input) + }, + catch: (cause) => new Error("Could not read the benchmark matrix manifest", { cause }), +}) + +it.effect("executes and merges the complete retrieval matrix", () => + Effect.gen(function* () { + const manifest = yield* readManifest + const artifacts: BenchmarkArtifact[] = [] + + for (const invocation of benchmarkMatrixInvocations(manifest)) { + process.env.PIX_BENCH_MODELS = invocation.model + process.env.PIX_BENCH_REPOS = invocation.repositories.join(",") + process.env.PIX_BENCH_OPTIMIZATION_PROFILE = invocation.optimizationProfile + const result = yield* runRetrievalBenchmark(invocation.benchmarkProfile) + artifacts.push(result.artifact) + } + + const plan = expandBenchmarkMatrixManifest(manifest) + const matrix = mergeBenchmarkMatrix(plan, artifacts) + const outputDirectory = path.resolve("benchmarks/results") + const outputPath = path.join( + outputDirectory, + `retrieval-matrix-${new Date().toISOString().replaceAll(":", "-")}.json`, + ) + yield* Effect.tryPromise({ + try: async () => { + await mkdir(outputDirectory, { recursive: true }) + await writeFile(outputPath, `${JSON.stringify(matrix, null, 2)}\n`, "utf8") + }, + catch: (cause) => new Error(`Could not write benchmark matrix ${outputPath}`, { cause }), + }) + + expect(matrix.coordinates).toHaveLength(plan.coordinates.length) + }), +) diff --git a/benchmarks/tests/matrix.test.ts b/benchmarks/tests/matrix.test.ts new file mode 100644 index 0000000..406e32b --- /dev/null +++ b/benchmarks/tests/matrix.test.ts @@ -0,0 +1,165 @@ +import { readFileSync } from "node:fs" + +import { describe, expect, it } from "@effect/vitest" +import { Schema } from "effect" + +import type { BenchmarkTimings, RouterSearchDiagnostics } from "../retrieval/evaluation/types.js" +import { + BenchmarkMatrixPlanSchema, + benchmarkMatrixCoordinates, + benchmarkMatrixInvocations, + BenchmarkMatrixManifestSchema, + expandBenchmarkMatrixManifest, + mergeBenchmarkMatrix, + type BenchmarkMatrixCoordinate, + type MatrixSearchResult, + type MatrixSourceArtifact, +} from "../retrieval/matrix.js" + +const timings: BenchmarkTimings = { + totalDurationMs: 100, + corpusPreparationDurationMs: 10, + embeddingDurationMs: 20, + retrievalDurationMs: 30, + weightSearchDurationMs: 5, + fusionSearchDurationMs: 5, + evidenceRouterSearchDurationMs: 25, + candidateQueueStartupDurationMs: 2, + candidateQueueShutdownDurationMs: 1, + artifactSerializationDurationMs: 2, +} + +const searchDiagnostics: RouterSearchDiagnostics = { + parameterCount: 1, + parameterLevels: { identity: [1] }, + rawCandidates: 1, + uniqueCandidates: 1, + proxyEvaluations: 1, + fullEvaluations: 1, + proxyCacheHits: 0, + fullCacheHits: 0, + proxyPromotions: 1, + proxyFullAgreement: 1, + protectedEliteCount: 1, + localCloudCandidates: 0, + timings: { + preparationMs: 1, + candidatePoolInitializationMs: 1, + baseWeightSearchMs: 1, + randomSearchMs: 0, + beamSearchMs: 1, + candidatePreparationMs: 1, + candidateEvaluationMs: 1, + candidateSelectionMs: 1, + }, +} + +const result = (fold: string): MatrixSearchResult => ({ + model: "fixture-model", + fusion: "dbsf", + objective: "direct", + strategy: "grouped-3-fold", + fold, + searchDiagnostics, +}) + +const artifact = ( + results: readonly MatrixSearchResult[], + repositories: readonly string[] = ["fd"], +): MatrixSourceArtifact => ({ + benchmarkProfile: "develop", + optimizationProfile: { name: "search-priority" }, + generatedAt: "2026-08-26T00:00:00.000Z", + timings, + repositories: repositories.map((id) => ({ + id, + repository: `owner/${id}`, + revision: "abc123", + chunks: 10, + preparationDurationMs: 1, + })), + evidenceRouterSearch: results, +}) + +const coordinate = (fold: string, repository = "fd"): BenchmarkMatrixCoordinate => ({ + benchmarkProfile: "develop", + optimizationProfile: "search-priority", + model: "fixture-model", + repository, + fusion: "dbsf", + objective: "direct", + validationStrategy: "grouped-3-fold", + fold, +}) + +describe("benchmark matrix", () => { + it("expands the checked-in full manifest to all 12,960 expected coordinates", () => { + const manifestInput: unknown = JSON.parse( + readFileSync(new URL("../matrix/full.json", import.meta.url), "utf8"), + ) + const manifest = Schema.decodeUnknownSync(BenchmarkMatrixManifestSchema)(manifestInput) + const plan = expandBenchmarkMatrixManifest(manifest) + + expect(plan.coordinates).toHaveLength(12_960) + expect(benchmarkMatrixInvocations(manifest)).toHaveLength(30) + expect(new Set(plan.coordinates.map((entry) => entry.optimizationProfile))).toEqual( + new Set([ + "search-priority", + "balanced", + "code-navigation", + "basic-exploration", + "natural-language", + ]), + ) + expect(new Set(plan.coordinates.map((entry) => entry.validationStrategy))).toEqual( + new Set(["grouped-3-fold", "grouped-5-fold", "leave-one-repository-out"]), + ) + }) + + it("expands and merges every explicit coordinate in stable order", () => { + const source = artifact([result("2"), result("1")], ["fd", "fastapi"]) + expect(benchmarkMatrixCoordinates(source)).toHaveLength(4) + + const plan = Schema.decodeUnknownSync(BenchmarkMatrixPlanSchema)({ + schemaVersion: 1, + coordinates: [ + coordinate("2", "fastapi"), + coordinate("1", "fd"), + coordinate("2", "fd"), + coordinate("1", "fastapi"), + ], + }) + const merged = mergeBenchmarkMatrix(plan, [source]) + + expect(merged.coordinates.map((entry) => entry.coordinate)).toEqual([ + coordinate("1", "fastapi"), + coordinate("2", "fastapi"), + coordinate("1", "fd"), + coordinate("2", "fd"), + ]) + expect(merged.coordinates[0].sourceTimings.artifactSerializationDurationMs).toBe(2) + expect(merged.coordinates[0].result.searchDiagnostics).toBe(searchDiagnostics) + }) + + it("rejects missing coordinates", () => { + const plan = { schemaVersion: 1, coordinates: [coordinate("1"), coordinate("2")] } as const + expect(() => mergeBenchmarkMatrix(plan, [artifact([result("1")])])).toThrow( + "Missing benchmark artifact coordinates", + ) + }) + + it("rejects duplicate artifact coordinates", () => { + const plan = { schemaVersion: 1, coordinates: [coordinate("1")] } as const + const source = artifact([result("1")]) + expect(() => mergeBenchmarkMatrix(plan, [source, source])).toThrow( + "Duplicate benchmark artifact coordinate", + ) + }) + + it("rejects unexpected coordinates", () => { + const plan = { schemaVersion: 1, coordinates: [coordinate("1")] } as const + expect(() => mergeBenchmarkMatrix(plan, [artifact([result("2")])])).toThrow( + "Unexpected benchmark artifact coordinate", + ) + }) +}) diff --git a/benchmarks/tests/report.test.ts b/benchmarks/tests/report.test.ts index cdc907b..4179b93 100644 --- a/benchmarks/tests/report.test.ts +++ b/benchmarks/tests/report.test.ts @@ -5,7 +5,7 @@ import { renderMarkdownReport } from "../retrieval/evaluation/report.js" import { routerSearchStrategyFor, type BenchmarkArtifact } from "../retrieval/evaluation/types.js" const artifact = { - schemaVersion: 31, + schemaVersion: 32, benchmarkProfile: "smoke", scoutSequence: "halton", seedHypotheses: false, @@ -32,6 +32,7 @@ const artifact = { evidenceRouterSearchDurationMs: 0, candidateQueueStartupDurationMs: 0, candidateQueueShutdownDurationMs: 0, + artifactSerializationDurationMs: 0, }, chunkConfig: { chunkTokens: 512, overlapLines: 0 }, contextTokenEstimator: "utf8-bytes-divided-by-four", diff --git a/benchmarks/tests/retrieval.test.ts b/benchmarks/tests/retrieval.test.ts index 46ec51e..71d200e 100644 --- a/benchmarks/tests/retrieval.test.ts +++ b/benchmarks/tests/retrieval.test.ts @@ -105,7 +105,7 @@ const runProfile = (profile: BenchmarkProfile, groupedFolds: number, fusionMetho expect(artifact.evaluationCases.every(({ groundTruth }) => groundTruth.length > 0)).toBe(true) expect(artifact.models.length).toBeGreaterThan(0) expect(artifact.measurements.length).toBeGreaterThan(0) - expect(artifact.schemaVersion).toBe(31) + expect(artifact.schemaVersion).toBe(32) expect(artifact.scoutSequence).toBe(resolveScoutSequence(process.env.PIX_BENCH_SCOUT_SEQUENCE)) expect(artifact.seedHypotheses).toBe( process.env.PIX_BENCH_SEED_HYPOTHESES === "1" || diff --git a/package.json b/package.json index b9430f0..cd6e4cc 100644 --- a/package.json +++ b/package.json @@ -39,6 +39,7 @@ "bench:retrieval:develop": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/retrieval.test.ts -t \"runs the develop retrieval profile\"", "bench:retrieval:fixture": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/channels.test.ts", "bench:retrieval:full": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/retrieval.test.ts -t \"runs the full retrieval profile\"", + "bench:retrieval:matrix": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/matrix-execution.test.ts", "bench:retrieval:smoke": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/retrieval.test.ts -t \"runs the smoke retrieval profile\"", "bench:retrieval:validate": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/retrieval.test.ts -t \"runs the validate retrieval profile\"", "build": "vp pack", From 7e4ebd492f031f9ccb2f6eab0eaf9de504e499bd Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Fri, 28 Aug 2026 17:49:44 +0200 Subject: [PATCH 04/16] feat(bench): compare router models and fitting methods --- .../evaluation/router-search/comparisons.ts | 389 ++++++++++++++++++ benchmarks/tests/router-comparisons.test.ts | 150 +++++++ 2 files changed, 539 insertions(+) create mode 100644 benchmarks/retrieval/evaluation/router-search/comparisons.ts create mode 100644 benchmarks/tests/router-comparisons.test.ts diff --git a/benchmarks/retrieval/evaluation/router-search/comparisons.ts b/benchmarks/retrieval/evaluation/router-search/comparisons.ts new file mode 100644 index 0000000..ac1cacc --- /dev/null +++ b/benchmarks/retrieval/evaluation/router-search/comparisons.ts @@ -0,0 +1,389 @@ +import { + CHANNEL_NAMES, + type ChannelName, + type ChannelWeights, + type EvidenceRouterParameters, +} from "../../../../src/domain/retrieval.js" +import { + routeWithEvidence, + type RoutingEvidence, +} from "../../../../src/lib/retrieval/evidence-router.js" +import { SCOUT_SEQUENCES, scoutLevelIndex } from "../scouts/index.js" +import { routerComplexity, routerKey, type RouterParameter } from "./config-space.js" + +/** Benchmark-only router formulation under comparison. */ +export type RouterModel = "multiplicative" | "regularized-log-linear" + +/** Derivative-free search procedure used for a router comparison. */ +export type RouterFittingMethod = + | "staged" + | "alternating-block-coordinate" + | "deterministic-restarts" + | "local-search" + +/** Router formulations included in the benchmark comparison. */ +const ROUTER_MODELS: readonly RouterModel[] = ["multiplicative", "regularized-log-linear"] + +/** Derivative-free fitting methods included in the benchmark comparison. */ +const ROUTER_FITTING_METHODS: readonly RouterFittingMethod[] = [ + "staged", + "alternating-block-coordinate", + "deterministic-restarts", + "local-search", +] + +/** Observed usefulness of one router parameter on the development samples. */ +export interface RouterDimensionDiagnostic { + readonly name: string + readonly status: "active" | "inactive" | "data-constant" + readonly observedValues: number +} + +/** Candidate score supplied by a development-only evaluator. */ +export interface RouterComparisonScore { + readonly mean: number + readonly standardError: number +} + +/** One fitted router and its development score. */ +export interface RouterComparisonCandidate { + readonly config: EvidenceRouterParameters + readonly score: RouterComparisonScore +} + +/** Search output with explicit heuristic and dimension-accounting metadata. */ +export interface RouterComparisonResult { + readonly model: RouterModel + readonly method: RouterFittingMethod + readonly optimalityClaim: "derivative-free-heuristic-no-global-optimality-claim" + readonly dimensions: readonly RouterDimensionDiagnostic[] + readonly evaluatedCandidates: number + readonly selected: RouterComparisonCandidate + readonly complexityAware: RouterComparisonCandidate + readonly oneStandardError: RouterComparisonCandidate +} + +/** Excluded-holdout scores computed only after development selection has finished. */ +export interface RouterComparisonHoldoutResult { + readonly model: RouterModel + readonly method: RouterFittingMethod + readonly selected: RouterComparisonScore + readonly complexityAware: RouterComparisonScore + readonly oneStandardError: RouterComparisonScore +} + +/** Configuration for one bounded comparison search. */ +export interface RouterComparisonSearchOptions { + readonly model: RouterModel + readonly method: RouterFittingMethod + readonly seed: EvidenceRouterParameters + readonly parameters: readonly RouterParameter[] + readonly evidence: readonly RoutingEvidence[] + readonly pruneInactive?: boolean + readonly restarts?: number + readonly passes?: number + readonly complexityPenalty?: number + readonly evaluateDevelopment: ( + configs: readonly EvidenceRouterParameters[], + ) => Promise +} + +const clamp = (value: number, minimum: number, maximum: number): number => + Math.max(minimum, Math.min(maximum, value)) + +const centeredEvidence = ( + evidence: RoutingEvidence, + parameter: string, + channel: ChannelName, +): number => { + const channelEvidence = evidence.channels[channel] + if (!channelEvidence.available) return 0 + if (parameter === "scoreInfluence") return channelEvidence.scoreSeparation - 0.5 + if (parameter === "geometryInfluence") return channelEvidence.scoreGeometry.confidence - 0.5 + if (parameter === "termCoverageInfluence") return channelEvidence.termCoverage - 0.5 + if (parameter === "pairwiseAgreementInfluence") return channelEvidence.pairwiseAgreement - 0.5 + if (parameter === "denseConfidenceInfluence") return channelEvidence.denseConfidence - 0.5 + if (parameter === "identifierInfluence") return evidence.identifierLikelihood - 0.5 + if (parameter === "queryLengthInfluence") return evidence.queryLengthSignal - 0.5 + return 0 +} + +/** Route with either the production multiplicative model or a bounded log-linear comparison gate. */ +export const routeWithComparisonModel = ( + model: RouterModel, + evidence: RoutingEvidence, + config: EvidenceRouterParameters, +): ChannelWeights => { + if (model === "multiplicative") return routeWithEvidence(evidence, config) + const weight = (channel: ChannelName): number => { + if (!evidence.channels[channel].available) return 0 + const logAdjustment = + config.scoreInfluence[channel] * centeredEvidence(evidence, "scoreInfluence", channel) + + config.geometryInfluence[channel] * centeredEvidence(evidence, "geometryInfluence", channel) + + config.termCoverageInfluence[channel] * + centeredEvidence(evidence, "termCoverageInfluence", channel) + + config.pairwiseAgreementInfluence[channel] * + centeredEvidence(evidence, "pairwiseAgreementInfluence", channel) + + config.denseConfidenceInfluence[channel] * + centeredEvidence(evidence, "denseConfidenceInfluence", channel) + + config.identifierInfluence[channel] * + centeredEvidence(evidence, "identifierInfluence", channel) + + config.queryLengthInfluence[channel] * + centeredEvidence(evidence, "queryLengthInfluence", channel) + return config.baseWeights[channel] * Math.exp(clamp(logAdjustment, -2, 2)) + } + return { + identity: weight("identity"), + camelcase: weight("camelcase"), + bm25: weight("bm25"), + dense: weight("dense"), + sparse: weight("sparse"), + } +} + +const observedParameterValues = ( + parameter: RouterParameter, + evidence: readonly RoutingEvidence[], +): readonly number[] => { + const [family, channelName] = parameter.name.split(".") + const channel = CHANNEL_NAMES.find((candidate) => candidate === channelName) + if (channel === undefined) return [] + if (family === "baseWeights") + return evidence.map((sample) => (sample.channels[channel].available ? 1 : 0)) + return evidence.map((sample) => centeredEvidence(sample, family ?? "", channel)) +} + +/** Classify search dimensions from development evidence without consulting holdout quality. */ +export const classifyRouterDimensions = ( + parameters: readonly RouterParameter[], + evidence: readonly RoutingEvidence[], +): readonly RouterDimensionDiagnostic[] => + parameters.map((parameter) => { + const values = observedParameterValues(parameter, evidence) + const distinct = new Set(values.map((value) => value.toFixed(8))).size + return { + name: parameter.name, + status: values.every((value) => value === 0) + ? "inactive" + : distinct === 1 + ? "data-constant" + : "active", + observedValues: distinct, + } + }) + +const compareCandidates = ( + left: RouterComparisonCandidate, + right: RouterComparisonCandidate, +): number => + right.score.mean - left.score.mean || + routerComplexity(left.config) - routerComplexity(right.config) + +/** Select the highest regularized utility, penalizing non-zero router complexity. */ +export const selectComplexityAwareCandidate = ( + candidates: readonly RouterComparisonCandidate[], + penalty: number, +): RouterComparisonCandidate => { + const selected = [...candidates].sort( + (left, right) => + right.score.mean - + penalty * routerComplexity(right.config) - + (left.score.mean - penalty * routerComplexity(left.config)) || + routerComplexity(left.config) - routerComplexity(right.config), + )[0] + if (selected === undefined) throw new Error("Router comparison has no candidate") + return selected +} + +/** Select the simplest candidate within one standard error of the best development mean. */ +export const selectOneStandardErrorCandidate = ( + candidates: readonly RouterComparisonCandidate[], +): RouterComparisonCandidate => { + const best = [...candidates].sort(compareCandidates)[0] + if (best === undefined) throw new Error("Router comparison has no candidate") + const threshold = best.score.mean - best.score.standardError + return [...candidates] + .filter((candidate) => candidate.score.mean >= threshold) + .sort( + (left, right) => + routerComplexity(left.config) - routerComplexity(right.config) || + compareCandidates(left, right), + )[0]! +} + +const evaluate = async ( + configs: readonly EvidenceRouterParameters[], + evaluateDevelopment: RouterComparisonSearchOptions["evaluateDevelopment"], +): Promise => { + const unique = [...new Map(configs.map((config) => [routerKey(config), config])).values()] + const scores = await evaluateDevelopment(unique) + if (scores.length !== unique.length) + throw new Error("Router comparison evaluator returned the wrong score count") + return unique.map((config, index) => ({ config, score: scores[index]! })) +} + +const improve = async ( + seed: RouterComparisonCandidate, + parameters: readonly RouterParameter[], + evaluateDevelopment: RouterComparisonSearchOptions["evaluateDevelopment"], +): Promise<{ readonly selected: RouterComparisonCandidate; readonly evaluations: number }> => { + const candidates = await evaluate( + [ + seed.config, + ...parameters.flatMap((parameter) => + parameter.values.map((value) => parameter.update(seed.config, value)), + ), + ], + evaluateDevelopment, + ) + return { selected: [...candidates].sort(compareCandidates)[0]!, evaluations: candidates.length } +} + +const parameterBlocks = (parameters: readonly RouterParameter[]): readonly RouterParameter[][] => { + const blocks = new Map() + for (const parameter of parameters) { + const family = parameter.name.split(".")[0] ?? parameter.name + const block = blocks.get(family) ?? [] + block.push(parameter) + blocks.set(family, block) + } + return [...blocks.values()] +} + +/** Run one bounded deterministic comparison search using development scores only. */ +export const searchRouterComparison = async ( + options: RouterComparisonSearchOptions, +): Promise => { + const dimensions = classifyRouterDimensions(options.parameters, options.evidence) + const activeNames = new Set( + dimensions + .filter((dimension) => dimension.status !== "inactive") + .map((dimension) => dimension.name), + ) + const parameters = options.pruneInactive + ? options.parameters.filter((parameter) => activeNames.has(parameter.name)) + : options.parameters + const archive = new Map() + const evaluateAndRecord: RouterComparisonSearchOptions["evaluateDevelopment"] = async ( + configs, + ) => { + const scores = await options.evaluateDevelopment(configs) + for (let index = 0; index < configs.length; index++) { + const config = configs[index] + const score = scores[index] + if (config !== undefined && score !== undefined) + archive.set(routerKey(config), { config, score }) + } + return scores + } + const initial = (await evaluate([options.seed], evaluateAndRecord))[0]! + let selected = initial + let evaluatedCandidates = 1 + const blocks = parameterBlocks(parameters) + const passes = options.passes ?? 2 + + if (options.method === "local-search") { + const result = await improve(selected, parameters, evaluateAndRecord) + selected = result.selected + evaluatedCandidates += result.evaluations + } else if (options.method === "staged") { + for (const block of blocks) { + const result = await improve(selected, block, evaluateAndRecord) + selected = result.selected + evaluatedCandidates += result.evaluations + } + } else if (options.method === "alternating-block-coordinate") { + for (let pass = 0; pass < passes; pass++) { + const ordered = pass % 2 === 0 ? blocks : [...blocks].reverse() + for (const block of ordered) { + const result = await improve(selected, block, evaluateAndRecord) + selected = result.selected + evaluatedCandidates += result.evaluations + } + } + } else { + const sequence = SCOUT_SEQUENCES.halton + const restartCount = options.restarts ?? 4 + const points = sequence.points(Math.max(0, restartCount - 1), parameters.length) + const starts = [ + options.seed, + ...points.map((point) => { + let config = options.seed + for (let index = 0; index < parameters.length; index++) { + const parameter = parameters[index]! + config = parameter.update( + config, + parameter.values[scoutLevelIndex(point[index]!, parameter.values.length)]!, + ) + } + return config + }), + ] + for (const start of await evaluate(starts, evaluateAndRecord)) { + const result = await improve(start, parameters, evaluateAndRecord) + if (compareCandidates(result.selected, selected) < 0) selected = result.selected + evaluatedCandidates += result.evaluations + 1 + } + } + + const candidates = [...archive.values()] + return { + model: options.model, + method: options.method, + optimalityClaim: "derivative-free-heuristic-no-global-optimality-claim", + dimensions, + evaluatedCandidates, + selected, + complexityAware: selectComplexityAwareCandidate(candidates, options.complexityPenalty ?? 0.001), + oneStandardError: selectOneStandardErrorCandidate(candidates), + } +} + +/** Score fixed development-selected candidates on an excluded holdout without reselection. */ +export const evaluateRouterComparisonHoldout = async ( + comparison: RouterComparisonResult, + evaluateExcludedHoldout: ( + configs: readonly EvidenceRouterParameters[], + ) => Promise, +): Promise => { + const scores = await evaluateExcludedHoldout([ + comparison.selected.config, + comparison.complexityAware.config, + comparison.oneStandardError.config, + ]) + const [selected, complexityAware, oneStandardError] = scores + if (selected === undefined || complexityAware === undefined || oneStandardError === undefined) + throw new Error("Router comparison holdout evaluator returned the wrong score count") + return { + model: comparison.model, + method: comparison.method, + selected, + complexityAware, + oneStandardError, + } +} + +/** Run the complete two-model, four-method comparison with a model-specific development evaluator. */ +export const compareRouterModelsAndMethods = async ( + options: Omit & { + readonly evaluateDevelopment: ( + model: RouterModel, + configs: readonly EvidenceRouterParameters[], + ) => Promise + }, +): Promise => { + const results: RouterComparisonResult[] = [] + for (const model of ROUTER_MODELS) { + for (const method of ROUTER_FITTING_METHODS) { + results.push( + await searchRouterComparison({ + ...options, + model, + method, + evaluateDevelopment: (configs) => options.evaluateDevelopment(model, configs), + }), + ) + } + } + return results +} diff --git a/benchmarks/tests/router-comparisons.test.ts b/benchmarks/tests/router-comparisons.test.ts new file mode 100644 index 0000000..932b4df --- /dev/null +++ b/benchmarks/tests/router-comparisons.test.ts @@ -0,0 +1,150 @@ +import { describe, expect, it } from "@effect/vitest" + +import { ZERO_CHANNEL_COEFFICIENTS, type ChannelRankings } from "../../src/domain/retrieval.js" +import { buildRoutingEvidence } from "../../src/lib/retrieval/evidence-router.js" +import { + classifyRouterDimensions, + compareRouterModelsAndMethods, + evaluateRouterComparisonHoldout, + routeWithComparisonModel, + searchRouterComparison, + selectComplexityAwareCandidate, + selectOneStandardErrorCandidate, + type RouterComparisonCandidate, + type RouterFittingMethod, +} from "../retrieval/evaluation/router-search/comparisons.js" +import { + emptyRouterConfig, + routerParameters, +} from "../retrieval/evaluation/router-search/config-space.js" + +const rankings: ChannelRankings = { + identity: [], + camelcase: [], + bm25: [ + { chunkIndex: 0, score: 10 }, + { chunkIndex: 1, score: 1 }, + ], + dense: [ + { chunkIndex: 0, score: 0.7 }, + { chunkIndex: 2, score: 0.5 }, + ], + sparse: [], +} +const evidence = [ + buildRoutingEvidence("target", rankings), + buildRoutingEvidence( + "find where the project configuration implementation loads all user settings", + rankings, + ), +] +const seed = emptyRouterConfig({ identity: 0, camelcase: 0, bm25: 1, dense: 1, sparse: 0 }) + +describe("router comparisons", () => { + it("routes with bounded multiplicative and log-linear benchmark models", () => { + const config = { + ...seed, + queryLengthInfluence: { ...ZERO_CHANNEL_COEFFICIENTS, dense: 1 }, + } + const multiplicative = routeWithComparisonModel("multiplicative", evidence[1]!, config) + const logLinear = routeWithComparisonModel("regularized-log-linear", evidence[1]!, config) + + expect(logLinear.dense).toBeGreaterThan(0) + expect(logLinear.dense).not.toBe(multiplicative.dense) + expect(logLinear.identity).toBe(0) + }) + + it("records active, inactive, and data-constant dimensions", () => { + const selectedParameters = routerParameters().filter((parameter) => + [ + "baseWeights.bm25", + "scoreInfluence.identity", + "scoreInfluence.bm25", + "queryLengthInfluence.bm25", + ].includes(parameter.name), + ) + const diagnostics = classifyRouterDimensions(selectedParameters, evidence) + + expect(diagnostics.find((entry) => entry.name === "scoreInfluence.identity")?.status).toBe( + "inactive", + ) + expect(diagnostics.find((entry) => entry.name === "scoreInfluence.bm25")?.status).toBe( + "data-constant", + ) + expect(diagnostics.find((entry) => entry.name === "queryLengthInfluence.bm25")?.status).toBe( + "active", + ) + }) + + it("compares all derivative-free fitting methods with development-only scores", async () => { + const parameters = routerParameters().filter((parameter) => + ["baseWeights.bm25", "queryLengthInfluence.bm25"].includes(parameter.name), + ) + const methods: readonly RouterFittingMethod[] = [ + "staged", + "alternating-block-coordinate", + "deterministic-restarts", + "local-search", + ] + + for (const method of methods) { + const result = await searchRouterComparison({ + model: "regularized-log-linear", + method, + seed, + parameters, + evidence, + pruneInactive: true, + evaluateDevelopment: async (configs) => + configs.map((config) => ({ + mean: config.baseWeights.bm25 + config.queryLengthInfluence.bm25, + standardError: 0.01, + })), + }) + expect(result.selected.score.mean).toBeGreaterThan(0) + expect(result.optimalityClaim).toBe("derivative-free-heuristic-no-global-optimality-claim") + const holdout = await evaluateRouterComparisonHoldout(result, async (configs) => + configs.map(() => ({ mean: 0.5, standardError: 0.1 })), + ) + expect(holdout.selected.mean).toBe(0.5) + } + }) + + it("supports complexity utility and one-standard-error simplicity selection", () => { + const simple: RouterComparisonCandidate = { + config: seed, + score: { mean: 0.79, standardError: 0.01 }, + } + const complex: RouterComparisonCandidate = { + config: { + ...seed, + scoreInfluence: { ...ZERO_CHANNEL_COEFFICIENTS, bm25: 1 }, + }, + score: { mean: 0.8, standardError: 0.02 }, + } + + expect(selectComplexityAwareCandidate([complex, simple], 0.02)).toBe(simple) + expect(selectOneStandardErrorCandidate([complex, simple])).toBe(simple) + }) + + it("runs the complete product and log-linear comparison grid", async () => { + const parameters = routerParameters().filter( + (parameter) => parameter.name === "queryLengthInfluence.bm25", + ) + const results = await compareRouterModelsAndMethods({ + seed, + parameters, + evidence, + evaluateDevelopment: async (model, configs) => + configs.map((config) => ({ + mean: config.queryLengthInfluence.bm25 + (model === "multiplicative" ? 0 : 0.01), + standardError: 0.01, + })), + }) + + expect(results).toHaveLength(8) + expect(new Set(results.map((result) => result.model))).toEqual( + new Set(["multiplicative", "regularized-log-linear"]), + ) + }) +}) From 3463d81daacef6446a70b4d6cecbc282a43cae48 Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Fri, 28 Aug 2026 17:58:11 +0200 Subject: [PATCH 05/16] feat(bench): fit corpus-size influence via sub-sampled corpora --- benchmarks/README.md | 12 + .../retrieval/evaluation/corpus-size.ts | 267 ++++++++++++++++++ .../tests/corpus-size-execution.test.ts | 117 ++++++++ benchmarks/tests/corpus-size.test.ts | 144 ++++++++++ package.json | 1 + 5 files changed, 541 insertions(+) create mode 100644 benchmarks/retrieval/evaluation/corpus-size.ts create mode 100644 benchmarks/tests/corpus-size-execution.test.ts create mode 100644 benchmarks/tests/corpus-size.test.ts diff --git a/benchmarks/README.md b/benchmarks/README.md index b2f2d50..008439d 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -104,6 +104,18 @@ vp run bench:retrieval:matrix This command runs 30 model-backed benchmarks. Use it for release evidence, not the development loop. +### Corpus-size influence + +`vp run bench:retrieval:corpus-size` plans deterministic sub-samples of a pinned corpus at 200, 500, +1000, 2500, 5000, and full chunk counts. Every sub-sample keeps the gold chunks of its evaluated +questions resolvable; questions whose gold no longer fits the size budget are dropped and recorded. + +The sweep module (`benchmarks/retrieval/evaluation/corpus-size.ts`) runs the same search protocol per +size, records the optimum series with exact `(corpus size, fusion, profile, objective, strategy, +fold)` coordinates, and fits a per-channel log-linear model over `log(chunkCount)`. A sensitivity +check compares weight shifts between adjacent sizes against the combined one-standard-error noise +band and recommends promoting a corpus-size factor only when a shift leaves that band. + | Profile | Repositories | Models | Validation | Static fusion | Router fusion | Diagnostics | | ---------- | ------------ | -------- | ----------------------------- | ------------- | ------------- | ---------------------- | | `smoke` | fd | MiniLM | grouped 5-fold | DBSF | DBSF | current router | diff --git a/benchmarks/retrieval/evaluation/corpus-size.ts b/benchmarks/retrieval/evaluation/corpus-size.ts new file mode 100644 index 0000000..4b0b5c7 --- /dev/null +++ b/benchmarks/retrieval/evaluation/corpus-size.ts @@ -0,0 +1,267 @@ +import type { Bm25Index } from "../../../src/domain/ports.js" +import { + CHANNEL_NAMES, + type ChannelName, + type ChannelWeights, +} from "../../../src/domain/retrieval.js" +import { buildBm25Index } from "../../../src/lib/retrieval/bm25.js" +import { buildIdentifierIndex } from "../../../src/lib/retrieval/identifier-index.js" +import type { PreparedCorpus } from "../corpus/prepare.js" +import type { ChunkIdentifiers } from "./metrics.js" + +/** Default corpus-size sweep; `Infinity` means the full corpus. */ +export const CORPUS_SIZE_STEPS: readonly number[] = [ + 200, + 500, + 1000, + 2500, + 5000, + Number.POSITIVE_INFINITY, +] + +/** Per-question gold chunk indices in the original corpus index space. */ +export interface CorpusSizeQuestionGold { + readonly questionId: string + readonly goldChunkIndices: readonly number[] +} + +/** One recorded question drop from a sub-sample. */ +export interface CorpusSizeDroppedQuestion { + readonly questionId: string + readonly reason: "gold-exceeds-size-budget" +} + +/** Deterministic sub-sample plan for one corpus size. */ +export interface CorpusSizeSubSample { + readonly targetSize: number + readonly chunkIndices: readonly number[] + readonly keptQuestionIds: readonly string[] + readonly droppedQuestions: readonly CorpusSizeDroppedQuestion[] +} + +/** + * Plan deterministic sub-samples that keep every evaluated question's gold chunks resolvable. + * Questions whose gold no longer fits a size budget are dropped and recorded. + */ +export const planCorpusSizeSubSamples = ( + questions: readonly CorpusSizeQuestionGold[], + chunkCount: number, + sizes: readonly number[] = CORPUS_SIZE_STEPS, +): readonly CorpusSizeSubSample[] => + [...sizes] + .sort((left, right) => left - right) + .map((targetSize) => { + if (!Number.isFinite(targetSize) || targetSize >= chunkCount) + return { + targetSize, + chunkIndices: Array.from({ length: chunkCount }, (_, index) => index), + keptQuestionIds: questions.map((question) => question.questionId), + droppedQuestions: [], + } + + const keptQuestionIds: string[] = [] + const droppedQuestions: CorpusSizeDroppedQuestion[] = [] + const mustKeep = new Set() + for (const question of questions) { + const missing = question.goldChunkIndices.filter((index) => !mustKeep.has(index)) + if (mustKeep.size + missing.length <= targetSize) { + keptQuestionIds.push(question.questionId) + for (const index of missing) mustKeep.add(index) + } else { + droppedQuestions.push({ + questionId: question.questionId, + reason: "gold-exceeds-size-budget", + }) + } + } + + const candidates = Array.from({ length: chunkCount }, (_, index) => index).filter( + (index) => !mustKeep.has(index), + ) + const remaining = targetSize - mustKeep.size + const picked = new Set() + for (let slot = 0; slot < remaining; slot++) + picked.add(candidates[Math.floor((slot * candidates.length) / remaining)]!) + const chunkIndices = [...mustKeep, ...picked].sort((left, right) => left - right) + return { targetSize, chunkIndices, keptQuestionIds, droppedQuestions } + }) + +/** Rebuilt indexes and remapped gold targets for one sub-sample. */ +export interface SubSampleCorpus { + readonly chunks: readonly PreparedCorpus["chunks"][number][] + readonly bm25Index: Bm25Index + readonly identifiersByChunk: ChunkIdentifiers + readonly identifierIndex: ReturnType + readonly chunkIndexMap: ReadonlyMap +} + +/** Build runnable indexes for a sub-sample, remapping every chunk index into dense 0..N-1 space. */ +export const buildSubSampleCorpus = ( + corpus: PreparedCorpus, + plan: CorpusSizeSubSample, +): SubSampleCorpus => { + const chunkIndexMap = new Map( + plan.chunkIndices.map((originalIndex, newIndex) => [originalIndex, newIndex]), + ) + const chunks = plan.chunkIndices.map((originalIndex) => corpus.chunks[originalIndex]!) + const identifiersByChunk: Map> = new Map() + const identifiers: { name: string; chunkIndex: number }[] = [] + for (const [originalIndex, newIndex] of chunkIndexMap) { + const names = corpus.identifiersByChunk.get(originalIndex) + if (names === undefined) continue + identifiersByChunk.set(newIndex, names) + for (const name of names) identifiers.push({ name, chunkIndex: newIndex }) + } + return { + chunks, + bm25Index: buildBm25Index(chunks.map((chunk, index) => ({ index, text: chunk.text }))), + identifiersByChunk, + identifierIndex: buildIdentifierIndex( + identifiers.map((identifier) => ({ ...identifier, kind: "value" as const })), + ), + chunkIndexMap, + } +} + +/** Remap gold chunk-index sets from the original index space into a sub-sample's dense space. */ +export const remapGoldTargets = ( + targets: readonly ReadonlySet[], + chunkIndexMap: ReadonlyMap, +): readonly ReadonlySet[] => + targets.map((target) => { + const remapped = new Set() + for (const index of target) { + const mapped = chunkIndexMap.get(index) + if (mapped !== undefined) remapped.add(mapped) + } + return remapped + }) + +/** Optimal router weights measured at one corpus size. */ +export interface CorpusSizeOptimum { + readonly corpusSize: number + readonly weights: ChannelWeights + /** One-standard-error noise band of the selected optimum. */ + readonly noise: number +} + +/** Fitted corpus-size relationship across channels. */ +export interface CorpusSizeFit { + readonly logLinear: readonly { channel: ChannelName; slope: number; intercept: number }[] + readonly largestRelativeShift: { + corpusSize: number + channel: ChannelName + relativeShift: number + } | null + readonly sensitivity: { + readonly shiftedOutsideNoise: boolean + readonly pairs: readonly { + readonly fromSize: number + readonly toSize: number + readonly channel: ChannelName + readonly delta: number + readonly noise: number + }[] + } + readonly recommendation: "promote-corpus-size-factor" | "do-not-promote-corpus-size-factor" +} + +const leastSquares = ( + points: readonly (readonly [number, number])[], +): { slope: number; intercept: number } => { + const n = points.length + if (n === 0) return { slope: 0, intercept: 0 } + const meanX = points.reduce((sum, [x]) => sum + x, 0) / n + const meanY = points.reduce((sum, [, y]) => sum + y, 0) / n + let covariance = 0 + let variance = 0 + for (const [x, y] of points) { + covariance += (x - meanX) * (y - meanY) + variance += (x - meanX) ** 2 + } + const slope = variance === 0 ? 0 : covariance / variance + return { slope, intercept: meanY - slope * meanX } +} + +/** + * Fit the corpus-size relationship from the per-size optimum series: a log-linear curve over + * log(chunkCount) per channel plus a sensitivity check against measurement noise. + */ +export const fitCorpusSizeModel = (optima: readonly CorpusSizeOptimum[]): CorpusSizeFit => { + const finite = optima.filter((optimum) => Number.isFinite(optimum.corpusSize)) + const logLinear = CHANNEL_NAMES.map((channel) => { + const { slope, intercept } = leastSquares( + finite.map((optimum) => [Math.log(optimum.corpusSize), optimum.weights[channel]] as const), + ) + return { channel, slope, intercept } + }) + + const pairs: { + fromSize: number + toSize: number + channel: ChannelName + delta: number + noise: number + }[] = [] + let largest: CorpusSizeFit["largestRelativeShift"] = null + for (let index = 1; index < optima.length; index++) { + const from = optima[index - 1]! + const to = optima[index]! + for (const channel of CHANNEL_NAMES) { + const delta = Math.abs(to.weights[channel] - from.weights[channel]) + const noise = Math.hypot(from.noise, to.noise) + pairs.push({ fromSize: from.corpusSize, toSize: to.corpusSize, channel, delta, noise }) + const base = Math.max(from.weights[channel], Number.EPSILON) + const relativeShift = delta / base + if (largest === null || relativeShift > largest.relativeShift) + largest = { corpusSize: to.corpusSize, channel, relativeShift } + } + } + const shiftedOutsideNoise = pairs.some((pair) => pair.delta > pair.noise) + return { + logLinear, + largestRelativeShift: largest, + sensitivity: { shiftedOutsideNoise, pairs }, + recommendation: shiftedOutsideNoise + ? "promote-corpus-size-factor" + : "do-not-promote-corpus-size-factor", + } +} + +/** Exact identity of one per-size sweep result. */ +export interface CorpusSizeSweepCoordinate { + readonly corpusSize: number + readonly fusion: string + readonly profile: string + readonly objective: string + readonly strategy: string + readonly fold: string +} + +/** One per-size sweep result with its optimal weights and score noise. */ +export interface CorpusSizeSweepRow { + readonly coordinate: CorpusSizeSweepCoordinate + readonly weights: ChannelWeights + readonly score: number + readonly noise: number +} + +/** Search protocol executed identically for every corpus size. */ +export type CorpusSizeSearch = (plan: CorpusSizeSubSample) => Promise + +/** Orchestrate the per-size sweep and derive the fitted model from its optima. */ +export const runCorpusSizeSweep = async ( + plans: readonly CorpusSizeSubSample[], + searchAtSize: CorpusSizeSearch, +): Promise<{ readonly rows: readonly CorpusSizeSweepRow[]; readonly fit: CorpusSizeFit }> => { + const rows: CorpusSizeSweepRow[] = [] + for (const plan of plans) rows.push(...(await searchAtSize(plan))) + const optima = plans.map((plan) => { + const planRows = rows.filter((row) => row.coordinate.corpusSize === plan.targetSize) + const best = [...planRows].sort((left, right) => right.score - left.score)[0] + if (best === undefined) + throw new Error(`Corpus-size sweep produced no result for size ${plan.targetSize}`) + return { corpusSize: plan.targetSize, weights: best.weights, noise: best.noise } + }) + return { rows, fit: fitCorpusSizeModel(optima) } +} diff --git a/benchmarks/tests/corpus-size-execution.test.ts b/benchmarks/tests/corpus-size-execution.test.ts new file mode 100644 index 0000000..b63ef5e --- /dev/null +++ b/benchmarks/tests/corpus-size-execution.test.ts @@ -0,0 +1,117 @@ +import { mkdir, writeFile } from "node:fs/promises" +import path from "node:path" + +import { expect, it } from "@effect/vitest" +import { Effect, Schema } from "effect" + +import type { ChunkingOptions } from "../../src/domain/ports.js" +import { ZERO_CHANNEL_COEFFICIENTS } from "../../src/domain/retrieval.js" +import { prepareCorpus } from "../retrieval/corpus/prepare.js" +import { prepareRepository } from "../retrieval/corpus/repository.js" +import { loadCorpusManifests } from "../retrieval/corpus/repository.js" +import { + CORPUS_SIZE_STEPS, + buildSubSampleCorpus, + planCorpusSizeSubSamples, + remapGoldTargets, + runCorpusSizeSweep, + type CorpusSizeQuestionGold, +} from "../retrieval/evaluation/corpus-size.js" +import { resolveGoldTargets } from "../retrieval/evaluation/metrics.js" + +const validationChunkingOptions: ChunkingOptions = { + maxTokens: Number.MAX_SAFE_INTEGER, + overlapLines: 0, + countTokens: () => Effect.succeed(0), + onDiagnostic: () => Effect.void, +} + +it.effect("plans, validates, and sweeps corpus-size sub-samples on a pinned corpus", () => + Effect.gen(function* () { + const manifests = yield* loadCorpusManifests() + const manifest = manifests.find((entry) => entry.id === "fd") + if (manifest === undefined) throw new Error("fd corpus manifest is missing") + const repositoryPath = yield* prepareRepository(manifest) + const corpus = yield* prepareCorpus(repositoryPath, manifest, validationChunkingOptions) + + const questions: CorpusSizeQuestionGold[] = manifest.questions.map((question) => ({ + questionId: question.id, + goldChunkIndices: [ + ...new Set( + resolveGoldTargets( + question.groundTruth, + corpus.chunks, + corpus.identifiersByChunk, + ).flatMap((targets) => [...targets]), + ), + ], + })) + expect(questions.every((question) => question.goldChunkIndices.length > 0)).toBe(true) + + const plans = planCorpusSizeSubSamples(questions, corpus.chunks.length, CORPUS_SIZE_STEPS) + for (const plan of plans) { + const subSample = buildSubSampleCorpus(corpus, plan) + const keptQuestions = questions.filter((question) => + plan.keptQuestionIds.includes(question.questionId), + ) + const resolved = remapGoldTargets( + keptQuestions.flatMap((question) => + resolveGoldTargets( + manifest.questions.find((entry) => entry.id === question.questionId)!.groundTruth, + corpus.chunks, + corpus.identifiersByChunk, + ), + ), + subSample.chunkIndexMap, + ) + expect(resolved.every((targets) => targets.size > 0)).toBe(true) + expect(subSample.chunks.length).toBe( + Math.min(plan.targetSize, corpus.chunks.length) || corpus.chunks.length, + ) + } + + const { rows, fit } = yield* Effect.promise(() => + runCorpusSizeSweep(plans, async (plan) => [ + { + coordinate: { + corpusSize: plan.targetSize, + fusion: "dbsf", + profile: "search-priority", + objective: "direct", + strategy: "grouped-3-fold", + fold: "1", + }, + weights: { ...ZERO_CHANNEL_COEFFICIENTS, bm25: 1 + Math.log10(plan.targetSize) / 10 }, + score: 0.8, + noise: 0.01, + }, + ]), + ) + + const outputDirectory = path.resolve("benchmarks/results") + const outputPath = path.join(outputDirectory, "retrieval-corpus-size-sweep.json") + yield* Effect.tryPromise({ + try: async () => { + await mkdir(outputDirectory, { recursive: true }) + await writeFile( + outputPath, + `${JSON.stringify({ schemaVersion: 1, rows, fit }, null, 2)}\n`, + "utf8", + ) + }, + catch: (cause) => new Error(`Could not write corpus-size sweep ${outputPath}`, { cause }), + }) + + expect(rows.length).toBe(plans.length) + expect( + Schema.is( + Schema.Struct({ + recommendation: Schema.Literals([ + "promote-corpus-size-factor", + "do-not-promote-corpus-size-factor", + ]), + }), + )({ recommendation: fit.recommendation }), + ).toBe(true) + }), +) diff --git a/benchmarks/tests/corpus-size.test.ts b/benchmarks/tests/corpus-size.test.ts new file mode 100644 index 0000000..24a8826 --- /dev/null +++ b/benchmarks/tests/corpus-size.test.ts @@ -0,0 +1,144 @@ +import { describe, expect, it } from "@effect/vitest" + +import { ZERO_CHANNEL_COEFFICIENTS } from "../../src/domain/retrieval.js" +import type { PreparedCorpus } from "../retrieval/corpus/prepare.js" +import { + buildSubSampleCorpus, + fitCorpusSizeModel, + planCorpusSizeSubSamples, + remapGoldTargets, + runCorpusSizeSweep, + type CorpusSizeQuestionGold, +} from "../retrieval/evaluation/corpus-size.js" + +const chunk = (file: string, startLine: number) => ({ + file, + startLine, + endLine: startLine + 5, + text: `symbol ${startLine}`, +}) + +const corpus = ( + chunkCount: number, + namesByChunk: ReadonlyMap, +): PreparedCorpus => + ({ + chunks: Array.from({ length: chunkCount }, (_, index) => chunk(`src/file${index}.ts`, index)), + bm25Index: { chunkLengths: [], docFreqs: {}, totalDocs: chunkCount }, + identifierIndex: { exact: {}, split: {} }, + identifiersByChunk: new Map([...namesByChunk].map(([index, names]) => [index, new Set(names)])), + preparationDurationMs: 0, + }) as unknown as PreparedCorpus + +const questions: readonly CorpusSizeQuestionGold[] = [ + { questionId: "q1", goldChunkIndices: [3] }, + { questionId: "q2", goldChunkIndices: [7, 9] }, + { questionId: "q3", goldChunkIndices: [50] }, +] + +describe("corpus-size influence", () => { + it("keeps gold chunks resolvable and records unavoidable drops", () => { + const plans = planCorpusSizeSubSamples(questions, 100, [3, 200]) + const small = plans[0]! + const full = plans[1]! + + expect(full.chunkIndices).toHaveLength(100) + expect(full.droppedQuestions).toEqual([]) + expect(small.chunkIndices).toHaveLength(3) + for (const question of questions.filter((entry) => + small.keptQuestionIds.includes(entry.questionId), + )) { + for (const index of question.goldChunkIndices) expect(small.chunkIndices).toContain(index) + } + expect(small.droppedQuestions.map((drop) => drop.questionId)).toEqual(["q3"]) + expect(small.keptQuestionIds).toEqual(["q1", "q2"]) + }) + + it("rebuilds dense indexes and remaps gold targets into sub-sample space", () => { + const source = corpus( + 60, + new Map([ + [50, ["targetSymbol"]], + [3, ["otherSymbol"]], + ]), + ) + const [plan] = planCorpusSizeSubSamples(questions, 60, [10]) + const subSample = buildSubSampleCorpus(source, plan!) + + expect(subSample.chunks).toHaveLength(10) + expect(subSample.chunkIndexMap.get(50)).toBeDefined() + expect(subSample.identifiersByChunk.get(subSample.chunkIndexMap.get(50)!)).toEqual( + new Set(["targetSymbol"]), + ) + expect(subSample.identifierIndex.exact["targetsymbol"]).toEqual([ + subSample.chunkIndexMap.get(50), + ]) + expect(subSample.bm25Index.chunkLengths).toHaveLength(10) + + const remapped = remapGoldTargets([new Set([50])], subSample.chunkIndexMap) + expect([...remapped[0]!]).toEqual([subSample.chunkIndexMap.get(50)]) + }) + + it("fits a log-linear relationship and detects shifts outside noise", () => { + const fit = fitCorpusSizeModel([ + { + corpusSize: 100, + weights: { identity: 1, camelcase: 1, bm25: 1, dense: 1, sparse: 1 }, + noise: 0.001, + }, + { + corpusSize: 1000, + weights: { identity: 1, camelcase: 1, bm25: 2, dense: 1, sparse: 1 }, + noise: 0.001, + }, + ]) + + expect(fit.logLinear.find((entry) => entry.channel === "bm25")?.slope).toBeGreaterThan(0) + expect(fit.sensitivity.shiftedOutsideNoise).toBe(true) + expect(fit.recommendation).toBe("promote-corpus-size-factor") + }) + + it("recommends no promotion when optima stay inside measurement noise", () => { + const fit = fitCorpusSizeModel([ + { + corpusSize: 100, + weights: { identity: 1, camelcase: 1, bm25: 1, dense: 1, sparse: 1 }, + noise: 0.5, + }, + { + corpusSize: 1000, + weights: { identity: 1, camelcase: 1, bm25: 1.1, dense: 1, sparse: 1 }, + noise: 0.5, + }, + ]) + + expect(fit.sensitivity.shiftedOutsideNoise).toBe(false) + expect(fit.recommendation).toBe("do-not-promote-corpus-size-factor") + }) + + it("runs the identical sweep protocol per size and derives the fit", async () => { + const plans = planCorpusSizeSubSamples(questions, 100, [10, 100]) + const { rows, fit } = await runCorpusSizeSweep(plans, async (plan) => [ + { + coordinate: { + corpusSize: plan.targetSize, + fusion: "dbsf", + profile: "search-priority", + objective: "direct", + strategy: "grouped-3-fold", + fold: "1", + }, + weights: { + ...ZERO_CHANNEL_COEFFICIENTS, + bm25: plan.targetSize >= 100 ? 1.5 : 1, + }, + score: 0.8, + noise: 0.01, + }, + ]) + + expect(rows).toHaveLength(2) + expect(new Set(rows.map((row) => row.coordinate.corpusSize))).toEqual(new Set([10, 100])) + expect(fit.recommendation).toBe("promote-corpus-size-factor") + }) +}) diff --git a/package.json b/package.json index cd6e4cc..0faf9b3 100644 --- a/package.json +++ b/package.json @@ -36,6 +36,7 @@ "scripts": { "bench:retrieval": "vp run bench:retrieval:validate", "bench:retrieval:corpus": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/corpus.test.ts", + "bench:retrieval:corpus-size": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/corpus-size-execution.test.ts", "bench:retrieval:develop": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/retrieval.test.ts -t \"runs the develop retrieval profile\"", "bench:retrieval:fixture": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/channels.test.ts", "bench:retrieval:full": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/retrieval.test.ts -t \"runs the full retrieval profile\"", From 5cc4b58e47a45bfa9a3ee03c2a070991cbf05c4a Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Fri, 28 Aug 2026 18:05:42 +0200 Subject: [PATCH 06/16] fix(bench): enforce gold retention in corpus-size sub-samples --- .../retrieval/evaluation/corpus-size.ts | 49 ++++++++++++--- .../tests/corpus-size-execution.test.ts | 35 ++++++----- benchmarks/tests/corpus-size.test.ts | 63 ++++++++++++++----- 3 files changed, 108 insertions(+), 39 deletions(-) diff --git a/benchmarks/retrieval/evaluation/corpus-size.ts b/benchmarks/retrieval/evaluation/corpus-size.ts index 4b0b5c7..b508510 100644 --- a/benchmarks/retrieval/evaluation/corpus-size.ts +++ b/benchmarks/retrieval/evaluation/corpus-size.ts @@ -41,28 +41,34 @@ export interface CorpusSizeSubSample { /** * Plan deterministic sub-samples that keep every evaluated question's gold chunks resolvable. - * Questions whose gold no longer fits a size budget are dropped and recorded. + * Questions whose gold no longer fits a size budget are dropped and recorded. Gold indices are + * deduplicated per question, and a post-condition throws if a kept question's gold ever falls + * outside the selected chunks. */ export const planCorpusSizeSubSamples = ( questions: readonly CorpusSizeQuestionGold[], chunkCount: number, sizes: readonly number[] = CORPUS_SIZE_STEPS, -): readonly CorpusSizeSubSample[] => - [...sizes] +): readonly CorpusSizeSubSample[] => { + const deduped = questions.map((question) => ({ + ...question, + goldChunkIndices: [...new Set(question.goldChunkIndices)], + })) + return [...sizes] .sort((left, right) => left - right) .map((targetSize) => { if (!Number.isFinite(targetSize) || targetSize >= chunkCount) return { targetSize, chunkIndices: Array.from({ length: chunkCount }, (_, index) => index), - keptQuestionIds: questions.map((question) => question.questionId), + keptQuestionIds: deduped.map((question) => question.questionId), droppedQuestions: [], } const keptQuestionIds: string[] = [] const droppedQuestions: CorpusSizeDroppedQuestion[] = [] const mustKeep = new Set() - for (const question of questions) { + for (const question of deduped) { const missing = question.goldChunkIndices.filter((index) => !mustKeep.has(index)) if (mustKeep.size + missing.length <= targetSize) { keptQuestionIds.push(question.questionId) @@ -83,8 +89,28 @@ export const planCorpusSizeSubSamples = ( for (let slot = 0; slot < remaining; slot++) picked.add(candidates[Math.floor((slot * candidates.length) / remaining)]!) const chunkIndices = [...mustKeep, ...picked].sort((left, right) => left - right) - return { targetSize, chunkIndices, keptQuestionIds, droppedQuestions } + const plan = { targetSize, chunkIndices, keptQuestionIds, droppedQuestions } + validateSubSampleGoldCoverage(deduped, plan) + return plan }) +} + +/** Fail fast when a kept question's gold chunks are not fully contained in the sub-sample. */ +export const validateSubSampleGoldCoverage = ( + questions: readonly CorpusSizeQuestionGold[], + plan: CorpusSizeSubSample, +): void => { + const selected = new Set(plan.chunkIndices) + const kept = new Set(plan.keptQuestionIds) + for (const question of questions) { + if (!kept.has(question.questionId)) continue + const missing = question.goldChunkIndices.filter((index) => !selected.has(index)) + if (missing.length > 0) + throw new Error( + `Corpus-size sub-sample ${plan.targetSize} lost gold chunks for ${question.questionId}: ${missing.join(", ")}`, + ) + } +} /** Rebuilt indexes and remapped gold targets for one sub-sample. */ export interface SubSampleCorpus { @@ -249,13 +275,20 @@ export interface CorpusSizeSweepRow { /** Search protocol executed identically for every corpus size. */ export type CorpusSizeSearch = (plan: CorpusSizeSubSample) => Promise -/** Orchestrate the per-size sweep and derive the fitted model from its optima. */ +/** + * Orchestrate the per-size sweep and derive the fitted model from its optima. When the question + * gold is supplied, every plan is validated to still resolve all kept questions' gold chunks. + */ export const runCorpusSizeSweep = async ( plans: readonly CorpusSizeSubSample[], searchAtSize: CorpusSizeSearch, + questions?: readonly CorpusSizeQuestionGold[], ): Promise<{ readonly rows: readonly CorpusSizeSweepRow[]; readonly fit: CorpusSizeFit }> => { const rows: CorpusSizeSweepRow[] = [] - for (const plan of plans) rows.push(...(await searchAtSize(plan))) + for (const plan of plans) { + if (questions !== undefined) validateSubSampleGoldCoverage(questions, plan) + rows.push(...(await searchAtSize(plan))) + } const optima = plans.map((plan) => { const planRows = rows.filter((row) => row.coordinate.corpusSize === plan.targetSize) const best = [...planRows].sort((left, right) => right.score - left.score)[0] diff --git a/benchmarks/tests/corpus-size-execution.test.ts b/benchmarks/tests/corpus-size-execution.test.ts index b63ef5e..8c8b536 100644 --- a/benchmarks/tests/corpus-size-execution.test.ts +++ b/benchmarks/tests/corpus-size-execution.test.ts @@ -71,21 +71,28 @@ it.effect("plans, validates, and sweeps corpus-size sub-samples on a pinned corp } const { rows, fit } = yield* Effect.promise(() => - runCorpusSizeSweep(plans, async (plan) => [ - { - coordinate: { - corpusSize: plan.targetSize, - fusion: "dbsf", - profile: "search-priority", - objective: "direct", - strategy: "grouped-3-fold", - fold: "1", + runCorpusSizeSweep( + plans, + async (plan) => [ + { + coordinate: { + corpusSize: plan.targetSize, + fusion: "dbsf", + profile: "search-priority", + objective: "direct", + strategy: "grouped-3-fold", + fold: "1", + }, + weights: { + ...ZERO_CHANNEL_COEFFICIENTS, + bm25: 1 + Math.log10(plan.targetSize) / 10, + }, + score: 0.8, + noise: 0.01, }, - weights: { ...ZERO_CHANNEL_COEFFICIENTS, bm25: 1 + Math.log10(plan.targetSize) / 10 }, - score: 0.8, - noise: 0.01, - }, - ]), + ], + questions, + ), ) const outputDirectory = path.resolve("benchmarks/results") diff --git a/benchmarks/tests/corpus-size.test.ts b/benchmarks/tests/corpus-size.test.ts index 24a8826..d68e322 100644 --- a/benchmarks/tests/corpus-size.test.ts +++ b/benchmarks/tests/corpus-size.test.ts @@ -118,27 +118,56 @@ describe("corpus-size influence", () => { it("runs the identical sweep protocol per size and derives the fit", async () => { const plans = planCorpusSizeSubSamples(questions, 100, [10, 100]) - const { rows, fit } = await runCorpusSizeSweep(plans, async (plan) => [ - { - coordinate: { - corpusSize: plan.targetSize, - fusion: "dbsf", - profile: "search-priority", - objective: "direct", - strategy: "grouped-3-fold", - fold: "1", - }, - weights: { - ...ZERO_CHANNEL_COEFFICIENTS, - bm25: plan.targetSize >= 100 ? 1.5 : 1, + const { rows, fit } = await runCorpusSizeSweep( + plans, + async (plan) => [ + { + coordinate: { + corpusSize: plan.targetSize, + fusion: "dbsf", + profile: "search-priority", + objective: "direct", + strategy: "grouped-3-fold", + fold: "1", + }, + weights: { + ...ZERO_CHANNEL_COEFFICIENTS, + bm25: plan.targetSize >= 100 ? 1.5 : 1, + }, + score: 0.8, + noise: 0.01, }, - score: 0.8, - noise: 0.01, - }, - ]) + ], + questions, + ) expect(rows).toHaveLength(2) expect(new Set(rows.map((row) => row.coordinate.corpusSize))).toEqual(new Set([10, 100])) expect(fit.recommendation).toBe("promote-corpus-size-factor") }) + + it("fails the sweep loudly when a kept question's gold falls outside the sub-sample", async () => { + const invalidPlan = { + targetSize: 5, + chunkIndices: [0, 1, 2], + keptQuestionIds: ["q1"], + droppedQuestions: [], + } + await expect(runCorpusSizeSweep([invalidPlan], async () => [], questions)).rejects.toThrow( + "lost gold chunks for q1", + ) + }) + + it("counts duplicate gold indices once against the size budget", () => { + const [plan] = planCorpusSizeSubSamples( + [{ questionId: "dup", goldChunkIndices: [3, 3, 7] }], + 100, + [10], + ) + + expect(plan!.keptQuestionIds).toEqual(["dup"]) + expect(plan!.chunkIndices).toHaveLength(10) + expect(plan!.chunkIndices.filter((index) => index === 3)).toHaveLength(1) + expect(plan!.droppedQuestions).toEqual([]) + }) }) From 85f352b089eeb8664a2a0db52928d662f856ae1c Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Fri, 28 Aug 2026 18:58:53 +0200 Subject: [PATCH 07/16] feat(bench): real corpus-size sweep and router model comparison --- benchmarks/retrieval/evaluation/collect.ts | 55 ++- .../retrieval/evaluation/corpus-size.ts | 2 +- .../evaluation/subsample-evidence.ts | 417 ++++++++++++++++++ benchmarks/tests/corpus-size-model.test.ts | 45 ++ benchmarks/tests/router-model-real.test.ts | 44 ++ ...21-evidence-router-stays-multiplicative.md | 48 ++ package.json | 2 + 7 files changed, 596 insertions(+), 17 deletions(-) create mode 100644 benchmarks/retrieval/evaluation/subsample-evidence.ts create mode 100644 benchmarks/tests/corpus-size-model.test.ts create mode 100644 benchmarks/tests/router-model-real.test.ts create mode 100644 docs/adr/0021-evidence-router-stays-multiplicative.md diff --git a/benchmarks/retrieval/evaluation/collect.ts b/benchmarks/retrieval/evaluation/collect.ts index 53de0e5..dc510bc 100644 --- a/benchmarks/retrieval/evaluation/collect.ts +++ b/benchmarks/retrieval/evaluation/collect.ts @@ -113,7 +113,7 @@ export interface CollectedBenchmarkData { const isLongInput = (text: string): boolean => Buffer.byteLength(text, "utf8") / 4 > SINGLE_ITEM_ESTIMATED_TOKENS -const embedTexts = ( +export const embedTexts = ( texts: readonly string[], model: string, embedder: BoundEmbedder, @@ -134,7 +134,7 @@ const embedTexts = ( return vectors }).pipe(Effect.mapError((cause) => new Error(`Embedding failed for ${model}`, { cause }))) -const embedSparseTexts = ( +export const embedSparseTexts = ( texts: readonly string[], embedder: typeof SparseEmbedder.Service, ): Effect.Effect => @@ -154,9 +154,9 @@ const toStoredChunk = (chunk: PreparedCorpus["chunks"][number]): StoredChunk => return { ...location, contentHash: contentHash(text) } } -const persistBenchmarkCorpus = ( +export const persistBenchmarkCorpus = ( store: typeof IndexStore.Service, - corpus: PreparedCorpus, + corpus: Pick, vectors: readonly Float32Array[], sparseVectors: readonly SparseVector[], dims: number, @@ -473,19 +473,18 @@ const inspectBenchmarkCache = ( cachePaths.databasePath, ) -const collectRepositoryMeasurements = ( - manifest: CorpusManifest, - models: readonly string[], - groupedFoldAssignments: ReadonlyMap, -): Effect.Effect => +/** Resolved model registry entry, loaded device embedder, and the effective chunk token budget. */ +export interface ResolvedModelContext { + readonly model: string + readonly info: (typeof MODEL_REGISTRY)[string] & object + readonly embedder: BoundEmbedder + readonly device: string + readonly maxTokens: number +} + +/** Resolve one embedding model against the registry and load it on the first working device. */ +export const resolveModelContext = (model: string): Effect.Effect => Effect.gen(function* () { - const repositoryPath = yield* prepareRepository(manifest) - if (models.length !== 1) { - return yield* Effect.fail( - new Error(`Retrieval benchmark requires exactly one model, received ${models.length}`), - ) - } - const model = models[0]! const info = MODEL_REGISTRY[model] if (info === undefined) return yield* Effect.fail(new Error(`Unknown embedding model ${model}`)) const sparseInfo = SPARSE_MODEL_REGISTRY[DEFAULT_CONFIG.sparseEmbedder.model] @@ -507,6 +506,30 @@ const collectRepositoryMeasurements = ( bound.embedder.limits, { model: sparseInfo.id, ...sparseInfo }, ]) + return { + model, + info, + embedder: bound.embedder, + device: bound.device, + maxTokens, + } + }) + +const collectRepositoryMeasurements = ( + manifest: CorpusManifest, + models: readonly string[], + groupedFoldAssignments: ReadonlyMap, +): Effect.Effect => + Effect.gen(function* () { + const repositoryPath = yield* prepareRepository(manifest) + if (models.length !== 1) { + return yield* Effect.fail( + new Error(`Retrieval benchmark requires exactly one model, received ${models.length}`), + ) + } + const context = yield* resolveModelContext(models[0]!) + const { model, info, embedder: boundEmbedder, maxTokens } = context + const bound = { device: context.device, embedder: boundEmbedder } const corpus = yield* prepareCorpus(repositoryPath, manifest, { maxTokens, overlapLines: DEFAULT_CONFIG.overlapLines, diff --git a/benchmarks/retrieval/evaluation/corpus-size.ts b/benchmarks/retrieval/evaluation/corpus-size.ts index b508510..927dfc6 100644 --- a/benchmarks/retrieval/evaluation/corpus-size.ts +++ b/benchmarks/retrieval/evaluation/corpus-size.ts @@ -237,7 +237,7 @@ export const fitCorpusSizeModel = (optima: readonly CorpusSizeOptimum[]): Corpus const delta = Math.abs(to.weights[channel] - from.weights[channel]) const noise = Math.hypot(from.noise, to.noise) pairs.push({ fromSize: from.corpusSize, toSize: to.corpusSize, channel, delta, noise }) - const base = Math.max(from.weights[channel], Number.EPSILON) + const base = Math.max(from.weights[channel], to.weights[channel], 0.1) const relativeShift = delta / base if (largest === null || relativeShift > largest.relativeShift) largest = { corpusSize: to.corpusSize, channel, relativeShift } diff --git a/benchmarks/retrieval/evaluation/subsample-evidence.ts b/benchmarks/retrieval/evaluation/subsample-evidence.ts new file mode 100644 index 0000000..b3be102 --- /dev/null +++ b/benchmarks/retrieval/evaluation/subsample-evidence.ts @@ -0,0 +1,417 @@ +import { Effect } from "effect" + +import { DEFAULT_CONFIG } from "../../../src/domain/config.js" +import type { EmbeddingDtype } from "../../../src/domain/dtype.js" +import type { BoundEmbedder } from "../../../src/domain/ports.js" +import { IndexStore, SparseEmbedder } from "../../../src/domain/ports.js" +import type { EvidenceRouterParameters } from "../../../src/domain/retrieval.js" +import { + buildQueryTermCoverage, + buildRoutingEvidence, + routeWithEvidence, +} from "../../../src/lib/retrieval/evidence-router.js" +import { fuseRankings } from "../../../src/lib/retrieval/fusion.js" +import type { QueryKind } from "../corpus/manifest.js" +import type { CorpusManifest } from "../corpus/manifest.js" +import { prepareCorpus, type PreparedCorpus } from "../corpus/prepare.js" +import { loadCorpusManifests, prepareRepository } from "../corpus/repository.js" +import { + getDefaultWorkerCount, + resolveWorkerCount, +} from "../execution/candidate-evaluation-pool.js" +import { withSqliteBenchmarkStore } from "../execution/sqlite-index.js" +import { + embedSparseTexts, + embedTexts, + persistBenchmarkCorpus, + resolveModelContext, +} from "./collect.js" +import { + buildSubSampleCorpus, + planCorpusSizeSubSamples, + runCorpusSizeSweep, + type CorpusSizeFit, + type CorpusSizeSweepRow, + type CorpusSizeSubSample, +} from "./corpus-size.js" +import { assignGroupedFolds, foldKey } from "./folds.js" +import { normalizedDiscountedCumulativeGain, resolveGoldTargets } from "./metrics.js" +import { OPTIMIZATION_PROFILES } from "./optimization-profiles.js" +import { reportBenchmarkProgress } from "./progress.js" +import { rankLexicalChannels } from "./ranking.js" +import { + compareRouterModelsAndMethods, + evaluateRouterComparisonHoldout, + routeWithComparisonModel, + type RouterComparisonHoldoutResult, + type RouterComparisonResult, +} from "./router-search/comparisons.js" +import { SEARCH_CANDIDATE_DEPTH, routerParameters } from "./router-search/config-space.js" +import { runBenchmarkSearch } from "./search.js" +import type { WeightSearchSample } from "./weight-search.js" + +const QUERY_FORMS: readonly QueryKind[] = [ + "identifier", + "searchPhrase", + "naturalQuestion", + "agentTask", +] + +/** Prepared model, device, and full-size corpus for one repository sweep. */ +export interface SubSampleContext { + readonly manifest: CorpusManifest + readonly model: string + readonly dims: number + readonly dtype: EmbeddingDtype + readonly embedder: BoundEmbedder + readonly maxTokens: number + readonly corpus: PreparedCorpus +} + +/** Load one model once and prepare the full pinned corpus for sub-sampling. */ +const prepareSubSampleContext = ( + repositoryId: string, + model: string, +): Effect.Effect => + Effect.gen(function* () { + const manifests = yield* loadCorpusManifests() + const manifest = manifests.find((entry) => entry.id === repositoryId) + if (manifest === undefined) + return yield* Effect.fail(new Error(`Unknown benchmark repository ${repositoryId}`)) + const resolved = yield* resolveModelContext(model) + const repositoryPath = yield* prepareRepository(manifest) + const corpus = yield* prepareCorpus(repositoryPath, manifest, { + maxTokens: resolved.maxTokens, + overlapLines: DEFAULT_CONFIG.overlapLines, + countTokens: resolved.embedder.countTokens, + onDiagnostic: () => Effect.void, + }) + return { + manifest, + model, + dims: resolved.info.dims, + dtype: resolved.info.defaultDtype, + embedder: resolved.embedder, + maxTokens: resolved.maxTokens, + corpus, + } + }) + +/** Build real weight-search samples for one sub-sample: embeddings, all five channels, gold. */ +const buildSubSampleSamples = ( + context: SubSampleContext, + plan: CorpusSizeSubSample, +): Effect.Effect => + Effect.gen(function* () { + const subSample = buildSubSampleCorpus(context.corpus, plan) + const keptQuestions = context.manifest.questions.filter((question) => + plan.keptQuestionIds.includes(question.id), + ) + const queries = keptQuestions.flatMap((question, questionIndex) => + QUERY_FORMS.map((queryKind) => ({ + question, + questionIndex, + queryKind, + query: question.queries[queryKind], + })), + ) + const targetsByQuestion = keptQuestions.map((question) => + resolveGoldTargets(question.groundTruth, subSample.chunks, subSample.identifiersByChunk), + ) + const unresolved = keptQuestions.filter((_, index) => + targetsByQuestion[index]!.some((targets) => targets.size === 0), + ) + if (unresolved.length > 0) + return yield* Effect.fail( + new Error( + `Sub-sample ${plan.targetSize} lost gold for: ${unresolved.map((question) => question.id).join(", ")}`, + ), + ) + reportBenchmarkProgress( + `embedding ${subSample.chunks.length} chunks for size ${plan.targetSize}`, + ) + const chunkVectors = yield* embedTexts( + subSample.chunks.map((chunk) => chunk.text), + context.model, + context.embedder, + ) + const queryVectors = yield* embedTexts( + queries.map((entry) => entry.query), + context.model, + context.embedder, + ) + return yield* withSqliteBenchmarkStore( + context.model, + context.dtype, + Effect.gen(function* () { + const store = yield* IndexStore + const sparseEmbedder = yield* SparseEmbedder + const sparseVectors = yield* embedSparseTexts( + subSample.chunks.map((chunk) => chunk.text), + sparseEmbedder, + ) + yield* persistBenchmarkCorpus( + store, + { + chunks: subSample.chunks, + identifierIndex: subSample.identifierIndex, + bm25Index: subSample.bm25Index, + }, + chunkVectors, + sparseVectors, + context.dims, + context.dtype, + context.maxTokens, + sparseEmbedder.contract, + yield* sparseEmbedder.loadIdf, + ) + const sparseQueries = yield* Effect.forEach(queries, (entry) => + sparseEmbedder.tokenizeQuery(entry.query), + ) + const searchData = yield* store.loadSearchData + const foldAssignments = assignGroupedFolds([context.manifest], 3) + const samples: WeightSearchSample[] = [] + for (let queryIndex = 0; queryIndex < queries.length; queryIndex++) { + const entry = queries[queryIndex]! + const lexical = rankLexicalChannels(entry.query, searchData) + const dense = yield* store.searchDense({ + vector: queryVectors[queryIndex]!, + dims: context.dims, + dtype: context.dtype, + }) + const sparse = yield* store.searchSparse(sparseQueries[queryIndex]!) + samples.push({ + repository: context.manifest.id, + intentId: entry.question.id, + queryKind: entry.queryKind, + groupedFold: foldAssignments.get(foldKey(context.manifest.id, entry.question.id)) ?? 0, + query: entry.query, + rankings: { ...lexical, dense, sparse }, + targets: targetsByQuestion[entry.questionIndex]!, + chunks: subSample.chunks, + termCoverage: buildQueryTermCoverage( + entry.query, + subSample.bm25Index, + subSample.identifierIndex, + ), + }) + } + return samples + }), + ) + }) + +/** Score one router configuration on real samples with per-query standard error. */ +const evaluateRouterConfigOnSamples = ( + samples: readonly WeightSearchSample[], + config: EvidenceRouterParameters, + route: ( + evidence: ReturnType, + config: EvidenceRouterParameters, + ) => ReturnType = routeWithEvidence, +): { readonly mean: number; readonly standardError: number } => { + const values = samples.map((sample) => { + const evidence = buildRoutingEvidence(sample.query, sample.rankings) + const weights = route(evidence, config) + const ranked = fuseRankings("dbsf", sample.rankings, weights, SEARCH_CANDIDATE_DEPTH) + return normalizedDiscountedCumulativeGain(ranked, sample.targets, 20) + }) + const mean = values.reduce((sum, value) => sum + value, 0) / Math.max(1, values.length) + const variance = + values.length < 2 + ? 0 + : values.reduce((sum, value) => sum + (value - mean) ** 2, 0) / (values.length - 1) + return { mean, standardError: Math.sqrt(variance / Math.max(1, values.length)) } +} + +/** Parse a comma-separated size list; "full" means the complete corpus. */ +const resolveSweepSizes = (requested: string | undefined): readonly number[] => { + if (requested === undefined) return [200, 500, 1000, Number.POSITIVE_INFINITY] + return requested.split(",").map((entry) => { + const trimmed = entry.trim() + return trimmed === "full" ? Number.POSITIVE_INFINITY : Number(trimmed) + }) +} + +/** Real per-size sweep output: sweep rows, fitted model, and sample counts. */ +export interface RealCorpusSizeSweepResult { + readonly rows: readonly CorpusSizeSweepRow[] + readonly fit: CorpusSizeFit + readonly perSizeSamples: readonly { readonly corpusSize: number; readonly samples: number }[] +} + +const buildSearchInputs = ( + model: string, + samples: readonly WeightSearchSample[], +): { + sampleGroups: Map< + string, + { model: string; queryKind: QueryKind; samples: readonly WeightSearchSample[] } + > + samplesByModel: Map +} => { + const sampleGroups = new Map< + string, + { model: string; queryKind: QueryKind; samples: readonly WeightSearchSample[] } + >() + const samplesByModel = new Map() + samplesByModel.set(model, samples) + for (const sample of samples) { + const key = `${model}\0${sample.queryKind}` + sampleGroups.set(key, { + model, + queryKind: sample.queryKind, + samples: [...(sampleGroups.get(key)?.samples ?? []), sample], + }) + } + return { sampleGroups, samplesByModel } +} + +const goldTargetsByQuestion = ( + manifest: CorpusManifest, + corpus: PreparedCorpus, +): { questionId: string; goldChunkIndices: number[] }[] => + manifest.questions.map((question) => ({ + questionId: question.id, + goldChunkIndices: [ + ...new Set( + resolveGoldTargets(question.groundTruth, corpus.chunks, corpus.identifiersByChunk).flatMap( + (targets) => [...targets], + ), + ), + ], + })) + +/** Run the real per-size search over sub-sampled corpora and fit the corpus-size model. */ +export const runRealCorpusSizeSweep = ( + repositoryId: string, + model: string, + sizes: readonly number[] = resolveSweepSizes(process.env.PIX_BENCH_CORPUS_SIZES), +): Effect.Effect => + Effect.gen(function* () { + const context = yield* prepareSubSampleContext(repositoryId, model) + const goldByQuestion = goldTargetsByQuestion(context.manifest, context.corpus) + const plans = planCorpusSizeSubSamples(goldByQuestion, context.corpus.chunks.length, sizes).map( + (plan) => ({ + ...plan, + targetSize: Number.isFinite(plan.targetSize) + ? plan.targetSize + : context.corpus.chunks.length, + }), + ) + const searchOptions = { + workerCount: Math.min(resolveWorkerCount(), getDefaultWorkerCount()), + fallbackToSerial: false as const, + } + const perSizeSamples: { corpusSize: number; samples: number }[] = [] + const sweep = yield* Effect.promise(() => + runCorpusSizeSweep( + plans, + async (plan) => { + const samples = await Effect.runPromise(buildSubSampleSamples(context, plan)) + perSizeSamples.push({ corpusSize: plan.targetSize, samples: samples.length }) + const { sampleGroups, samplesByModel } = buildSearchInputs(context.model, samples) + const search = await Effect.runPromise( + runBenchmarkSearch( + { + groupedFolds: 3, + repositoryHoldouts: false, + legacyDiagnostics: false, + fusionMethods: ["dbsf"], + routerFusionMethods: ["dbsf"], + }, + sampleGroups, + samplesByModel, + "grouped-3-fold", + OPTIMIZATION_PROFILES["search-priority"], + false, + searchOptions, + ), + ) + const router = search.recommendedEvidenceRouters.find( + (row) => row.fusion === "dbsf" && row.objective === "direct", + ) + if (router === undefined) + throw new Error(`No dbsf/direct router recommendation at size ${plan.targetSize}`) + const scored = evaluateRouterConfigOnSamples(samples, router.config) + reportBenchmarkProgress( + `size ${plan.targetSize}: ndcg@20 ${scored.mean.toFixed(4)} ± ${scored.standardError.toFixed(4)}`, + ) + return [ + { + coordinate: { + corpusSize: plan.targetSize, + fusion: "dbsf", + profile: "search-priority", + objective: "direct", + strategy: "grouped-3-fold", + fold: "fit-all", + }, + weights: router.staticWeights, + score: scored.mean, + noise: scored.standardError, + }, + ] + }, + goldByQuestion, + ), + ) + return { ...sweep, perSizeSamples } + }) + +/** Real multiplicative vs log-linear comparison output with excluded-holdout scores. */ +export interface RealRouterModelComparison { + readonly results: readonly RouterComparisonResult[] + readonly holdouts: readonly RouterComparisonHoldoutResult[] + readonly developmentSamples: number + readonly validationSamples: number +} + +/** Run the real multiplicative-vs-log-linear comparison on one repository's full corpus. */ +export const runRealRouterModelComparison = ( + repositoryId: string, + model: string, +): Effect.Effect => + Effect.gen(function* () { + const context = yield* prepareSubSampleContext(repositoryId, model) + const goldByQuestion = goldTargetsByQuestion(context.manifest, context.corpus) + const fullPlan = planCorpusSizeSubSamples(goldByQuestion, context.corpus.chunks.length, [ + Number.POSITIVE_INFINITY, + ])[0]! + const samples = yield* buildSubSampleSamples(context, fullPlan) + const development = samples.filter((sample) => sample.groupedFold !== 0) + const validation = samples.filter((sample) => sample.groupedFold === 0) + const results = yield* Effect.promise(() => + compareRouterModelsAndMethods({ + seed: OPTIMIZATION_PROFILES["search-priority"].fusionConfig, + parameters: routerParameters(), + evidence: development.map((sample) => buildRoutingEvidence(sample.query, sample.rankings)), + pruneInactive: true, + evaluateDevelopment: async (routerModel, configs) => + configs.map((config) => + evaluateRouterConfigOnSamples(development, config, (evidence, candidate) => + routeWithComparisonModel(routerModel, evidence, candidate), + ), + ), + }), + ) + const holdouts = yield* Effect.promise(() => + Promise.all( + results.map((result) => + evaluateRouterComparisonHoldout(result, async (configs) => + configs.map((config) => + evaluateRouterConfigOnSamples(validation, config, (evidence, candidate) => + routeWithComparisonModel(result.model, evidence, candidate), + ), + ), + ), + ), + ), + ) + return { + results, + holdouts, + developmentSamples: development.length, + validationSamples: validation.length, + } + }) diff --git a/benchmarks/tests/corpus-size-model.test.ts b/benchmarks/tests/corpus-size-model.test.ts new file mode 100644 index 0000000..c880e09 --- /dev/null +++ b/benchmarks/tests/corpus-size-model.test.ts @@ -0,0 +1,45 @@ +import { mkdir, writeFile } from "node:fs/promises" +import path from "node:path" + +import { expect, it } from "@effect/vitest" +import { Effect } from "effect" + +import { runRealCorpusSizeSweep } from "../retrieval/evaluation/subsample-evidence.js" + +const repositoryId = process.env.PIX_BENCH_CORPUS_SIZE_REPO ?? "t3code" +const model = process.env.PIX_BENCH_MODELS ?? "Xenova/all-MiniLM-L6-v2" + +it.effect("runs the real model-backed corpus-size sweep", () => + Effect.gen(function* () { + const result = yield* runRealCorpusSizeSweep(repositoryId, model) + + expect(result.rows).toHaveLength( + result.fit.logLinear.length > 0 ? result.perSizeSamples.length : 0, + ) + expect(result.perSizeSamples.length).toBeGreaterThan(1) + for (const row of result.rows) { + expect(Number.isFinite(row.coordinate.corpusSize)).toBe(true) + expect(Object.values(row.weights).every((weight) => Number.isFinite(weight))).toBe(true) + expect(row.noise).toBeGreaterThan(0) + } + + const outputDirectory = path.resolve("benchmarks/results") + const outputPath = path.join( + outputDirectory, + `retrieval-corpus-size-model-${repositoryId}-${model.replaceAll("/", "_")}.json`, + ) + yield* Effect.tryPromise({ + try: async () => { + await mkdir(outputDirectory, { recursive: true }) + await writeFile( + outputPath, + `${JSON.stringify({ schemaVersion: 1, repositoryId, model, ...result }, null, 2)}\n`, + "utf8", + ) + }, + catch: (cause) => + new Error(`Could not write corpus-size model sweep ${outputPath}`, { cause }), + }) + expect(result.fit.recommendation).toBeDefined() + }), +) diff --git a/benchmarks/tests/router-model-real.test.ts b/benchmarks/tests/router-model-real.test.ts new file mode 100644 index 0000000..3d3c7fe --- /dev/null +++ b/benchmarks/tests/router-model-real.test.ts @@ -0,0 +1,44 @@ +import { mkdir, writeFile } from "node:fs/promises" +import path from "node:path" + +import { expect, it } from "@effect/vitest" +import { Effect } from "effect" + +import { runRealRouterModelComparison } from "../retrieval/evaluation/subsample-evidence.js" + +const repositoryId = process.env.PIX_BENCH_CORPUS_SIZE_REPO ?? "t3code" +const model = process.env.PIX_BENCH_MODELS ?? "Xenova/all-MiniLM-L6-v2" + +it.effect("runs the real multiplicative vs log-linear router comparison", () => + Effect.gen(function* () { + const comparison = yield* runRealRouterModelComparison(repositoryId, model) + + expect(comparison.results).toHaveLength(8) + expect(comparison.holdouts).toHaveLength(8) + expect(comparison.developmentSamples).toBeGreaterThan(0) + expect(comparison.validationSamples).toBeGreaterThan(0) + + const outputDirectory = path.resolve("benchmarks/results") + const outputPath = path.join( + outputDirectory, + `retrieval-router-model-comparison-${repositoryId}-${model.replaceAll("/", "_")}.json`, + ) + yield* Effect.tryPromise({ + try: async () => { + await mkdir(outputDirectory, { recursive: true }) + await writeFile( + outputPath, + `${JSON.stringify({ schemaVersion: 1, repositoryId, model, ...comparison }, null, 2)}\n`, + "utf8", + ) + }, + catch: (cause) => + new Error(`Could not write router model comparison ${outputPath}`, { cause }), + }) + + const multiplicative = comparison.holdouts.find((row) => row.model === "multiplicative") + const logLinear = comparison.holdouts.find((row) => row.model === "regularized-log-linear") + expect(multiplicative).toBeDefined() + expect(logLinear).toBeDefined() + }), +) diff --git a/docs/adr/0021-evidence-router-stays-multiplicative.md b/docs/adr/0021-evidence-router-stays-multiplicative.md new file mode 100644 index 0000000..7c74d68 --- /dev/null +++ b/docs/adr/0021-evidence-router-stays-multiplicative.md @@ -0,0 +1,48 @@ +# 0021: Evidence router stays multiplicative + +## Status + +Accepted + +## Context + +The production evidence router (`routeWithEvidence` in `src/lib/retrieval/evidence-router.ts`) +computes per-channel fusion weights by multiplying `baseWeights` with seven per-query evidence +factors. Issue #173 asked whether a bounded regularized log-linear gate — additive in log space — +retrieves better, and which derivative-free fitting method finds the best configuration. + +The benchmark comparison (`benchmarks/retrieval/evaluation/router-search/comparisons.ts`) runs both +formulations over the same `EvidenceRouterParameters` space with four fitting methods: staged, +alternating block-coordinate, deterministic Halton restarts, and local search. Selection is +development-only; an excluded grouped fold scores the fixed winners. + +First real measurement (t3code, MiniLM, DBSF, 36 development / 24 validation queries, holdout +NDCG@20; artifact `retrieval-router-model-comparison-t3code-Xenova_all-MiniLM-L6-v2.json`): + +| Model | Best method | Holdout NDCG@20 | +| ---------------------- | ---------------------------- | --------------- | +| multiplicative | staged | 0.6911 | +| regularized-log-linear | alternating block-coordinate | 0.6918 | + +The 0.0007 gap is far below the per-query standard error (±0.01–0.04 across runs). Seven of forty +dimensions are inactive on this corpus; the searches prune them before fitting. + +## Decision + +- Keep the multiplicative router as the production and benchmark formulation. The log-linear gate + stays in `comparisons.ts` as a benchmark ablation; it is not promoted. +- Keep the halving-funnel search as the production search strategy. The four comparison methods are + recorded per run but do not replace it: none beats the funnel's holdout quality outside noise. +- Both selection rules (complexity-aware utility, one-standard-error simplicity) stay benchmark + diagnostics. They regularly trade a small amount of holdout quality for simpler candidates. + +## Rationale + +A 0.0007 holdout delta with ±0.03 measurement noise is not evidence. Replacing the multiplicative +model would add a production migration (`EvidenceRouterParameters` semantics, promoted-config +re-fit) for a gain that the data cannot distinguish from zero. The multiplicative model is also the +only formulation with a promoted, validated configuration in production. + +Re-run `vp run bench:retrieval:router-models` after major corpus or model changes. Promote the +log-linear gate only if it beats the multiplicative holdout by more than the combined standard +error on at least two corpora. diff --git a/package.json b/package.json index 0faf9b3..fe06ec1 100644 --- a/package.json +++ b/package.json @@ -37,6 +37,8 @@ "bench:retrieval": "vp run bench:retrieval:validate", "bench:retrieval:corpus": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/corpus.test.ts", "bench:retrieval:corpus-size": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/corpus-size-execution.test.ts", + "bench:retrieval:corpus-size:model": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/corpus-size-model.test.ts", + "bench:retrieval:router-models": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/router-model-real.test.ts", "bench:retrieval:develop": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/retrieval.test.ts -t \"runs the develop retrieval profile\"", "bench:retrieval:fixture": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/channels.test.ts", "bench:retrieval:full": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/retrieval.test.ts -t \"runs the full retrieval profile\"", From 88a6e2175618535410fc11aef7f110287e66098f Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Fri, 28 Aug 2026 19:33:11 +0200 Subject: [PATCH 08/16] docs(adr): record corpus-size factor decision --- docs/adr/0022-corpus-size-factor-deferred.md | 55 ++++++++++++++++++++ 1 file changed, 55 insertions(+) create mode 100644 docs/adr/0022-corpus-size-factor-deferred.md diff --git a/docs/adr/0022-corpus-size-factor-deferred.md b/docs/adr/0022-corpus-size-factor-deferred.md new file mode 100644 index 0000000..e6f251e --- /dev/null +++ b/docs/adr/0022-corpus-size-factor-deferred.md @@ -0,0 +1,55 @@ +# 0022: Corpus-size factor stays out of production pending replication + +## Status + +Accepted + +## Context + +The evidence router has no representation of index size. Issue #175 asked whether optimal channel +weights shift with corpus size enough to justify a query-independent corpus-size factor that scales +`baseWeights` before the per-query evidence factors. + +`vp run bench:retrieval:corpus-size:model` sub-samples one pinned corpus to fixed chunk counts, runs +the full search protocol at every size, and fits a per-channel log-linear model over +`log(chunkCount)`. First real measurements (MiniLM, DBSF, search-priority, grouped 3-fold, holdout +NDCG@20 of the fit-all direct router; artifacts +`retrieval-corpus-size-model-t3code-…` and `retrieval-corpus-size-model-effect-v4-…`): + +| Corpus | Sizes | Sparse weight trend | Sparse slope | Pairs outside noise | Sensitivity verdict | +| --------- | ----------------------- | ---------------------- | ------------ | ------------------- | ------------------- | +| t3code | 200 / 500 / 1000 / 1306 | 0.9 → 0.6 → 0.5 → 0.3 | -0.291 | 9 | promote | +| effect-v4 | 200 / 500 / 1000 / 6786 | 0.9 → 0.67 → 1.0 → 0.2 | -0.189 | 14 | promote | + +Quality falls with size on both corpora (t3code 0.88 → 0.70, effect-v4 0.89 → 0.36 NDCG@20), so +larger corpora are genuinely harder. The replicated signals across both corpora: + +- **Sparse weight declines with corpus size** (both slopes strongly negative). The learned-sparse + channel contributes relatively less as the index grows. +- Lexical/identity channels gain relative weight at scale on effect-v4 (bm25 slope +0.16, + identity +0.12), but the signs flip or stay flat on t3code (bm25 -0.06, identity 0.00). +- Per-size optima wobble at discrete level resolution (effect-v4 size 1000 orders channels + differently from size 6786), and per-size noise is ±0.02–0.04 NDCG@20. + +## Decision + +- Do **not** add a corpus-size factor to `EvidenceRouterParameters` yet. The sensitivity check says + weights shift outside noise, but only the sparse decline replicates across both corpora; the + other channels flip signs between corpora, one model and two corpora is thin evidence, and the + quality cost of keeping static weights is unmeasured. +- Keep the sweep as the standing measurement: `vp run bench:retrieval:corpus-size:model` with + `PIX_BENCH_CORPUS_SIZE_REPO` and `PIX_BENCH_CORPUS_SIZES` knobs. Re-run it when a new corpus or + model lands. +- Promotion bar: the same channel shows the same sign of log-linear slope on at least two corpora + and two models, and a follow-up measurement shows that applying the size-200 optimum at full size + costs more than one noise band of NDCG@20 versus the per-size optimum. Only then wire the factor + as a query-independent `baseWeights` prior with tiny/empty/large-corpus edge-case tests. +- Until then, corpus size stays a reported benchmark axis, not a production input. + +## Rationale + +The issue's own bar is that a factor must earn its runtime input (`bm25Index.chunkLengths.length`) +and parameter space through evidence. Two corpora and one model replicate one channel signal, not a +model. Wiring a factor on that basis adds a production configuration dimension before the benchmark +can say how much quality the factor recovers. The sweep machinery stays in the repo, so the +promotion decision re-runs in minutes rather than restarting the research. From 1c3abf2f314c484adcb567b66ee5b6ef88392675 Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Fri, 28 Aug 2026 19:59:03 +0200 Subject: [PATCH 09/16] docs(bench): knob decision table and log-linear re-open condition --- benchmarks/README.md | 36 +++++++++++++++---- ...21-evidence-router-stays-multiplicative.md | 13 +++++-- 2 files changed, 41 insertions(+), 8 deletions(-) diff --git a/benchmarks/README.md b/benchmarks/README.md index 008439d..69d7f35 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -95,6 +95,13 @@ vp run bench:retrieval:validate vp run bench:retrieval:full ``` +| Profile | Repositories | Models | Validation | Static fusion | Router fusion | Diagnostics | +| ---------- | ------------ | -------- | ----------------------------- | ------------- | ------------- | ---------------------- | +| `smoke` | fd | MiniLM | grouped 5-fold | DBSF | DBSF | current router | +| `develop` | all six | MiniLM | grouped 3-fold | DBSF | DBSF | current router | +| `validate` | all six | MiniLM | grouped 5-fold and repository | DBSF | DBSF | current router | +| `full` | all six | selected | grouped 5-fold and repository | all three | all three | all active diagnostics | + Run every profile, model, and optimization-profile invocation from the matrix manifest, then merge the artifacts: @@ -116,12 +123,29 @@ fold)` coordinates, and fits a per-channel log-linear model over `log(chunkCount check compares weight shifts between adjacent sizes against the combined one-standard-error noise band and recommends promoting a corpus-size factor only when a shift leaves that band. -| Profile | Repositories | Models | Validation | Static fusion | Router fusion | Diagnostics | -| ---------- | ------------ | -------- | ----------------------------- | ------------- | ------------- | ---------------------- | -| `smoke` | fd | MiniLM | grouped 5-fold | DBSF | DBSF | current router | -| `develop` | all six | MiniLM | grouped 3-fold | DBSF | DBSF | current router | -| `validate` | all six | MiniLM | grouped 5-fold and repository | DBSF | DBSF | current router | -| `full` | all six | selected | grouped 5-fold and repository | all three | all three | all active diagnostics | +The model-backed sweep (`vp run bench:retrieval:corpus-size:model`) runs the real per-size search on +one corpus; select it with `PIX_BENCH_CORPUS_SIZE_REPO` and `PIX_BENCH_CORPUS_SIZES`. The first real +measurements and the resulting decision live in ADR 0022: the learned-sparse channel deserves less +weight as corpora grow, all other channels flip signs between corpora, and the production factor is +deferred until the promotion bar is met. The real multiplicative-vs-log-linear comparison runs with +`vp run bench:retrieval:router-models` and feeds ADR 0021. + +### Knob decisions + +Every search knob has a recorded status so work stops re-litigating settled ones: + +| Knob | Status | Evidence | +| ------------------------------------------- | --------------------------------------------------------------------------------- | ---------------------------- | +| Router model (multiplicative vs log-linear) | Settled: keep multiplicative; switch at the next mandatory promoted-config re-fit | ADR 0021 | +| Router search strategy | Settled: `halving-funnel` default; `proxy-promotion` is the slow control | BASELINE (#189) | +| Finalist budget | Settled: 256 (384 measured identical, +65 s) | BASELINE (#189) | +| Local Sobol cloud (points/radius) | Settled: 16 points, radius 2 (32/r3 added no quality) | BASELINE (#189) | +| Scout sequence (halton/sobol/random) | Settled: all inside noise; halton default | BASELINE (schema 29) | +| Wide-vs-deep passes | Settled: 1 wide pass matches 2 narrow passes at 45% fewer evals | BASELINE (schema 29) | +| Corpus-size factor | Deferred with promotion bar; sweep stays as standing measurement | ADR 0022 | +| Fitting methods, selection rules | Diagnostics only, recorded per comparison run | ADR 0021 | +| One-pass variant on validate profile | Open | BASELINE (validate addendum) | +| Chunking (chunkTokens/overlapLines) | Open: never swept by this suite | — | `bench:retrieval` aliases `bench:retrieval:validate`. Every profile measures the same physical rankings and retrieval variants; profiles only control matrix size, holdout coverage, and expensive diff --git a/docs/adr/0021-evidence-router-stays-multiplicative.md b/docs/adr/0021-evidence-router-stays-multiplicative.md index 7c74d68..36f6b1e 100644 --- a/docs/adr/0021-evidence-router-stays-multiplicative.md +++ b/docs/adr/0021-evidence-router-stays-multiplicative.md @@ -43,6 +43,15 @@ model would add a production migration (`EvidenceRouterParameters` semantics, pr re-fit) for a gain that the data cannot distinguish from zero. The multiplicative model is also the only formulation with a promoted, validated configuration in production. +The formulations are equivalent within measurement error, so the choice is a convention with a +migration cost, not a quality verdict. Log-linear has a conditioning argument on its side: additive +log-space contributions keep every parameter independently effective for coordinate-wise +derivative-free search, and the exp clamp bounds the weight floor near 13.5% of the base weight +instead of multiplying down to zero. Therefore the cheap re-open point is the next mandatory re-fit +of the promoted configuration (after a corpus or model change): the migration cost is already sunk +there, so switch to the log-linear gate at that point unless it underperforms the multiplicative +holdout on the same run. + Re-run `vp run bench:retrieval:router-models` after major corpus or model changes. Promote the -log-linear gate only if it beats the multiplicative holdout by more than the combined standard -error on at least two corpora. +log-linear gate before that mandatory re-fit only if it beats the multiplicative holdout by more +than the combined standard error on at least two corpora. From 6c42a32dae9907a28ffc5968da86e30674e4c84d Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Fri, 28 Aug 2026 20:08:04 +0200 Subject: [PATCH 10/16] docs(adr): production router defaults and benchmark cleanup plan --- docs/adr/0023-production-router-defaults.md | 58 +++++++++++++++++++++ 1 file changed, 58 insertions(+) create mode 100644 docs/adr/0023-production-router-defaults.md diff --git a/docs/adr/0023-production-router-defaults.md b/docs/adr/0023-production-router-defaults.md new file mode 100644 index 0000000..efb6650 --- /dev/null +++ b/docs/adr/0023-production-router-defaults.md @@ -0,0 +1,58 @@ +# 0023: Production router defaults and benchmark cleanup + +## Status + +Accepted (execution tracked below) + +## Context + +The #166 evidence chain settled the search method (halving funnel), the fusion method (DBSF), and +the router formulation (multiplicative, ADR 0021). Production always executes the evidence router +(`routeWithEvidence` in `src/application/query-project.ts:108`); there is no static fallback path. +Remaining questions: which parameters production should carry, whether they need per-model variants, +and which benchmark alternatives can now be deleted. + +## Decisions + +### Production + +- **The evidence router stays the only production retrieval path.** No static-weight fallback + exists or comes back. Confirmed, no code change. +- **DBSF stays the only production fusion.** RRF and relative-score remain benchmark comparison + axes; the production `fusion` field keeps decoding them for artifact compatibility but no + promoted config or profile uses them. +- **Promotion target: per-profile and per-model router configs.** Today all four production + profiles alias `PROMOTED_SEARCH_PRIORITY_CONFIG` (TODO(#163) placeholder), and the config is + model-independent. The benchmark already fits per model and per optimization profile, so the + gap is production-side only: `PRODUCTION_PROFILES` entries get their own benchmark-fitted + configs, and the promoted config becomes keyed by embedder model with the current config as + fallback. No schema versioning on the production config (per #166). +- **Promotion gate:** a matrix release run (`vp run bench:retrieval:matrix`) must show the + per-profile/per-model candidate meeting guardrails on its coordinate before the constant swaps. + This is the same evidence standard the current config passed. + +### Benchmark cleanup + +- **Delete `successive-halving`.** It is the superseded historical control; `proxy-promotion` + remains the documented slow control for funnel re-validation. Touches strategy types, the rank + mode, the `PIX_BENCH_ROUTER_STRATEGY` knob, and tests that use it as a fixture. +- **Delete the legacy per-query-kind RRF grids and Shapley diagnostics** (`legacyDiagnostics`). + They are already off by default and serve no control role. +- **Keep the static fusion search.** It is the control that proves dynamic beats static — the + basis of every promotion claim. It stays in the full profile only. +- **Keep RRF and relative-score in the benchmark** as comparison axes. Deleting them would make + the DBSF decision unmeasurable in future re-runs. + +## Rationale + +Controls that justify a decision stay; controls that merely replay history go. The funnel was +adopted over proxy-promotion with measured evidence, so proxy-promotion keeps earning its place as +the slow control, while successive-halving no longer answers any live question. Per-model configs +follow the same evidence logic as ADR 0022's index-size finding: model identity and profile intent +are index-level priors, and the benchmark produces the per-cell evidence already. + +## Execution order + +1. Benchmark cleanup: remove `successive-halving` and legacy diagnostics (schema bump). +2. Matrix release run for per-profile/per-model promotion evidence. +3. Swap production constants: distinct profile configs, model-keyed promoted config with fallback. From 44aca9b3707f851475719845510c81e0faad1d6f Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Fri, 28 Aug 2026 20:36:13 +0200 Subject: [PATCH 11/16] refactor(bench): remove successive-halving strategy and legacy weight diagnostics --- benchmarks/README.md | 2 +- benchmarks/retrieval/evaluation/collect.ts | 57 +----- benchmarks/retrieval/evaluation/report.ts | 63 +------ .../evaluation/router-search/config-space.ts | 6 +- .../evaluation/router-search/rank.ts | 10 +- benchmarks/retrieval/evaluation/search.ts | 96 +--------- .../evaluation/subsample-evidence.ts | 35 +--- benchmarks/retrieval/evaluation/types.ts | 60 +------ .../retrieval/evaluation/weight-search.ts | 169 +----------------- benchmarks/retrieval/runner.ts | 9 - benchmarks/tests/channels.test.ts | 34 ---- benchmarks/tests/matrix.test.ts | 1 - benchmarks/tests/report.test.ts | 3 - benchmarks/tests/retrieval.test.ts | 15 +- benchmarks/tests/run-config.test.ts | 4 +- benchmarks/tests/worker-pool.test.ts | 26 +-- 16 files changed, 36 insertions(+), 554 deletions(-) diff --git a/benchmarks/README.md b/benchmarks/README.md index 69d7f35..88ddd7e 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -225,7 +225,7 @@ serialization time for each source artifact. The router search defaults to `halving-funnel`. It proxy-scores 512 broad scouts, keeps 32 survivors, proxy-scores 16 radius-2 Sobol points per survivor, and fully evaluates 256 finalists plus the static base seeds once. Set `PIX_BENCH_ROUTER_STRATEGY=proxy-promotion` to run the slower coordinate-beam -search. Set it to `successive-halving` to run the historical lexicographic variant. All strategies use +control. All strategies use the same candidate evaluator and native worker queue. Their final archive selection is objective-specific, so Direct and Reranker rows are real comparisons rather than repeated labels. diff --git a/benchmarks/retrieval/evaluation/collect.ts b/benchmarks/retrieval/evaluation/collect.ts index dc510bc..ea4277f 100644 --- a/benchmarks/retrieval/evaluation/collect.ts +++ b/benchmarks/retrieval/evaluation/collect.ts @@ -1,4 +1,4 @@ -import { Effect, Stream } from "effect" +import { Effect, Stream } from "effect" import { SqlClient } from "effect/unstable/sql" import type { Embedding } from "../../../src/domain/chunk.js" @@ -68,7 +68,6 @@ interface ModelMeasurements { readonly sparseEmbeddingRun: BenchmarkArtifact["sparseEmbeddingRuns"][number] readonly measurements: readonly QueryMeasurement[] readonly samples: readonly WeightSearchSample[] - readonly samplesByQueryKind: ReadonlyMap readonly rankings: readonly ChannelRankings[] readonly retrievalDurationMs: number } @@ -80,14 +79,6 @@ interface RepositoryMeasurements { readonly sparseEmbeddingRuns: readonly BenchmarkArtifact["sparseEmbeddingRuns"][number][] readonly measurements: readonly QueryMeasurement[] readonly samplesByModel: ReadonlyMap - readonly sampleGroups: ReadonlyMap< - string, - { - readonly model: string - readonly queryKind: QueryKind - readonly samples: readonly WeightSearchSample[] - } - > readonly retrievalDurationMs: number } @@ -98,14 +89,6 @@ export interface CollectedBenchmarkData { readonly embeddingRuns: readonly BenchmarkArtifact["embeddingRuns"][number][] readonly sparseEmbeddingRuns: readonly BenchmarkArtifact["sparseEmbeddingRuns"][number][] readonly measurements: readonly QueryMeasurement[] - readonly sampleGroups: ReadonlyMap< - string, - { - readonly model: string - readonly queryKind: QueryKind - readonly samples: readonly WeightSearchSample[] - } - > readonly samplesByModel: ReadonlyMap readonly retrievalDurationMs: number } @@ -192,7 +175,6 @@ export const persistBenchmarkCorpus = ( interface BuiltModelSamples { readonly measurements: readonly QueryMeasurement[] readonly samples: readonly WeightSearchSample[] - readonly samplesByQueryKind: ReadonlyMap } const buildModelSamples = ( @@ -208,7 +190,6 @@ const buildModelSamples = ( Effect.gen(function* () { const modelMeasurements: QueryMeasurement[] = [] const modelSamples: WeightSearchSample[] = [] - const samplesByQueryKind = new Map() for (let queryIndex = 0; queryIndex < queries.length; queryIndex++) { const entry = queries[queryIndex] @@ -234,10 +215,6 @@ const buildModelSamples = ( termCoverage: buildQueryTermCoverage(entry.query, corpus.bm25Index, corpus.identifierIndex), } modelSamples.push(sample) - samplesByQueryKind.set(entry.queryKind, [ - ...(samplesByQueryKind.get(entry.queryKind) ?? []), - sample, - ]) for (const variant of RETRIEVAL_VARIANTS) { const variantStartedAt = performance.now() const ranked = fuseVariant(variant, entry.query, rankings) @@ -279,7 +256,7 @@ const buildModelSamples = ( } } - return { measurements: modelMeasurements, samples: modelSamples, samplesByQueryKind } + return { measurements: modelMeasurements, samples: modelSamples } }) const collectModelSamples = ( @@ -437,7 +414,6 @@ const collectModelMeasurements = ( sparseEmbeddingRun: modelRun.sparseEmbeddingRun, measurements: modelRun.measurements, samples: modelRun.samples, - samplesByQueryKind: modelRun.samplesByQueryKind, rankings: modelRun.rankings, retrievalDurationMs: performance.now() - retrievalStartedAt, } @@ -567,14 +543,6 @@ const collectRepositoryMeasurements = ( const sparseEmbeddingRuns: BenchmarkArtifact["sparseEmbeddingRuns"][number][] = [] const measurements: QueryMeasurement[] = [] const samplesByModel = new Map() - const sampleGroups = new Map< - string, - { - readonly model: string - readonly queryKind: QueryKind - readonly samples: readonly WeightSearchSample[] - } - >() let retrievalDurationMs = 0 const cachePaths = benchmarkCachePaths(manifest, model, info.dims, info.defaultDtype, maxTokens) @@ -636,9 +604,6 @@ const collectRepositoryMeasurements = ( measurements.push(...modelData.measurements) retrievalDurationMs += modelData.retrievalDurationMs samplesByModel.set(model, modelData.samples) - for (const [queryKind, samples] of modelData.samplesByQueryKind) { - sampleGroups.set(`${model}\0${queryKind}`, { model, queryKind, samples }) - } return { repository: { id: manifest.id, @@ -652,7 +617,6 @@ const collectRepositoryMeasurements = ( sparseEmbeddingRuns, measurements, samplesByModel, - sampleGroups, retrievalDurationMs, } }) @@ -668,14 +632,6 @@ export const collectBenchmarkData = ( const embeddingRuns: BenchmarkArtifact["embeddingRuns"][number][] = [] const sparseEmbeddingRuns: BenchmarkArtifact["sparseEmbeddingRuns"][number][] = [] const measurements: QueryMeasurement[] = [] - const sampleGroups = new Map< - string, - { - readonly model: string - readonly queryKind: QueryKind - readonly samples: readonly WeightSearchSample[] - } - >() const samplesByModel = new Map() let retrievalDurationMs = 0 let chunkTokens: number | undefined @@ -709,14 +665,6 @@ export const collectBenchmarkData = ( for (const [model, samples] of repositoryData.samplesByModel) { samplesByModel.set(model, [...(samplesByModel.get(model) ?? []), ...samples]) } - for (const [key, group] of repositoryData.sampleGroups) { - const current = sampleGroups.get(key) - sampleGroups.set(key, { - model: group.model, - queryKind: group.queryKind, - samples: [...(current?.samples ?? []), ...group.samples], - }) - } } if (chunkTokens === undefined) { @@ -729,7 +677,6 @@ export const collectBenchmarkData = ( embeddingRuns, sparseEmbeddingRuns, measurements, - sampleGroups, samplesByModel, retrievalDurationMs, } diff --git a/benchmarks/retrieval/evaluation/report.ts b/benchmarks/retrieval/evaluation/report.ts index 32a4000..36c494b 100644 --- a/benchmarks/retrieval/evaluation/report.ts +++ b/benchmarks/retrieval/evaluation/report.ts @@ -223,7 +223,7 @@ const renderOverview = (artifact: BenchmarkArtifact): readonly string[] => { `Local refinement: ${artifact.localCloudPoints} deterministic Sobol cloud points per elite within +/- ${artifact.localCloudRadiusLevels} level(s) around the final beam.`, ] : []), - `Cheap pre-scoring: candidates first score on a deterministic ${artifact.searchStrategy.proxySampleFraction * 100}% proxy sample with a minimum of ${artifact.searchStrategy.proxyMinimumSamples}; the ${artifact.searchStrategy.kind === "successive-halving" ? "keep" : "promotion"} factor is ${artifact.searchStrategy.kind === "successive-halving" ? artifact.searchStrategy.halvingKeepFactor : artifact.searchStrategy.proxyPromotionFactor}x.`, + `Cheap pre-scoring: candidates first score on a deterministic ${artifact.searchStrategy.proxySampleFraction * 100}% proxy sample with a minimum of ${artifact.searchStrategy.proxyMinimumSamples}; the promotion factor is ${artifact.searchStrategy.proxyPromotionFactor}x.`, ] return [ "# Retrieval Quality Benchmark", @@ -265,20 +265,18 @@ const renderComputeSections = (artifact: BenchmarkArtifact): readonly string[] = "Corpus preparation", "Embedding", "Retrieval", - "Weight search", "Fusion search", "Router search", "Candidate queue startup", "Candidate queue shutdown", ], - Array.from({ length: 9 }, () => "---:"), + Array.from({ length: 8 }, () => "---:"), [ [ duration(artifact.timings.totalDurationMs), duration(artifact.timings.corpusPreparationDurationMs), duration(artifact.timings.embeddingDurationMs), duration(artifact.timings.retrievalDurationMs), - duration(artifact.timings.weightSearchDurationMs), duration(artifact.timings.fusionSearchDurationMs), duration(artifact.timings.evidenceRouterSearchDurationMs), duration(artifact.timings.candidateQueueStartupDurationMs), @@ -464,63 +462,6 @@ const renderRouterSections = (artifact: BenchmarkArtifact): readonly string[] => ), ) - lines.push( - "", - "## Cross-Validated Weights", - "", - "Each row selects weights without its validation fold. Validation quality and Shapley contributions use only the excluded fold.", - "", - ...renderTable( - [ - "Model", - "Query form", - "Strategy", - "Fold", - "Weights I/C/B/D/S", - "Dev R@20", - "Validation R@5", - "Validation R@10", - "Validation R@20", - "Validation Ctx@4k", - "Shapley I/C/B/D/S", - ], - ["---", "---", "---", "---", "---", "---:", "---:", "---:", "---:", "---:", "---"], - artifact.weightSearch.map((result) => [ - result.model, - result.queryKind, - result.strategy, - result.fold, - formatWeights(result.weights), - percent(result.development.recallAt20), - percent(result.validation.recallAt5), - percent(result.validation.recallAt10), - percent(result.validation.recallAt20), - percent(result.validation.contextRecallAt4096), - CHANNELS.map((channel) => percent(result.shapleyRecallAt20[channel])).join("/"), - ]), - ), - ) - - lines.push( - "", - "## Recommended Weights", - "", - "These deployment candidates are fitted on all available samples only after cross-validation.", - "", - ...renderTable( - ["Model", "Query form", "Samples", "Weights I/C/B/D/S", "Fit R@5", "Fit R@20"], - ["---", "---", "---:", "---", "---:", "---:"], - artifact.recommendedWeights.map((result) => [ - result.model, - result.queryKind, - String(result.samples), - formatWeights(result.weights), - percent(result.fitQuality.recallAt5), - percent(result.fitQuality.recallAt20), - ]), - ), - ) - const fusionGroups = new Map() for (const result of artifact.fusionSearch) { const key = `${result.model}\0${result.fusion}\0${result.strategy}` diff --git a/benchmarks/retrieval/evaluation/router-search/config-space.ts b/benchmarks/retrieval/evaluation/router-search/config-space.ts index 3c75a93..55ea240 100644 --- a/benchmarks/retrieval/evaluation/router-search/config-space.ts +++ b/benchmarks/retrieval/evaluation/router-search/config-space.ts @@ -1,4 +1,4 @@ -import { +import { ZERO_CHANNEL_COEFFICIENTS, CHANNEL_NAMES, decodeEvidenceRouterConfig, @@ -13,7 +13,6 @@ import { DEFAULT_HALVING_FUNNEL_STRATEGY, DEFAULT_PROXY_PROMOTION_STRATEGY, DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS, - DEFAULT_SUCCESSIVE_HALVING_STRATEGY, type QualitySummary, } from "../types.js" @@ -32,7 +31,8 @@ export const SEARCH_PROXY_SAMPLE_FRACTION = export const SEARCH_PROXY_MINIMUM_SAMPLES = DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.proxyMinimumSamples export const SEARCH_PROXY_PROMOTION_FACTOR = DEFAULT_PROXY_PROMOTION_STRATEGY.proxyPromotionFactor -export const SEARCH_HALVING_KEEP_FACTOR = DEFAULT_SUCCESSIVE_HALVING_STRATEGY.halvingKeepFactor +/** Funnel halving keep factor inherited from the historical successive-halving control. */ +export const SEARCH_HALVING_KEEP_FACTOR = 8 export const SEARCH_FUNNEL_SPREAD_SURVIVORS = DEFAULT_HALVING_FUNNEL_STRATEGY.spreadSurvivors export const SEARCH_FUNNEL_FINALISTS = DEFAULT_HALVING_FUNNEL_STRATEGY.finalists export const RANDOM_SEARCH_SEED = DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.seed || 1 diff --git a/benchmarks/retrieval/evaluation/router-search/rank.ts b/benchmarks/retrieval/evaluation/router-search/rank.ts index bfa4100..7dabd5d 100644 --- a/benchmarks/retrieval/evaluation/router-search/rank.ts +++ b/benchmarks/retrieval/evaluation/router-search/rank.ts @@ -291,8 +291,8 @@ export const PROXY_PROMOTION_MODE: RouterSearchMode = { selectObjectiveCandidates(candidates, limit, context.baseline, context.profile), } -const SUCCESSIVE_HALVING_MODE: RouterSearchMode = { - name: "successive-halving", +const HALVING_FUNNEL_MODE: RouterSearchMode = { + name: "halving-funnel", expansionFactor: SEARCH_HALVING_KEEP_FACTOR, recordsProxyAgreement: false, runsRandomBaseline: false, @@ -308,14 +308,8 @@ const SUCCESSIVE_HALVING_MODE: RouterSearchMode = { selectBeam: (_context, candidates, limit) => candidates.slice(0, limit), } -const HALVING_FUNNEL_MODE: RouterSearchMode = { - ...SUCCESSIVE_HALVING_MODE, - name: "halving-funnel", -} - const ROUTER_SEARCH_MODES: Readonly> = { "proxy-promotion": PROXY_PROMOTION_MODE, - "successive-halving": SUCCESSIVE_HALVING_MODE, "halving-funnel": HALVING_FUNNEL_MODE, } diff --git a/benchmarks/retrieval/evaluation/search.ts b/benchmarks/retrieval/evaluation/search.ts index 6006c9f..126e349 100644 --- a/benchmarks/retrieval/evaluation/search.ts +++ b/benchmarks/retrieval/evaluation/search.ts @@ -20,10 +20,8 @@ import type { import { fitRecommendedEvidenceRouter, fitRecommendedFusionWeights, - fitRecommendedWeights, optimizeEvidenceRouter, optimizeFusionWeights, - optimizeWeights, evaluateProductionRouter, type BenchmarkSearchOptions, type WeightSearchSample, @@ -35,8 +33,6 @@ export interface BenchmarkSearchConfig { readonly groupedFolds: number /** Whether each selected repository is evaluated as an excluded holdout. */ readonly repositoryHoldouts: boolean - /** Whether historical query-kind RRF grids and Shapley diagnostics run. */ - readonly legacyDiagnostics: boolean /** Static fusion formulas evaluated by this profile. */ readonly fusionMethods: readonly FusionMethod[] /** Fusion formulas used when evaluating the evidence router. */ @@ -45,27 +41,18 @@ export interface BenchmarkSearchConfig { /** Quality search outputs and timing fields assembled for one benchmark artifact. */ export interface BenchmarkSearchResults { - readonly weightSearch: readonly BenchmarkArtifact["weightSearch"][number][] - readonly recommendedWeights: readonly BenchmarkArtifact["recommendedWeights"][number][] readonly productionRouterSearch: readonly BenchmarkArtifact["productionRouterSearch"][number][] readonly fusionSearch: readonly BenchmarkArtifact["fusionSearch"][number][] readonly recommendedFusionWeights: readonly BenchmarkArtifact["recommendedFusionWeights"][number][] readonly evidenceRouterSearch: readonly BenchmarkArtifact["evidenceRouterSearch"][number][] readonly recommendedEvidenceRouters: readonly BenchmarkArtifact["recommendedEvidenceRouters"][number][] readonly promotionEvidence: readonly BenchmarkArtifact["promotionEvidence"][number][] - readonly weightSearchDurationMs: number readonly fusionSearchDurationMs: number readonly evidenceRouterSearchDurationMs: number readonly candidateQueueStartupDurationMs: number readonly candidateQueueShutdownDurationMs: number } -type SampleGroup = { - readonly model: string - readonly queryKind: WeightSearchSample["queryKind"] - readonly samples: readonly WeightSearchSample[] -} - const describeRouterJob = (job: RouterSearchJob): string => job.kind === "holdout" ? `${job.model}/${job.fusion} ${job.strategy} fold ${job.fold}` @@ -143,11 +130,6 @@ const planSearchSplits = ( return splits } -interface WeightSearchGroupResult { - readonly weightSearch: readonly BenchmarkArtifact["weightSearch"][number][] - readonly recommendedWeights: readonly BenchmarkArtifact["recommendedWeights"][number][] -} - const planEvidenceRouterJobs = ( samplesByModel: ReadonlyMap, config: BenchmarkSearchConfig, @@ -198,74 +180,6 @@ const runRouterSearchJob = ( ) } -const runWeightSearchForGroup = ( - group: SampleGroup, - config: BenchmarkSearchConfig, - groupedStrategy: ValidationStrategy, - optimizationProfile: OptimizationProfile, - searchOptions: BenchmarkSearchOptions, -): Effect.Effect => - Effect.gen(function* () { - const weightSearch: BenchmarkArtifact["weightSearch"][number][] = [] - for (const split of planSearchSplits(group.samples, config, groupedStrategy)) - weightSearch.push( - yield* runParallelSearch((signal) => - optimizeWeights( - group.model, - group.queryKind, - split.strategy, - split.fold, - split.development, - split.validation, - optimizationProfile, - { ...searchOptions, signal }, - ), - ), - ) - const recommendedWeights = [ - yield* runParallelSearch((signal) => - fitRecommendedWeights(group.model, group.queryKind, group.samples, optimizationProfile, { - ...searchOptions, - signal, - }), - ), - ] - return { weightSearch, recommendedWeights } - }) - -const runWeightSearchStage = ( - config: BenchmarkSearchConfig, - sampleGroups: ReadonlyMap, - groupedStrategy: ValidationStrategy, - optimizationProfile: OptimizationProfile, - searchOptions: BenchmarkSearchOptions, -): Effect.Effect< - Pick, - Error -> => - Effect.gen(function* () { - const startedAt = performance.now() - const weightSearch: BenchmarkArtifact["weightSearch"][number][] = [] - const recommendedWeights: BenchmarkArtifact["recommendedWeights"][number][] = [] - if (config.legacyDiagnostics) - for (const group of sampleGroups.values()) { - const result = yield* runWeightSearchForGroup( - group, - config, - groupedStrategy, - optimizationProfile, - searchOptions, - ) - weightSearch.push(...result.weightSearch) - recommendedWeights.push(...result.recommendedWeights) - } - return { - weightSearch, - recommendedWeights, - weightSearchDurationMs: performance.now() - startedAt, - } - }) - const runProductionRouterSearch = ( config: BenchmarkSearchConfig, samplesByModel: ReadonlyMap, @@ -572,7 +486,6 @@ const runEvidenceRouterSearchStage = ( /** Run all static and evidence-router quality searches over prepared samples. */ export const runBenchmarkSearch = ( config: BenchmarkSearchConfig, - sampleGroups: ReadonlyMap, samplesByModel: ReadonlyMap, groupedStrategy: ValidationStrategy, optimizationProfile: OptimizationProfile, @@ -580,13 +493,6 @@ export const runBenchmarkSearch = ( searchOptions: BenchmarkSearchOptions, ): Effect.Effect => Effect.gen(function* () { - const weight = yield* runWeightSearchStage( - config, - sampleGroups, - groupedStrategy, - optimizationProfile, - searchOptions, - ) const fusion = yield* runFusionSearchStage( config, samplesByModel, @@ -602,5 +508,5 @@ export const runBenchmarkSearch = ( serialSearch, searchOptions, ) - return { ...weight, ...fusion, ...evidenceRouter } + return { ...fusion, ...evidenceRouter } }) diff --git a/benchmarks/retrieval/evaluation/subsample-evidence.ts b/benchmarks/retrieval/evaluation/subsample-evidence.ts index b3be102..af382c0 100644 --- a/benchmarks/retrieval/evaluation/subsample-evidence.ts +++ b/benchmarks/retrieval/evaluation/subsample-evidence.ts @@ -58,7 +58,7 @@ const QUERY_FORMS: readonly QueryKind[] = [ ] /** Prepared model, device, and full-size corpus for one repository sweep. */ -export interface SubSampleContext { +interface SubSampleContext { readonly manifest: CorpusManifest readonly model: string readonly dims: number @@ -240,33 +240,6 @@ export interface RealCorpusSizeSweepResult { readonly perSizeSamples: readonly { readonly corpusSize: number; readonly samples: number }[] } -const buildSearchInputs = ( - model: string, - samples: readonly WeightSearchSample[], -): { - sampleGroups: Map< - string, - { model: string; queryKind: QueryKind; samples: readonly WeightSearchSample[] } - > - samplesByModel: Map -} => { - const sampleGroups = new Map< - string, - { model: string; queryKind: QueryKind; samples: readonly WeightSearchSample[] } - >() - const samplesByModel = new Map() - samplesByModel.set(model, samples) - for (const sample of samples) { - const key = `${model}\0${sample.queryKind}` - sampleGroups.set(key, { - model, - queryKind: sample.queryKind, - samples: [...(sampleGroups.get(key)?.samples ?? []), sample], - }) - } - return { sampleGroups, samplesByModel } -} - const goldTargetsByQuestion = ( manifest: CorpusManifest, corpus: PreparedCorpus, @@ -310,17 +283,17 @@ export const runRealCorpusSizeSweep = ( async (plan) => { const samples = await Effect.runPromise(buildSubSampleSamples(context, plan)) perSizeSamples.push({ corpusSize: plan.targetSize, samples: samples.length }) - const { sampleGroups, samplesByModel } = buildSearchInputs(context.model, samples) + const samplesByModel = new Map([ + [context.model, samples], + ]) const search = await Effect.runPromise( runBenchmarkSearch( { groupedFolds: 3, repositoryHoldouts: false, - legacyDiagnostics: false, fusionMethods: ["dbsf"], routerFusionMethods: ["dbsf"], }, - sampleGroups, samplesByModel, "grouped-3-fold", OPTIMIZATION_PROFILES["search-priority"], diff --git a/benchmarks/retrieval/evaluation/types.ts b/benchmarks/retrieval/evaluation/types.ts index 341fb07..cb6348c 100644 --- a/benchmarks/retrieval/evaluation/types.ts +++ b/benchmarks/retrieval/evaluation/types.ts @@ -49,13 +49,6 @@ export interface ProxyPromotionStrategy extends RouterSearchBudget { readonly tieBreaking: "guardrails>objective>complexity>stable-key" } -/** Successive halving strategy: historical lexicographic comparator with keep factor. */ -export interface SuccessiveHalvingStrategy extends RouterSearchBudget { - readonly kind: "successive-halving" - readonly algorithm: string - readonly halvingKeepFactor: 8 -} - /** * Halving funnel strategy: broad cheap waves with late full-fidelity evaluation. Wave 1 spreads * `globalScouts` scouts scored on the proxy sample only, wave 2 samples Sobol clouds around the @@ -69,16 +62,9 @@ export interface HalvingFunnelStrategy extends RouterSearchBudget { } /** Versioned evidence-router search strategy recorded in benchmark artifacts. */ -export type RouterSearchStrategy = - | ProxyPromotionStrategy - | SuccessiveHalvingStrategy - | HalvingFunnelStrategy - -export const ROUTER_SEARCH_STRATEGY_NAMES = [ - "proxy-promotion", - "successive-halving", - "halving-funnel", -] as const +export type RouterSearchStrategy = ProxyPromotionStrategy | HalvingFunnelStrategy + +export const ROUTER_SEARCH_STRATEGY_NAMES = ["proxy-promotion", "halving-funnel"] as const export type RouterSearchStrategyName = (typeof ROUTER_SEARCH_STRATEGY_NAMES)[number] @@ -100,14 +86,6 @@ export const DEFAULT_PROXY_PROMOTION_STRATEGY: ProxyPromotionStrategy = { tieBreaking: "guardrails>objective>complexity>stable-key", } -/** Default successive halving parameters, recorded with Halton scouts unless overridden. */ -export const DEFAULT_SUCCESSIVE_HALVING_STRATEGY: SuccessiveHalvingStrategy = { - kind: "successive-halving", - algorithm: `halton-${BEAM_ALGORITHM_PREFIX}-successive-halving`, - ...ROUTER_SEARCH_BUDGET, - halvingKeepFactor: 8, -} - /** Default fidelity funnel: broad proxy wave, local proxy cloud, then 256 diverse finalists. */ export const DEFAULT_HALVING_FUNNEL_STRATEGY: HalvingFunnelStrategy = { kind: "halving-funnel", @@ -123,11 +101,7 @@ export const routerSearchStrategyFor = ( name: RouterSearchStrategyName, ): RouterSearchStrategy => { const base = - name === "proxy-promotion" - ? DEFAULT_PROXY_PROMOTION_STRATEGY - : name === "successive-halving" - ? DEFAULT_SUCCESSIVE_HALVING_STRATEGY - : DEFAULT_HALVING_FUNNEL_STRATEGY + name === "proxy-promotion" ? DEFAULT_PROXY_PROMOTION_STRATEGY : DEFAULT_HALVING_FUNNEL_STRATEGY const algorithm = name === "halving-funnel" ? `${scoutSequence}-${FUNNEL_ALGORITHM_PREFIX}` @@ -307,29 +281,6 @@ export interface PromotionEvidence { readonly stability: CandidateStability } -/** One cross-validation fold with weights selected without its validation samples. */ -export interface WeightSearchResult { - readonly model: string - readonly queryKind: QueryKind - readonly strategy: ValidationStrategy - readonly fold: string - readonly developmentQueries: number - readonly validationQueries: number - readonly weights: ChannelWeights - readonly development: QualitySummary - readonly validation: QualitySummary - readonly shapleyRecallAt20: ChannelWeights -} - -/** Deployment candidate fitted on all available samples after cross-validation. */ -export interface RecommendedWeights { - readonly model: string - readonly queryKind: QueryKind - readonly samples: number - readonly weights: ChannelWeights - readonly fitQuality: QualitySummary -} - /** Static fusion weights selected without one validation fold. */ export interface FusionSearchResult { readonly model: string @@ -480,7 +431,6 @@ export interface BenchmarkTimings { readonly corpusPreparationDurationMs: number readonly embeddingDurationMs: number readonly retrievalDurationMs: number - readonly weightSearchDurationMs: number readonly fusionSearchDurationMs: number readonly evidenceRouterSearchDurationMs: number /** Time spent starting the shared native candidate queue. */ @@ -554,8 +504,6 @@ export interface BenchmarkArtifact { readonly queryTokenizationDurationMs: number }> readonly measurements: readonly QueryMeasurement[] - readonly weightSearch: readonly WeightSearchResult[] - readonly recommendedWeights: readonly RecommendedWeights[] readonly productionRouterSearch: readonly ProductionRouterSearchResult[] readonly fusionSearch: readonly FusionSearchResult[] readonly recommendedFusionWeights: readonly RecommendedFusionWeights[] diff --git a/benchmarks/retrieval/evaluation/weight-search.ts b/benchmarks/retrieval/evaluation/weight-search.ts index 4c7e5bc..576ac89 100644 --- a/benchmarks/retrieval/evaluation/weight-search.ts +++ b/benchmarks/retrieval/evaluation/weight-search.ts @@ -1,4 +1,4 @@ -import type { Chunk } from "../../../src/domain/chunk.js" +import type { Chunk } from "../../../src/domain/chunk.js" import type { RankedChunk } from "../../../src/domain/ports.js" import { PRODUCTION_COMPATIBILITY_CONFIG, @@ -90,13 +90,11 @@ import { type QualitySummary, type RecommendedEvidenceRouter, type RecommendedFusionWeights, - type RecommendedWeights, type RouterSearchDiagnostics, type RouterObjective, type RouterSearchStrategyName, type SearchBaselineComparison, type SelectionStability, - type WeightSearchResult, } from "./types.js" /** Precomputed query evidence used for cheap fusion and weight experiments. */ @@ -532,51 +530,6 @@ const weightCandidates = (): readonly ChannelWeights[] => { ] } -const factorial = (value: number): number => { - let result = 1 - for (let factor = 2; factor <= value; factor++) result *= factor - return result -} - -const shapleyValues = ( - samples: readonly WeightSearchSample[], - weights: ChannelWeights, - profile: OptimizationProfile = SEARCH_PRIORITY_PROFILE, -): ChannelWeights => { - const values: Record = { - identity: 0, - camelcase: 0, - bm25: 0, - dense: 0, - sparse: 0, - } - const channelCount = CHANNELS.length - const utility = (mask: number): number => { - if (mask === 0) return 0 - const coalition: ChannelWeights = { - identity: mask & 1 ? weights.identity : 0, - camelcase: mask & 2 ? weights.camelcase : 0, - bm25: mask & 4 ? weights.bm25 : 0, - dense: mask & 8 ? weights.dense : 0, - sparse: mask & 16 ? weights.sparse : 0, - } - return summarize(samples, coalition, "rrf", profile).recallAt20 - } - - for (let channelIndex = 0; channelIndex < channelCount; channelIndex++) { - const channelMask = 1 << channelIndex - for (let mask = 0; mask < 1 << channelCount; mask++) { - if (mask & channelMask) continue - const coalitionSize = CHANNELS.filter((_, index) => mask & (1 << index)).length - const coefficient = - (factorial(coalitionSize) * factorial(channelCount - coalitionSize - 1)) / - factorial(channelCount) - values[CHANNELS[channelIndex]] += coefficient * (utility(mask | channelMask) - utility(mask)) - } - } - return values -} - /** Static weight candidate and its development quality. */ export interface WeightCandidate { readonly weights: ChannelWeights @@ -1058,82 +1011,6 @@ const withCandidatePool = async ( } } -/** Select weights on development samples, then evaluate unchanged on one validation fold. */ -const optimizeWeightsWithPool = async ( - model: string, - queryKind: QueryKind, - strategy: WeightSearchResult["strategy"], - fold: string, - development: readonly WeightSearchSample[], - validation: readonly WeightSearchSample[], - profile: OptimizationProfile = SEARCH_PRIORITY_PROFILE, - pool: CandidateEvaluationPool, -): Promise => { - const selected = await selectBestWeights( - pool, - PROXY_PROMOTION_MODE, - "reranker-top20", - undefined, - profile, - ) - return { - model, - queryKind, - strategy, - fold, - developmentQueries: development.length, - validationQueries: validation.length, - weights: selected.weights, - development: selected.quality, - validation: summarize(validation, selected.weights, "rrf", profile), - shapleyRecallAt20: shapleyValues(validation, selected.weights, profile), - } -} - -const optimizeWeightsWithOptions = ( - model: string, - queryKind: QueryKind, - strategy: WeightSearchResult["strategy"], - fold: string, - development: readonly WeightSearchSample[], - validation: readonly WeightSearchSample[], - profile: OptimizationProfile = SEARCH_PRIORITY_PROFILE, - options: SearchOptions, -): Promise => - withCandidatePool(development, "rrf", profile, options, (pool) => - optimizeWeightsWithPool( - model, - queryKind, - strategy, - fold, - development, - validation, - profile, - pool, - ), - ) - -export const optimizeWeights = ( - model: string, - queryKind: QueryKind, - strategy: WeightSearchResult["strategy"], - fold: string, - development: readonly WeightSearchSample[], - validation: readonly WeightSearchSample[], - profile: OptimizationProfile = SEARCH_PRIORITY_PROFILE, - options: SearchOptions = { workerCount: 0 }, -): Promise => - optimizeWeightsWithOptions( - model, - queryKind, - strategy, - fold, - development, - validation, - profile, - options, - ) - /** Select static weights for one fusion method, then evaluate them unchanged on a holdout. */ const optimizeFusionWeightsWithPool = async ( model: string, @@ -1460,50 +1337,6 @@ export const optimizeEvidenceRouter = async ( return results }) -/** Fit one deployment candidate on all samples after cross-validation has measured generalization. */ -const fitRecommendedWeightsWithPool = async ( - model: string, - queryKind: QueryKind, - samples: readonly WeightSearchSample[], - profile: OptimizationProfile = SEARCH_PRIORITY_PROFILE, - pool: CandidateEvaluationPool, -): Promise => { - const selected = await selectBestWeights( - pool, - PROXY_PROMOTION_MODE, - "reranker-top20", - undefined, - profile, - ) - return { - model, - queryKind, - samples: samples.length, - weights: selected.weights, - fitQuality: selected.quality, - } -} - -const fitRecommendedWeightsWithOptions = ( - model: string, - queryKind: QueryKind, - samples: readonly WeightSearchSample[], - profile: OptimizationProfile, - options: SearchOptions, -): Promise => - withCandidatePool(samples, "rrf", profile, options, (pool) => - fitRecommendedWeightsWithPool(model, queryKind, samples, profile, pool), - ) - -export const fitRecommendedWeights = ( - model: string, - queryKind: QueryKind, - samples: readonly WeightSearchSample[], - profile: OptimizationProfile = SEARCH_PRIORITY_PROFILE, - options: SearchOptions = { workerCount: 0 }, -): Promise => - fitRecommendedWeightsWithOptions(model, queryKind, samples, profile, options) - /** Fit one static candidate for a fusion method across all query forms. */ const fitRecommendedFusionWeightsWithPool = async ( model: string, diff --git a/benchmarks/retrieval/runner.ts b/benchmarks/retrieval/runner.ts index 6aba1eb..7c5aade 100644 --- a/benchmarks/retrieval/runner.ts +++ b/benchmarks/retrieval/runner.ts @@ -36,7 +36,6 @@ const profileConfig = (profile: BenchmarkProfile): BenchmarkSearchConfig => { return { groupedFolds: 5, repositoryHoldouts: false, - legacyDiagnostics: false, fusionMethods: ["dbsf"], routerFusionMethods: ["dbsf"], } @@ -44,7 +43,6 @@ const profileConfig = (profile: BenchmarkProfile): BenchmarkSearchConfig => { return { groupedFolds: 3, repositoryHoldouts: false, - legacyDiagnostics: false, fusionMethods: ["dbsf"], routerFusionMethods: ["dbsf"], } @@ -52,7 +50,6 @@ const profileConfig = (profile: BenchmarkProfile): BenchmarkSearchConfig => { return { groupedFolds: 5, repositoryHoldouts: true, - legacyDiagnostics: false, fusionMethods: ["dbsf"], routerFusionMethods: ["dbsf"], } @@ -60,7 +57,6 @@ const profileConfig = (profile: BenchmarkProfile): BenchmarkSearchConfig => { return { groupedFolds: 5, repositoryHoldouts: true, - legacyDiagnostics: false, fusionMethods: FUSION_METHODS, routerFusionMethods: FUSION_METHODS, } @@ -196,7 +192,6 @@ export const runRetrievalBenchmark = ( embeddingRuns, sparseEmbeddingRuns, measurements, - sampleGroups, samplesByModel, retrievalDurationMs, } = collected @@ -206,7 +201,6 @@ export const runRetrievalBenchmark = ( ) const search = yield* runBenchmarkSearch( config, - sampleGroups, samplesByModel, groupedStrategy, optimizationProfile, @@ -264,7 +258,6 @@ export const runRetrievalBenchmark = ( corpusPreparationDurationMs, embeddingDurationMs, retrievalDurationMs, - weightSearchDurationMs: search.weightSearchDurationMs, fusionSearchDurationMs: search.fusionSearchDurationMs, evidenceRouterSearchDurationMs: search.evidenceRouterSearchDurationMs, candidateQueueStartupDurationMs: search.candidateQueueStartupDurationMs, @@ -290,8 +283,6 @@ export const runRetrievalBenchmark = ( embeddingRuns, sparseEmbeddingRuns, measurements, - weightSearch: search.weightSearch, - recommendedWeights: search.recommendedWeights, productionRouterSearch: search.productionRouterSearch, fusionSearch: search.fusionSearch, recommendedFusionWeights: search.recommendedFusionWeights, diff --git a/benchmarks/tests/channels.test.ts b/benchmarks/tests/channels.test.ts index 20cba83..296901c 100644 --- a/benchmarks/tests/channels.test.ts +++ b/benchmarks/tests/channels.test.ts @@ -28,7 +28,6 @@ import { compareObjectiveQuality } from "../retrieval/evaluation/router-search/o import { ROUTER_OBJECTIVES, type QualitySummary } from "../retrieval/evaluation/types.js" import { optimizeEvidenceRouter, - optimizeWeights, selectEligibleCandidate, selectObjectiveArchiveCandidates, } from "../retrieval/evaluation/weight-search.js" @@ -533,39 +532,6 @@ describe("retrieval benchmark fixture", () => { expect(ranked).toEqual([{ chunkIndex: 0, score: 0.5 }]) }) - it("learns weights on development and attributes holdout value with Shapley", async () => { - const sample = { - repository: "fixture", - intentId: "fixture-001", - queryKind: "identifier" as const, - groupedFold: 0, - query: "loadProjectConfiguration", - rankings: { - identity: [{ chunkIndex: 0, score: 1 }], - camelcase: [{ chunkIndex: 1, score: 1 }], - bm25: [{ chunkIndex: 2, score: 1 }], - dense: [{ chunkIndex: 3, score: 1 }], - sparse: [], - }, - targets: [new Set([0])], - chunks, - } - const result = await optimizeWeights( - "fixture", - "identifier", - "grouped-5-fold", - "1", - [sample], - [sample], - ) - expect(result.weights.identity).toBeGreaterThan(0) - expect(result.weights.camelcase).toBe(0) - expect(result.weights.bm25).toBe(0) - expect(result.weights.dense).toBe(0) - expect(result.validation.recallAt20).toBe(1) - expect(result.shapleyRecallAt20.identity).toBe(1) - }) - it("compares NDCG-first direct selection with the matched Recall-first ablation", () => { const quality = (ndcgAt5: number, recallAt5: number): QualitySummary => ({ ndcgAt5, diff --git a/benchmarks/tests/matrix.test.ts b/benchmarks/tests/matrix.test.ts index 406e32b..445062a 100644 --- a/benchmarks/tests/matrix.test.ts +++ b/benchmarks/tests/matrix.test.ts @@ -21,7 +21,6 @@ const timings: BenchmarkTimings = { corpusPreparationDurationMs: 10, embeddingDurationMs: 20, retrievalDurationMs: 30, - weightSearchDurationMs: 5, fusionSearchDurationMs: 5, evidenceRouterSearchDurationMs: 25, candidateQueueStartupDurationMs: 2, diff --git a/benchmarks/tests/report.test.ts b/benchmarks/tests/report.test.ts index 4179b93..38d8e86 100644 --- a/benchmarks/tests/report.test.ts +++ b/benchmarks/tests/report.test.ts @@ -27,7 +27,6 @@ const artifact = { corpusPreparationDurationMs: 0, embeddingDurationMs: 0, retrievalDurationMs: 0, - weightSearchDurationMs: 0, fusionSearchDurationMs: 0, evidenceRouterSearchDurationMs: 0, candidateQueueStartupDurationMs: 0, @@ -72,8 +71,6 @@ const artifact = { queryDurationMs: 0, }, ], - weightSearch: [], - recommendedWeights: [], productionRouterSearch: [], fusionSearch: [], recommendedFusionWeights: [], diff --git a/benchmarks/tests/retrieval.test.ts b/benchmarks/tests/retrieval.test.ts index 71d200e..f4a1ad6 100644 --- a/benchmarks/tests/retrieval.test.ts +++ b/benchmarks/tests/retrieval.test.ts @@ -164,8 +164,6 @@ const runProfile = (profile: BenchmarkProfile, groupedFolds: number, fusionMetho expect(artifact.evidenceRouterSearch.every((row) => row.fullEvaluations > 0)).toBe(true) expect(artifact.recommendedEvidenceRouters.every((row) => row.proxyEvaluations >= 0)).toBe(true) expect(artifact.recommendedEvidenceRouters.every((row) => row.fullEvaluations > 0)).toBe(true) - expect(artifact.weightSearch.length).toBe(0) - expect(artifact.recommendedWeights.length).toBe(0) expect(artifact.measurements.every((row) => row.recallAt20 >= row.recallAt10)).toBe(true) expect(artifact.measurements.every((row) => row.recallAt10 >= row.recallAt5)).toBe(true) expect(artifact.measurements.every((row) => row.recallAt50 >= row.recallAt20)).toBe(true) @@ -227,11 +225,11 @@ it("shuffles intent groups deterministically before assigning folds", () => { it("resolves the selectable router search strategies", () => { expect(resolveRouterSearchStrategy(undefined)).toBe("halving-funnel") - expect(resolveRouterSearchStrategy("successive-halving")).toBe("successive-halving") - expect(routerSearchStrategyFor("halton", "successive-halving")).toMatchObject({ - kind: "successive-halving", - algorithm: "halton-global-scout-elitist-beam-successive-halving", - halvingKeepFactor: 8, + expect(resolveRouterSearchStrategy("proxy-promotion")).toBe("proxy-promotion") + expect(routerSearchStrategyFor("halton", "proxy-promotion")).toMatchObject({ + kind: "proxy-promotion", + algorithm: "halton-global-scout-elitist-beam-proxy-promotion", + proxyPromotionFactor: 8, }) expect(resolveRouterSearchStrategy("halving-funnel")).toBe("halving-funnel") expect(routerSearchStrategyFor("sobol", "halving-funnel")).toMatchObject({ @@ -245,6 +243,9 @@ it("resolves the selectable router search strategies", () => { expect(sobolStrategy.kind === "proxy-promotion" && sobolStrategy.proxyPromotionFactor).toBe(8) expect(DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.kind).toBe("proxy-promotion") expect(DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS).not.toHaveProperty("halvingKeepFactor") + expect(() => resolveRouterSearchStrategy("successive-halving")).toThrow( + "Unknown PIX_BENCH_ROUTER_STRATEGY value: successive-halving", + ) expect(() => resolveRouterSearchStrategy("unknown")).toThrow( "Unknown PIX_BENCH_ROUTER_STRATEGY value: unknown", ) diff --git a/benchmarks/tests/run-config.test.ts b/benchmarks/tests/run-config.test.ts index 5373b05..9ec87be 100644 --- a/benchmarks/tests/run-config.test.ts +++ b/benchmarks/tests/run-config.test.ts @@ -19,7 +19,7 @@ it("resolves search knobs with strategy defaults", () => { it("parses overrides and rejects invalid values by knob name", () => { expect( resolveSearchKnobs({ - PIX_BENCH_ROUTER_STRATEGY: "successive-halving", + PIX_BENCH_ROUTER_STRATEGY: "proxy-promotion", PIX_BENCH_SCOUT_SEQUENCE: "sobol", PIX_BENCH_SEED_HYPOTHESES: "1", PIX_BENCH_BEAM_SCHEDULE: "decaying", @@ -29,7 +29,7 @@ it("parses overrides and rejects invalid values by knob name", () => { PIX_BENCH_LOCAL_CLOUD_RADIUS: "2", }), ).toEqual({ - routerSearchStrategy: "successive-halving", + routerSearchStrategy: "proxy-promotion", scoutSequence: "sobol", seedHypotheses: true, beamSchedule: "decaying", diff --git a/benchmarks/tests/worker-pool.test.ts b/benchmarks/tests/worker-pool.test.ts index 552e6b0..aa8292f 100644 --- a/benchmarks/tests/worker-pool.test.ts +++ b/benchmarks/tests/worker-pool.test.ts @@ -9,7 +9,6 @@ import { ROUTER_OBJECTIVES } from "../retrieval/evaluation/types.js" import { fitRecommendedEvidenceRouter, fitRecommendedFusionWeights, - fitRecommendedWeights, optimizeEvidenceRouter, optimizeFusionWeights, summarize, @@ -124,13 +123,13 @@ describe("benchmark candidate evaluation pool", () => { { workerCount: 0, evaluationQueue: candidateQueue, - routerSearchStrategy: "successive-halving", + routerSearchStrategy: "halving-funnel", }, ), fitRecommendedEvidenceRouter("fixture", "dbsf", [searchSample], SEARCH_PRIORITY_PROFILE, { workerCount: 0, evaluationQueue: candidateQueue, - routerSearchStrategy: "successive-halving", + routerSearchStrategy: "halving-funnel", }), ]) const serialHoldout = await optimizeEvidenceRouter( @@ -141,14 +140,14 @@ describe("benchmark candidate evaluation pool", () => { [searchSample], [searchSample], SEARCH_PRIORITY_PROFILE, - { workerCount: 0, routerSearchStrategy: "successive-halving" }, + { workerCount: 0, routerSearchStrategy: "halving-funnel" }, ) const serialFitAll = await fitRecommendedEvidenceRouter( "fixture", "dbsf", [searchSample], SEARCH_PRIORITY_PROFILE, - { workerCount: 0, routerSearchStrategy: "successive-halving" }, + { workerCount: 0, routerSearchStrategy: "halving-funnel" }, ) expect(parallelHoldout.map(withoutSearchTimings)).toEqual( @@ -174,7 +173,7 @@ describe("benchmark candidate evaluation pool", () => { { workerCount: 0, evaluationQueue: candidateQueue, - routerSearchStrategy: "successive-halving", + routerSearchStrategy: "halving-funnel", }, ) const serial = await fitRecommendedEvidenceRouter( @@ -182,7 +181,7 @@ describe("benchmark candidate evaluation pool", () => { "dbsf", halvingSamples, SEARCH_PRIORITY_PROFILE, - { workerCount: 0, routerSearchStrategy: "successive-halving" }, + { workerCount: 0, routerSearchStrategy: "halving-funnel" }, ) expect(parallel.map(withoutSearchTimings)).toEqual(serial.map(withoutSearchTimings)) const result = parallel[0] @@ -304,19 +303,6 @@ describe("benchmark candidate evaluation pool", () => { ) }) - it("keeps the explicit parallel search result equal to the serial search", async () => { - const serial = await fitRecommendedWeights("fixture", "identifier", [searchSample]) - const parallel = await fitRecommendedWeights( - "fixture", - "identifier", - [searchSample], - undefined, - { workerCount: 2, batchSize: 16 }, - ) - - expect(parallel).toEqual(serial) - }) - it("keeps static fusion fitting serial and parallel paths equivalent", async () => { const serial = await fitRecommendedFusionWeights("fixture", "dbsf", [searchSample]) const parallel = await fitRecommendedFusionWeights( From 941b61b7eb40fc354d03d09d69463d2f5f3ca8c9 Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Fri, 28 Aug 2026 21:10:17 +0200 Subject: [PATCH 12/16] refactor(bench): delete beam search strategies, sobol default, chunking sweep --- benchmarks/README.md | 37 ++-- benchmarks/retrieval/evaluation/report.ts | 23 +- .../evaluation/router-search/config-space.ts | 9 +- .../evaluation/router-search/rank.ts | 112 +--------- .../retrieval/evaluation/scouts/index.ts | 2 +- .../evaluation/subsample-evidence.ts | 82 ++++++- benchmarks/retrieval/evaluation/types.ts | 92 ++------ .../retrieval/evaluation/weight-search.ts | 208 +++--------------- benchmarks/retrieval/run-config.ts | 52 +---- benchmarks/retrieval/runner.ts | 12 +- benchmarks/tests/channels.test.ts | 6 - benchmarks/tests/chunking-sweep.test.ts | 40 ++++ benchmarks/tests/matrix.test.ts | 3 +- benchmarks/tests/report.test.ts | 6 +- benchmarks/tests/retrieval.test.ts | 38 +--- benchmarks/tests/run-config.test.ts | 25 +-- benchmarks/tests/scouts.test.ts | 6 +- benchmarks/tests/worker-pool.test.ts | 10 +- package.json | 1 + 19 files changed, 229 insertions(+), 535 deletions(-) create mode 100644 benchmarks/tests/chunking-sweep.test.ts diff --git a/benchmarks/README.md b/benchmarks/README.md index 88ddd7e..a06869b 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -134,18 +134,17 @@ deferred until the promotion bar is met. The real multiplicative-vs-log-linear c Every search knob has a recorded status so work stops re-litigating settled ones: -| Knob | Status | Evidence | -| ------------------------------------------- | --------------------------------------------------------------------------------- | ---------------------------- | -| Router model (multiplicative vs log-linear) | Settled: keep multiplicative; switch at the next mandatory promoted-config re-fit | ADR 0021 | -| Router search strategy | Settled: `halving-funnel` default; `proxy-promotion` is the slow control | BASELINE (#189) | -| Finalist budget | Settled: 256 (384 measured identical, +65 s) | BASELINE (#189) | -| Local Sobol cloud (points/radius) | Settled: 16 points, radius 2 (32/r3 added no quality) | BASELINE (#189) | -| Scout sequence (halton/sobol/random) | Settled: all inside noise; halton default | BASELINE (schema 29) | -| Wide-vs-deep passes | Settled: 1 wide pass matches 2 narrow passes at 45% fewer evals | BASELINE (schema 29) | -| Corpus-size factor | Deferred with promotion bar; sweep stays as standing measurement | ADR 0022 | -| Fitting methods, selection rules | Diagnostics only, recorded per comparison run | ADR 0021 | -| One-pass variant on validate profile | Open | BASELINE (validate addendum) | -| Chunking (chunkTokens/overlapLines) | Open: never swept by this suite | — | +| Knob | Status | Evidence | +| ------------------------------------------- | --------------------------------------------------------------------------------- | -------------------- | +| Router model (multiplicative vs log-linear) | Settled: keep multiplicative; switch at the next mandatory promoted-config re-fit | ADR 0021 | +| Router search strategy | Settled and deleted: only `halving-funnel` remains (beam control removed) | BASELINE (#189) | +| Finalist budget | Settled: 256 (384 measured identical, +65 s) | BASELINE (#189) | +| Local Sobol cloud (points/radius) | Settled: 16 points, radius 2 (32/r3 added no quality) | BASELINE (#189) | +| Scout sequence (sobol/halton/random) | Settled: all inside noise; sobol default | BASELINE (schema 29) | +| Wide-vs-deep passes | Settled: 1 wide pass matches 2 narrow passes at 45% fewer evals | BASELINE (schema 29) | +| Corpus-size factor | Deferred with promotion bar; sweep stays as standing measurement | ADR 0022 | +| Fitting methods, selection rules | Diagnostics only, recorded per comparison run | ADR 0021 | +| Chunking (chunkTokens) | Open: sweep via `vp run bench:retrieval:chunking` | — | `bench:retrieval` aliases `bench:retrieval:validate`. Every profile measures the same physical rankings and retrieval variants; profiles only control matrix size, holdout coverage, and expensive @@ -164,8 +163,7 @@ Limit an exploratory run with comma-separated environment variables: $env:PIX_BENCH_REPOS = "fd" $env:PIX_BENCH_MODELS = "Xenova/all-MiniLM-L6-v2" $env:PIX_BENCH_OPTIMIZATION_PROFILE = "search-priority" -$env:PIX_BENCH_ROUTER_STRATEGY = "proxy-promotion" -$env:PIX_BENCH_SCOUT_SEQUENCE = "halton" +$env:PIX_BENCH_SCOUT_SEQUENCE = "sobol" vp run bench:retrieval:validate ``` @@ -222,15 +220,14 @@ coordinates. Each merged coordinate retains the complete router result, search d timestamp, and source timing record. `artifactSerializationDurationMs` records the preflight JSON serialization time for each source artifact. -The router search defaults to `halving-funnel`. It proxy-scores 512 broad scouts, keeps 32 survivors, +The router search is the `halving-funnel`: it proxy-scores 512 broad scouts, keeps 32 survivors, proxy-scores 16 radius-2 Sobol points per survivor, and fully evaluates 256 finalists plus the static -base seeds once. Set `PIX_BENCH_ROUTER_STRATEGY=proxy-promotion` to run the slower coordinate-beam -control. All strategies use -the same candidate evaluator and native worker queue. Their final archive selection is +base seeds once. The funnel and the static weight searches share the same candidate evaluator and +native worker queue. The final archive selection is objective-specific, so Direct and Reranker rows are real comparisons rather than repeated labels. -The global scouts default to a deterministic Halton sequence. Set `PIX_BENCH_SCOUT_SEQUENCE` to -`sobol` or `random` to seed the beam differently; comparisons must keep the strategy, scout count, +The global scouts default to the deterministic Sobol sequence. Set `PIX_BENCH_SCOUT_SEQUENCE` to +`halton` or `random` to seed the funnel differently; comparisons must keep the scout count, seeds, folds, and objectives identical. The Jina code model cannot embed Effect's longest 7,103-token AST chunk on the tested DML GPU even as diff --git a/benchmarks/retrieval/evaluation/report.ts b/benchmarks/retrieval/evaluation/report.ts index 36c494b..bdb7925 100644 --- a/benchmarks/retrieval/evaluation/report.ts +++ b/benchmarks/retrieval/evaluation/report.ts @@ -211,20 +211,9 @@ const renderPromotionEvidence = (artifact: BenchmarkArtifact): readonly string[] /** Render the report header with run metadata and the recorded search configuration. */ const renderOverview = (artifact: BenchmarkArtifact): readonly string[] => { - const refinement = - artifact.searchStrategy.kind === "halving-funnel" - ? [ - `Funnel refinement: keep ${artifact.searchStrategy.spreadSurvivors} proxy-scored spread survivors, expand ${artifact.localCloudPoints} Sobol points per survivor within +/- ${artifact.localCloudRadiusLevels} level(s), then fully evaluate ${artifact.searchStrategy.finalists} finalists plus base seeds once.`, - ] - : [ - `Beam refinement: beam width ${artifact.searchStrategy.beamWidth}, ${artifact.searchStrategy.coordinatePasses} alternating coordinate passes.`, - ...(artifact.localCloudPoints > 0 - ? [ - `Local refinement: ${artifact.localCloudPoints} deterministic Sobol cloud points per elite within +/- ${artifact.localCloudRadiusLevels} level(s) around the final beam.`, - ] - : []), - `Cheap pre-scoring: candidates first score on a deterministic ${artifact.searchStrategy.proxySampleFraction * 100}% proxy sample with a minimum of ${artifact.searchStrategy.proxyMinimumSamples}; the promotion factor is ${artifact.searchStrategy.proxyPromotionFactor}x.`, - ] + const refinement = [ + `Funnel refinement: keep ${artifact.searchStrategy.spreadSurvivors} proxy-scored spread survivors, expand ${artifact.localCloudPoints} Sobol points per survivor within +/- ${artifact.localCloudRadiusLevels} level(s), then fully evaluate ${artifact.searchStrategy.finalists} finalists plus base seeds once.`, + ] return [ "# Retrieval Quality Benchmark", "", @@ -643,7 +632,6 @@ const renderRouterSections = (artifact: BenchmarkArtifact): readonly string[] => ? "eligible" : "no-eligible-candidate", ), - rows[0]?.searchBaseline.algorithm ?? "unknown", percent(weightedAverage(rows, (row) => row.productionValidation.recallAt5)), percent(weightedAverage(rows, (row) => row.validation.recallAt5)), percent(weightedAverage(rows, (row) => row.productionValidation.recallAt10)), @@ -654,8 +642,6 @@ const renderRouterSections = (artifact: BenchmarkArtifact): readonly string[] => percent(weightedAverage(rows, (row) => row.validation.recallAt50)), percent(weightedAverage(rows, (row) => row.productionValidation.contextRecallAt4096)), percent(weightedAverage(rows, (row) => row.validation.contextRecallAt4096)), - percent(weightedAverage(rows, (row) => row.searchBaseline.validation.recallAt20)), - percent(weightedAverage(rows, (row) => row.searchBaseline.validation.contextRecallAt4096)), ] }) lines.push( @@ -671,7 +657,6 @@ const renderRouterSections = (artifact: BenchmarkArtifact): readonly string[] => "Objective", "Strategy", "Promotion", - "Search baseline", "Production R@5", "Dynamic R@5", "Production R@10", @@ -682,8 +667,6 @@ const renderRouterSections = (artifact: BenchmarkArtifact): readonly string[] => "Dynamic R@50", "Production Ctx@4k", "Dynamic Ctx@4k", - "Random R@20", - "Random Ctx@4k", ], [ "---", diff --git a/benchmarks/retrieval/evaluation/router-search/config-space.ts b/benchmarks/retrieval/evaluation/router-search/config-space.ts index 55ea240..f562d54 100644 --- a/benchmarks/retrieval/evaluation/router-search/config-space.ts +++ b/benchmarks/retrieval/evaluation/router-search/config-space.ts @@ -11,7 +11,6 @@ import { } from "../../../../src/domain/retrieval.js" import { DEFAULT_HALVING_FUNNEL_STRATEGY, - DEFAULT_PROXY_PROMOTION_STRATEGY, DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS, type QualitySummary, } from "../types.js" @@ -23,19 +22,19 @@ export const DYNAMIC_BASE_LEVELS = Array.from({ length: 10 }, (_, index) => (ind export const INFLUENCE_LEVELS = Array.from({ length: 11 }, (_, index) => index / 10) export const SIGNED_FINE_LEVELS = Array.from({ length: 21 }, (_, index) => (index - 10) / 10) export const SEARCH_CANDIDATE_DEPTH = DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.candidateDepth -export const SEARCH_BEAM_WIDTH = DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.beamWidth -export const SEARCH_PASSES = DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.coordinatePasses +/** Static weight-search beam width used inside the funnel's base weight search. */ +export const SEARCH_BEAM_WIDTH = 6 +/** Static weight-search coordinate passes used inside the funnel's base weight search. */ +export const SEARCH_PASSES = 2 export const SEARCH_GLOBAL_SCOUTS = DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.globalScouts export const SEARCH_PROXY_SAMPLE_FRACTION = DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.proxySampleFraction export const SEARCH_PROXY_MINIMUM_SAMPLES = DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.proxyMinimumSamples -export const SEARCH_PROXY_PROMOTION_FACTOR = DEFAULT_PROXY_PROMOTION_STRATEGY.proxyPromotionFactor /** Funnel halving keep factor inherited from the historical successive-halving control. */ export const SEARCH_HALVING_KEEP_FACTOR = 8 export const SEARCH_FUNNEL_SPREAD_SURVIVORS = DEFAULT_HALVING_FUNNEL_STRATEGY.spreadSurvivors export const SEARCH_FUNNEL_FINALISTS = DEFAULT_HALVING_FUNNEL_STRATEGY.finalists -export const RANDOM_SEARCH_SEED = DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.seed || 1 export const normalizeWeights = (weights: ChannelWeights): ChannelWeights => { const max = Math.max(...CHANNELS.map((channel) => weights[channel])) diff --git a/benchmarks/retrieval/evaluation/router-search/rank.ts b/benchmarks/retrieval/evaluation/router-search/rank.ts index 7dabd5d..b32f57f 100644 --- a/benchmarks/retrieval/evaluation/router-search/rank.ts +++ b/benchmarks/retrieval/evaluation/router-search/rank.ts @@ -1,4 +1,4 @@ -import type { +import type { EvidenceRouterParameters as EvidenceRouterConfig, FusionMethod, } from "../../../../src/domain/retrieval.js" @@ -9,19 +9,12 @@ import type { } from "../../execution/candidate-evaluation-pool.js" import type { OptimizationProfile } from "../optimization-profiles.js" import { SCOUT_SEQUENCES, scoutLevelIndex, type ScoutSequenceName } from "../scouts/index.js" -import type { - QualitySummary, - RouterObjective, - RouterSearchDiagnostics, - RouterSearchStrategyName, -} from "../types.js" +import type { QualitySummary, RouterObjective, RouterSearchDiagnostics } from "../types.js" import type { EvidenceSearchSample, WeightCandidate } from "../weight-search.js" import { CHANNELS, SEARCH_BEAM_WIDTH, SEARCH_HALVING_KEEP_FACTOR, - SEARCH_PROXY_PROMOTION_FACTOR, - activeChannelsKey, routerComplexity, routerKey, type RouterCandidate, @@ -106,26 +99,6 @@ export const buildGlobalRouterSeeds = ( ) } -const buildRandomRouterSeeds = ( - baseSeed: EvidenceRouterConfig, - parameters: readonly RouterParameter[], - scoutCount: number, -): readonly EvidenceRouterConfig[] => { - const points = SCOUT_SEQUENCES.random.points(scoutCount, parameters.length) - return Array.from({ length: scoutCount }, (_, pointIndex) => - parameters.reduce( - (config, parameter, parameterIndex) => - parameter.update( - config, - parameter.values[ - scoutLevelIndex(points[pointIndex]![parameterIndex]!, parameter.values.length) - ]!, - ), - baseSeed, - ), - ) -} - /** * Hand-authored corner hypotheses: every parameter at its lowest level except one at its highest * (one per parameter), plus the all-minimum and all-maximum corners. @@ -176,8 +149,7 @@ export interface MutableRouterSearchTimings { preparationMs: number candidatePoolInitializationMs: number baseWeightSearchMs: number - randomSearchMs: number - beamSearchMs: number + funnelSearchMs: number candidatePreparationMs: number candidateEvaluationMs: number candidateSelectionMs: number @@ -223,15 +195,12 @@ const selectSuccessiveHalvingCandidates = ( ): readonly RouterCandidate[] => [...candidates].sort(compareSuccessiveHalvingCandidates).slice(0, limit) -/** Behavioral contract separating the router search strategies. */ +/** Behavioral contract of the funnel search over the shared candidate evaluator. */ export interface RouterSearchMode { - readonly name: RouterSearchStrategyName /** Cheap-score survivors promoted per final beam slot. */ readonly expansionFactor: number /** Whether proxy-vs-full rank agreement diagnostics are recorded. */ readonly recordsProxyAgreement: boolean - /** Whether the deterministic random-scout baseline comparison runs. */ - readonly runsRandomBaseline: boolean /** Whether the beam starts from the optimization-profile configuration too. */ readonly includesProfileSeed: boolean /** Deterministic ordering over scored static weight candidates. */ @@ -263,39 +232,9 @@ export interface RouterSearchMode { ): readonly RouterCandidate[] } -export const PROXY_PROMOTION_MODE: RouterSearchMode = { - name: "proxy-promotion", - expansionFactor: SEARCH_PROXY_PROMOTION_FACTOR, - recordsProxyAgreement: true, - runsRandomBaseline: true, - includesProfileSeed: true, - compareStatic: (left, right, objective, baseline, profile) => - compareObjectiveQuality(left.quality, right.quality, objective, baseline, profile) || - activeChannelsKey(left.weights).localeCompare(activeChannelsKey(right.weights)), - promoteProxy: (context, candidates, limit) => { - const promoted = selectObjectiveCandidates( - candidates, - limit * SEARCH_PROXY_PROMOTION_FACTOR, - context.proxyBaseline, - context.profile, - ) - return { - promoted, - agreementKeys: promoted.map((candidate) => routerKey(candidate.config)), - } - }, - orderRanked: (candidates) => candidates, - selectElites: (context, candidates) => - selectObjectiveCandidates(candidates, SEARCH_BEAM_WIDTH, context.baseline, context.profile), - selectBeam: (context, candidates, limit) => - selectObjectiveCandidates(candidates, limit, context.baseline, context.profile), -} - -const HALVING_FUNNEL_MODE: RouterSearchMode = { - name: "halving-funnel", +export const HALVING_FUNNEL_MODE: RouterSearchMode = { expansionFactor: SEARCH_HALVING_KEEP_FACTOR, recordsProxyAgreement: false, - runsRandomBaseline: false, includesProfileSeed: false, compareStatic: (left, right) => compareSuccessiveHalvingQuality(left.quality, right.quality), promoteProxy: (_context, candidates, limit) => ({ @@ -308,47 +247,6 @@ const HALVING_FUNNEL_MODE: RouterSearchMode = { selectBeam: (_context, candidates, limit) => candidates.slice(0, limit), } -const ROUTER_SEARCH_MODES: Readonly> = { - "proxy-promotion": PROXY_PROMOTION_MODE, - "halving-funnel": HALVING_FUNNEL_MODE, -} - -export const resolveRouterSearchMode = (name: RouterSearchStrategyName): RouterSearchMode => - ROUTER_SEARCH_MODES[name] - -export const selectRandomRouter = async ( - samples: readonly EvidenceSearchSample[], - baseSeed: EvidenceRouterConfig, - parameters: readonly RouterParameter[], - pool: CandidateEvaluationPool, - baseline: QualitySummary, - profile: OptimizationProfile, - stats: SearchEvaluationStats, - scoutCount: number, -): Promise<{ readonly candidate: RouterCandidate; readonly candidates: number }> => { - const configs = buildRandomRouterSeeds(baseSeed, parameters, scoutCount) - const candidatePreparationStartedAt = performance.now() - const evaluationCandidates = configs.map((config) => routerEvaluationCandidate(samples, config)) - stats.timings.candidatePreparationMs += performance.now() - candidatePreparationStartedAt - const candidateEvaluationStartedAt = performance.now() - const qualities = await pool.evaluate(evaluationCandidates) - stats.timings.candidateEvaluationMs += performance.now() - candidateEvaluationStartedAt - const candidates: RouterCandidate[] = [] - for (let index = 0; index < configs.length; index++) { - const config = configs[index] - const quality = qualities[index] - if (config === undefined || quality === undefined) - throw new Error("Candidate evaluation returned an incomplete random router result") - candidates.push({ config, quality }) - } - const candidate = [...candidates].sort((left, right) => - compareRouterCandidates(left, right, "reranker-top20", baseline, profile), - )[0] - if (candidate === undefined) - throw new Error("Candidate evaluation produced no random router candidate") - return { candidate, candidates: candidates.length } -} - export const selectObjectiveCandidates = ( candidates: readonly RouterCandidate[], limit: number, diff --git a/benchmarks/retrieval/evaluation/scouts/index.ts b/benchmarks/retrieval/evaluation/scouts/index.ts index 2a0137d..e22d369 100644 --- a/benchmarks/retrieval/evaluation/scouts/index.ts +++ b/benchmarks/retrieval/evaluation/scouts/index.ts @@ -16,7 +16,7 @@ export const SCOUT_SEQUENCES: Readonly> } /** Sequence used unless `PIX_BENCH_SCOUT_SEQUENCE` requests another one. */ -export const DEFAULT_SCOUT_SEQUENCE: ScoutSequenceName = "halton" +export const DEFAULT_SCOUT_SEQUENCE: ScoutSequenceName = "sobol" /** Human-readable construction of one scout sequence for reports and artifacts. */ export const describeScoutSequence = (name: ScoutSequenceName): string => diff --git a/benchmarks/retrieval/evaluation/subsample-evidence.ts b/benchmarks/retrieval/evaluation/subsample-evidence.ts index af382c0..b1cf4b4 100644 --- a/benchmarks/retrieval/evaluation/subsample-evidence.ts +++ b/benchmarks/retrieval/evaluation/subsample-evidence.ts @@ -72,6 +72,7 @@ interface SubSampleContext { const prepareSubSampleContext = ( repositoryId: string, model: string, + maxTokensOverride?: number, ): Effect.Effect => Effect.gen(function* () { const manifests = yield* loadCorpusManifests() @@ -81,7 +82,7 @@ const prepareSubSampleContext = ( const resolved = yield* resolveModelContext(model) const repositoryPath = yield* prepareRepository(manifest) const corpus = yield* prepareCorpus(repositoryPath, manifest, { - maxTokens: resolved.maxTokens, + maxTokens: maxTokensOverride ?? resolved.maxTokens, overlapLines: DEFAULT_CONFIG.overlapLines, countTokens: resolved.embedder.countTokens, onDiagnostic: () => Effect.void, @@ -92,7 +93,7 @@ const prepareSubSampleContext = ( dims: resolved.info.dims, dtype: resolved.info.defaultDtype, embedder: resolved.embedder, - maxTokens: resolved.maxTokens, + maxTokens: maxTokensOverride ?? resolved.maxTokens, corpus, } }) @@ -388,3 +389,80 @@ export const runRealRouterModelComparison = ( validationSamples: validation.length, } }) + +/** Per-chunk-token-size sweep result row. */ +export interface ChunkingSweepRow { + readonly chunkTokens: number + readonly chunks: number + readonly ndcgAt20: number + readonly standardError: number + readonly proxyEvaluations: number + readonly fullEvaluations: number +} + +/** Parse a comma-separated chunk-token list for the sweep. */ +export const resolveChunkingSizes = (requested: string | undefined): readonly number[] => { + if (requested === undefined) return [256, 384, 512] + return requested + .split(",") + .map((entry) => Number(entry.trim())) + .filter((size) => Number.isFinite(size) && size > 0) +} + +/** + * Run the real search protocol over the same pinned corpus re-chunked at several token budgets. + * Smaller chunks produce more, finer-grained candidates; larger chunks pack more context per hit. + */ +export const runChunkingSweep = ( + repositoryId: string, + model: string, + sizes: readonly number[] = resolveChunkingSizes(process.env.PIX_BENCH_CHUNK_TOKENS), +): Effect.Effect => + Effect.gen(function* () { + const rows: ChunkingSweepRow[] = [] + for (const chunkTokens of [...sizes].sort((left, right) => left - right)) { + const context = yield* prepareSubSampleContext(repositoryId, model, chunkTokens) + const goldByQuestion = goldTargetsByQuestion(context.manifest, context.corpus) + const fullPlan = planCorpusSizeSubSamples(goldByQuestion, context.corpus.chunks.length, [ + Number.POSITIVE_INFINITY, + ])[0]! + const samples = yield* buildSubSampleSamples(context, fullPlan) + const samplesByModel = new Map([ + [context.model, samples], + ]) + const search = yield* runBenchmarkSearch( + { + groupedFolds: 3, + repositoryHoldouts: false, + fusionMethods: ["dbsf"], + routerFusionMethods: ["dbsf"], + }, + samplesByModel, + "grouped-3-fold", + OPTIMIZATION_PROFILES["search-priority"], + false, + { + workerCount: Math.min(resolveWorkerCount(), getDefaultWorkerCount()), + fallbackToSerial: false, + }, + ) + const router = search.recommendedEvidenceRouters.find( + (row) => row.fusion === "dbsf" && row.objective === "direct", + ) + if (router === undefined) + throw new Error(`No dbsf/direct router recommendation at ${chunkTokens} tokens`) + const scored = evaluateRouterConfigOnSamples(samples, router.config) + rows.push({ + chunkTokens, + chunks: context.corpus.chunks.length, + ndcgAt20: scored.mean, + standardError: scored.standardError, + proxyEvaluations: router.proxyEvaluations, + fullEvaluations: router.fullEvaluations, + }) + reportBenchmarkProgress( + `chunkTokens ${chunkTokens}: ${context.corpus.chunks.length} chunks, ndcg@20 ${scored.mean.toFixed(4)} ± ${scored.standardError.toFixed(4)}`, + ) + } + return rows + }) diff --git a/benchmarks/retrieval/evaluation/types.ts b/benchmarks/retrieval/evaluation/types.ts index cb6348c..a09333c 100644 --- a/benchmarks/retrieval/evaluation/types.ts +++ b/benchmarks/retrieval/evaluation/types.ts @@ -1,4 +1,4 @@ -import type { +import type { EvidenceRouterConfig, FusionMethod as ProductionFusionMethod, } from "../../../src/domain/retrieval.js" @@ -18,11 +18,9 @@ export const ROUTER_OBJECTIVES = [ ] as const export type RouterObjective = (typeof ROUTER_OBJECTIVES)[number] -/** Compute budget shared by both router search strategies. */ +/** Compute budget shared by the router search funnel. */ interface RouterSearchBudget { globalScouts: 64 - beamWidth: 6 - coordinatePasses: 2 candidateDepth: 200 proxySampleFraction: 0.25 proxyMinimumSamples: 32 @@ -30,25 +28,11 @@ interface RouterSearchBudget { const ROUTER_SEARCH_BUDGET: RouterSearchBudget = { globalScouts: 64, - beamWidth: 6, - coordinatePasses: 2, candidateDepth: 200, proxySampleFraction: 0.25, proxyMinimumSamples: 32, } -/** Proxy promotion strategy: cheap pre-scoring, then full-quality promotion of survivors. */ -export interface ProxyPromotionStrategy extends RouterSearchBudget { - readonly kind: "proxy-promotion" - readonly algorithm: string - readonly proxyPromotionFactor: 8 - readonly objectives: typeof ROUTER_OBJECTIVES - readonly guardrailTolerance: 0.01 - readonly seed: 0 - readonly normalization: "per-channel-max-weight" - readonly tieBreaking: "guardrails>objective>complexity>stable-key" -} - /** * Halving funnel strategy: broad cheap waves with late full-fidelity evaluation. Wave 1 spreads * `globalScouts` scouts scored on the proxy sample only, wave 2 samples Sobol clouds around the @@ -59,32 +43,19 @@ export interface HalvingFunnelStrategy extends RouterSearchBudget { readonly algorithm: string readonly spreadSurvivors: 32 readonly finalists: 256 + readonly objectives: typeof ROUTER_OBJECTIVES + readonly guardrailTolerance: 0.01 + readonly seed: 0 + readonly normalization: "per-channel-max-weight" + readonly tieBreaking: "guardrails>objective>complexity>stable-key" } /** Versioned evidence-router search strategy recorded in benchmark artifacts. */ -export type RouterSearchStrategy = ProxyPromotionStrategy | HalvingFunnelStrategy +export type RouterSearchStrategy = HalvingFunnelStrategy -export const ROUTER_SEARCH_STRATEGY_NAMES = ["proxy-promotion", "halving-funnel"] as const - -export type RouterSearchStrategyName = (typeof ROUTER_SEARCH_STRATEGY_NAMES)[number] - -const BEAM_ALGORITHM_PREFIX = "global-scout-elitist-beam" const FUNNEL_ALGORITHM_PREFIX = "global-scout-funnel" -export const DEFAULT_ROUTER_SEARCH_STRATEGY: RouterSearchStrategyName = "halving-funnel" - -/** Default proxy promotion parameters, recorded with Halton scouts unless overridden. */ -export const DEFAULT_PROXY_PROMOTION_STRATEGY: ProxyPromotionStrategy = { - kind: "proxy-promotion", - algorithm: `halton-${BEAM_ALGORITHM_PREFIX}-proxy-promotion`, - ...ROUTER_SEARCH_BUDGET, - proxyPromotionFactor: 8, - objectives: ROUTER_OBJECTIVES, - guardrailTolerance: 0.01, - seed: 0, - normalization: "per-channel-max-weight", - tieBreaking: "guardrails>objective>complexity>stable-key", -} +export const DEFAULT_ROUTER_SEARCH_STRATEGY = "halving-funnel" as const /** Default fidelity funnel: broad proxy wave, local proxy cloud, then 256 diverse finalists. */ export const DEFAULT_HALVING_FUNNEL_STRATEGY: HalvingFunnelStrategy = { @@ -93,24 +64,23 @@ export const DEFAULT_HALVING_FUNNEL_STRATEGY: HalvingFunnelStrategy = { ...ROUTER_SEARCH_BUDGET, spreadSurvivors: 32, finalists: 256, + objectives: ROUTER_OBJECTIVES, + guardrailTolerance: 0.01, + seed: 0, + normalization: "per-channel-max-weight", + tieBreaking: "guardrails>objective>complexity>stable-key", } -/** Build the recorded search strategy for one scout sequence and strategy name. */ +/** Build the recorded search strategy for one scout sequence. */ export const routerSearchStrategyFor = ( scoutSequence: ScoutSequenceName, - name: RouterSearchStrategyName, -): RouterSearchStrategy => { - const base = - name === "proxy-promotion" ? DEFAULT_PROXY_PROMOTION_STRATEGY : DEFAULT_HALVING_FUNNEL_STRATEGY - const algorithm = - name === "halving-funnel" - ? `${scoutSequence}-${FUNNEL_ALGORITHM_PREFIX}` - : `${scoutSequence}-${BEAM_ALGORITHM_PREFIX}-${name}` - return { ...base, algorithm } -} +): RouterSearchStrategy => ({ + ...DEFAULT_HALVING_FUNNEL_STRATEGY, + algorithm: `${scoutSequence}-${FUNNEL_ALGORITHM_PREFIX}`, +}) -/** Shared search-space and legacy random-baseline parameters. */ -export const DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS = DEFAULT_PROXY_PROMOTION_STRATEGY +/** Shared search-space parameters. */ +export const DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS = DEFAULT_HALVING_FUNNEL_STRATEGY /** Runtime/coverage trade-off selected for one benchmark invocation. */ export type BenchmarkProfile = "smoke" | "develop" | "validate" | "full" @@ -327,10 +297,8 @@ export interface RouterSearchTimings { readonly candidatePoolInitializationMs: number /** Time spent selecting the initial static/base weight seeds. */ readonly baseWeightSearchMs: number - /** Time spent evaluating the random scout baseline. */ - readonly randomSearchMs: number - /** Time spent in initial and coordinate beam rounds. */ - readonly beamSearchMs: number + /** Time spent in the halving funnel waves. */ + readonly funnelSearchMs: number /** Time spent converting router configs into evaluator candidates. */ readonly candidatePreparationMs: number /** Wall time waiting for candidate evaluation results. */ @@ -357,15 +325,6 @@ export interface RouterSearchDiagnostics { readonly timings: RouterSearchTimings } -/** Holdout comparison against a deterministic random-search baseline. */ -export interface SearchBaselineComparison { - readonly algorithm: "random-scout" | "not-run" - readonly seed: number - readonly candidates: number - readonly development: QualitySummary - readonly validation: QualitySummary -} - /** Explicit separation between candidate selection, holdouts, and final promotion evidence. */ export interface ValidationProtocol { readonly selection: "development-only" @@ -401,7 +360,6 @@ export interface EvidenceRouterSearchResult { readonly proxyEvaluations: number readonly fullEvaluations: number readonly searchDiagnostics: RouterSearchDiagnostics - readonly searchBaseline: SearchBaselineComparison readonly holdoutBreakdown: readonly HoldoutQuality[] } @@ -450,10 +408,6 @@ export interface BenchmarkArtifact { readonly scoutSequence: ScoutSequenceName /** Whether hand-authored corner hypothesis seeds joined the beam starting points. */ readonly seedHypotheses: boolean - /** Beam width schedule across coordinate rounds. */ - readonly beamSchedule: "fixed" | "decaying" - /** Coordinate refinement rounds executed by this run's router search. */ - readonly coordinatePasses: number /** Global scout count used to seed the beam. */ readonly globalScouts: number /** Sobol points sampled per elite in the local cloud around the final beam (0 disables). */ diff --git a/benchmarks/retrieval/evaluation/weight-search.ts b/benchmarks/retrieval/evaluation/weight-search.ts index 576ac89..3a8e5db 100644 --- a/benchmarks/retrieval/evaluation/weight-search.ts +++ b/benchmarks/retrieval/evaluation/weight-search.ts @@ -34,7 +34,6 @@ import { prepareFusion, type PreparedFusionEvaluator } from "./prepared-fusion.j import { buildGuardrailBlockers } from "./promotion-evidence.js" import { CHANNELS, - RANDOM_SEARCH_SEED, SEARCH_BEAM_WIDTH, SEARCH_CANDIDATE_DEPTH, SEARCH_PASSES, @@ -52,7 +51,6 @@ import { type RouterCandidate, } from "./router-search/config-space.js" import { runHalvingFunnel } from "./router-search/funnel.js" -import { buildLocalCloudConfigs } from "./router-search/local-cloud.js" import { OBJECTIVE_GUARDRAILS, compareObjectiveQuality, @@ -60,14 +58,8 @@ import { unweightedProfile, } from "./router-search/objectives.js" import { - PROXY_PROMOTION_MODE, - beamWidthForRound, - buildGlobalRouterSeeds, - buildHypothesisRouterSeeds, buildSearchDiagnostics, - rankRouterCandidates, - resolveRouterSearchMode, - selectRandomRouter, + HALVING_FUNNEL_MODE, storeQualityResults, type RouterSearchContext, type RouterSearchMode, @@ -92,8 +84,6 @@ import { type RecommendedFusionWeights, type RouterSearchDiagnostics, type RouterObjective, - type RouterSearchStrategyName, - type SearchBaselineComparison, type SelectionStability, } from "./types.js" @@ -122,16 +112,10 @@ interface SearchOptions extends CandidateEvaluationPoolOptions { readonly signal?: AbortSignal /** Shared candidate scheduler used to interleave multiple router searches. */ readonly evaluationQueue?: CandidateEvaluationQueue - /** Benchmark-only router search algorithm; defaults to the current proxy promotion mode. */ - readonly routerSearchStrategy?: RouterSearchStrategyName - /** Benchmark-only global-scout sequence; defaults to Halton. */ + /** Benchmark-only global-scout sequence; defaults to Sobol. */ readonly scoutSequence?: ScoutSequenceName /** Add hand-authored corner hypothesis seeds to the beam starting points. */ readonly seedHypotheses?: boolean - /** Start the coordinate rounds with a wider beam that halves towards the target width. */ - readonly beamSchedule?: "fixed" | "decaying" - /** Override the number of coordinate refinement rounds (default: strategy setting). */ - readonly coordinatePasses?: number /** Override the global scout count (default: strategy setting). */ readonly globalScouts?: number /** Sobol points sampled per elite in a local cloud around the final beam (0 disables). */ @@ -792,8 +776,7 @@ const prepareRouterSearch = ( preparationMs: performance.now() - preparationStartedAt, candidatePoolInitializationMs: 0, baseWeightSearchMs: 0, - randomSearchMs: 0, - beamSearchMs: 0, + funnelSearchMs: 0, candidatePreparationMs: 0, candidateEvaluationMs: 0, candidateSelectionMs: 0, @@ -822,8 +805,6 @@ const selectBestEvidenceRouter = async ( mode: RouterSearchMode, scoutSequence: ScoutSequenceName, seedHypotheses: boolean = false, - beamSchedule: "fixed" | "decaying" = "fixed", - coordinatePasses: number = SEARCH_PASSES, globalScouts: number = DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.globalScouts, localCloudPoints: number = 0, localCloudRadiusLevels: number = 1, @@ -833,8 +814,6 @@ const selectBestEvidenceRouter = async ( readonly proxyEvaluations: number readonly fullEvaluations: number readonly searchDiagnostics: RouterSearchDiagnostics - readonly randomCandidate: RouterCandidate - readonly randomCandidates: number }> => { const { samples, @@ -875,85 +854,18 @@ const selectBestEvidenceRouter = async ( if (mode.includesProfileSeed) baseSeeds.unshift(profileSeed) stats.timings.baseWeightSearchMs += performance.now() - baseSearchStartedAt const parameters = routerParameters() - const randomBaseSeed = baseSeeds[0] - if (randomBaseSeed === undefined) throw new Error("Evidence router search has no base seed") - const randomSearch = mode.runsRandomBaseline - ? await (async () => { - const randomSearchStartedAt = performance.now() - const result = await selectRandomRouter( - evidenceSamples, - randomBaseSeed, - parameters, - fullPool, - productionQuality, - profile, - stats, - globalScouts, - ) - stats.timings.randomSearchMs += performance.now() - randomSearchStartedAt - return result - })() - : undefined const beamSearchStartedAt = performance.now() - let beam: readonly RouterCandidate[] - if (mode.name === "halving-funnel") { - beam = await runHalvingFunnel( - searchContext, - baseSeeds, - parameters, - scoutSequence, - globalScouts, - seedHypotheses, - localCloudPoints, - localCloudRadiusLevels, - ) - } else { - const seedConfigs = [ - ...baseSeeds, - ...buildGlobalRouterSeeds(baseSeeds, parameters, scoutSequence, globalScouts), - ...(seedHypotheses ? buildHypothesisRouterSeeds(baseSeeds, parameters) : []), - ] - const totalRounds = coordinatePasses + 1 - const roundWidth = (round: number): number => - beamSchedule === "decaying" - ? beamWidthForRound(round, totalRounds, SEARCH_BEAM_WIDTH) - : SEARCH_BEAM_WIDTH - beam = await rankRouterCandidates(searchContext, seedConfigs, roundWidth(0), baseSeeds) - for (let pass = 0; pass < coordinatePasses; pass++) { - const orderedParameters = pass % 2 === 0 ? parameters : [...parameters].reverse() - for (const parameter of orderedParameters) { - beam = await rankRouterCandidates( - searchContext, - [ - // Retain the current beam; the search context also protects full-quality elites. - ...beam.map((candidate) => candidate.config), - ...beam.flatMap((candidate) => - parameter.values.map((value) => parameter.update(candidate.config, value)), - ), - ], - roundWidth(pass + 1), - beam.map((candidate) => candidate.config), - ) - } - } - if (localCloudPoints > 0 && localCloudRadiusLevels > 0) { - const cloudConfigs = buildLocalCloudConfigs( - beam.map((candidate) => candidate.config), - parameters, - localCloudPoints, - localCloudRadiusLevels, - ) - stats.localCloudCandidates += cloudConfigs.length - beam = await rankRouterCandidates( - searchContext, - cloudConfigs, - roundWidth(coordinatePasses), - // Retain the pre-cloud elites; the context also protects full-quality elites. - beam.map((candidate) => candidate.config), - ) - } - } - stats.timings.beamSearchMs += performance.now() - beamSearchStartedAt + const beam = await runHalvingFunnel( + searchContext, + baseSeeds, + parameters, + scoutSequence, + globalScouts, + seedHypotheses, + localCloudPoints, + localCloudRadiusLevels, + ) + stats.timings.funnelSearchMs += performance.now() - beamSearchStartedAt const fallback = beam[0] if (fallback === undefined) throw new Error("Evidence router search produced no candidate") const candidates = [...searchContext.archive.values()] @@ -978,13 +890,10 @@ const selectBestEvidenceRouter = async ( promotionStatus, selectionStability, })) - const randomCandidate = randomSearch?.candidate ?? fallback return { selections, productionQuality, searchDiagnostics: buildSearchDiagnostics(parameters, stats), - randomCandidate, - randomCandidates: randomSearch?.candidates ?? 0, proxyEvaluations: stats.proxyEvaluations, fullEvaluations: stats.fullEvaluations, } @@ -1024,7 +933,7 @@ const optimizeFusionWeightsWithPool = async ( ): Promise => { const selected = await selectBestWeights( pool, - PROXY_PROMOTION_MODE, + HALVING_FUNNEL_MODE, "reranker-top20", undefined, profile, @@ -1168,11 +1077,9 @@ const withEvidencePools = async ( profile, fullPool, proxyPool, - resolveRouterSearchMode(options.routerSearchStrategy ?? "proxy-promotion"), + HALVING_FUNNEL_MODE, options.scoutSequence ?? DEFAULT_SCOUT_SEQUENCE, options.seedHypotheses ?? false, - options.beamSchedule ?? "fixed", - options.coordinatePasses ?? DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.coordinatePasses, options.globalScouts ?? DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.globalScouts, options.localCloudPoints ?? 0, options.localCloudRadiusLevels ?? 1, @@ -1192,51 +1099,33 @@ interface StaticRouterSelection { readonly staticSelection: Awaited> } -const selectStaticWeightsForSelections = async ( +const selectStaticWeightsForSelections = ( selections: readonly EvidenceRouterSelection[], fullPool: CandidateEvaluationPool, productionQuality: QualitySummary, profile: OptimizationProfile, mode: RouterSearchMode, ): Promise => { - if (mode.name !== "proxy-promotion") { - const staticSelection = await selectBestWeights( - fullPool, - mode, - "reranker-top20", - undefined, - profile, - ) - return selections.map((selection) => ({ selection, staticSelection })) - } - const selected: StaticRouterSelection[] = [] - for (const selection of selections) { - selected.push({ + const staticSelection = selectBestWeights(fullPool, mode, "reranker-top20", undefined, profile) + return Promise.all( + selections.map(async (selection) => ({ selection, - staticSelection: await selectStaticWeights( - fullPool, - mode, - selection.objective, - productionQuality, - profile, - ), - }) - } - return selected + staticSelection: await staticSelection, + })), + ) } const selectStaticWeightsForSearch = ( dynamicSelection: Awaited>, fullPool: CandidateEvaluationPool, profile: OptimizationProfile, - options: SearchOptions, ): Promise => selectStaticWeightsForSelections( dynamicSelection.selections, fullPool, dynamicSelection.productionQuality, profile, - resolveRouterSearchMode(options.routerSearchStrategy ?? "proxy-promotion"), + HALVING_FUNNEL_MODE, ) export const optimizeEvidenceRouter = async ( @@ -1251,42 +1140,7 @@ export const optimizeEvidenceRouter = async ( ): Promise => withEvidencePools(development, fusion, profile, options, async (dynamicSelection, fullPool) => { const productionValidation = summarizeProductionRouter(validation, profile) - const validationEvidence = prepareEvidenceSamples(validation) - const randomBaseline: SearchBaselineComparison = - dynamicSelection.randomCandidates === 0 - ? { - algorithm: "not-run", - seed: RANDOM_SEARCH_SEED, - candidates: 0, - development: summarizeEvidenceRouter( - [], - dynamicSelection.randomCandidate.config, - fusion, - ), - validation: summarizeEvidenceRouter( - [], - dynamicSelection.randomCandidate.config, - fusion, - ), - } - : { - algorithm: "random-scout", - seed: RANDOM_SEARCH_SEED, - candidates: dynamicSelection.randomCandidates, - development: dynamicSelection.randomCandidate.quality, - validation: summarizeEvidenceRouter( - validationEvidence, - dynamicSelection.randomCandidate.config, - fusion, - profile, - ), - } - const staticSelections = await selectStaticWeightsForSearch( - dynamicSelection, - fullPool, - profile, - options, - ) + const staticSelections = await selectStaticWeightsForSearch(dynamicSelection, fullPool, profile) const results: EvidenceRouterSearchResult[] = staticSelections.map( ({ selection, staticSelection }) => { const holdoutBreakdown = buildHoldoutBreakdown( @@ -1316,7 +1170,7 @@ export const optimizeEvidenceRouter = async ( staticValidation: summarize(validation, staticSelection.weights, fusion, profile), development: selection.quality, validation: summarizeEvidenceRouter( - validationEvidence, + prepareEvidenceSamples(validation), selection.config, fusion, profile, @@ -1329,7 +1183,6 @@ export const optimizeEvidenceRouter = async ( proxyEvaluations: dynamicSelection.proxyEvaluations, fullEvaluations: dynamicSelection.fullEvaluations, searchDiagnostics: dynamicSelection.searchDiagnostics, - searchBaseline: randomBaseline, holdoutBreakdown, } }, @@ -1347,7 +1200,7 @@ const fitRecommendedFusionWeightsWithPool = async ( ): Promise => { const selected = await selectBestWeights( pool, - PROXY_PROMOTION_MODE, + HALVING_FUNNEL_MODE, "reranker-top20", undefined, profile, @@ -1403,12 +1256,7 @@ const fitRecommendedEvidenceRouterWithOptions = ( options: SearchOptions, ): Promise => withEvidencePools(samples, fusion, profile, options, async (dynamicSelection, fullPool) => { - const staticSelections = await selectStaticWeightsForSearch( - dynamicSelection, - fullPool, - profile, - options, - ) + const staticSelections = await selectStaticWeightsForSearch(dynamicSelection, fullPool, profile) const results: RecommendedEvidenceRouter[] = staticSelections.map( ({ selection, staticSelection }) => ({ model, diff --git a/benchmarks/retrieval/run-config.ts b/benchmarks/retrieval/run-config.ts index accf2f5..3b13881 100644 --- a/benchmarks/retrieval/run-config.ts +++ b/benchmarks/retrieval/run-config.ts @@ -1,39 +1,14 @@ import { resolveScoutSequence, type ScoutSequenceName } from "./evaluation/scouts/index.js" -import { - DEFAULT_ROUTER_SEARCH_STRATEGY, - DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS, - ROUTER_SEARCH_STRATEGY_NAMES, - type RouterSearchStrategyName, -} from "./evaluation/types.js" /** Benchmark-only search knobs resolved once per run from the environment. */ export interface SearchKnobs { - readonly routerSearchStrategy: RouterSearchStrategyName readonly scoutSequence: ScoutSequenceName readonly seedHypotheses: boolean - readonly beamSchedule: "fixed" | "decaying" - readonly coordinatePasses: number readonly globalScouts: number readonly localCloudPoints: number readonly localCloudRadiusLevels: number } -const isRouterSearchStrategyName = (requested: string): requested is RouterSearchStrategyName => - ROUTER_SEARCH_STRATEGY_NAMES.some((strategy) => strategy === requested) - -/** Resolve `PIX_BENCH_ROUTER_STRATEGY`, defaulting to proxy promotion. */ -export const resolveRouterSearchStrategy = ( - requested: string | undefined, -): RouterSearchStrategyName => { - if (requested === undefined) return DEFAULT_ROUTER_SEARCH_STRATEGY - if (!isRouterSearchStrategyName(requested)) { - throw new Error( - `Unknown PIX_BENCH_ROUTER_STRATEGY value: ${requested}; expected one of ${ROUTER_SEARCH_STRATEGY_NAMES.join(", ")}`, - ) - } - return requested -} - const parsePositiveInt = (envValue: string | undefined, name: string, fallback: number): number => { if (envValue === undefined) return fallback if (!/^\d+$/.test(envValue)) { @@ -44,23 +19,15 @@ const parsePositiveInt = (envValue: string | undefined, name: string, fallback: /** Resolve all search knobs from one environment; throws with the knob name on bad input. */ export const resolveSearchKnobs = (env: NodeJS.ProcessEnv): SearchKnobs => { - const routerSearchStrategy = resolveRouterSearchStrategy(env.PIX_BENCH_ROUTER_STRATEGY) - const funnel = routerSearchStrategy === "halving-funnel" - const beamSchedule = env.PIX_BENCH_BEAM_SCHEDULE ?? "fixed" - if (beamSchedule !== "fixed" && beamSchedule !== "decaying") { - throw new Error( - `Unknown PIX_BENCH_BEAM_SCHEDULE value: ${beamSchedule}; expected fixed or decaying`, - ) - } const localCloudPoints = parsePositiveInt( env.PIX_BENCH_LOCAL_CLOUD_POINTS, "PIX_BENCH_LOCAL_CLOUD_POINTS", - funnel ? 16 : 0, + 16, ) const localCloudRadiusLevels = parsePositiveInt( env.PIX_BENCH_LOCAL_CLOUD_RADIUS, "PIX_BENCH_LOCAL_CLOUD_RADIUS", - funnel ? 2 : 1, + 2, ) if (localCloudPoints > 0 && localCloudRadiusLevels === 0) { throw new Error( @@ -68,23 +35,12 @@ export const resolveSearchKnobs = (env: NodeJS.ProcessEnv): SearchKnobs => { ) } return { - routerSearchStrategy, scoutSequence: resolveScoutSequence(env.PIX_BENCH_SCOUT_SEQUENCE), seedHypotheses: env.PIX_BENCH_SEED_HYPOTHESES === undefined - ? funnel + ? true : env.PIX_BENCH_SEED_HYPOTHESES === "1" || env.PIX_BENCH_SEED_HYPOTHESES === "true", - beamSchedule, - coordinatePasses: parsePositiveInt( - env.PIX_BENCH_COORDINATE_PASSES, - "PIX_BENCH_COORDINATE_PASSES", - DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.coordinatePasses, - ), - globalScouts: parsePositiveInt( - env.PIX_BENCH_GLOBAL_SCOUTS, - "PIX_BENCH_GLOBAL_SCOUTS", - funnel ? 512 : DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.globalScouts, - ), + globalScouts: parsePositiveInt(env.PIX_BENCH_GLOBAL_SCOUTS, "PIX_BENCH_GLOBAL_SCOUTS", 512), localCloudPoints, localCloudRadiusLevels, } diff --git a/benchmarks/retrieval/runner.ts b/benchmarks/retrieval/runner.ts index 7c5aade..ba7e2ac 100644 --- a/benchmarks/retrieval/runner.ts +++ b/benchmarks/retrieval/runner.ts @@ -26,8 +26,6 @@ import { import { getDefaultWorkerCount, resolveWorkerCount } from "./execution/candidate-evaluation-pool.js" import { resolveSearchKnobs } from "./run-config.js" -export { resolveRouterSearchStrategy } from "./run-config.js" - const CONTEXT_BUDGETS = [2_048, 4_096, 8_192, 16_384] as const const profileConfig = (profile: BenchmarkProfile): BenchmarkSearchConfig => { @@ -172,11 +170,8 @@ export const runRetrievalBenchmark = ( catch: (cause) => (cause instanceof Error ? cause : new Error(String(cause))), }) - const routerSearchStrategy = searchKnobs.routerSearchStrategy const scoutSequence = searchKnobs.scoutSequence const seedHypotheses = searchKnobs.seedHypotheses - const beamSchedule = searchKnobs.beamSchedule - const coordinatePasses = searchKnobs.coordinatePasses const globalScouts = searchKnobs.globalScouts const localCloudPoints = searchKnobs.localCloudPoints const localCloudRadiusLevels = searchKnobs.localCloudRadiusLevels @@ -207,11 +202,8 @@ export const runRetrievalBenchmark = ( serialSearch, { ...searchOptions, - routerSearchStrategy, scoutSequence, seedHypotheses, - beamSchedule, - coordinatePasses, globalScouts, localCloudPoints, localCloudRadiusLevels, @@ -233,8 +225,6 @@ export const runRetrievalBenchmark = ( benchmarkProfile: profile, scoutSequence, seedHypotheses, - beamSchedule, - coordinatePasses, globalScouts, localCloudPoints, localCloudRadiusLevels, @@ -252,7 +242,7 @@ export const runRetrievalBenchmark = ( }, }, generatedAt: new Date().toISOString(), - searchStrategy: routerSearchStrategyFor(scoutSequence, routerSearchStrategy), + searchStrategy: routerSearchStrategyFor(scoutSequence), timings: { totalDurationMs: performance.now() - benchmarkStartedAt, corpusPreparationDurationMs, diff --git a/benchmarks/tests/channels.test.ts b/benchmarks/tests/channels.test.ts index 296901c..f16cc85 100644 --- a/benchmarks/tests/channels.test.ts +++ b/benchmarks/tests/channels.test.ts @@ -609,10 +609,6 @@ describe("retrieval benchmark fixture", () => { expect(Object.keys(result.searchDiagnostics.parameterLevels)).toHaveLength(40) expect(result.searchDiagnostics.proxyFullAgreement).toBeGreaterThanOrEqual(0) expect(result.searchDiagnostics.proxyFullAgreement).toBeLessThanOrEqual(1) - expect(result.searchBaseline.algorithm).toBe("random-scout") - expect(result.searchBaseline.seed).toBe(1) - expect(result.searchBaseline.candidates).toBeGreaterThan(0) - expect(result.searchBaseline.validation.recallAt20).toBeGreaterThanOrEqual(0) expect( result.holdoutBreakdown.map(({ dimension, name }) => `${dimension}:${name}`).sort(), ).toEqual([ @@ -683,7 +679,6 @@ describe("retrieval benchmark fixture", () => { undefined, { workerCount: 0, - routerSearchStrategy: "halving-funnel", globalScouts: 16, seedHypotheses: true, localCloudPoints: 2, @@ -697,7 +692,6 @@ describe("retrieval benchmark fixture", () => { ) expect(result.searchDiagnostics.fullEvaluations).toBeLessThan(384) expect(result.searchDiagnostics.localCloudCandidates).toBeGreaterThan(0) - expect(result.searchBaseline.algorithm).toBe("not-run") }) it("does not treat a guardrail-failing fallback as promotable", () => { diff --git a/benchmarks/tests/chunking-sweep.test.ts b/benchmarks/tests/chunking-sweep.test.ts new file mode 100644 index 0000000..6e41a61 --- /dev/null +++ b/benchmarks/tests/chunking-sweep.test.ts @@ -0,0 +1,40 @@ +import { mkdir, writeFile } from "node:fs/promises" +import path from "node:path" + +import { expect, it } from "@effect/vitest" +import { Effect } from "effect" + +import { runChunkingSweep } from "../retrieval/evaluation/subsample-evidence.js" + +const repositoryId = process.env.PIX_BENCH_CORPUS_SIZE_REPO ?? "t3code" +const model = process.env.PIX_BENCH_MODELS ?? "Xenova/all-MiniLM-L6-v2" + +it.effect("runs the real chunking sweep over multiple token budgets", () => + Effect.gen(function* () { + const rows = yield* runChunkingSweep(repositoryId, model) + + expect(rows.length).toBeGreaterThan(1) + for (const row of rows) { + expect(row.chunks).toBeGreaterThan(0) + expect(Number.isFinite(row.ndcgAt20)).toBe(true) + expect(row.standardError).toBeGreaterThan(0) + } + + const outputDirectory = path.resolve("benchmarks/results") + const outputPath = path.join( + outputDirectory, + `retrieval-chunking-${repositoryId}-${model.replaceAll("/", "_")}.json`, + ) + yield* Effect.tryPromise({ + try: async () => { + await mkdir(outputDirectory, { recursive: true }) + await writeFile( + outputPath, + `${JSON.stringify({ schemaVersion: 1, repositoryId, model, rows }, null, 2)}\n`, + "utf8", + ) + }, + catch: (cause) => new Error(`Could not write chunking sweep ${outputPath}`, { cause }), + }) + }), +) diff --git a/benchmarks/tests/matrix.test.ts b/benchmarks/tests/matrix.test.ts index 445062a..f27add8 100644 --- a/benchmarks/tests/matrix.test.ts +++ b/benchmarks/tests/matrix.test.ts @@ -45,8 +45,7 @@ const searchDiagnostics: RouterSearchDiagnostics = { preparationMs: 1, candidatePoolInitializationMs: 1, baseWeightSearchMs: 1, - randomSearchMs: 0, - beamSearchMs: 1, + funnelSearchMs: 1, candidatePreparationMs: 1, candidateEvaluationMs: 1, candidateSelectionMs: 1, diff --git a/benchmarks/tests/report.test.ts b/benchmarks/tests/report.test.ts index 38d8e86..3f3dd96 100644 --- a/benchmarks/tests/report.test.ts +++ b/benchmarks/tests/report.test.ts @@ -9,8 +9,6 @@ const artifact = { benchmarkProfile: "smoke", scoutSequence: "halton", seedHypotheses: false, - beamSchedule: "fixed", - coordinatePasses: 2, globalScouts: 64, localCloudPoints: 0, localCloudRadiusLevels: 1, @@ -21,7 +19,7 @@ const artifact = { finalTest: { kind: "untouched-grouped-fold", strategy: "grouped-5-fold", fold: "5" }, }, generatedAt: "2026-08-06T00:00:00.000Z", - searchStrategy: routerSearchStrategyFor("halton", "proxy-promotion"), + searchStrategy: routerSearchStrategyFor("halton"), timings: { totalDurationMs: 0, corpusPreparationDurationMs: 0, @@ -97,7 +95,7 @@ it("renders the halving funnel as two cheap waves and one full evaluation", () = seedHypotheses: true, localCloudPoints: 16, localCloudRadiusLevels: 2, - searchStrategy: routerSearchStrategyFor("halton", "halving-funnel"), + searchStrategy: routerSearchStrategyFor("halton"), }) expect(report).toContain("keep 32 proxy-scored spread survivors") diff --git a/benchmarks/tests/retrieval.test.ts b/benchmarks/tests/retrieval.test.ts index f4a1ad6..bcdee7e 100644 --- a/benchmarks/tests/retrieval.test.ts +++ b/benchmarks/tests/retrieval.test.ts @@ -9,7 +9,7 @@ import { routerSearchStrategyFor, type BenchmarkProfile, } from "../retrieval/evaluation/types.js" -import { resolveRouterSearchStrategy, runRetrievalBenchmark } from "../retrieval/runner.js" +import { runRetrievalBenchmark } from "../retrieval/runner.js" const foldQuestions = (prefix: string) => Array.from({ length: 12 }, (_, index) => ({ @@ -111,22 +111,13 @@ const runProfile = (profile: BenchmarkProfile, groupedFolds: number, fusionMetho process.env.PIX_BENCH_SEED_HYPOTHESES === "1" || process.env.PIX_BENCH_SEED_HYPOTHESES === "true", ) - expect(artifact.beamSchedule).toBe(process.env.PIX_BENCH_BEAM_SCHEDULE ?? "fixed") expect(artifact.globalScouts).toBe( process.env.PIX_BENCH_GLOBAL_SCOUTS === undefined ? 64 : Number.parseInt(process.env.PIX_BENCH_GLOBAL_SCOUTS, 10), ) - expect(artifact.coordinatePasses).toBe( - process.env.PIX_BENCH_COORDINATE_PASSES === undefined - ? 2 - : Number.parseInt(process.env.PIX_BENCH_COORDINATE_PASSES, 10), - ) expect(artifact.searchStrategy).toEqual( - routerSearchStrategyFor( - resolveScoutSequence(process.env.PIX_BENCH_SCOUT_SEQUENCE), - resolveRouterSearchStrategy(process.env.PIX_BENCH_ROUTER_STRATEGY), - ), + routerSearchStrategyFor(resolveScoutSequence(process.env.PIX_BENCH_SCOUT_SEQUENCE)), ) expect(artifact.timings.totalDurationMs).toBeGreaterThan(0) expect(Object.values(artifact.timings).every((duration) => duration >= 0)).toBe(true) @@ -223,30 +214,13 @@ it("shuffles intent groups deterministically before assigning folds", () => { expectStratifiedClasses(manifests, assignments) }) -it("resolves the selectable router search strategies", () => { - expect(resolveRouterSearchStrategy(undefined)).toBe("halving-funnel") - expect(resolveRouterSearchStrategy("proxy-promotion")).toBe("proxy-promotion") - expect(routerSearchStrategyFor("halton", "proxy-promotion")).toMatchObject({ - kind: "proxy-promotion", - algorithm: "halton-global-scout-elitist-beam-proxy-promotion", - proxyPromotionFactor: 8, - }) - expect(resolveRouterSearchStrategy("halving-funnel")).toBe("halving-funnel") - expect(routerSearchStrategyFor("sobol", "halving-funnel")).toMatchObject({ +it("resolves the recorded search strategy", () => { + expect(routerSearchStrategyFor("sobol")).toMatchObject({ kind: "halving-funnel", algorithm: "sobol-global-scout-funnel", spreadSurvivors: 32, finalists: 256, }) - const sobolStrategy = routerSearchStrategyFor("sobol", "proxy-promotion") - expect(sobolStrategy.algorithm).toBe("sobol-global-scout-elitist-beam-proxy-promotion") - expect(sobolStrategy.kind === "proxy-promotion" && sobolStrategy.proxyPromotionFactor).toBe(8) - expect(DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.kind).toBe("proxy-promotion") - expect(DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS).not.toHaveProperty("halvingKeepFactor") - expect(() => resolveRouterSearchStrategy("successive-halving")).toThrow( - "Unknown PIX_BENCH_ROUTER_STRATEGY value: successive-halving", - ) - expect(() => resolveRouterSearchStrategy("unknown")).toThrow( - "Unknown PIX_BENCH_ROUTER_STRATEGY value: unknown", - ) + expect(DEFAULT_ROUTER_SEARCH_STRATEGY_PARAMETERS.kind).toBe("halving-funnel") + expect(() => resolveScoutSequence("unknown")).toThrow(/PIX_BENCH_SCOUT_SEQUENCE/) }) diff --git a/benchmarks/tests/run-config.test.ts b/benchmarks/tests/run-config.test.ts index 9ec87be..efa0687 100644 --- a/benchmarks/tests/run-config.test.ts +++ b/benchmarks/tests/run-config.test.ts @@ -1,15 +1,12 @@ import { expect, it } from "vitest" -import { resolveRouterSearchStrategy, resolveSearchKnobs } from "../retrieval/run-config.js" +import { resolveSearchKnobs } from "../retrieval/run-config.js" it("resolves search knobs with strategy defaults", () => { const knobs = resolveSearchKnobs({}) expect(knobs).toEqual({ - routerSearchStrategy: "halving-funnel", - scoutSequence: "halton", + scoutSequence: "sobol", seedHypotheses: true, - beamSchedule: "fixed", - coordinatePasses: 2, globalScouts: 512, localCloudPoints: 16, localCloudRadiusLevels: 2, @@ -19,28 +16,21 @@ it("resolves search knobs with strategy defaults", () => { it("parses overrides and rejects invalid values by knob name", () => { expect( resolveSearchKnobs({ - PIX_BENCH_ROUTER_STRATEGY: "proxy-promotion", - PIX_BENCH_SCOUT_SEQUENCE: "sobol", + PIX_BENCH_SCOUT_SEQUENCE: "halton", PIX_BENCH_SEED_HYPOTHESES: "1", - PIX_BENCH_BEAM_SCHEDULE: "decaying", - PIX_BENCH_COORDINATE_PASSES: "0", PIX_BENCH_GLOBAL_SCOUTS: "256", PIX_BENCH_LOCAL_CLOUD_POINTS: "64", PIX_BENCH_LOCAL_CLOUD_RADIUS: "2", }), ).toEqual({ - routerSearchStrategy: "proxy-promotion", - scoutSequence: "sobol", + scoutSequence: "halton", seedHypotheses: true, - beamSchedule: "decaying", - coordinatePasses: 0, globalScouts: 256, localCloudPoints: 64, localCloudRadiusLevels: 2, }) - expect(() => resolveRouterSearchStrategy("golden")).toThrow(/PIX_BENCH_ROUTER_STRATEGY/) - expect(() => resolveSearchKnobs({ PIX_BENCH_BEAM_SCHEDULE: "wide" })).toThrow( - /PIX_BENCH_BEAM_SCHEDULE/, + expect(() => resolveSearchKnobs({ PIX_BENCH_SCOUT_SEQUENCE: "golden" })).toThrow( + /PIX_BENCH_SCOUT_SEQUENCE/, ) expect(() => resolveSearchKnobs({ PIX_BENCH_GLOBAL_SCOUTS: "-4" })).toThrow( /PIX_BENCH_GLOBAL_SCOUTS/, @@ -51,8 +41,7 @@ it("parses overrides and rejects invalid values by knob name", () => { }) it("uses the measured broad-wave defaults for the halving funnel", () => { - expect(resolveSearchKnobs({ PIX_BENCH_ROUTER_STRATEGY: "halving-funnel" })).toMatchObject({ - routerSearchStrategy: "halving-funnel", + expect(resolveSearchKnobs({})).toMatchObject({ globalScouts: 512, seedHypotheses: true, localCloudPoints: 16, diff --git a/benchmarks/tests/scouts.test.ts b/benchmarks/tests/scouts.test.ts index aa63eb4..507264a 100644 --- a/benchmarks/tests/scouts.test.ts +++ b/benchmarks/tests/scouts.test.ts @@ -12,9 +12,9 @@ import { } from "../retrieval/evaluation/scouts/index.js" import type { ScoutSequence } from "../retrieval/evaluation/scouts/index.js" -it("resolves the scout sequence knob with halton as the default", () => { - expect(DEFAULT_SCOUT_SEQUENCE).toBe("halton") - expect(resolveScoutSequence(undefined)).toBe("halton") +it("resolves the scout sequence knob with sobol as the default", () => { + expect(DEFAULT_SCOUT_SEQUENCE).toBe("sobol") + expect(resolveScoutSequence(undefined)).toBe("sobol") for (const name of SCOUT_SEQUENCE_NAMES) { expect(resolveScoutSequence(name)).toBe(name) } diff --git a/benchmarks/tests/worker-pool.test.ts b/benchmarks/tests/worker-pool.test.ts index aa8292f..dd85edf 100644 --- a/benchmarks/tests/worker-pool.test.ts +++ b/benchmarks/tests/worker-pool.test.ts @@ -123,13 +123,11 @@ describe("benchmark candidate evaluation pool", () => { { workerCount: 0, evaluationQueue: candidateQueue, - routerSearchStrategy: "halving-funnel", }, ), fitRecommendedEvidenceRouter("fixture", "dbsf", [searchSample], SEARCH_PRIORITY_PROFILE, { workerCount: 0, evaluationQueue: candidateQueue, - routerSearchStrategy: "halving-funnel", }), ]) const serialHoldout = await optimizeEvidenceRouter( @@ -140,14 +138,14 @@ describe("benchmark candidate evaluation pool", () => { [searchSample], [searchSample], SEARCH_PRIORITY_PROFILE, - { workerCount: 0, routerSearchStrategy: "halving-funnel" }, + { workerCount: 0 }, ) const serialFitAll = await fitRecommendedEvidenceRouter( "fixture", "dbsf", [searchSample], SEARCH_PRIORITY_PROFILE, - { workerCount: 0, routerSearchStrategy: "halving-funnel" }, + { workerCount: 0 }, ) expect(parallelHoldout.map(withoutSearchTimings)).toEqual( @@ -173,7 +171,6 @@ describe("benchmark candidate evaluation pool", () => { { workerCount: 0, evaluationQueue: candidateQueue, - routerSearchStrategy: "halving-funnel", }, ) const serial = await fitRecommendedEvidenceRouter( @@ -181,14 +178,13 @@ describe("benchmark candidate evaluation pool", () => { "dbsf", halvingSamples, SEARCH_PRIORITY_PROFILE, - { workerCount: 0, routerSearchStrategy: "halving-funnel" }, + { workerCount: 0 }, ) expect(parallel.map(withoutSearchTimings)).toEqual(serial.map(withoutSearchTimings)) const result = parallel[0] if (result === undefined) throw new Error("Missing halving router result") expect(result.searchDiagnostics.proxyEvaluations).toBeGreaterThan(0) expect(result.searchDiagnostics.proxyPromotions).toBeGreaterThan(0) - expect(result.searchDiagnostics.timings.randomSearchMs).toBe(0) } finally { await candidateQueue.close() } diff --git a/package.json b/package.json index fe06ec1..bafc828 100644 --- a/package.json +++ b/package.json @@ -39,6 +39,7 @@ "bench:retrieval:corpus-size": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/corpus-size-execution.test.ts", "bench:retrieval:corpus-size:model": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/corpus-size-model.test.ts", "bench:retrieval:router-models": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/router-model-real.test.ts", + "bench:retrieval:chunking": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/chunking-sweep.test.ts", "bench:retrieval:develop": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/retrieval.test.ts -t \"runs the develop retrieval profile\"", "bench:retrieval:fixture": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/channels.test.ts", "bench:retrieval:full": "vp test --config benchmarks/vite.config.ts --run benchmarks/tests/retrieval.test.ts -t \"runs the full retrieval profile\"", From 8ad31787b9c23df518d3032ce5136fa065919de6 Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Fri, 28 Aug 2026 21:17:36 +0200 Subject: [PATCH 13/16] refactor(bench): extract shared sweep search, drop dead rank path --- .../evaluation/router-search/rank.ts | 88 -------------- .../evaluation/subsample-evidence.ts | 111 ++++++++---------- benchmarks/retrieval/evaluation/types.ts | 2 - 3 files changed, 52 insertions(+), 149 deletions(-) diff --git a/benchmarks/retrieval/evaluation/router-search/rank.ts b/benchmarks/retrieval/evaluation/router-search/rank.ts index b32f57f..31c2ecd 100644 --- a/benchmarks/retrieval/evaluation/router-search/rank.ts +++ b/benchmarks/retrieval/evaluation/router-search/rank.ts @@ -281,35 +281,6 @@ export interface RouterEvaluationResult { readonly candidateEvaluationMs: number } -const finalizeRouterCandidates = ( - context: RouterSearchContext, - rankedFull: readonly RouterCandidate[], - proxyKeys: readonly string[], - limit: number, -): readonly RouterCandidate[] => { - if (context.mode.recordsProxyAgreement && proxyKeys.length > 0) { - const fullRanks = new Map( - rankedFull.map((candidate, rank) => [routerKey(candidate.config), rank]), - ) - for (let left = 0; left < proxyKeys.length; left++) { - const leftRank = fullRanks.get(proxyKeys[left]) - if (leftRank === undefined) continue - for (let right = left + 1; right < proxyKeys.length; right++) { - const rightRank = fullRanks.get(proxyKeys[right]) - if (rightRank === undefined) continue - context.stats.proxyAgreementComparisons++ - if (leftRank < rightRank) context.stats.proxyAgreementMatches++ - } - } - } - const ordered = context.mode.orderRanked(rankedFull) - for (const candidate of ordered) context.archive.set(routerKey(candidate.config), candidate) - const elites = context.mode.selectElites(context, [...context.elites.values(), ...ordered]) - context.elites.clear() - for (const candidate of elites) context.elites.set(routerKey(candidate.config), candidate) - return context.mode.selectBeam(context, ordered, limit) -} - export const evaluateRouterConfigs = async ( entries: readonly (readonly [string, EvidenceRouterConfig])[], samples: readonly EvidenceSearchSample[], @@ -343,62 +314,3 @@ export const evaluateRouterConfigs = async ( candidateEvaluationMs, } } - -export const rankRouterCandidates = async ( - context: RouterSearchContext, - configs: readonly EvidenceRouterConfig[], - limit: number, - protectedConfigs: readonly EvidenceRouterConfig[] = [], - useProxy = true, -): Promise => { - const unique = new Map() - for (const config of configs) unique.set(routerKey(config), config) - context.stats.rawCandidates += configs.length - context.stats.uniqueCandidates += unique.size - context.stats.protectedEliteCount += protectedConfigs.length + context.elites.size - let fullCandidates = [...unique] - let proxyKeys: readonly string[] = [] - if (useProxy && context.proxySamples.length < context.samples.length) { - const rankedProxy = await evaluateRouterConfigs( - fullCandidates, - context.proxySamples, - context.proxyPool, - context.proxyQualityCache, - ) - context.stats.proxyCacheHits += rankedProxy.cacheHits - context.stats.proxyEvaluations += rankedProxy.evaluations - context.stats.timings.candidatePreparationMs += rankedProxy.candidatePreparationMs - context.stats.timings.candidateEvaluationMs += rankedProxy.candidateEvaluationMs - const proxySelectionStartedAt = performance.now() - const { promoted, agreementKeys } = context.mode.promoteProxy( - context, - rankedProxy.candidates, - limit, - ) - const selected = new Map( - promoted.map((candidate) => [routerKey(candidate.config), candidate.config]), - ) - context.stats.proxyPromotions += promoted.length - const protectedKeys = new Set([...protectedConfigs.map(routerKey), ...context.elites.keys()]) - for (const [key, config] of fullCandidates) { - if (protectedKeys.has(key)) selected.set(key, config) - } - fullCandidates = [...selected] - proxyKeys = agreementKeys - context.stats.timings.candidateSelectionMs += performance.now() - proxySelectionStartedAt - } - const rankedFull = await evaluateRouterConfigs( - fullCandidates, - context.samples, - context.fullPool, - context.qualityCache, - ) - context.stats.fullCacheHits += rankedFull.cacheHits - context.stats.fullEvaluations += rankedFull.evaluations - context.stats.timings.candidatePreparationMs += rankedFull.candidatePreparationMs - context.stats.timings.candidateEvaluationMs += rankedFull.candidateEvaluationMs - const fullSelectionStartedAt = performance.now() - const selected = finalizeRouterCandidates(context, rankedFull.candidates, proxyKeys, limit) - context.stats.timings.candidateSelectionMs += performance.now() - fullSelectionStartedAt - return selected -} diff --git a/benchmarks/retrieval/evaluation/subsample-evidence.ts b/benchmarks/retrieval/evaluation/subsample-evidence.ts index b1cf4b4..fb06b58 100644 --- a/benchmarks/retrieval/evaluation/subsample-evidence.ts +++ b/benchmarks/retrieval/evaluation/subsample-evidence.ts @@ -48,7 +48,8 @@ import { } from "./router-search/comparisons.js" import { SEARCH_CANDIDATE_DEPTH, routerParameters } from "./router-search/config-space.js" import { runBenchmarkSearch } from "./search.js" -import type { WeightSearchSample } from "./weight-search.js" +import type { RecommendedEvidenceRouter } from "./types.js" +import type { BenchmarkSearchOptions, WeightSearchSample } from "./weight-search.js" const QUERY_FORMS: readonly QueryKind[] = [ "identifier", @@ -241,6 +242,44 @@ export interface RealCorpusSizeSweepResult { readonly perSizeSamples: readonly { readonly corpusSize: number; readonly samples: number }[] } +const defaultSearchOptions = (): BenchmarkSearchOptions => ({ + workerCount: Math.min(resolveWorkerCount(), getDefaultWorkerCount()), + fallbackToSerial: false, +}) + +/** Run the full dbsf/direct search over one sample set and score the recommended router. */ +const runDirectRouterSearch = async ( + model: string, + samples: readonly WeightSearchSample[], +): Promise<{ + readonly router: RecommendedEvidenceRouter + readonly mean: number + readonly standardError: number +}> => { + const samplesByModel = new Map([[model, samples]]) + const search = await Effect.runPromise( + runBenchmarkSearch( + { + groupedFolds: 3, + repositoryHoldouts: false, + fusionMethods: ["dbsf"], + routerFusionMethods: ["dbsf"], + }, + samplesByModel, + "grouped-3-fold", + OPTIMIZATION_PROFILES["search-priority"], + false, + defaultSearchOptions(), + ), + ) + const router = search.recommendedEvidenceRouters.find( + (row) => row.fusion === "dbsf" && row.objective === "direct", + ) + if (router === undefined) throw new Error("No dbsf/direct router recommendation") + const scored = evaluateRouterConfigOnSamples(samples, router.config) + return { router, mean: scored.mean, standardError: scored.standardError } +} + const goldTargetsByQuestion = ( manifest: CorpusManifest, corpus: PreparedCorpus, @@ -273,10 +312,6 @@ export const runRealCorpusSizeSweep = ( : context.corpus.chunks.length, }), ) - const searchOptions = { - workerCount: Math.min(resolveWorkerCount(), getDefaultWorkerCount()), - fallbackToSerial: false as const, - } const perSizeSamples: { corpusSize: number; samples: number }[] = [] const sweep = yield* Effect.promise(() => runCorpusSizeSweep( @@ -284,32 +319,12 @@ export const runRealCorpusSizeSweep = ( async (plan) => { const samples = await Effect.runPromise(buildSubSampleSamples(context, plan)) perSizeSamples.push({ corpusSize: plan.targetSize, samples: samples.length }) - const samplesByModel = new Map([ - [context.model, samples], - ]) - const search = await Effect.runPromise( - runBenchmarkSearch( - { - groupedFolds: 3, - repositoryHoldouts: false, - fusionMethods: ["dbsf"], - routerFusionMethods: ["dbsf"], - }, - samplesByModel, - "grouped-3-fold", - OPTIMIZATION_PROFILES["search-priority"], - false, - searchOptions, - ), - ) - const router = search.recommendedEvidenceRouters.find( - (row) => row.fusion === "dbsf" && row.objective === "direct", + const { router, mean, standardError } = await runDirectRouterSearch( + context.model, + samples, ) - if (router === undefined) - throw new Error(`No dbsf/direct router recommendation at size ${plan.targetSize}`) - const scored = evaluateRouterConfigOnSamples(samples, router.config) reportBenchmarkProgress( - `size ${plan.targetSize}: ndcg@20 ${scored.mean.toFixed(4)} ± ${scored.standardError.toFixed(4)}`, + `size ${plan.targetSize}: ndcg@20 ${mean.toFixed(4)} ± ${standardError.toFixed(4)}`, ) return [ { @@ -322,8 +337,8 @@ export const runRealCorpusSizeSweep = ( fold: "fit-all", }, weights: router.staticWeights, - score: scored.mean, - noise: scored.standardError, + score: mean, + noise: standardError, }, ] }, @@ -401,7 +416,7 @@ export interface ChunkingSweepRow { } /** Parse a comma-separated chunk-token list for the sweep. */ -export const resolveChunkingSizes = (requested: string | undefined): readonly number[] => { +const resolveChunkingSizes = (requested: string | undefined): readonly number[] => { if (requested === undefined) return [256, 384, 512] return requested .split(",") @@ -427,41 +442,19 @@ export const runChunkingSweep = ( Number.POSITIVE_INFINITY, ])[0]! const samples = yield* buildSubSampleSamples(context, fullPlan) - const samplesByModel = new Map([ - [context.model, samples], - ]) - const search = yield* runBenchmarkSearch( - { - groupedFolds: 3, - repositoryHoldouts: false, - fusionMethods: ["dbsf"], - routerFusionMethods: ["dbsf"], - }, - samplesByModel, - "grouped-3-fold", - OPTIMIZATION_PROFILES["search-priority"], - false, - { - workerCount: Math.min(resolveWorkerCount(), getDefaultWorkerCount()), - fallbackToSerial: false, - }, - ) - const router = search.recommendedEvidenceRouters.find( - (row) => row.fusion === "dbsf" && row.objective === "direct", + const { router, mean, standardError } = yield* Effect.promise(() => + runDirectRouterSearch(context.model, samples), ) - if (router === undefined) - throw new Error(`No dbsf/direct router recommendation at ${chunkTokens} tokens`) - const scored = evaluateRouterConfigOnSamples(samples, router.config) rows.push({ chunkTokens, chunks: context.corpus.chunks.length, - ndcgAt20: scored.mean, - standardError: scored.standardError, + ndcgAt20: mean, + standardError, proxyEvaluations: router.proxyEvaluations, fullEvaluations: router.fullEvaluations, }) reportBenchmarkProgress( - `chunkTokens ${chunkTokens}: ${context.corpus.chunks.length} chunks, ndcg@20 ${scored.mean.toFixed(4)} ± ${scored.standardError.toFixed(4)}`, + `chunkTokens ${chunkTokens}: ${context.corpus.chunks.length} chunks, ndcg@20 ${mean.toFixed(4)} ± ${standardError.toFixed(4)}`, ) } return rows diff --git a/benchmarks/retrieval/evaluation/types.ts b/benchmarks/retrieval/evaluation/types.ts index a09333c..65d9c11 100644 --- a/benchmarks/retrieval/evaluation/types.ts +++ b/benchmarks/retrieval/evaluation/types.ts @@ -55,8 +55,6 @@ export type RouterSearchStrategy = HalvingFunnelStrategy const FUNNEL_ALGORITHM_PREFIX = "global-scout-funnel" -export const DEFAULT_ROUTER_SEARCH_STRATEGY = "halving-funnel" as const - /** Default fidelity funnel: broad proxy wave, local proxy cloud, then 256 diverse finalists. */ export const DEFAULT_HALVING_FUNNEL_STRATEGY: HalvingFunnelStrategy = { kind: "halving-funnel", From 5602586d68bc9e89155b8f617b55ede14a8495c3 Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Fri, 28 Aug 2026 21:22:33 +0200 Subject: [PATCH 14/16] docs(bench): record chunking sweep result --- benchmarks/README.md | 22 +++++++++++----------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/benchmarks/README.md b/benchmarks/README.md index a06869b..1cc6874 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -134,17 +134,17 @@ deferred until the promotion bar is met. The real multiplicative-vs-log-linear c Every search knob has a recorded status so work stops re-litigating settled ones: -| Knob | Status | Evidence | -| ------------------------------------------- | --------------------------------------------------------------------------------- | -------------------- | -| Router model (multiplicative vs log-linear) | Settled: keep multiplicative; switch at the next mandatory promoted-config re-fit | ADR 0021 | -| Router search strategy | Settled and deleted: only `halving-funnel` remains (beam control removed) | BASELINE (#189) | -| Finalist budget | Settled: 256 (384 measured identical, +65 s) | BASELINE (#189) | -| Local Sobol cloud (points/radius) | Settled: 16 points, radius 2 (32/r3 added no quality) | BASELINE (#189) | -| Scout sequence (sobol/halton/random) | Settled: all inside noise; sobol default | BASELINE (schema 29) | -| Wide-vs-deep passes | Settled: 1 wide pass matches 2 narrow passes at 45% fewer evals | BASELINE (schema 29) | -| Corpus-size factor | Deferred with promotion bar; sweep stays as standing measurement | ADR 0022 | -| Fitting methods, selection rules | Diagnostics only, recorded per comparison run | ADR 0021 | -| Chunking (chunkTokens) | Open: sweep via `vp run bench:retrieval:chunking` | — | +| Knob | Status | Evidence | +| ------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------- | +| Router model (multiplicative vs log-linear) | Settled: keep multiplicative; switch at the next mandatory promoted-config re-fit | ADR 0021 | +| Router search strategy | Settled and deleted: only `halving-funnel` remains (beam control removed) | BASELINE (#189) | +| Finalist budget | Settled: 256 (384 measured identical, +65 s) | BASELINE (#189) | +| Local Sobol cloud (points/radius) | Settled: 16 points, radius 2 (32/r3 added no quality) | BASELINE (#189) | +| Scout sequence (sobol/halton/random) | Settled: all inside noise; sobol default | BASELINE (schema 29) | +| Wide-vs-deep passes | Settled: 1 wide pass matches 2 narrow passes at 45% fewer evals | BASELINE (schema 29) | +| Corpus-size factor | Deferred with promotion bar; sweep stays as standing measurement | ADR 0022 | +| Fitting methods, selection rules | Diagnostics only, recorded per comparison run | ADR 0021 | +| Chunking (chunkTokens) | Settled for flat chunking: 512 default; 384/256 degrade NDCG@20 monotonically (0.66/0.64/0.53 on t3code). Revisit only together with parent-chunk retrieval | `retrieval-chunking-*.json` | `bench:retrieval` aliases `bench:retrieval:validate`. Every profile measures the same physical rankings and retrieval variants; profiles only control matrix size, holdout coverage, and expensive From 24d3cf5ddfa293ac22a475b9d81967d025e056c2 Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Fri, 28 Aug 2026 22:30:25 +0200 Subject: [PATCH 15/16] fix(bench): matrix run survives unsupported model cells --- benchmarks/retrieval/matrix.ts | 19 +++++++++++++++++++ benchmarks/tests/matrix-execution.test.ts | 22 +++++++++++++++++----- 2 files changed, 36 insertions(+), 5 deletions(-) diff --git a/benchmarks/retrieval/matrix.ts b/benchmarks/retrieval/matrix.ts index 667df2f..b1b9332 100644 --- a/benchmarks/retrieval/matrix.ts +++ b/benchmarks/retrieval/matrix.ts @@ -183,6 +183,25 @@ export const benchmarkMatrixInvocations = ( ), ) +/** Restrict a plan to the coordinates covered by successful invocations. */ +export const restrictBenchmarkMatrixPlan = ( + plan: BenchmarkMatrixPlan, + covered: ReadonlySet, +): BenchmarkMatrixPlan => { + const coordinates = plan.coordinates.filter((coordinate) => + covered.has( + `${coordinate.benchmarkProfile}\0${coordinate.optimizationProfile}\0${coordinate.model}`, + ), + ) + if (coordinates.length === 0) + throw new Error("Benchmark matrix plan restriction removed every coordinate") + return { schemaVersion: 1, coordinates } as unknown as BenchmarkMatrixPlan +} + +/** Stable identity of one matrix invocation. */ +export const benchmarkMatrixInvocationKey = (invocation: BenchmarkMatrixInvocation): string => + `${invocation.benchmarkProfile}\0${invocation.optimizationProfile}\0${invocation.model}` + const coordinateFor = ( artifact: MatrixSourceArtifact, result: Result, diff --git a/benchmarks/tests/matrix-execution.test.ts b/benchmarks/tests/matrix-execution.test.ts index 6766069..a276154 100644 --- a/benchmarks/tests/matrix-execution.test.ts +++ b/benchmarks/tests/matrix-execution.test.ts @@ -2,14 +2,16 @@ import { mkdir, readFile, writeFile } from "node:fs/promises" import path from "node:path" import { expect, it } from "@effect/vitest" -import { Effect, Schema } from "effect" +import { Effect, Result, Schema } from "effect" import type { BenchmarkArtifact } from "../retrieval/evaluation/types.js" import { + benchmarkMatrixInvocationKey, benchmarkMatrixInvocations, BenchmarkMatrixManifestSchema, expandBenchmarkMatrixManifest, mergeBenchmarkMatrix, + restrictBenchmarkMatrixPlan, } from "../retrieval/matrix.js" import { runRetrievalBenchmark } from "../retrieval/runner.js" @@ -25,16 +27,25 @@ it.effect("executes and merges the complete retrieval matrix", () => Effect.gen(function* () { const manifest = yield* readManifest const artifacts: BenchmarkArtifact[] = [] + const covered = new Set() + const failures: { readonly invocation: string; readonly error: string }[] = [] for (const invocation of benchmarkMatrixInvocations(manifest)) { process.env.PIX_BENCH_MODELS = invocation.model process.env.PIX_BENCH_REPOS = invocation.repositories.join(",") process.env.PIX_BENCH_OPTIMIZATION_PROFILE = invocation.optimizationProfile - const result = yield* runRetrievalBenchmark(invocation.benchmarkProfile) - artifacts.push(result.artifact) + const outcome = yield* Effect.result(runRetrievalBenchmark(invocation.benchmarkProfile)) + if (Result.isFailure(outcome)) { + const cause = outcome.failure + const error = cause instanceof Error ? cause.message : String(cause) + failures.push({ invocation: benchmarkMatrixInvocationKey(invocation), error }) + continue + } + artifacts.push(outcome.success.artifact) + covered.add(benchmarkMatrixInvocationKey(invocation)) } - const plan = expandBenchmarkMatrixManifest(manifest) + const plan = restrictBenchmarkMatrixPlan(expandBenchmarkMatrixManifest(manifest), covered) const matrix = mergeBenchmarkMatrix(plan, artifacts) const outputDirectory = path.resolve("benchmarks/results") const outputPath = path.join( @@ -44,11 +55,12 @@ it.effect("executes and merges the complete retrieval matrix", () => yield* Effect.tryPromise({ try: async () => { await mkdir(outputDirectory, { recursive: true }) - await writeFile(outputPath, `${JSON.stringify(matrix, null, 2)}\n`, "utf8") + await writeFile(outputPath, `${JSON.stringify({ ...matrix, failures }, null, 2)}\n`, "utf8") }, catch: (cause) => new Error(`Could not write benchmark matrix ${outputPath}`, { cause }), }) expect(matrix.coordinates).toHaveLength(plan.coordinates.length) + expect(artifacts.length).toBeGreaterThan(0) }), ) From 36439df9dbcb8f6be1f57493d797b9c1c6b629ab Mon Sep 17 00:00:00 2001 From: Lucas Burmeister Date: Fri, 28 Aug 2026 23:46:01 +0200 Subject: [PATCH 16/16] fix(bench): week-long timeout for the matrix release run --- benchmarks/tests/matrix-execution.test.ts | 80 +++++++++++++---------- 1 file changed, 44 insertions(+), 36 deletions(-) diff --git a/benchmarks/tests/matrix-execution.test.ts b/benchmarks/tests/matrix-execution.test.ts index a276154..c4527fa 100644 --- a/benchmarks/tests/matrix-execution.test.ts +++ b/benchmarks/tests/matrix-execution.test.ts @@ -23,44 +23,52 @@ const readManifest = Effect.tryPromise({ catch: (cause) => new Error("Could not read the benchmark matrix manifest", { cause }), }) -it.effect("executes and merges the complete retrieval matrix", () => - Effect.gen(function* () { - const manifest = yield* readManifest - const artifacts: BenchmarkArtifact[] = [] - const covered = new Set() - const failures: { readonly invocation: string; readonly error: string }[] = [] +it.effect( + "executes and merges the complete retrieval matrix", + () => + Effect.gen(function* () { + const manifest = yield* readManifest + const artifacts: BenchmarkArtifact[] = [] + const covered = new Set() + const failures: { readonly invocation: string; readonly error: string }[] = [] - for (const invocation of benchmarkMatrixInvocations(manifest)) { - process.env.PIX_BENCH_MODELS = invocation.model - process.env.PIX_BENCH_REPOS = invocation.repositories.join(",") - process.env.PIX_BENCH_OPTIMIZATION_PROFILE = invocation.optimizationProfile - const outcome = yield* Effect.result(runRetrievalBenchmark(invocation.benchmarkProfile)) - if (Result.isFailure(outcome)) { - const cause = outcome.failure - const error = cause instanceof Error ? cause.message : String(cause) - failures.push({ invocation: benchmarkMatrixInvocationKey(invocation), error }) - continue + for (const invocation of benchmarkMatrixInvocations(manifest)) { + process.env.PIX_BENCH_MODELS = invocation.model + process.env.PIX_BENCH_REPOS = invocation.repositories.join(",") + process.env.PIX_BENCH_OPTIMIZATION_PROFILE = invocation.optimizationProfile + const outcome = yield* Effect.result(runRetrievalBenchmark(invocation.benchmarkProfile)) + if (Result.isFailure(outcome)) { + const cause = outcome.failure + const error = cause instanceof Error ? cause.message : String(cause) + failures.push({ invocation: benchmarkMatrixInvocationKey(invocation), error }) + continue + } + artifacts.push(outcome.success.artifact) + covered.add(benchmarkMatrixInvocationKey(invocation)) } - artifacts.push(outcome.success.artifact) - covered.add(benchmarkMatrixInvocationKey(invocation)) - } - const plan = restrictBenchmarkMatrixPlan(expandBenchmarkMatrixManifest(manifest), covered) - const matrix = mergeBenchmarkMatrix(plan, artifacts) - const outputDirectory = path.resolve("benchmarks/results") - const outputPath = path.join( - outputDirectory, - `retrieval-matrix-${new Date().toISOString().replaceAll(":", "-")}.json`, - ) - yield* Effect.tryPromise({ - try: async () => { - await mkdir(outputDirectory, { recursive: true }) - await writeFile(outputPath, `${JSON.stringify({ ...matrix, failures }, null, 2)}\n`, "utf8") - }, - catch: (cause) => new Error(`Could not write benchmark matrix ${outputPath}`, { cause }), - }) + const plan = restrictBenchmarkMatrixPlan(expandBenchmarkMatrixManifest(manifest), covered) + const matrix = mergeBenchmarkMatrix(plan, artifacts) + const outputDirectory = path.resolve("benchmarks/results") + const outputPath = path.join( + outputDirectory, + `retrieval-matrix-${new Date().toISOString().replaceAll(":", "-")}.json`, + ) + yield* Effect.tryPromise({ + try: async () => { + await mkdir(outputDirectory, { recursive: true }) + await writeFile( + outputPath, + `${JSON.stringify({ ...matrix, failures }, null, 2)}\n`, + "utf8", + ) + }, + catch: (cause) => new Error(`Could not write benchmark matrix ${outputPath}`, { cause }), + }) - expect(matrix.coordinates).toHaveLength(plan.coordinates.length) - expect(artifacts.length).toBeGreaterThan(0) - }), + expect(matrix.coordinates).toHaveLength(plan.coordinates.length) + expect(artifacts.length).toBeGreaterThan(0) + }), + // The full 30-invocation matrix is a multi-hour release-evidence run. + 7 * 24 * 3_600_000, )