diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..2f77e91 --- /dev/null +++ b/.gitattributes @@ -0,0 +1 @@ +*.ipynb linguist-documentation diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..b58254c --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,48 @@ +# wads CI — calls the reusable workflow hosted in i2mint/wads. +# +# All configuration comes from this repo's pyproject.toml [tool.wads.ci.*]. +# To customize the workflow itself (rare), replace this file with the +# full inline template `wads/data/github_ci_uv.yml` from i2mint/wads. +# +# Pinning: `@master` floats with wads. If you need version stability for +# a release-sensitive repo, change `@master` to a wads tag (e.g. `@v0.1.81`). +# CI failure does not block a published release — it blocks the publish +# step itself — so floating master is generally safe. +# +# Permissions: GitHub validates that the caller grants AT LEAST the +# permissions any job in the called workflow requests — at workflow-parse +# time, not at run-time, even if the job would be skipped via `if:`. +# The reusable workflow needs: +# contents: write for the publish job's version-bump push-back +# and for the github-pages job's gh-pages branch push +# pages: write for the github-pages job's REST API Pages config +# Both default to `write` on org-account GITHUB_TOKEN and need to be +# granted explicitly on personal-account callers (where the default is +# read-only). No `id-token: write` needed — the publish-github-pages +# action uses peaceiris/actions-gh-pages (branch-based) + REST API, +# not the OIDC `actions/deploy-pages` flow. +name: Continuous Integration +on: [push, pull_request] +jobs: + ci: + uses: i2mint/wads/.github/workflows/uv-ci.yml@master + permissions: + contents: write + pages: write + # Explicit pass-through (not `secrets: inherit`) because `inherit` does + # not reliably propagate caller-repo secrets to a reusable workflow owned + # by a different account (verified empirically: personal-account caller + + # i2mint-org workflow → `${{ secrets.PYPI_PASSWORD }}` resolved to empty). + # + # This list is the per-repo *transport*: it should contain PYPI_PASSWORD + # (for publishing) plus every secret your tests/CI need. It is generated + # from [tool.wads.ci.env] in pyproject.toml. To add one, run + # wads-secrets add VAR_NAME # updates pyproject + this block + # or just append a line below. *Which* of these become job env vars (and + # which are required) is controlled by [tool.wads.ci.env] — passing a + # secret here does not by itself put it in the environment. + # + # A secret name must also be declared in the reusable workflow's superset + # (wads/ci_secrets.py). `wads-secrets add` warns if it is not. + secrets: + PYPI_PASSWORD: ${{ secrets.PYPI_PASSWORD }} diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..22feac4 --- /dev/null +++ b/.gitignore @@ -0,0 +1,120 @@ +.claude/handoffs/ +.claude/scratch/ + +# Byte-compiled / optimized / DLL files +__pycache__/ +*.py[cod] +*$py.class + + +.DS_Store +# C extensions +*.so + +# TLS certificates +## Ignore all PEM files anywhere +*.pem +## Also ignore any certs directory +certs/ + +# Distribution / packaging +.Python +build/ +develop-eggs/ +dist/ +downloads/ +eggs/ +.eggs/ +lib/ +lib64/ +parts/ +sdist/ +var/ +wheels/ +*.egg-info/ +.installed.cfg +*.egg +MANIFEST +_build + +# PyInstaller +# Usually these files are written by a python script from a template +# before PyInstaller builds the exe, so as to inject date/other infos into it. +*.manifest +*.spec + +# Installer logs +pip-log.txt +pip-delete-this-directory.txt + +# Unit test / coverage reports +htmlcov/ +.tox/ +.coverage +.coverage.* +.cache +nosetests.xml +coverage.xml +*.cover +.hypothesis/ +.pytest_cache/ + +# Translations +*.mo +*.pot + +# Django stuff: +*.log +local_settings.py +db.sqlite3 + +# Flask stuff: +instance/ +.webassets-cache + +# Scrapy stuff: +.scrapy + +# Sphinx documentation +docs/_build/ +docs/* + +# PyBuilder +target/ + +# Jupyter Notebook +.ipynb_checkpoints + +# pyenv +.python-version + +# celery beat schedule file +celerybeat-schedule + +# SageMath parsed files +*.sage.py + +# Environments +.env +.venv +env/ +venv/ +ENV/ +env.bak/ +venv.bak/ + +# Spyder project settings +.spyderproject +.spyproject + +# Rope project settings +.ropeproject + +# mkdocs documentation +/site + +# mypy +.mypy_cache/ + +# PyCharm +.idea diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..9cae6ea --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 i2mint + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/ir/__init__.py b/ir/__init__.py new file mode 100644 index 0000000..45af6e1 --- /dev/null +++ b/ir/__init__.py @@ -0,0 +1,61 @@ +"""``ir`` — an information-retrieval substrate for agentic systems. + +One uniform "find the relevant things in this corpus" contract that scales from +an ad-hoc search over an ephemeral list to a maintained search engine. Retrieval +is the core; generation/selection/reranking are layered on top. + +Quick start:: + + import ir + + # Define a corpus source (abstract strategy + parameters, smart defaults): + source = ir.CorpusSource.from_md_reports() # project docs/ reports + corpus = ir.build(source) # index (incremental) + hits = ir.search(corpus, "how do I deploy the app") # ranked SearchHits + + # Light, dependency-free embedding for fast tests: + corpus = ir.build(source, embedder="light") + +A corpus source is defined by a ``scope`` (what is in the corpus), a +``change_signal`` (what counts as stale), an ``indexing_strategy`` (how a raw +item becomes filter fields + embeddable surfaces), and an ``embedder``. The +default embedder is a decent *local* model (``all-MiniLM-L6-v2``); ``"light"`` +selects a numpy-only hashing embedder. Data persists under XDG dirs through a +``dol`` repository layer. +""" + +from __future__ import annotations + +from . import embed as _embed # noqa: F401 (sets USE_TF=0 before transformers) +from .base import Artifact, IndexPlan, Record, SearchHit, Surface +from .index import Corpus, build, open_corpus +from .retrieve import search as _search +from .sources import CorpusSource +from .store import CorpusStore +from .strategy import Chunked, IndexingStrategy, Package, Skill, WholeText + +__all__ = [ + "Artifact", + "Surface", + "Record", + "SearchHit", + "IndexPlan", + "IndexingStrategy", + "WholeText", + "Chunked", + "Skill", + "Package", + "CorpusSource", + "CorpusStore", + "Corpus", + "build", + "open_corpus", + "search", +] + + +def search(corpus, query, **kwargs): + """Search a :class:`~ir.index.Corpus`, or a corpus *name* (reopened lazily).""" + if isinstance(corpus, str): + corpus = open_corpus(corpus) + return _search(corpus, query, **kwargs) diff --git a/ir/base.py b/ir/base.py new file mode 100644 index 0000000..b8f3804 --- /dev/null +++ b/ir/base.py @@ -0,0 +1,117 @@ +"""Core data model for ``ir``. + +Retrieval in ``ir`` flows through four small, explicit types: + +- :class:`Artifact` — a logical item in a corpus (a file, a skill, a package). + Opaque ``raw`` payload plus ``metadata``. +- :class:`Surface` — one *embeddable unit* derived from an artifact. A single + artifact may yield several heterogeneous surfaces (a short description, an + AI-authored synopsis, a list of problem classes, body chunks). The + artifact→surfaces decomposition is the job of an + :class:`~ir.strategy.IndexingStrategy`. +- :class:`IndexPlan` — a strategy's output for one artifact: the + ``filter_fields`` (hard-filterable metadata, *not* embedded) and the list of + surfaces (embedded). +- :class:`Record` — a stored, embedded surface (one row in the index; maps + directly to a ``vd`` ``Document``). +- :class:`SearchHit` — a scored record returned by retrieval, with a helper to + collapse multiple surface-hits of the same artifact. + +The split between **filter_fields** (metadata you filter on) and **surfaces** +(text you embed) is deliberate and central: good retrieval is hard metadata +filtering *and* semantic ranking, not only embeddings. +""" + +from __future__ import annotations + +import hashlib +from collections.abc import Mapping, Sequence +from dataclasses import dataclass, field +from typing import Any + +import numpy as np + +FilterFields = Mapping[str, Any] +"""Non-embedded, hard-filterable metadata for an artifact (name, owner, tags).""" + + +def storage_key(*parts: str) -> str: + """Stable, filesystem-safe id from arbitrary string parts (truncated SHA-256).""" + h = hashlib.sha256("␟".join(parts).encode("utf-8")).hexdigest() + return h[:24] + + +@dataclass +class Artifact: + """A logical corpus item before decomposition into surfaces.""" + + id: str + raw: Any + metadata: dict = field(default_factory=dict) + + +@dataclass(frozen=True) +class Surface: + """One embeddable unit derived from an artifact. + + ``kind`` names the surface type (e.g. ``"description"``, ``"synopsis"``, + ``"problem_class"``, ``"chunk"``) so a query can match the *right part* of + an artifact. ``granularity`` is a coarse hint (``"document"`` / ``"chunk"`` + / ``"field"``). ``metadata`` is surface-local (e.g. chunk offsets). + """ + + artifact_id: str + kind: str + text: str + granularity: str = "document" + metadata: Mapping[str, Any] = field(default_factory=dict) + + +@dataclass +class IndexPlan: + """An :class:`~ir.strategy.IndexingStrategy`'s output for one artifact.""" + + filter_fields: dict = field(default_factory=dict) + surfaces: list[Surface] = field(default_factory=list) + + +@dataclass(frozen=True) +class Record: + """A stored, embedded surface — one row of the index.""" + + id: str + artifact_id: str + surface_kind: str + surface_index: int + text: str + vector: np.ndarray + metadata: dict = field(default_factory=dict) + + @staticmethod + def make_id(artifact_id: str, surface_kind: str, surface_index: int) -> str: + """Deterministic storage id for a surface of an artifact.""" + return storage_key(artifact_id, surface_kind, str(surface_index)) + + +@dataclass(frozen=True) +class SearchHit: + """A scored record returned by retrieval (higher score = closer).""" + + artifact_id: str + surface_kind: str + score: float + text: str + metadata: Mapping[str, Any] = field(default_factory=dict) + + +def best_per_artifact(hits: Sequence[SearchHit]) -> list[SearchHit]: + """Collapse hits to the highest-scoring surface per artifact. + + Returns the surviving hits sorted by score (descending). + """ + seen: dict[str, SearchHit] = {} + for h in hits: + cur = seen.get(h.artifact_id) + if cur is None or h.score > cur.score: + seen[h.artifact_id] = h + return sorted(seen.values(), key=lambda h: h.score, reverse=True) diff --git a/ir/config.py b/ir/config.py new file mode 100644 index 0000000..7de3a6f --- /dev/null +++ b/ir/config.py @@ -0,0 +1,78 @@ +"""Filesystem locations and process-wide defaults for ``ir``. + +``ir`` separates three kinds of on-disk state, each under an XDG-standard base +(overridable per-install via ``IR_CONFIG_DIR`` / ``IR_DATA_DIR`` / +``IR_CACHE_DIR``, then ``XDG_CONFIG_HOME`` / ``XDG_DATA_HOME`` / +``XDG_CACHE_HOME``, then ``~/.config`` / ``~/.local/share`` / ``~/.cache``): + +- **config** (``~/.config/ir``) — the named-corpus registry and user settings. +- **data** (``~/.local/share/ir``) — durable corpus stores (record metadata, + vectors, ledgers). The source of truth; losing it means rebuilding the index. +- **cache** (``~/.cache/ir``) — regenerable derived data, chiefly the embedding + cache keyed by ``(model, content_hash)``. + +Every path is a plain :class:`~pathlib.Path` created on demand, so callers can +treat the directories as guaranteed to exist. +""" + +from __future__ import annotations + +import os +import re +from pathlib import Path + +APP_NAME = "ir" + +_SAFE = re.compile(r"[^A-Za-z0-9_.-]+") + + +def _ensure(path: Path) -> Path: + """Create *path* (and parents) if missing and return it.""" + path.mkdir(parents=True, exist_ok=True) + return path + + +def _base(env_specific: str, xdg_var: str, home_subdir: str) -> Path: + """Resolve an XDG base dir for ``ir`` from env vars, with a home fallback.""" + specific = os.environ.get(env_specific) + if specific: + return Path(specific).expanduser() + xdg = os.environ.get(xdg_var) + if xdg: + return Path(xdg).expanduser() / APP_NAME + return Path.home() / home_subdir / APP_NAME + + +def config_dir() -> Path: + """User configuration directory (default ``~/.config/ir``).""" + return _ensure(_base("IR_CONFIG_DIR", "XDG_CONFIG_HOME", ".config")) + + +def data_dir() -> Path: + """Durable data directory (default ``~/.local/share/ir``).""" + return _ensure(_base("IR_DATA_DIR", "XDG_DATA_HOME", ".local/share")) + + +def cache_dir() -> Path: + """Regenerable cache directory (default ``~/.cache/ir``).""" + return _ensure(_base("IR_CACHE_DIR", "XDG_CACHE_HOME", ".cache")) + + +def safe_name(name: str) -> str: + """Filesystem-safe slug for a corpus/model identifier.""" + return _SAFE.sub("_", name).strip("_") or "unnamed" + + +def corpus_dir(name: str) -> Path: + """Durable directory holding one corpus's stores.""" + return _ensure(data_dir() / "corpora" / safe_name(name)) + + +def embeddings_cache_dir(model_id: str) -> Path: + """Cache directory for one embedding model's vectors.""" + return _ensure(cache_dir() / "embeddings" / safe_name(model_id)) + + +def registry_path() -> Path: + """JSON file mapping registered corpus names to their build settings.""" + return config_dir() / "corpora.json" diff --git a/ir/embed.py b/ir/embed.py new file mode 100644 index 0000000..cb836f3 --- /dev/null +++ b/ir/embed.py @@ -0,0 +1,102 @@ +"""Embedder resolution for ``ir`` — a decent local default, a light fallback. + +``ir`` favors a *decent local* embedding so retrieval works offline and tests +need no API keys, while keeping a *light* (numpy-only) option for when semantic +power is not what's under test. + +- ``"default"`` / ``"local"`` / ``"minilm"`` → ``all-MiniLM-L6-v2`` (384-dim) + via :func:`ef.embedder_adapters.sentence_transformers_embedder` + (``normalize=True``), wrapped in :class:`ef.CachedEmbedder` over a ``dol`` + cache under ``~/.cache/ir/embeddings/``. If ``sentence-transformers`` + is unavailable, it degrades to the hashing embedder with a warning. +- ``"light"`` / ``"hashing"`` → :class:`ef.HashingEmbedder` (numpy only). +- any other string → treated as a sentence-transformers model name. +- a callable / existing ``Embedder`` → passed through ``ef.as_embedder``. + +:func:`make_embedder` returns ``(embedder, embedder_id)``; the id pins the +model in the corpus ledger so a model change triggers a re-embed (the SSOT +discipline that keeps the index from silently drifting). + +Importing this module sets ``USE_TF=0`` so ``sentence-transformers`` (via +``transformers``) does not import TensorFlow, which crashes on this stack's +numpy ABI. Import ``ir`` before anything that imports ``transformers``. +""" + +from __future__ import annotations + +import os +import warnings +from collections.abc import Callable +from typing import Any + +# Force-disable TensorFlow in transformers (it crashes on this stack's numpy +# ABI). Assignment (not setdefault) so a stray ``USE_TF=1`` in the shell can't +# re-enable it. Must precede any transformers import; ``ir`` never uses TF. +os.environ["USE_TF"] = "0" + +DEFAULT_MODEL = "all-MiniLM-L6-v2" + +_LIGHT = {"light", "hashing", "hash"} +_LOCAL = {"default", "local", "minilm", "st", "sentence-transformers"} + + +def _hashing(): + import ef + + return ef.HashingEmbedder(), "hashing__dim512" + + +def _sentence_transformers(model_name: str): + import ef + from ef import embedder_adapters + + emb = embedder_adapters.sentence_transformers_embedder(model_name, normalize=True) + return emb, f"st__{model_name}" + + +def _with_cache(emb, model_id: str): + """Wrap *emb* in a disk-cached embedder keyed under this model's cache dir.""" + import ef + + from .config import embeddings_cache_dir + from .store import _ndarray_store + + store = _ndarray_store(embeddings_cache_dir(model_id)) + return ef.CachedEmbedder(emb, store) + + +def make_embedder(spec: Any = "default", *, cache: bool = True) -> tuple[Callable, str]: + """Resolve *spec* to ``(embedder, embedder_id)``. + + ``embedder`` is a batch callable ``Iterable[str] -> ndarray(n, dim)`` that + also accepts ``input_type=`` (``"query"`` / ``"document"``). + """ + import ef + + if callable(spec) and not isinstance(spec, str): + return ef.as_embedder(spec), getattr(spec, "_ir_id", "custom") + + key = (spec or "default").strip().lower() + + if key in _LIGHT: + return _hashing() + + if key in _LOCAL: + model_name = DEFAULT_MODEL + else: + model_name = spec # any other string is a model name + + try: + emb, model_id = _sentence_transformers(model_name) + except Exception as e: # ImportError or model-load failure + warnings.warn( + f"Local embedder {model_name!r} unavailable ({e}); " + f"falling back to the light hashing embedder. " + f"Install with `pip install sentence-transformers` for semantic search.", + stacklevel=2, + ) + return _hashing() + + if cache: + emb = _with_cache(emb, model_id) + return emb, model_id diff --git a/ir/index.py b/ir/index.py new file mode 100644 index 0000000..d3cc241 --- /dev/null +++ b/ir/index.py @@ -0,0 +1,191 @@ +"""The indexing pipeline and incremental maintenance. + +:func:`build` turns a :class:`~ir.sources.CorpusSource` into a queryable +:class:`Corpus`, persisting records through a +:class:`~ir.store.CorpusStore`. It is **incremental and idempotent**: each +artifact's change signal is compared to the ledger, and only new, changed, or +re-modeled artifacts are decomposed and embedded. Artifacts that vanished from +the source are pruned (full-refresh). Re-running ``build`` on an unchanged +source is a near-no-op. + +This is the light maintenance path — content-hash CRUD over a flat store, which +is the right tool for the small corpora ``ir`` targets. The heavier +content-addressed artifact-graph (``ef.artifact_graph``) is the documented +upgrade for corpora large enough that recomputing surfaces is the bottleneck. +""" + +from __future__ import annotations + +import json +from collections import defaultdict +from dataclasses import dataclass +from typing import Any, Callable + +import numpy as np + +from .base import Record, storage_key +from .embed import make_embedder +from .sources import CorpusSource +from .store import CorpusStore + + +def _strategy_id(strategy) -> str: + """Stable id for a strategy: class name + its simple parameters. + + Changing the strategy (or its parameters) changes this id, so an unchanged + corpus rebuilt under a different strategy is correctly re-decomposed rather + than skipped. + """ + params = { + k: v + for k, v in vars(strategy).items() + if isinstance(v, (str, int, float, bool, type(None))) + } + return f"{type(strategy).__name__}:{json.dumps(params, sort_keys=True)}" + + +def _embed(emb: Callable, texts: list[str], input_type: str) -> np.ndarray: + """Call an embedder, tolerating those that don't accept ``input_type``.""" + if not texts: + return np.zeros((0, 0), dtype=np.float32) + try: + out = emb(texts, input_type=input_type) + except TypeError: + out = emb(texts) + return np.asarray(out, dtype=np.float32) + + +def _embed_batched(emb, texts, input_type, batch_size): + out = [] + for i in range(0, len(texts), batch_size): + out.append(_embed(emb, texts[i : i + batch_size], input_type)) + return np.vstack(out) if out else np.zeros((0, 0), dtype=np.float32) + + +@dataclass +class Corpus: + """A built, queryable corpus: a store plus its embedder.""" + + name: str + store: CorpusStore + embedder: Callable + embedder_id: str + + def search(self, query, **kwargs): + from .retrieve import search + + return search(self, query, **kwargs) + + def __len__(self) -> int: + return len(self.store) + + +def build( + source: CorpusSource, + *, + store: CorpusStore | None = None, + embedder: Any = None, + full: bool = True, + batch_size: int = 256, +) -> Corpus: + """Build or incrementally update *source* into a :class:`Corpus`. + + Parameters + ---------- + store : the persistence backend (default: file-backed under XDG data dir). + embedder : override the source's embedder spec. + full : when True (default), prune artifacts no longer in the source. + batch_size : embedding batch size. + """ + store = CorpusStore.local(source.name) if store is None else store + spec = embedder if embedder is not None else source.embedder + emb, emb_id = make_embedder(spec) + strat_id = _strategy_id(source.indexing_strategy) + + seen: set[str] = set() + changed: dict[str, tuple] = {} + + for artifact_id, raw in source.items(): + seen.add(artifact_id) + version = source.change_signal(artifact_id, raw) + key = storage_key(artifact_id) + prev = store.get_ledger_entry(key) + if ( + prev + and prev.get("version") == version + and prev.get("embedder_id") == emb_id + and prev.get("strategy_id") == strat_id + ): + continue # unchanged content + same model + same strategy + meta_extra = source.metadata_of(artifact_id, raw) if source.metadata_of else {} + plan = source.indexing_strategy.decompose(artifact_id, raw, meta_extra) + changed[artifact_id] = (key, version, prev, plan) + + # Embed all surfaces of changed artifacts together (cache makes this cheap). + flat = [ + (artifact_id, i, surface) + for artifact_id, (_k, _v, _p, plan) in changed.items() + for i, surface in enumerate(plan.surfaces) + ] + vectors = _embed_batched(emb, [s.text for _, _, s in flat], "document", batch_size) + + # Replace records of changed artifacts (delete-then-write). + for artifact_id, (_k, _v, prev, _plan) in changed.items(): + if prev: + for rid in prev.get("record_ids", []): + store.delete_record(rid) + + record_ids: dict[str, list[str]] = defaultdict(list) + for (artifact_id, i, surface), vec in zip(flat, vectors): + rid = Record.make_id(artifact_id, surface.kind, i) + plan = changed[artifact_id][3] + metadata = {**plan.filter_fields, **dict(surface.metadata)} + store.put_record( + Record( + id=rid, + artifact_id=artifact_id, + surface_kind=surface.kind, + surface_index=i, + text=surface.text, + vector=vec, + metadata=metadata, + ) + ) + record_ids[artifact_id].append(rid) + + for artifact_id, (key, version, _prev, _plan) in changed.items(): + store.set_ledger_entry( + key, + { + "artifact_id": artifact_id, + "version": version, + "embedder_id": emb_id, + "strategy_id": strat_id, + "record_ids": record_ids.get(artifact_id, []), + }, + ) + + if full: + for key, entry in store.ledger_items(): + if entry.get("artifact_id") not in seen: + for rid in entry.get("record_ids", []): + store.delete_record(rid) + store.delete_ledger_entry(key) + + store.set_config( + { + "name": source.name, + "embedder_spec": spec if isinstance(spec, str) else "custom", + "embedder_id": emb_id, + } + ) + return Corpus(source.name, store, emb, emb_id) + + +def open_corpus(name: str, *, embedder: Any = None) -> Corpus: + """Reopen a previously built corpus by name (resolves its embedder).""" + store = CorpusStore.local(name) + cfg = store.get_config() + spec = embedder if embedder is not None else cfg.get("embedder_spec", "default") + emb, emb_id = make_embedder(spec) + return Corpus(name, store, emb, cfg.get("embedder_id", emb_id)) diff --git a/ir/retrieve.py b/ir/retrieve.py new file mode 100644 index 0000000..3b2db49 --- /dev/null +++ b/ir/retrieve.py @@ -0,0 +1,102 @@ +"""Retrieval — hard metadata filtering + dense brute-force ranking. + +:func:`search` embeds the query, applies a hard metadata filter to narrow the +candidate set (the ``vd`` Mongo-style filter language — ownership, name, tags), +ranks the survivors by cosine similarity (exact brute force; correct and +instant at ``ir``'s corpus sizes), and collapses surface-hits to the best +surface per artifact. + +Hybrid lexical fusion (BM25 + ``vd.reciprocal_rank_fusion``) and cross-encoder +reranking are deliberate next-step seams (see the retrieval issue): dense + +filter is the shippable baseline. +""" + +from __future__ import annotations + +from collections.abc import Iterable, Mapping +from typing import Any + +import numpy as np + +from .base import SearchHit, best_per_artifact + + +def _matches(metadata: Mapping[str, Any], filter: Mapping[str, Any]) -> bool: + """Mongo-style metadata match, via ``vd`` when available.""" + try: + from vd.filters import matches_filter + + return matches_filter(dict(metadata), dict(filter)) + except Exception: + # Minimal equality fallback if vd is unavailable. + return all(metadata.get(k) == v for k, v in filter.items()) + + +def _embed_query(embedder, query: str) -> np.ndarray: + from .index import _embed + + vec = _embed(embedder, [query], "query")[0] + norm = float(np.linalg.norm(vec)) + return vec / norm if norm else vec + + +def search( + corpus, + query: str, + *, + k: int = 10, + filter: Mapping[str, Any] | None = None, + surfaces: Iterable[str] | None = None, + per_artifact: bool = True, +) -> list[SearchHit]: + """Return the top-*k* :class:`~ir.base.SearchHit` for *query*. + + Parameters + ---------- + k : number of results. + filter : a ``vd`` Mongo-style filter over record metadata (hard filter). + surfaces : restrict to these surface kinds (e.g. ``{"description"}``). + per_artifact : collapse to the best surface per artifact (default True). + """ + ids, mat, metas = corpus.store.matrix() + if not ids: + return [] + + surface_set = set(surfaces) if surfaces is not None else None + candidates = [] + for j, m in enumerate(metas): + if surface_set is not None and m["surface_kind"] not in surface_set: + continue + if filter is not None and not _matches(m.get("metadata", {}), filter): + continue + candidates.append(j) + if not candidates: + return [] + + qv = _embed_query(corpus.embedder, query) + if qv.shape[0] != mat.shape[1]: + raise ValueError( + f"Query embedding dim {qv.shape[0]} != index dim {mat.shape[1]}. " + f"The corpus was built with a different embedder than the one " + f"querying it; rebuild the corpus or use its original embedder." + ) + sub = mat[candidates] + scores = sub @ qv + + # Over-fetch before collapsing to artifacts so dedupe doesn't starve top-k. + take = min(len(candidates), max(k * 5, 50) if per_artifact else k) + order = np.argsort(-scores)[:take] + + hits = [ + SearchHit( + artifact_id=metas[candidates[o]]["artifact_id"], + surface_kind=metas[candidates[o]]["surface_kind"], + score=float(scores[o]), + text=metas[candidates[o]]["text"], + metadata=metas[candidates[o]].get("metadata", {}), + ) + for o in order + ] + if per_artifact: + hits = best_per_artifact(hits) + return hits[:k] diff --git a/ir/sources.py b/ir/sources.py new file mode 100644 index 0000000..5f1d3a9 --- /dev/null +++ b/ir/sources.py @@ -0,0 +1,305 @@ +"""Defining a corpus source — an abstract strategy plus parameters. + +A :class:`CorpusSource` says *what* a corpus is and *how to keep it current*, +independent of how it is indexed or stored: + +- ``scope`` — a ``Mapping[id -> raw]`` enumerating the corpus (a dict, + a ``dol`` file store, or any mapping). This is the "what is in the corpus" slot. +- ``change_signal`` — ``(id, raw) -> version_str``; default is the content + hash. This is the "what counts as stale" slot, driving incremental reindex. +- ``indexing_strategy``— how a raw item becomes filter fields + surfaces. +- ``embedder`` — embedder spec (default: the decent local model). +- ``metadata_of`` — optional ``(id, raw) -> dict`` of extra filter metadata. + +Smart-default constructors cover the common ways to define a source: +:meth:`from_mapping`, :meth:`from_files`, :meth:`from_md_reports`, +:meth:`from_skills`, :meth:`from_packages`. +""" + +from __future__ import annotations + +import os +import re +from collections.abc import Callable, Mapping +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any + +from .strategy import Chunked, IndexingStrategy, Package, Skill, WholeText + +ALLCAPS_MD = re.compile(r"^[A-Z0-9_ ]+\.md$") + + +def content_hash_signal(artifact_id: str, raw: Any) -> str: + """Default change signal: a content hash of the raw payload.""" + import ef + + return ef.content_hash(raw) + + +@dataclass +class CorpusSource: + """A corpus definition: scope + change signal + strategy + embedder.""" + + name: str + scope: Mapping[str, Any] + indexing_strategy: IndexingStrategy = field(default_factory=WholeText) + change_signal: Callable[[str, Any], str] = content_hash_signal + embedder: Any = "default" + metadata_of: Callable[[str, Any], Mapping[str, Any]] | None = None + + def items(self): + return self.scope.items() + + # ----- smart-default constructors ------------------------------------ # + + @classmethod + def from_mapping( + cls, + mapping: Mapping[str, Any], + *, + name: str, + strategy: IndexingStrategy | None = None, + **kwargs, + ) -> "CorpusSource": + """Any mapping ``{id -> raw}`` (dict, ``dol`` store) as a corpus.""" + return cls( + name=name, + scope=mapping, + indexing_strategy=strategy or WholeText(), + **kwargs, + ) + + @classmethod + def from_files( + cls, + root: str | Path, + *, + name: str | None = None, + pattern: str = r".*\.md$", + exclude: Callable[[str], bool] | None = None, + strategy: IndexingStrategy | None = None, + **kwargs, + ) -> "CorpusSource": + """A directory tree of text files as a corpus (lazy ``dol`` scope).""" + import dol + + root = Path(root).expanduser() + keep = re.compile(pattern) + store = dol.TextFiles(str(root)) + + def predicate(k: str) -> bool: + if not keep.search(k): + return False + return not (exclude and exclude(k)) + + scope = dol.filt_iter(store, filt=predicate) + + def metadata_of(aid, raw): + return {"path": str(root / aid), "filename": os.path.basename(aid)} + + return cls( + name=name or root.name, + scope=scope, + indexing_strategy=strategy or Chunked(), + metadata_of=metadata_of, + **kwargs, + ) + + @classmethod + def from_md_reports( + cls, + *, + name: str = "reports", + projects_root: str | Path | None = None, + strategy: IndexingStrategy | None = None, + **kwargs, + ) -> "CorpusSource": + """Markdown reports under projects' ``docs/`` and ``misc/docs/``. + + Excludes ALL-CAPS filenames (README/CLAUDE/MEMORY/SKILL...). Each record + is a project-tagged document; ids are paths relative to the projects root. + """ + root = Path(projects_root or _projects_root()) + scope: dict[str, dict] = {} + meta: dict[str, dict] = {} + for path in _iter_md_reports(root): + rel = str(path.relative_to(root)) + try: + text = path.read_text(encoding="utf-8", errors="ignore") + except OSError: + continue + if not text.strip(): + continue + scope[rel] = {"text": text} + meta[rel] = { + "project": _project_of(rel), + "path": str(path), + "filename": path.name, + } + + return cls( + name=name, + scope=scope, + indexing_strategy=strategy or Chunked(), + metadata_of=lambda aid, raw: meta.get(aid, {}), + **kwargs, + ) + + @classmethod + def from_skills( + cls, + *, + name: str = "skills", + filter: Any = None, + fetcher: Callable[[], list] | None = None, + strategy: IndexingStrategy | None = None, + **kwargs, + ) -> "CorpusSource": + """The agent-skills corpus, via ``priv.skills_index``. + + ``fetcher`` overrides the source of skill records (each a mapping with + ``name``/``description``/``parent``) — inject a test double to avoid the + ``priv`` dependency. + """ + if fetcher is not None: + records = list(fetcher()) + else: + from priv.skills_index import skills_index as _skills_index + + records = _skills_index(filter=filter, egress="raw") + # Preserve same-named skills from different packages (collision-safe ids). + scope: dict[str, dict] = {} + for r in records: + key = r["name"] + if key in scope: + key = f"{r['name']}@{r.get('parent')}" + scope[key] = r + + def metadata_of(aid, raw): + return { + "parent": raw.get("parent"), + "skill_path": raw.get("skill_path"), + "base_path": raw.get("base_path"), + } + + return cls( + name=name, + scope=scope, + indexing_strategy=strategy or Skill(), + metadata_of=metadata_of, + **kwargs, + ) + + @classmethod + def from_packages( + cls, + *, + name: str = "packages", + manifest: str | Path | None = None, + readme_chars: int = 20000, + strategy: IndexingStrategy | None = None, + **kwargs, + ) -> "CorpusSource": + """The local package ecosystem, scanned from the ``.pth`` manifest.""" + scope = _scan_packages(manifest, readme_chars=readme_chars) + + def metadata_of(aid, raw): + return {"path": raw.get("path")} + + return cls( + name=name, + scope=scope, + indexing_strategy=strategy or Package(), + metadata_of=metadata_of, + **kwargs, + ) + + +# --------------------------------------------------------------------------- # +# Source-specific scope helpers +# --------------------------------------------------------------------------- # + + +def _projects_root() -> Path: + """Locate the projects folder (``$PP``), preferring ``priv.config``.""" + try: + from priv.config import projects_folder + + return Path(projects_folder()) + except Exception: + pp = os.environ.get("PP") + if pp: + return Path(pp).expanduser() + raise RuntimeError( + "Cannot locate the projects folder; set $PP or install priv.config." + ) + + +def _iter_md_reports(root: Path): + """Yield non-ALLCAPS ``*.md`` files under projects' docs/ and misc/docs/.""" + for sub in root.glob("*/*/docs"): + yield from _md_in(sub) + for sub in root.glob("*/*/misc/docs"): + yield from _md_in(sub) + + +def _md_in(folder: Path): + if not folder.is_dir(): + return + for f in folder.glob("*.md"): + if f.is_file() and not ALLCAPS_MD.match(f.name): + yield f + + +def _project_of(rel_path: str) -> str: + """``i/dol/docs/x.md`` -> ``dol`` (the package folder name).""" + parts = Path(rel_path).parts + return parts[1] if len(parts) >= 2 else (parts[0] if parts else "") + + +def _manifest_paths(manifest: str | Path | None) -> list[Path]: + """Absolute package directories listed in the ``.pth`` manifest.""" + manifest = manifest or os.environ.get("PTH_FILEPATH") + if not manifest: + raise RuntimeError("No manifest given and $PTH_FILEPATH is unset.") + paths = [] + for line in Path(manifest).read_text().splitlines(): + line = line.strip() + if line.startswith("/") and Path(line).is_dir(): + paths.append(Path(line)) + return paths + + +def _scan_packages(manifest, *, readme_chars: int) -> dict[str, dict]: + """Build ``{package_name: {name, description, readme, owner, deps, path}}``.""" + import tomllib + + scope: dict[str, dict] = {} + for path in _manifest_paths(manifest): + name = path.name + description, deps = "", [] + pyproject = path / "pyproject.toml" + if pyproject.is_file(): + try: + data = tomllib.loads(pyproject.read_text(encoding="utf-8")) + proj = data.get("project", {}) + description = proj.get("description", "") or "" + deps = list(proj.get("dependencies", []) or []) + except Exception: + pass + readme = "" + for cand in ("README.md", "README.rst", "README.txt"): + rp = path / cand + if rp.is_file(): + readme = rp.read_text(encoding="utf-8", errors="ignore")[:readme_chars] + break + scope[name] = { + "name": name, + "description": description, + "readme": readme, + "owner": "ours", + "deps": deps, + "path": str(path), + } + return scope diff --git a/ir/store.py b/ir/store.py new file mode 100644 index 0000000..feccf10 --- /dev/null +++ b/ir/store.py @@ -0,0 +1,178 @@ +"""Persistence for ``ir`` — the repository layer over ``dol`` key-value views. + +A :class:`CorpusStore` bundles three ``MutableMapping`` views, so *where* and +*how* data is persisted is swappable without touching the rest of ``ir``: + +- ``meta`` : ``record_id -> dict`` (text + metadata + filter fields), JSON. +- ``vectors``: ``record_id -> ndarray`` (the embedding), numpy bytes. +- ``ledger`` : ``artifact_id -> dict`` (version, embedder id, record ids) — + drives incremental maintenance. +- ``config`` : ``key -> dict`` (one entry: the corpus build settings). + +The default factory :meth:`CorpusStore.local` roots all four under +``~/.local/share/ir/corpora/`` via ``dol`` file stores; +:meth:`CorpusStore.memory` gives a dependency-free in-memory store for tests. +Brute-force search reads vectors into a single normalized matrix +(:meth:`matrix`), cached in-process and invalidated on writes. +""" + +from __future__ import annotations + +import io +import os +from collections.abc import Iterator, Mapping, MutableMapping +from typing import Any + +import numpy as np + +from .base import Record + + +def _ndarray_store(rootdir) -> MutableMapping[str, np.ndarray]: + """A ``dol`` file store whose values are float32 ``ndarray``s.""" + import dol + + rootdir = str(rootdir) + os.makedirs(rootdir, exist_ok=True) + files = dol.mk_dirs_if_missing(dol.Files(rootdir)) + + def encode(arr: np.ndarray) -> bytes: + buf = io.BytesIO() + np.save(buf, np.asarray(arr, dtype=np.float32), allow_pickle=False) + return buf.getvalue() + + def decode(data: bytes) -> np.ndarray: + return np.load(io.BytesIO(data), allow_pickle=False) + + return dol.wrap_kvs(files, obj_of_data=decode, data_of_obj=encode) + + +def _json_store(rootdir) -> MutableMapping[str, Any]: + """A ``dol`` file store whose values are JSON objects.""" + import dol + + rootdir = str(rootdir) + os.makedirs(rootdir, exist_ok=True) + return dol.mk_dirs_if_missing(dol.JsonFiles(rootdir)) + + +class CorpusStore: + """Repository bundling the meta/vectors/ledger/config views of one corpus.""" + + def __init__( + self, + meta: MutableMapping[str, Any], + vectors: MutableMapping[str, np.ndarray], + ledger: MutableMapping[str, Any], + config: MutableMapping[str, Any], + ): + self.meta = meta + self.vectors = vectors + self.ledger = ledger + self.config = config + self._matrix_cache: tuple | None = None + + # ----- factories ------------------------------------------------------ # + + @classmethod + def local(cls, name: str) -> "CorpusStore": + """File-backed store under ``~/.local/share/ir/corpora/``.""" + from .config import corpus_dir + + root = corpus_dir(name) + return cls( + meta=_json_store(root / "meta"), + vectors=_ndarray_store(root / "vectors"), + ledger=_json_store(root / "ledger"), + config=_json_store(root / "config"), + ) + + @classmethod + def memory(cls) -> "CorpusStore": + """In-memory store (no dependencies); ideal for tests.""" + return cls(meta={}, vectors={}, ledger={}, config={}) + + # ----- record CRUD ---------------------------------------------------- # + + def put_record(self, record: Record) -> None: + self.meta[record.id] = { + "artifact_id": record.artifact_id, + "surface_kind": record.surface_kind, + "surface_index": record.surface_index, + "text": record.text, + "metadata": dict(record.metadata), + } + self.vectors[record.id] = np.asarray(record.vector, dtype=np.float32) + self._matrix_cache = None + + def delete_record(self, record_id: str) -> None: + self.meta.pop(record_id, None) + try: + del self.vectors[record_id] + except KeyError: + pass + self._matrix_cache = None + + def record_ids(self) -> Iterator[str]: + return iter(self.meta) + + def get_record(self, record_id: str) -> Record: + m = self.meta[record_id] + return Record( + id=record_id, + artifact_id=m["artifact_id"], + surface_kind=m["surface_kind"], + surface_index=m["surface_index"], + text=m["text"], + vector=np.asarray(self.vectors[record_id], dtype=np.float32), + metadata=m.get("metadata", {}), + ) + + def __len__(self) -> int: + return len(self.meta) + + # ----- ledger --------------------------------------------------------- # + + def get_ledger_entry(self, key: str) -> dict | None: + return self.ledger.get(key) + + def set_ledger_entry(self, key: str, entry: Mapping[str, Any]) -> None: + self.ledger[key] = dict(entry) + + def delete_ledger_entry(self, key: str) -> None: + self.ledger.pop(key, None) + + def ledger_items(self) -> Iterator[tuple[str, dict]]: + # Materialize to a list so callers may mutate the ledger while iterating. + return iter(list(self.ledger.items())) + + # ----- config --------------------------------------------------------- # + + def get_config(self) -> dict: + return dict(self.config.get("config", {})) + + def set_config(self, settings: Mapping[str, Any]) -> None: + self.config["config"] = dict(settings) + + # ----- search matrix -------------------------------------------------- # + + def matrix(self) -> tuple[list[str], np.ndarray, list[dict]]: + """Return ``(record_ids, normalized_matrix, metas)`` for brute force. + + Rows are L2-normalized so cosine similarity is a dot product. Empty + corpora return a ``(0, 0)`` matrix. Cached until the next write. + """ + if self._matrix_cache is not None: + return self._matrix_cache + ids = list(self.meta) + if not ids: + self._matrix_cache = ([], np.zeros((0, 0), dtype=np.float32), []) + return self._matrix_cache + rows = [np.asarray(self.vectors[rid], dtype=np.float32) for rid in ids] + mat = np.vstack(rows) + norms = np.linalg.norm(mat, axis=1, keepdims=True) + norms[norms == 0] = 1.0 + mat = mat / norms + metas = [self.meta[rid] for rid in ids] + self._matrix_cache = (ids, mat, metas) + return self._matrix_cache diff --git a/ir/strategy.py b/ir/strategy.py new file mode 100644 index 0000000..6f9ffe2 --- /dev/null +++ b/ir/strategy.py @@ -0,0 +1,212 @@ +"""Indexing strategies — the "what do we index?" seam. + +An :class:`IndexingStrategy` decomposes one artifact into an +:class:`~ir.base.IndexPlan`: the ``filter_fields`` (hard-filterable metadata, +*not* embedded) and a list of :class:`~ir.base.Surface` (embeddable units). This +is the central extensibility point of ``ir``: a naive corpus uses +:class:`WholeText`; a structured corpus (a package) decomposes into several +heterogeneous surfaces so a query can match the *right part* of an artifact, +and constrains candidates by metadata *before* semantic ranking. + +Shipped strategies: + +- :class:`WholeText` — one surface = the whole text. The out-of-the-box default. +- :class:`Chunked` — split the text into overlapping chunks (one surface each). +- :class:`Skill` — embed ``name + description`` only (the body stays on disk, + per the capability-discovery research); name/parent become filter fields. +- :class:`Package` — ``name + description`` plus README chunks as surfaces; + name/owner/deps become filter fields (AI synopsis / problem-class surfaces + are a documented extension). + +Every strategy is a plain callable-ish object with a ``decompose`` method, so +custom strategies need only match the :class:`IndexingStrategy` protocol. +""" + +from __future__ import annotations + +import re +from collections.abc import Mapping +from typing import Any, Protocol, runtime_checkable + +from .base import IndexPlan, Surface + + +@runtime_checkable +class IndexingStrategy(Protocol): + """Decompose one artifact into filter fields + embeddable surfaces.""" + + def decompose( + self, artifact_id: str, raw: Any, metadata: Mapping[str, Any] | None = None + ) -> IndexPlan: ... + + +def _text_of(raw: Any, text_key: str | None = None) -> str: + """Best-effort text extraction from a raw artifact payload.""" + if isinstance(raw, str): + return raw + if isinstance(raw, Mapping): + if text_key is not None: + return str(raw.get(text_key, "") or "") + if "text" in raw: + return str(raw.get("text", "") or "") + # join string-valued fields as a fallback + return "\n".join(str(v) for v in raw.values() if isinstance(v, str)) + return str(raw) + + +def _split(text: str, *, chunk_size: int, overlap: int) -> list[str]: + """Paragraph-packing chunker: greedily fill ~``chunk_size`` chunks. + + Splits on blank lines, then packs whole paragraphs up to ``chunk_size`` + (carrying an ``overlap`` tail between chunks). Paragraphs longer than + ``chunk_size`` are hard-split. This packs to a target size rather than + emitting one chunk per paragraph. + """ + text = text.strip() + if not text: + return [] + if len(text) <= chunk_size: + return [text] + + paras = [p.strip() for p in re.split(r"\n\s*\n", text) if p.strip()] + chunks: list[str] = [] + cur = "" + for para in paras: + if len(para) > chunk_size: # hard-split an oversized paragraph + if cur: + chunks.append(cur) + cur = "" + step = max(1, chunk_size - overlap) + chunks.extend(para[i : i + chunk_size] for i in range(0, len(para), step)) + continue + if cur and len(cur) + 2 + len(para) > chunk_size: + chunks.append(cur) + tail = cur[-overlap:] if overlap else "" + cur = f"{tail}\n\n{para}".strip() if tail else para + else: + cur = f"{cur}\n\n{para}".strip() if cur else para + if cur: + chunks.append(cur) + return chunks + + +class WholeText: + """One surface = the entire text. Sensible default for a naive corpus.""" + + def __init__(self, *, text_key: str | None = None, kind: str = "document"): + self.text_key = text_key + self.kind = kind + + def decompose(self, artifact_id, raw, metadata=None) -> IndexPlan: + meta = dict(metadata or {}) + text = _text_of(raw, self.text_key) + surfaces = ( + [Surface(artifact_id, self.kind, text, granularity="document")] + if text.strip() + else [] + ) + return IndexPlan(filter_fields=meta, surfaces=surfaces) + + +class Chunked: + """Split the artifact's text into overlapping chunk surfaces.""" + + def __init__( + self, + *, + chunk_size: int = 1200, + overlap: int = 200, + text_key: str | None = None, + kind: str = "chunk", + ): + self.chunk_size = chunk_size + self.overlap = overlap + self.text_key = text_key + self.kind = kind + + def decompose(self, artifact_id, raw, metadata=None) -> IndexPlan: + meta = dict(metadata or {}) + text = _text_of(raw, self.text_key) + chunks = _split(text, chunk_size=self.chunk_size, overlap=self.overlap) + surfaces = [ + Surface( + artifact_id, + self.kind, + chunk, + granularity="chunk", + metadata={"chunk_index": i, "n_chunks": len(chunks)}, + ) + for i, chunk in enumerate(chunks) + ] + return IndexPlan(filter_fields=meta, surfaces=surfaces) + + +class Skill: + """Capability strategy: embed ``name + description`` only. + + The body (SKILL.md) is loaded post-selection and is *not* indexed; name and + parent are filter fields. + """ + + def decompose(self, artifact_id, raw, metadata=None) -> IndexPlan: + meta = dict(metadata or {}) + raw = raw if isinstance(raw, Mapping) else {"name": str(raw)} + name = str(raw.get("name", artifact_id)) + description = str(raw.get("description", "") or "") + filter_fields = { + "name": name, + "parent": raw.get("parent"), + **meta, + } + text = f"{name}\n\n{description}".strip() + surfaces = [Surface(artifact_id, "capability", text, granularity="field")] + return IndexPlan(filter_fields=filter_fields, surfaces=surfaces) + + +class Package: + """Package strategy: ``name + description`` surface plus README chunks. + + Filter fields capture ownership (ours vs third-party), name, deps. AI + synopsis / problem-class surfaces are a documented extension point. + """ + + def __init__(self, *, chunk_size: int = 1500, overlap: int = 200): + self.chunk_size = chunk_size + self.overlap = overlap + + def decompose(self, artifact_id, raw, metadata=None) -> IndexPlan: + meta = dict(metadata or {}) + raw = raw if isinstance(raw, Mapping) else {"name": str(raw)} + name = str(raw.get("name", artifact_id)) + description = str(raw.get("description", "") or "") + readme = str(raw.get("readme", "") or "") + filter_fields = { + "name": name, + "owner": raw.get("owner"), + "has_readme": bool(readme.strip()), + "deps": list(raw.get("deps", []) or []), + **meta, + } + surfaces = [ + Surface( + artifact_id, + "description", + f"{name}: {description}".strip(": ").strip(), + granularity="field", + ) + ] + for i, chunk in enumerate( + _split(readme, chunk_size=self.chunk_size, overlap=self.overlap) + ): + surfaces.append( + Surface( + artifact_id, + "readme_chunk", + chunk, + granularity="chunk", + metadata={"chunk_index": i}, + ) + ) + # Drop empty surfaces (e.g. no description and no README). + surfaces = [s for s in surfaces if s.text.strip()] + return IndexPlan(filter_fields=filter_fields, surfaces=surfaces) diff --git a/misc/docs/ir_01 -- Progressive Capability Discovery for AI Agents -- Give the Agent One Search Tool.md.md b/misc/docs/ir_01 -- Progressive Capability Discovery for AI Agents -- Give the Agent One Search Tool.md.md new file mode 100644 index 0000000..29086af --- /dev/null +++ b/misc/docs/ir_01 -- Progressive Capability Discovery for AI Agents -- Give the Agent One Search Tool.md.md @@ -0,0 +1,206 @@ +# Progressive Capability Discovery for AI Agents: Give the Agent One Search Tool + +*A comparative landscape survey of tool, skill, and subagent discovery architectures (2024 H2 – 2026)* + +**Author: Thor Whalen** + +--- + +## TL;DR + +- **The single-search-tool pattern is now validated by primary-source numbers, not just intuition.** Anthropic's own internal testing shows tool-selection accuracy on MCP evaluations jumping from 49% to 74% (Opus 4) and 79.5% to 88.1% (Opus 4.5) when its Tool Search Tool is enabled, with an 85% token reduction; independent papers (RAG-MCP, Toolshed, ScaleMCP) show 3× accuracy gains and >50% token cuts. The pattern is real and worth adopting above ~30–50 tools. +- **Architecture should be staged, not maximal.** Single-shot semantic retrieval + a reranker is the right default; the planner→fan-out→selector multi-stage pipeline (Toolshed's "Advanced RAG-Tool Fusion") buys measurable recall gains but is over-engineered for most catalogs under a few hundred tools. Retrieval — not the LLM's final selection — is the dominant failure mode (LiveMCPBench: "retrieval errors account for nearly half of all failures"). +- **Tools, skills, and subagents share one discovery primitive (metadata-first progressive disclosure) but diverge at the execution boundary.** A unified index can serve all three for *retrieval*; the difference is what gets loaded (a schema, a SKILL.md document, or an agent card). Watch the real gotchas: tool-description quality is the actual bottleneck, dynamic tool injection silently breaks prompt caching, and dynamic loading opens rug-pull/tool-poisoning attack surface. + +--- + +## Key Findings + +1. **Context degradation is the underlying physics.** Chroma's 2025 "context rot" study (Hong, Troynikov & Huber, July 14, 2025) evaluated 18 LLMs "including the state-of-the-art GPT-4.1, Claude 4, Gemini 2.5, and Qwen3 models" and found "their performance grows increasingly unreliable as input length grows" — well before the window is full — with distractors having outsized, non-uniform impact. This is *why* loading 50–100 tool schemas hurts: it is not just token cost, it is signal-to-noise collapse. + +2. **The "49% baseline" family of findings is robust and converging.** RAG-MCP (arXiv 2505.03275, Gan & Sun) "significantly cuts prompt tokens (e.g., by over 50%) and more than triples tool selection accuracy (43.13% vs 13.62% baseline)." Anthropic's production numbers (49%→74%) corroborate the same shape. The breakpoint where the pattern starts paying off is widely cited at 30–50 tools. + +3. **Retrieval is the bottleneck, not selection.** LiveMCPBench (70 servers, 527 tools) finds Claude-Sonnet-4 at 78.95% task success while "most models achieve only 30-50%," and "retrieval errors account for nearly half of all failures—highlighting retrieval as the dominant bottleneck." MCP-Universe (Salesforce, 231 tasks) finds even the best model (GPT-5, 43.72%) fails the majority of realistic multi-tool tasks. + +4. **Skills and the single-search-tool pattern are the same idea at different layers.** Anthropic's Agent Skills (SKILL.md) implement three-level progressive disclosure — metadata (~30–50 tokens) → SKILL.md body (<5k tokens) → bundled files — which is exactly the defer-then-load shape of Tool Search. + +5. **Prompt caching is the hidden cost of dynamic loading.** Anthropic's cache hierarchy is tools → system → messages; changing any tool definition invalidates the entire downstream cache. Naively injecting discovered tools mid-conversation therefore destroys cache hits. Pydantic AI's native tool-search path is explicitly designed to avoid this by keeping discovery append-only. + +6. **Security is underweighted.** Dynamic tool registries enable rug-pull and tool-poisoning attacks (Invariant Labs, OWASP GenAI). The more dynamic your discovery, the larger the attack surface — and the more you need signed/versioned tool definitions (ETDI) and host-side allow-listing. + +--- + +## Details + +### 1. The single-search-tool pattern in practice + +**Anthropic Tool Search Tool + `defer_loading`.** Released in beta on November 24, 2025 as part of "advanced tool use," the pattern is: provide all tool definitions to the API but mark most with `defer_loading: true`. Deferred tools are not loaded into context; Claude sees only the search tool plus your 3–5 always-on tools. Two variants ship: `tool_search_tool_regex_20251119` (Python `re.search` patterns) and `tool_search_tool_bm25_20251119` (natural-language BM25). Anthropic also documents a client-side embeddings path using `tool_reference` content blocks, so you can swap in your own vector retriever while keeping the same wire protocol. + +The measured impact, from Anthropic's own internal testing (first-party, self-reported — treat as vendor claims): tool-selection accuracy on MCP evaluations rose from **49% to 74% on Opus 4** and **79.5% to 88.1% on Opus 4.5**, with the search tool preserving **191,300 tokens vs 122,800** under the traditional approach — an **85% reduction in token usage**. Anthropic notes it has "seen tool definitions consume 134K tokens before optimization," and a representative five-server setup is 58 tools / ~55K tokens before the conversation begins. + +The companion **Programmatic Tool Calling** lets Claude orchestrate tools in a code-execution sandbox so intermediate results never hit context: average usage "dropped from 43,588 to 27,297 tokens, a 37% reduction on complex research tasks," with accuracy on internal knowledge retrieval improving 25.6%→28.5% and GAIA-style benchmarks 46.5%→51.2%. The related **code-execution-with-MCP** pattern reports a headline case of 150,000 → ~2,000 tokens (98.7% reduction) by turning MCP servers into code-level APIs. + +**Independent corroboration on production readiness is mixed.** One marketing-automation practitioner (Growth Method) reported only ~60% retrieval accuracy in their own testing and judged it "not production-ready" for high-stakes actions — a useful counterweight to vendor numbers. + +**MCP progressive disclosure proposals.** Importantly, progressive disclosure is *not* in the MCP spec — it is a pattern layered on top of `tools/list`, `tools/call`, and `notifications/tools/list_changed`. Active proposals: **SEP-1888** (Harshal Patil's "Progressive Disclosure for Typed Library Discovery & Introspection," reference impl ProDisco) proposes a standard `.searchTools` meta-tool with `operations`/`types` modes; **SEP-1576** proposes embedding-based host-side tool filtering; and a community MCP "Progressive Disclosure" extension moves full tool descriptions into MCP resources fetched on demand. A widely cited MCP discussion measured a MySQL MCP server with 106 tools consuming 207KB / ~54,600 tokens on every initialization. Claude Code began rolling out MCP Tool Search (v2.1.7) that auto-triggers when MCP tool descriptions would exceed 10% of context. + +**Equivalent patterns across ecosystems:** +- **OpenAI Agents SDK** added hosted **Tool Search** (`ToolSearchTool()`, requires `openai>=2.25.0`): mark function tools `defer_loading=True`, group with `tool_namespace(...)`, or defer `HostedMCPTool`. OpenAI's official guidance is "use namespaces where possible" rather than many individually deferred functions. +- **LangGraph** ships **`langgraph-bigtool`**, a library that stores tool metadata in LangGraph's long-term-memory store (in-memory or Postgres) and equips the agent with a `retrieve_tools` search tool; LangChain demonstrated it working reliably with >50 tools on local models via Ollama. +- **Pydantic AI** exposes `defer_loading=True` on tools/toolsets and `.defer_loading()` on MCP servers and `FastMCPToolset`, with an auto-injected `ToolSearch` capability that uses native provider search (Anthropic, OpenAI Responses) when available and a custom callable strategy otherwise. +- **smolagents** is tool-agnostic (MCP, LangChain, Hub Spaces as tools) but has no built-in semantic tool-retrieval layer — discovery is left to the developer. + +### 2. Architecture of the search tool itself + +Four reference architectures, in increasing complexity: + +**(a) Flat single-shot semantic retrieval.** Embed tool docs once, embed the query, return top-k. This is RAG-MCP's core design and `langgraph-bigtool`'s default. Cheapest, lowest latency, and already captures most of the gain — RAG-MCP's jump to 43.13% accuracy (from a 13.62% blank-conditioning baseline and 18.20% actual-match), while cutting average prompt tokens to 1,084 from 2,133.84, comes from this alone. The verdict from the evidence: this is the correct default for catalogs up to a few hundred tools. + +**(b) Retrieve-then-rerank.** Add a cross-encoder/LLM reranker over the top-k candidates. Toolshed (arXiv 2410.14594) folds this into "post-retrieval" corrective strategies. Worth the extra latency when tool descriptions are similar/confusable (the `notification-send-user` vs `notification-send-channel` problem Anthropic flags). + +**(c) Agentic/iterative search (ReAct-style).** The agent searches, inspects, optionally re-searches. This is LiveMCPBench's MCP Copilot Agent and Pydantic AI's "search → discover → call → search again" loop. Necessary when a single query cannot express the need (multi-hop tasks), but it reintroduces latency and per-iteration inference cost. + +**(d) Planner → fan-out → selector decomposition.** A query-planner subagent decomposes the request, parallel searches fan out, selector/filter subagents read enriched candidate results, and a manager makes the final selection. This is the spirit of Toolshed's "Advanced RAG-Tool Fusion" (pre-/intra-/post-retrieval ensemble), which reports **46%/56%/47% absolute Recall@5 improvements** on ToolE-single, ToolE-multi, and Seal-Tools respectively — without fine-tuning. HGMF (arXiv 2508.07602) offers a cheaper middle path: hierarchical Gaussian-mixture clustering of servers then tools. + +**Verdict (opinionated):** The multi-stage pipeline is *state of the art on recall metrics* but *over-engineered relative to single-shot + rerank* for the typical practitioner. The marginal recall gains are real but come at multiplied latency, token cost, and operational complexity (multiple models, multiple failure points). Adopt (a)+(b) first; escalate to (c) only for genuinely multi-hop tasks; reserve (d) for very large (thousands of tools) catalogs where Recall@5 is business-critical. ScaleMCP's contribution is orthogonal and arguably more valuable than pipeline depth: its **Tool Document Weighted Average (TDWA)** embedding strategy and CRUD auto-synchronization (MCP servers as single source of truth) address index *freshness*, which matters more in production than squeezing recall. + +### 3. Selection vs. retrieval distinction + +Once candidates are retrieved, final selection strategies and their failure modes: + +- **Top-k cutoff.** Simplest; Toolshed explicitly tunes top-k against the tool-count (tool-M) to trade recall vs. token cost. Failure mode: a fixed k either over-selects (distractors) or misses the needed tool when the right tool ranks k+1. +- **LLM-as-selector.** The model reasons over retrieved candidates and commits. Failure mode: distractor tools. The "Tool Preferences in Agentic LLMs are Unreliable" paper (arXiv 2505.18135) shows selection can be swung by trivial description edits — assertive phrasing alone disproportionately boosts a tool's selection rate, making the protocol "exploitable." BiasBusters (arXiv 2510.00307) documents the same metadata/ordering bias. +- **Confidence thresholds.** Return only candidates above a similarity threshold. Failure mode: brittle calibration across query types (Chroma shows needle-question similarity has non-uniform effects). +- **Learned routers / intent classification.** GeckOpt maps tasks to intents offline then restricts the toolset; AutoTool (arXiv 2511.14650) argues many tool selections are patterned enough to bypass full LLM reasoning. Failure mode: requires maintenance as the catalog drifts. + +The empirically grounded conclusion: **the distractor problem is the central selection risk**, and it is worsened, not solved, by retrieving more candidates. MCP-Atlas (Scale AI) deliberately injects 5–10 distractors per task alongside 3–7 target tools precisely to measure this. Fewer, higher-precision candidates beat more candidates. + +### 4. Tools vs. skills vs. subagents — different discovery strategies? + +The three capability types: +- **Tool** = a typed callable with a JSON schema (name, description, input schema). +- **Skill** = a document folder (SKILL.md + optional scripts/references/assets) following the agentskills.io open standard. +- **Subagent** = a delegatable capability with its own context window, discovered via registries or A2A agent cards. + +**Discovery primitive is shared; execution boundary differs.** All three use *metadata-first progressive disclosure*: load a lightweight descriptor (tool name+description ~200–800 tokens; SKILL.md frontmatter ~30–50 tokens; A2A agent card JSON) and load the heavy payload only on activation. A single vector index *can* serve retrieval across all three — you are embedding descriptions either way. + +Where they diverge: +- **Skills** load a *document* into context (the SKILL.md body, then referenced files), and can bundle executable scripts the agent runs without reading into context. Discovery is fundamentally about *when to read more*, governed entirely by the quality of the name+description. The format is an open standard with **32 adopters as of March 2026** per agentskills.io — Anthropic published the spec on December 18, 2025, and within 48 hours Microsoft (VS Code) and OpenAI (ChatGPT, Codex CLI) added support; adopters also include Google (Gemini CLI), JetBrains (Junie), Sourcegraph (Amp), Block (Goose), Snowflake, Databricks, ByteDance, Mistral AI, and Spring AI. It is now governed under the Linux Foundation's Agentic AI Foundation (AAIF), which had grown to 146 member organizations by February 2026 — important reassurance for lock-in-averse readers. +- **Tools** load a *schema* the model must populate correctly; the discovery payoff is both context savings and parameterization accuracy. +- **Subagents** are delegated to, not loaded. **A2A** (Agent2Agent, now under the A2A Project) standardizes the **Agent Card** — a JSON descriptor with identity, service endpoint, capabilities, skills, and auth schemes — discoverable via a well-known URI or a central **registry/catalog** that clients query by skill/tag/capability. Delegation routing in the OpenAI Agents SDK is via **handoffs** (represented to the LLM as `transfer_to_` tools, with `handoff_description` as the discovery hint) vs. **agent-as-tool** (the manager retains control and calls subagents as tools). This is the cleanest illustration that *subagent discovery is just tool discovery with a different execution semantics* — a handoff is literally surfaced to the model as a tool. + +**Verdict:** Use one unified retrieval index for discovery, but keep three distinct loaders. Treat subagent discovery as a first-class peer: register agent cards in the same searchable catalog as tools and skills, route via handoff-as-tool, and let the manager's final selection logic be identical to tool selection. + +### 5. Latency, caching, and the cold-path problem + +**The prompt-cache interaction is the most underappreciated production gotcha.** Anthropic's cache is a prefix cache built strictly in order tools → system → messages. Cache hits require byte-identical prefixes. Therefore: +- Changing *any* tool definition invalidates the tools cache *and* everything downstream (system + messages). Adding an MCP tool mid-session can 5× the cost of that turn. +- Non-deterministic tool serialization (dict/set ordering across Python runs) silently creates cache misses; always serialize tools in a fixed, tested order. +- The naive dynamic-loading anti-pattern — discover a tool, then inject its full definition into the tools array — invalidates the cache on every discovery turn. + +Anthropic's `defer_loading` is designed to dodge this: the system expands `tool_reference` blocks in the *message* history (the cheap, append-only end of the prefix) rather than mutating the tools block, "so Claude can reuse discovered tools in subsequent turns without re-searching." Pydantic AI documents this explicitly: native provider search keeps discovery append-only and preserves the cache, whereas its local fallback (flipping `defer_loading=False` between rounds) "changes the tool-definition prefix and invalidates the cached request prefix on every discovery turn." Newer models (Opus 4.5+, Sonnet 4.6+) also support mid-conversation system messages that don't invalidate the cached prefix. + +**Practical guidance:** keep your 3–5 highest-frequency tools always-loaded and stable; do discovery *once per session* where possible rather than per-turn; place per-request/dynamic data *after* the last cache breakpoint; and treat model switches as a hard cache boundary. Cache reads cost 10% of base input; writes cost 25% more — so a thrashing cache is strictly worse than no cache. + +### 6. Libraries and frameworks (Python-first) — maturity, license, lock-in + +| Project | What it does | Maturity | License | Lock-in risk | +|---|---|---|---|---| +| **`langgraph-bigtool`** (LangChain) | Semantic tool retrieval via LangGraph long-term store; in-memory/Postgres backends; custom retrieval fns | Released, maintained, narrow scope | MIT (OSI) | Low–medium: tied to LangGraph runtime but components swappable; Postgres backend is standard | +| **Pydantic AI toolsets** (`defer_loading`, `ToolSearch`, `FastMCPToolset`) | Native tool-search across providers + custom strategies; MCP via FastMCP | Actively developed, production-oriented | MIT (OSI) | Low: provider-agnostic, custom search callable, swappable models | +| **RAG-MCP** (reference impls: memoverflow, fintools-ai) | Tool-RAG over MCP via vector index | Research-grade / community | Open (varies by repo) | Low: pattern, not a platform | +| **ScaleMCP** (Lumer et al.) | Auto-synchronizing MCP tool retriever, TDWA embeddings, CRUD index | Research paper + concepts | Paper (no canonical OSI lib) | N/A (pattern) | +| **Toolshed / Advanced RAG-Tool Fusion** (Lumer et al.) | Tool knowledge base + pre/intra/post-retrieval ensemble | Research paper + sample code | Open sample code | Low (pattern) | +| **agentgateway** (Solo.io) | MCP gateway with progressive disclosure (`get_tool`/`invoke_tool`), reported 91% token cut | Vendor OSS | Open-source core | Medium: gateway is infra you operate; pattern portable | +| **FastMCP** | MCP server/client framework, Streamable HTTP transport | Mature, widely adopted | Apache-2.0 (OSI) | Low: standard MCP, swappable | +| **Anthropic Tool Search Tool / Skills API** | Native defer-loading + sandboxed skills | GA-track beta | Proprietary API (Skills *format* is open) | **High for the API; low for SKILL.md format** | +| **OpenAI Agents SDK ToolSearch** | Hosted tool search, namespaces, MCP | Released (`openai>=2.25.0`) | Apache-2.0 SDK, proprietary hosted search | Medium–high: hosted search is OpenAI-only | + +**Lock-in verdict for a lock-in-averse reader:** The maximally swappable stack is FastMCP (Apache-2.0) for transport + Pydantic AI (MIT) for the agent/toolset layer with a *custom* `ToolSearch` strategy backed by your own embeddings/vector DB, falling back to provider-native search only as an optimization. The SKILL.md format is safe to adopt (open standard, 32 adopters, Linux Foundation governance). Anthropic's Tool Search Tool and OpenAI's hosted ToolSearch are the highest lock-in components — use them behind an abstraction so the discovery layer can be re-pointed at an OSS retriever. `langgraph-bigtool` is fine if you are already on LangGraph but couples you to that runtime. + +### 7. The questions not being asked but should be + +1. **Tool-description quality is the real bottleneck.** Both the BM25 and embedding paths are only as good as the descriptions. "Tool Preferences in Agentic LLMs are Unreliable" shows selection can be gamed by description edits; Anthropic's own guidance stresses documenting return formats and using clear, natural-language descriptions. Most "retrieval failures" are description failures. This deserves a dedicated quality bar and review process — descriptions are now load-bearing infrastructure, not docs. + +2. **Embedding drift and index versioning.** When the tool catalog changes, embeddings computed with one model become inconsistent if you later switch embedding models, and stale indexes mis-route. ScaleMCP's CRUD auto-sync (servers as single source of truth) is the only project explicitly treating this. You need: versioned tool indexes, a re-embedding strategy on catalog change, and a pinned embedding model with migration discipline. + +3. **Evaluation of discovery quality itself.** Recall@k and tool-selection accuracy (BFCL, LiveMCPBench, MCP-Universe, MCP-Atlas) measure end-task success but rarely isolate *retrieval* quality from *selection* quality. LiveMCPBench's finding that ~half of failures are retrieval errors is the exception. Inspect AI (UK AISI) ports BFCL and is a strong OSS harness for building discovery-specific evals; instrument retrieval recall separately from final-answer accuracy. + +4. **Security: dynamic loading widens the attack surface.** Tool poisoning (malicious instructions in descriptions/schemas), rug pulls (a server swaps a clean tool for a malicious one after approval), and cross-server shadowing (a malicious server registers a trusted tool's name) all exploit dynamic registries — and clients typically don't re-prompt on description changes. Defenses: host-side allow-listing, static manifest scanning (Invariant Labs' mcp-scan), runtime sandboxing, and signed/versioned definitions (ETDI, arXiv 2506.01333, using OAuth-issued signed JWTs). The more dynamic your discovery, the more these matter. + +5. **Multi-tenant catalog isolation.** Largely unaddressed in the literature. If one index serves multiple tenants, retrieval must be tenant-scoped (filtered vector search), descriptions must not leak across tenants, and per-tenant tool subsets must be enforced before anything reaches the model. This is an open problem practitioners hit immediately in SaaS deployments. + +6. **The CLI-and-skills alternative.** A recurring thread in MCP discussions argues the answer to tool bloat isn't "5 core tools + progressive discovery" but "expose a CLI + teach workflows via skills." For some domains this is genuinely simpler and gives better progressive disclosure than MCP — worth evaluating before committing to a retrieval pipeline. + +--- + +## Recommendations + +**Stage 1 — Adopt the pattern once you cross ~30 tools.** Below ~30 tools, load everything; the retrieval overhead isn't worth it. Above it, the 49%→74%-class accuracy gains and 85%-class token savings justify a search tool. Start with single-shot semantic retrieval + a reranker (architectures a+b). Keep 3–5 high-frequency tools always-loaded. + +*Threshold to escalate:* if Recall@5 on a held-out eval drops below ~0.8 or distractor-induced wrong-tool selections exceed your error budget, move to Stage 2. + +**Stage 2 — Add iterative search and description hardening.** Introduce ReAct-style re-search for multi-hop tasks. Simultaneously, treat tool descriptions as load-bearing: enforce a description style guide, add consistent prefixes (`github_*`, `slack_*`), and put user-search keywords in descriptions. Most accuracy gains here come from descriptions, not pipeline depth. + +*Threshold to escalate:* only if you exceed ~1,000 tools and Recall@5 is business-critical should you consider the full planner→fan-out→selector pipeline (Stage 3). + +**Stage 3 — Multi-stage pipeline only for very large catalogs.** Adopt Toolshed-style pre/intra/post-retrieval fusion or HGMF hierarchical clustering. Pair with ScaleMCP-style CRUD index auto-sync. Budget for the multiplied latency and operational complexity. + +**Cross-cutting, do from day one:** +- **Protect the cache.** Serialize tools in a fixed order (unit-test it); do discovery once per session; never mutate the tools block mid-conversation — use append-only `tool_reference`/native search. Prefer provider-native search *behind an abstraction* so you can swap to your own retriever. +- **Build the swappable stack:** FastMCP (Streamable HTTP) + Pydantic AI + a custom `ToolSearch` strategy over your own vector DB; adopt SKILL.md for skills; treat subagent cards (A2A) as first-class entries in the same index, routed via handoff-as-tool. +- **Instrument discovery evals separately** (Inspect AI): measure retrieval recall and selection accuracy as distinct metrics. +- **Secure the registry:** allow-list servers, scan manifests (mcp-scan), version and ideally sign tool definitions (ETDI), and enforce tenant-scoped retrieval. + +--- + +## Caveats + +- **Vendor numbers are self-reported.** Anthropic's 49%→74%, 85%-reduction, and PTC 37%-reduction figures are first-party internal tests with no published methodology or sample sizes. The independent 60%-retrieval-accuracy counter-report and the RAG-MCP/Toolshed academic numbers (which *do* publish methods) are the more trustworthy anchors for the *shape* of the gains; treat absolute percentages as indicative, not guaranteed. +- **Benchmarks measure end tasks, not discovery in isolation.** BFCL, LiveMCPBench, MCP-Universe, and MCP-Atlas conflate retrieval and selection quality except where explicitly separated. Leaderboards also move fast (MCP-Universe's top entry shifted from GPT-5 43.72% in the paper to Gemini-3-Pro-Preview 44.59% on the live board); cite the version. +- **The pattern is young.** `defer_loading` shipped in beta in late November 2025; framework support (LangChain, Pydantic AI, OpenAI SDK) was still landing through early 2026, and progressive disclosure is not yet in the MCP spec. Expect API churn. +- **"Progressive disclosure" is overloaded.** It refers to tool search, SKILL.md three-level loading, MCP resource-based lazy descriptions, and UI design simultaneously; verify which layer a given source means. +- **Some sources are vendor blogs and Medium posts.** Where used (Solo.io, Arcade, Unified.to, practitioner Medium/DEV posts), they corroborate primary docs and papers but carry promotional or anecdotal bias; the load-bearing claims here are anchored to arXiv papers and official Anthropic/OpenAI/Pydantic/MCP/A2A documentation. + +--- + +## References + +[1] [Introducing advanced tool use on the Claude Developer Platform — Anthropic](https://www.anthropic.com/engineering/advanced-tool-use) +[2] [Tool search tool — Claude API Docs](https://platform.claude.com/docs/en/agents-and-tools/tool-use/tool-search-tool) +[3] [RAG-MCP: Mitigating Prompt Bloat in LLM Tool Selection via Retrieval-Augmented Generation (arXiv 2505.03275)](https://arxiv.org/abs/2505.03275) +[4] [Toolshed: Scale Tool-Equipped Agents with Advanced RAG-Tool Fusion and Tool Knowledge Bases (arXiv 2410.14594)](https://arxiv.org/abs/2410.14594) +[5] [ScaleMCP: Dynamic and Auto-Synchronizing Model Context Protocol Tools for LLM Agents (arXiv 2505.06416)](https://arxiv.org/abs/2505.06416) +[6] [LiveMCPBench: Can Agents Navigate an Ocean of MCP Tools? (arXiv 2508.01780)](https://arxiv.org/abs/2508.01780) +[7] [MCP-Universe: Benchmarking Large Language Models with Real-World Model Context Protocol Servers (arXiv 2508.14704)](https://arxiv.org/abs/2508.14704) +[8] [MCP-Atlas: A Large-Scale Benchmark for Tool-Use Competency with Real MCP Servers — Scale AI](https://scale.com/blog/open-sourcing-mcp-atlas) +[9] [Context Rot: How Increasing Input Tokens Impacts LLM Performance — Chroma](https://www.trychroma.com/research/context-rot) +[10] [langgraph-bigtool — GitHub (LangChain)](https://github.com/langchain-ai/langgraph-bigtool) +[11] [Equipping agents for the real world with Agent Skills — Anthropic](https://www.anthropic.com/engineering/equipping-agents-for-the-real-world-with-agent-skills) +[12] [Agent Skills Overview — agentskills.io](https://agentskills.io/home) +[13] [Agent Discovery — Agent2Agent (A2A) Protocol](https://a2a-protocol.org/latest/topics/agent-discovery/) +[14] [Handoffs — OpenAI Agents SDK](https://openai.github.io/openai-agents-python/handoffs/) +[15] [Tools — OpenAI Agents SDK](https://openai.github.io/openai-agents-python/tools/) +[16] [Toolsets — Pydantic AI Docs](https://ai.pydantic.dev/toolsets/) +[17] [Advanced Tool Features — Pydantic AI Docs](https://pydantic.dev/docs/ai/tools-toolsets/tools-advanced/) +[18] [Prompt caching — Claude API Docs](https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching) +[19] [Code execution with MCP: building more efficient AI agents — Anthropic](https://www.anthropic.com/engineering/code-execution-with-mcp) +[20] [SEP-1888: Progressive Disclosure for Typed Library Discovery & Introspection — MCP GitHub](https://github.com/modelcontextprotocol/modelcontextprotocol/issues/1888) +[21] [MCP Tool Schema Bloat: The Hidden Token Tax — Layered System](https://layered.dev/mcp-tool-schema-bloat-the-hidden-token-tax-and-how-to-fix-it/) +[22] [Tool Preferences in Agentic LLMs are Unreliable (arXiv 2505.18135)](https://arxiv.org/abs/2505.18135) +[23] [Less is More: Optimizing Function Calling for LLM Execution on Edge Devices (arXiv 2411.15399)](https://arxiv.org/abs/2411.15399) +[24] [Berkeley Function Calling Leaderboard (BFCL) — Gorilla](https://gorilla.cs.berkeley.edu/leaderboard.html) +[25] [MCP Security Notification: Tool Poisoning Attacks — Invariant Labs](https://invariantlabs.ai/blog/mcp-security-notification-tool-poisoning-attacks) +[26] [ETDI: Mitigating Tool Squatting and Rug Pull Attacks in MCP (arXiv 2506.01333)](https://arxiv.org/abs/2506.01333) +[27] [HGMF: A Hierarchical Gaussian Mixture Framework for Scalable Tool Invocation within MCP (arXiv 2508.07602)](https://arxiv.org/abs/2508.07602) +[28] [MCP Progressive Disclosure: Save Tokens, Retrieve Schemas — Solo.io](https://www.solo.io/blog/mcp-progressive-disclosure) +[29] [Anthropic's Tool Search: Not Ready for Production Marketing Workflows — Growth Method](https://growthmethod.com/anthropic-tool-search/) +[30] [smolagents — GitHub (Hugging Face)](https://github.com/huggingface/smolagents) +[31] [FastMCP Client — Pydantic AI Docs](https://pydantic.dev/docs/ai/mcp/fastmcp-client/) +[32] [AutoTool: Efficient Tool Selection for Large Language Model Agents (arXiv 2511.14650)](https://arxiv.org/abs/2511.14650) +[33] [BiasBusters: Uncovering and Mitigating Tool Selection Bias in Large Language Models (arXiv 2510.00307)](https://arxiv.org/abs/2510.00307) +[34] [Progressive Disclosure for Typed Library Discovery & Introspection — MCP Discussion #631](https://github.com/orgs/modelcontextprotocol/discussions/631) + +--- + +*Note for the reader: This report is delivered as Markdown and can be saved directly as a `.md` file (e.g., `progressive-capability-discovery.md`). All references include live hyperlinks in `[name](url)` format per Vancouver-style numbering.* \ No newline at end of file diff --git a/misc/docs/ir_02 -- Indexing & Embedding Strategy for Agentic Capability Artifacts -- SKILL.md, MCP Tools, and Subagent Specs.md b/misc/docs/ir_02 -- Indexing & Embedding Strategy for Agentic Capability Artifacts -- SKILL.md, MCP Tools, and Subagent Specs.md new file mode 100644 index 0000000..a3aff26 --- /dev/null +++ b/misc/docs/ir_02 -- Indexing & Embedding Strategy for Agentic Capability Artifacts -- SKILL.md, MCP Tools, and Subagent Specs.md @@ -0,0 +1,114 @@ +# Indexing & Embedding Strategy for Agentic Capability Artifacts: SKILL.md, MCP Tools, and Subagent Specs + +## TL;DR +- **Embed intent, not implementation.** For all three artifact types, the highest-signal vector is a *concatenation of name + description + an LLM-generated "when-to-use" / synthetic-query expansion*; raw bodies, parameter schemas, and JSON typing add noise more often than signal. The strongest empirical result in the literature is that document expansion — not model swaps — moves capability retrieval the most: Tool-DE's expansion-trained reranker beat the MTEB SoTA open model by +10.23 NDCG@10 on its own benchmark. +- **Run hybrid (BM25 + dense) with RRF and a reranker; default the reranker to Voyage rerank-2.5.** Capability descriptions are short and jargon-dense (exact tool/function names matter), the exact regime where dense-only retrieval fails silently; ToolRet showed ColBERT underperforming plain BM25, and instruction-conditioning added +9 to +17 NDCG@10. rerank-2.5's instruction-following is a natural fit for the query→capability asymmetry. +- **Do not index SKILL.md bodies; treat the Git registry as SSOT with content-hash invalidation.** Progressive disclosure means the body is loaded *after* selection, so it belongs on disk, not in the retrieval index. Sync the index by hashing (name + description + params) per ScaleMCP's SHA-256 CRUD pipeline and re-embedding only on hash change. + +## Key Findings + +### 1. Tool/skill retrieval is a distinct, hard problem — generic MTEB scores do not transfer +The single most important framing result comes from ToolRet (Shi et al., ACL Findings 2025): across 7.6k retrieval tasks over a 43k-tool corpus, **even the best embedding model, NV-Embed-v1, reached only NDCG@10 = 33.83**, and *"retrieval methods that demonstrate strong performance in conventional IR tasks, such as ColBERT, even underperform compared to simple lexical-based matching approaches like BM25"* (ColBERT 19.46 vs BM25s 22.32, no-instruction). The root cause is measured: query↔tool lexical overlap (ROUGE-L) is just **0.06** in ToolRet versus 0.27–0.34 for MSMARCO/MTEB. You therefore cannot pick an embedder off the MTEB leaderboard and assume it works for capability retrieval — the task has near-zero surface overlap between a task-phrased query and a capability-phrased document, and ToolRet's correlation analysis (Pearson β = 0.790 with MTEB but uniformly lower absolute scores) confirms the gap is systematic. + +This is what underlies the "tool selection collapses at scale" narrative. Anthropic's own engineering data, reported in *Advanced tool use*, shows that with the Tool Search Tool enabled on large MCP libraries, **Opus 4 tool-selection accuracy improved from 49% to 74%, and Opus 4.5 from 79.5% to 88.1%** — i.e., roughly a quarter of selections still failed even on the better baseline. RAG-MCP's controlled stress test is starker still: schema-dump ("Blank Conditioning") selection scored **13.62%**, "Actual Match" 18.20%, and retrieval-based MCP-RAG **43.13%** — the highest of the three — while cutting average prompt tokens from 2,133.84 to 1,084 (≈49% reduction). The qualitative degradation past 30–50 tools is robust across sources; the exact percentages are setup-dependent. + +### 2. What to embed: expansion beats field-juggling +Two convergent findings dominate: +- **Document expansion is the highest-leverage lever.** Tool-DE (Lu et al., arXiv:2510.22670, ICLR 2026) enriches each tool doc with LLM-generated structured fields and reports field-level contribution (radar-plot deltas, §4.1): `function_description` and `tags` contribute most (their removal causes noticeable NDCG drops), `when_to_use` contributes large improvements, `limitations` is retained, and **`example_usage` is least useful — smallest, often negative gains, and it actively hurt GritLM** — so the authors *excluded* it from the final profile. Their expansion-trained Tool-Embed-4B (Qwen3-Embedding base) hit NDCG@10 = 52.23 / Recall@10 = 63.13 on Tool-DE; the Tool-Rank-4B reranker reached **56.44 / 67.81 / 56.60 (N/R/C@10), +10.23 / +10.29 / +9.08 over Qwen3-Embedding-8B**, the MTEB SoTA open model as of Sept 2025. Expansion roughly *doubled* the 4B retriever's gain versus a non-expansion control (+6.69 vs +3.67 NDCG@10). +- **Query-side expansion closes the asymmetry.** Re-Invoke (Chen et al., Google, EMNLP 2024 Findings, arXiv:2408.01875) generates synthetic queries per tool at index time and extracts intents from the user query at inference time, achieving *"a 20% relative improvement in nDCG@5 for single-tool retrieval and a 39% improvement for multi-tool retrieval"* on ToolE — fully unsupervised, over BM25 and Vertex AI text-embedding baselines. + +Practical synthesis: the embedded text for a capability should be `name + description + LLM-generated(when_to_use, tags, synthetic queries)`, *excluding* verbose usage examples and raw parameter dumps. + +### 3. Multi-vector / late-interaction: promising but not yet justified here +Late-interaction (ColBERTv2, Jina-ColBERT-v2, GTE-ModernColBERT) preserves token-level signal and excels at exact-term matching — attractive for jargon-dense identifiers. But ToolRet found vanilla ColBERT *underperforming BM25* on tool retrieval, and the multi-vector index carries a 6–10× storage penalty even after ColBERTv2's residual compression. For a catalog of hundreds-to-thousands of capabilities, that storage and operational cost is not yet repaid by accuracy gains; a hybrid sparse+dense pipeline with a strong cross-encoder reranker dominates the current evidence. The genuinely useful multi-vector pattern here is **field-level multi-vector** (separate embeddings for "what it does" vs "when to use it"), not token-level late interaction — ScaleMCP's TDWA (Tool Document Weighted Average) is the closest published instance, selectively weighting tool-doc components (e.g., name, synthetic questions) during embedding. + +### 4. SKILL.md body: keep it out of the retrieval index +The agentskills.io spec's progressive-disclosure design is decisive: metadata (~100 tokens: name + description) loads at startup for all skills; the full body (<5k tokens recommended, <500 lines) loads *only after* the skill is selected; and `references/`, `scripts/`, `assets/` load on demand. Since selection happens on metadata alone and the runtime loads the body into context *after* selection, **the body carries no retrieval responsibility.** Indexing it risks fragmenting intent — a query matching one subsection of a long body even when the skill as a whole is the wrong choice. Recommendation: index `name + description (+ expansion)`; leave the body on disk. AST/structure-aware chunking of the body for *selection* is an anti-pattern; it is only justified if you separately serve intra-skill search (e.g., to retrieve a specific reference section after activation). The same logic applies to subagent specs: the Markdown system-prompt body is appended to the spawned agent post-delegation, and Claude Code routes purely on the `description` field — so the body should not be in the routing index. + +### 5. Hybrid retrieval and reranker selection +For short, identifier-heavy capability text, BM25 is not optional: dense-only retrieval "fails silently on exact identifiers, code, and rare terms," and rare/new identifiers are poorly represented by embeddings (cold-start). The production-consensus stack is BM25 + dense ANN fused with **Reciprocal Rank Fusion** (rank-based, sidestepping the BM25/cosine score-scale mismatch), then a cross-encoder reranker over the top 20–50 candidates. Note the practitioner caveat that naive RRF can undersell tuned hybrid (one Elasticsearch benchmark: plain RRF +1.3% NDCG vs a term-match-boosted tier +7.5%) — so weight lexical matches up for exact tool/function names. + +Reranker choice (anchored on Voyage rerank-2.5, swappable): +- **Voyage rerank-2.5 / 2.5-lite** *(default)* — first reranker with instruction-following, 32K context. Per Voyage's benchmark post, *"rerank-2.5 and rerank-2.5-lite outperform Cohere Rerank v3.5 by 12.70% and 10.36%, respectively"* on MAIR, and +7.94% across 93 datasets. Instruction-following is the killer feature for capability retrieval: you can steer "prefer read-only tools" or "rank tools matching this task intent" at query time. +- **Qwen3-Reranker-8B** — strongest open-weight option; Voyage reports rerank-2.5-lite edging it by **+1.01% NDCG@10 "despite being over an order of magnitude smaller."** Pick Qwen3 for self-hosting. +- **bge-reranker-v2-gemma** — the *single best reranker on ToolRet itself* (NDCG@10 = 47.52 with instruction), giving it direct, domain-specific benchmark support; strong open default for tool catalogs specifically. +- **Cohere Rerank 3.5** — solid generalist, shorter context, no instruction steering. + +Critical caution from ToolRet: reranking can *hurt*. MonoT5 reranking dropped NV-Embed-v1 from 33.83 to 28.92 NDCG@10. Rerankers must be evaluated on your own capability set, never assumed beneficial. + +### 6. SSOT and index maintenance +The canonical pattern is ScaleMCP's auto-synchronization pipeline: treat the Git-first file registry as single source of truth, compute a **SHA-256 hash over (name + description + parameters)** per capability, diff against stored hashes, and issue CRUD ops — re-embedding only on hash mismatch (matched hashes are no-ops; changed hashes trigger discard-and-re-embed). This cleanly handles description edits, additions, and removals, and eliminates the "stored catalog drifts from live registry" bug. Anthropic's Tool Search BM25 variant takes a complementary stateless route for smaller catalogs: the catalog is rebuilt from the current tool-defs on every assembly, which "prevents drift bugs where a stored catalog goes out of sync." Embedding-model version migration requires a *full* re-embed (vectors are model-specific and non-comparable across models, and even Voyage notes a model upgrade "requires re-embedding your corpus") — so store model+version in index metadata and trigger a bulk re-embed on version bump. + +### 7. Embedding model choice with tool-specific benchmark evidence +Ranked by *tool-retrieval-specific* evidence, not generic MTEB: +- **Instruction-tuned embedders win on tool retrieval — the strongest single lever after expansion.** ToolRet's with-instruction setting: gte-Qwen2-1.5B-instruct hit NDCG@10 = 45.96 (best embedding model), and instruction-conditioning lifted every model **+9 to +17 NDCG@10** (gte-Qwen2 +17.00, bge-large +15.47, BM25s +14.14, e5-mistral +12.91, GritLM +11.11, NV-Embed +8.88). +- **Qwen3-Embedding (0.6B/4B/8B)** — current open default and the base for Tool-DE's SoTA Tool-Embed; supports instructions and Matryoshka dims (32–1024+, +1–5% from instructions per Qwen). +- **Voyage-3.5 / voyage-3-large** — strong proprietary default in the anchored stack; voyage-3.5 beats OpenAI v3-large by 8.26% across eight domains, with int8@2048 cutting vector-DB cost ~83%. Use **voyage-code-3** for code-identifier-heavy catalogs. +- **NV-Embed-v1** — best *without* instructions on ToolRet (33.83); reasonable non-instruction baseline. +- **pplx-embed (Perplexity, arXiv:2602.11151, Feb 2026)** — newer; Perplexity reports the family *"leads a range of public benchmarks including MTEB(Multilingual, v2), BERGEN, ToolRet, and ConTEB,"* with pplx-embed-v1-4B (INT8) at nDCG@10 69.66% on MTEB Multilingual v2 (vs Qwen3-Embedding-4B 69.60%, gemini-embedding-001 67.71%), plus INT8/binary compression. Worth evaluating; figures are vendor-reported. + +### 8. Gotchas and the questions you should be asking but haven't +- **Multi-tool / multi-hop queries break single-vector retrieval.** ScaleMCP found a single query embedding "often fails to capture multiple distinct retrieval targets"; a query needing 3–12 tools (e.g., "revenue + net income") won't be served by one vector. You need query decomposition or agentic multi-retrieval, not a better embedder. (Re-Invoke's multi-tool gain of +39% vs +20% single-tool quantifies how much harder the multi-target case is.) +- **The "shifting bottleneck."** Retrieval/Tool Search doesn't fix bad capability hygiene — "bad tool definitions lead to bad tool selection." Retrieval makes description quality *more* load-bearing, not less. Anthropic's own evidence: refining tool *descriptions* alone drove Claude Sonnet 3.5 to SoTA on SWE-bench Verified. +- **Embedding the schema dilutes the description.** Parameter names, enums, and JSON typing are noise for *selection* (they matter for *invocation*, which is post-selection). Anthropic's tool-search BM25 variant searches names + descriptions + arg names/descriptions for lexical recall — but for *dense* embedding, concatenating full schemas dilutes the intent vector. Expose arg names to BM25; keep them out of the dense vector. +- **Annotations (readOnlyHint/destructiveHint/idempotentHint/openWorldHint)** carry near-zero semantic retrieval signal but high policy signal — use them as *metadata filters and reranker instructions*, not embedded text. +- **Completeness@K, not Recall@K, is the metric that matters.** ToolRet's Completeness@10 (=1 only if *all* required tools are in top-K) is far stricter than recall, and all retrievers scored <35% on it. If tasks need tool *sets*, evaluate on completeness. +- **Description-as-trigger collision.** For SKILL.md and subagent specs the description *is* the routing signal; near-duplicate descriptions cause silent mis-routing — Claude Code "keeps one and discards the other without warning" on name collisions. Contrastive description-writing and dedup are retrieval-critical, not cosmetic. +- **Raw retrieval accuracy ≠ end-to-end eval.** Anthropic's 49%→74% figures are *end-to-end* selection-with-reasoning; an independent cross-check (Arcade, relayed via Stacklok) reported Tool Search hitting only ~56% (regex) / ~64% (BM25) *raw retrieval* accuracy across 4,027 tools. Headline accuracy gains can mask weaker first-stage recall at very large scale. + +## Details: Comparison Matrix + +| Dimension | SKILL.md (agentskills.io) | MCP tool definition (FastMCP) | Subagent spec (Claude Code) | +|---|---|---|---| +| **Primary embed field** | `name + description` | `name + description` | `name + description` | +| **Recommended expansion** | + LLM `when_to_use` + tags + synthetic queries | + `when_to_use` + tags + synthetic queries (Re-Invoke / Tool-DE) | + trigger phrases ("use proactively when…") | +| **Index the body?** | **No** — loaded post-selection via progressive disclosure | N/A (no body; schema is invocation-time) | **No** — system-prompt body is post-delegation | +| **Schema / params** | n/a | Exclude from dense vector; expose arg names to BM25 only | n/a (`tools` field is a policy allowlist) | +| **Annotations / metadata** | license, compatibility → filters | readOnly/destructive/idempotent/openWorld → filters + reranker instructions | tools, model, permissionMode → filters | +| **Multi-vector?** | Optional: "what" / "when" split | Optional: TDWA-weighted fields | Low value (few specs) | +| **Hash key for SSOT** | name+description+frontmatter | name+description+parameters (SHA-256, ScaleMCP) | name+description+tools+model | +| **Retrieval need** | Often small (50–100); BM25+regex may suffice | Large (50–1000s) → full hybrid + rerank | Usually small; description-routing, light retrieval | +| **Best-evidence stack** | Hybrid + rerank-2.5; index metadata only | Hybrid + RRF + rerank-2.5; expansion-trained embedder | Description-match + dedup; retrieval optional | + +## Recommendations + +**Stage 1 — Baseline (ship first):** Index `name + description` per capability with a hybrid pipeline — BM25 + dense (voyage-3.5 or Qwen3-Embedding-4B) fused via RRF, top-50 → Voyage rerank-2.5 with a task-intent instruction. Git registry as SSOT, SHA-256 content-hash invalidation. Do **not** index SKILL.md or subagent bodies. For MCP catalogs under ~30 tools, Anthropic Tool Search (BM25/regex variant) alone is sufficient — defer the vector index. + +**Stage 2 — Expansion (highest ROI; when recall@k or completeness@k underperforms):** Add offline LLM document expansion (`function_description`, `when_to_use`, `tags`; *omit* `example_usage`) per Tool-DE, plus Re-Invoke-style synthetic-query indexing. This should precede any model swap. **Threshold to trigger:** Completeness@10 below ~0.6 on a held-out task set, or observed mis-routing among similar capabilities. + +**Stage 3 — Instruction-conditioning & model upgrade:** Move to an instruction-tuned embedder (gte-Qwen2-instruct / Qwen3-Embedding) and pass per-query instructions; ToolRet shows +9–17 NDCG@10. A/B bge-reranker-v2-gemma against rerank-2.5 on *your* catalog (bge was ToolRet's best reranker). **Threshold:** if instruction-conditioning doesn't lift held-out NDCG@10, the gap is description quality, not the model — return to Stage 2 hygiene. + +**Stage 4 — Multi-hop handling:** If tasks routinely need tool *sets* (3+ capabilities), add agentic query decomposition / iterative retrieval (ScaleMCP pattern); a single query vector will not solve this regardless of embedder quality. + +**What would change these recommendations:** a multi-vector / late-interaction model showing tool-retrieval gains net of its 6–10× storage cost (none yet); catalogs small enough (<30 tools) that BM25/regex tool search removes the need for a vector index entirely; or a code-identifier-dominated domain, favoring voyage-code-3 and heavier BM25 weighting. + +## Caveats +- Tool-DE's §4.1 field ablation is reported graphically (radar plots); the directional conclusions quoted here are explicit, but per-field absolute NDCG values are not printed in the text. +- ToolRet and Tool-DE are built on web APIs, code functions, and customized apps — *not* specifically on MCP tool definitions, SKILL.md, or subagent specs. The transfer is well-motivated (all are short capability descriptions with low query overlap) but not directly benchmarked; treat the SKILL.md/subagent guidance as reasoned extrapolation, not measured fact. +- Vendor benchmark numbers (Voyage, Qwen, Perplexity, Anthropic Tool Search) are self-reported; the load-bearing claims here are anchored to peer-reviewed papers (ToolRet ACL 2025, Re-Invoke EMNLP 2024, Tool-DE ICLR 2026, RAG-MCP) with vendor figures explicitly flagged. +- The Anthropic Tool Search "49%→74%" figures are end-to-end selection evals relayed via the Anthropic engineering post; independent raw-retrieval measurements at large scale (~56–64%) are lower, so interpret accuracy claims by stage. +- Some sourced material (Hermes Agent, MarkTechPost, "MCP is dead" practitioner posts) carries marketing/speculative framing and was used only for corroboration, not as primary evidence. + +## References +1. Shi Z, Wang Y, Yan L, Ren P, Wang S, Yin D, Ren Z. Retrieval Models Aren't Tool-Savvy: Benchmarking Tool Retrieval for Large Language Models. Findings of ACL 2025. Available from: https://aclanthology.org/2025.findings-acl.1258/ and https://arxiv.org/abs/2503.01763 +2. Lu X, Huang H, Meng R, Jin Y, Zeng W, Shen X. Tools are Under-Documented: Simple Document Expansion Boosts Tool Retrieval. arXiv:2510.22670 (ICLR 2026). Available from: https://arxiv.org/abs/2510.22670 +3. Chen Y, Yoon J, Sachan DS, Wang Q, Cohen-Addad V, Bateni M, et al. Re-Invoke: Tool Invocation Rewriting for Zero-Shot Tool Retrieval. Findings of EMNLP 2024. Available from: https://aclanthology.org/2024.findings-emnlp.270/ and https://arxiv.org/abs/2408.01875 +4. Gan Q, Sun Q. RAG-MCP: Mitigating Prompt Bloat in LLM Tool Selection via Retrieval-Augmented Generation. arXiv:2505.03275. Available from: https://arxiv.org/abs/2505.03275 +5. Lumer E, et al. ScaleMCP: Dynamic and Auto-Synchronizing Model Context Protocol Tools for LLM Agents. arXiv:2505.06416. Available from: https://arxiv.org/abs/2505.06416 +6. Anthropic. Introducing advanced tool use on the Claude Developer Platform. 2025. Available from: https://www.anthropic.com/engineering/advanced-tool-use +7. Anthropic. Writing effective tools for AI agents—using AI agents. 2025. Available from: https://www.anthropic.com/engineering/writing-tools-for-agents +8. Anthropic. Effective context engineering for AI agents. 2025. Available from: https://www.anthropic.com/engineering/effective-context-engineering-for-ai-agents +9. Anthropic / Claude API Docs. Tool search tool. 2025. Available from: https://platform.claude.com/docs/en/agents-and-tools/tool-use/tool-search-tool +10. Agent Skills. Specification (SKILL.md format). 2025. Available from: https://agentskills.io/specification and https://github.com/agentskills/agentskills/blob/main/docs/specification.mdx +11. Claude Code Docs. Create custom subagents. 2026. Available from: https://code.claude.com/docs/en/sub-agents +12. Voyage AI. rerank-2.5 and rerank-2.5-lite: Instruction-Following Rerankers. 2025 Aug 11. Available from: https://blog.voyageai.com/2025/08/11/rerank-2-5/ +13. Voyage AI. voyage-3.5 and voyage-3.5-lite: improved quality for a new retrieval frontier. 2025 May 20. Available from: https://blog.voyageai.com/2025/05/20/voyage-3-5/ +14. Voyage AI. voyage-3-large: the new state-of-the-art general-purpose embedding model. 2025 Jan 7. Available from: https://blog.voyageai.com/2025/01/07/voyage-3-large/ +15. Zhang Y, et al. (Qwen Team). Qwen3 Embedding: Advancing Text Embedding and Reranking Through Foundation Models. arXiv:2506.05176. Available from: https://arxiv.org/abs/2506.05176 +16. Santhanam K, Khattab O, Saad-Falcon J, Potts C, Zaharia M. ColBERTv2: Effective and Efficient Retrieval via Lightweight Late Interaction. arXiv:2112.01488. Available from: https://arxiv.org/abs/2112.01488 +17. Jha A, et al. Jina-ColBERT-v2: A General-Purpose Multilingual Late Interaction Retriever. arXiv:2408.16672. Available from: https://arxiv.org/abs/2408.16672 +18. Luo R, et al. MCP-Universe: Benchmarking Large Language Models with Real-World Model Context Protocol Servers. arXiv:2508.14704. Available from: https://arxiv.org/abs/2508.14704 +19. Perplexity AI. pplx-embed: State-of-the-Art Embedding Models for Web-Scale Retrieval. 2026. Available from: https://research.perplexity.ai/articles/pplx-embed-state-of-the-art-embedding-models-for-web-scale-retrieval +20. Lee C, et al. (NVIDIA). NV-Embed: Improved Techniques for Training LLMs as Generalist Embedding Models. arXiv:2405.17428. Available from: https://arxiv.org/abs/2405.17428 +21. Digital Applied. Hybrid Search: BM25, Vector & Reranking — 2026 Reference. 2026. Available from: https://www.digitalapplied.com/blog/hybrid-search-bm25-vector-reranking-reference-2026 +22. Tessl. Anthropic brings MCP tool search to Claude Code. 2026. Available from: https://tessl.io/blog/anthropic-brings-mcp-tool-search-to-claude-code/ \ No newline at end of file diff --git a/misc/docs/ir_03 -- Evaluating the Capability-Discovery Layer of Agentic AI -- Benchmarks, Metrics, and a Failure-Mode Taxonomy for Tool Retrieval & Selection.md b/misc/docs/ir_03 -- Evaluating the Capability-Discovery Layer of Agentic AI -- Benchmarks, Metrics, and a Failure-Mode Taxonomy for Tool Retrieval & Selection.md new file mode 100644 index 0000000..09770c5 --- /dev/null +++ b/misc/docs/ir_03 -- Evaluating the Capability-Discovery Layer of Agentic AI -- Benchmarks, Metrics, and a Failure-Mode Taxonomy for Tool Retrieval & Selection.md @@ -0,0 +1,422 @@ +# Evaluating the Capability-Discovery Layer of Agentic AI: Benchmarks, Metrics, and a Failure-Mode Taxonomy for Tool Retrieval & Selection + +*Author: **Thor Whalen** — Prompt C in the agentic-platform eval series. Audience: eval-methodology experts (Inspect AI, deepdiff scorers, LLM-as-judge via G-Eval/DAG). Eval fundamentals are assumed and skipped. Sources are 2024–2026, prioritizing 2025–2026.* + +> **How to save this file:** copy everything below the title into a `.md` file (e.g. `capability-discovery-eval.md`). The document is self-contained Markdown with Vancouver-style numbered references and hyperlinks. + +--- + +## TL;DR + +- **Almost no existing benchmark measures what you actually need.** The field conflates three distinct things — retrieval recall (surfacing the right tool from a large catalog), selection accuracy (picking the right tool from candidates), and end-to-end task success. Only **ToolRet** [1] cleanly isolates retrieval; **MetaTool** [2] isolates selection/awareness; everything else (BFCL [3], τ²-bench [4], AppWorld [5], ToolHop [6], ACEBench [7]) bundles selection into end-task success. Build your harness around the **retrieval/selection decoupling** that the literature still lacks. +- **Distractor-robustness is the dominant, under-measured failure axis, and it directly explains your ~49% baseline.** Anthropic's own engineering data shows Claude Opus 4 going from **49%→74%** and Opus 4.5 from **79.5%→88.1%** once the Tool Search Tool (progressive disclosure via `defer_loading`) is enabled [8]; RAG-MCP shows raw selection accuracy collapsing to **13.62%** (full-schema-dump baseline) and recovering to **43.13%** with retrieval pre-filtering in a needle-in-a-haystack MCP stress test [9]; BFCL-derived experiments show calendar-task accuracy dropping **43%→2%** going from 4 tools to 51 tools across 7 domains [10]. Your 49% is a structural artifact of static tool-loading, not a model-quality ceiling. +- **Recommended stack:** treat retrieval as an IR problem scored with **recall@k / NDCG@k** (BEIR/MTEB-style [11,12]), score selection **conditionally on correct retrieval**, score parameters with a **deepdiff-based structural scorer**, and wrap everything in **Inspect AI** scorers (MIT) [13]. Generate eval cases by **back-translation from your `@command` registry** with LLM-judge validation and explicit distractor injection. Use **τ²-bench** [4] and **AppWorld** [5] for end-task sanity checks only — not for discovery-layer measurement. + +--- + +## Key Findings + +1. **The retrieval-vs-selection distinction is real and consequential.** ToolRet [1] demonstrated that IR models which excel on conventional benchmarks perform poorly on tool retrieval: "even the best model (i.e., NV-embed-v1)… achieves an nDCG@10 of only 33.83" on a 43k-tool corpus across 7.6k tasks. Most tool-use benchmarks "simplify this step by manually pre-annotating a small set of relevant tools," meaning they never test retrieval at all. With 50+ tools, the pre-annotation shortcut hides your single biggest risk. + +2. **Selection accuracy degrades monotonically with catalog size and distractor count.** This is the most robustly replicated finding in the 2025–2026 literature, with consistent evidence from RAG-MCP [9], BFCL-derived studies [10], and Anthropic [8]. The mechanism: the model rarely abstains; it picks a plausible-but-wrong tool, hallucinates parameters, or confuses similarly-named tools (`notification-send-user` vs `notification-send-channel`) [8]. + +3. **Progressive disclosure / tool-search is the proven mitigation, and it is itself an evaluation target.** Anthropic's Tool Search Tool [8], Claude Code's MCP Tool Search (`defer_loading: true`, auto-triggers when tool schemas exceed ~10% of context) [14], and the meta-tool pattern [15] all convert the problem from "select among N" to "retrieve k then select among k." This means your harness must evaluate the *retrieval query the model writes*, not just the final selection. + +4. **Published failure taxonomies now exist and map cleanly onto your selection/parameterization/ordering/hallucination classes** [16,17,18,19] — extend them with discovery-specific classes (query-planning failure, retrieval miss, selector false-negative, over-selection, namespace/granularity confusion, tool hallucination). + +5. **Synthetic case generation is mature.** APIGen (three-stage verification) [20], ToolACE (self-evolution + dual-layer verification) [21], Hammer (function masking + irrelevance augmentation) [22], and back-translation pipelines [23,24] give you a well-trodden path to generate (intent → correct-capability) pairs from your registry. + +--- + +## Details + +### 1. Benchmark landscape — what each actually measures + +The single most important analytical move is to classify each benchmark by which of the three sub-problems it measures: + +- **(R) Retrieval recall** — can the system surface the right tool from a large, *unfiltered* catalog? +- **(S) Selection accuracy** — given a small candidate set, does the model pick the right tool / abstain correctly? +- **(E) End-to-end** — does the whole task succeed (selection bundled with parameterization, execution, ordering, multi-turn)? + +#### Comparison matrix + +| Benchmark | Year | Measures | Catalog scale | Credibility / maintenance | License | +|---|---|---|---|---|---| +| **ToolRet** [1] | 2025 | **R** (pure retrieval) | 43k tools, 7.6k tasks | High — only true large-scale retrieval benchmark; actively maintained | research code (GitHub) | +| **MetaTool / ToolE** [2] | ICLR 2024 | **S** + tool-use awareness/abstention | ~390 tools, 21k queries | Medium — influential for "whether to use a tool"; somewhat dated | open (GitHub) | +| **BFCL v1–v4** [3] | 2024–2026 | **S** + **E** (AST + executable + multi-turn + agentic) | thousands (AST scales) | High — canonical, updated (v4 = `bfcl-eval 2025.12.17`); some saturation | Apache-2.0 (gorilla) | +| **τ-bench / τ²-bench / τ³** [4,25] | 2024–2026 | **E** (policy adherence, multi-turn, dual-control) | small per-domain (~tens) | High — best enterprise-realism eval; actively maintained | open (sierra-research) | +| **AppWorld** [5] | ACL 2024 | **E** (stateful, code-gen, collateral damage) | 457 APIs, 9 apps | High — best stateful execution env; train/test split guards leakage | open (encrypted bundles) | +| **ToolHop** [6] | 2025 | **E** (multi-hop) | query-driven, executable | Medium-high — GPT-4o only ~49% | open | +| **ComplexFuncBench** [26] | 2025 | **E** (multi-step, long-context, constrained) | 1,000 samples | Medium — narrow (live Booking API) | open (THUDM) | +| **NESTFUL** [27] | 2025 | **E** (nested API sequences) | 1,800+ sequences | Medium — niche but rigorous; GPT-4o only 28% full-match | Apache-2.0 (IBM) | +| **ACEBench** [7] | 2025 | **S** + **E** (+ imperfect-instruction handling) | ~4,500 APIs, 8 domains | Medium-high — no-GPT eval, good error taxonomy | open | +| **StableToolBench** [28] | 2024 | **E** (stabilized ToolBench via virtual API) | 16k+ tools (simulated) | Medium — solves API-instability but GPT-judge variance | open | +| **ToolSandbox** [29] | 2024 | **S** + **E** (stateful, conversational, on-policy) | on-device sim + RapidAPI | Medium-high — Apple; strong state-dependency tasks | custom Apple license | +| **MTU-Bench** [30] | 2024 | **S** + **E** (multi-granularity, no-GPT metrics) | 136 tools, 54.8k dialogues | Medium — cheap fine-grained metrics incl. tool selection/order accuracy | open | +| **MCP-RADAR** [31] | 2025 | **S** + **E** (MCP framework) | 507 tasks, 6 domains | Medium — MCP-specific, new | open | +| **MCP-Atlas** [32] | 2026 | **E** (large-scale real MCP, claims-based rubric) | 220 tools, 36 servers, 1k tasks | New — Claude Opus 4.5 best at 62.3% | open | +| **GTA / GTA-2** [33] | 2024/2026 | **E** (real queries, deployed tools, workflow) | hierarchical | High — now evaluates harnesses too | open (open-compass) | +| **ToolBench / ToolLLM** [24] | ICLR 2024 | **R** (neural retriever) + **E** | 16,464 APIs | Declining — live-API instability; superseded by StableToolBench/ToolRet | Apache-2.0 / CC-BY-NC | +| **API-Bank** [34] | 2023 | **S** + **E** (planning/retrieving/calling) | 73 tools | Dated baseline | open | + +**Opinionated assessment:** + +- **For the retrieval sub-problem, ToolRet [1] is the only credible large-scale benchmark, and you should adopt its BEIR/MTEB-style framing directly.** ToolRet standardizes tasks "into a retrieval format akin to MTEB and BEIR." That is exactly the right abstraction for your `@command` registry: treat each command's `(name, description, params)` as a document, the user intent as a query, and score with NDCG@10 / Recall@10. +- **For selection, MetaTool [2] and ACEBench [7] are the most relevant** because they explicitly test "whether to use a tool" (abstention) and "which to use" with confusable/irrelevant candidates. Hammer's irrelevance augmentation (the xLAM-7.5k-Irrelevance set) is the key training-side complement [22]. +- **BFCL remains the canonical function-calling leaderboard but is showing saturation** (GLM-4.5 at 0.778 on v3; top reasoning models clustered) [3,35]. It is also at contamination risk given its age and popularity; BFCL-v2 explicitly added "live, user-contributed scenarios" to combat contamination and bias [36]. Use BFCL for cross-model sanity, not for your discovery-layer signal. +- **τ²-bench [4] and AppWorld [5] are your end-task anchors.** Neither isolates discovery, but both have excellent state-based scoring (AppWorld's hash-based state diffing; τ²'s database-state comparison and `pass^k` reliability metric). Use them to confirm that discovery-layer gains translate to task success. +- **Reproducibility / contamination flags:** ToolBench's live RapidAPI dependency makes it non-reproducible (hence StableToolBench's virtual server + MirrorAPI [28]). ComplexFuncBench requires a live Booking API subscription [26]. Treat any benchmark requiring live third-party APIs as non-reproducible for regression testing. + +### 2. Metrics specific to the discovery layer + +Core design principle: **decouple retrieval from selection and score each conditionally.** Define: + +- **Retrieval recall@k** — fraction of cases where the gold tool is in the top-k retrieved set. +- **NDCG@k** — rank-quality of retrieval (the BEIR/MTEB standard [11,12]; ToolRet's primary metric [1]). +- **Conditional selection accuracy** — `P(correct selection | gold tool was retrieved)`. This isolates the "searched, retrieved the right tool into context, but still failed to pick it" case you specifically called out. +- **End-to-end selection accuracy** = recall@k × conditional selection accuracy (the decomposition that makes the bottleneck visible). +- **Distractor-robustness curve** — selection accuracy as a function of N distractors / catalog size. The single most diagnostic metric for your platform. +- **Over-selection rate** — fraction of cases with unnecessary/extra tool calls. +- **Abstention accuracy** — correct refusal-to-act when no tool applies (MetaTool's reliability subtask [2]; Hammer's irrelevance detection [22]). +- **Parameter-filling accuracy** — scored separately and structurally (deepdiff). + +MTU-Bench [30] is worth emulating here: it ships **tool selection accuracy, parameter selection accuracy, tool number accuracy, and tool order accuracy** as separate axes with *no* GPT cost — a good template for a cheap, deterministic discovery-layer scorecard. + +#### recall@k, NDCG@k, and conditional selection accuracy (production Python) + +```python +from __future__ import annotations +from dataclasses import dataclass +from collections.abc import Sequence +import math + +@dataclass(frozen=True) +class DiscoveryCase: + intent: str + gold_tool: str + retrieved: Sequence[str] # ranked list from the retriever + selected: str | None # tool the agent actually chose (None = abstained) + gold_is_none: bool = False # True when the correct behavior is to abstain + +def recall_at_k(cases: Sequence[DiscoveryCase], k: int) -> float: + """Fraction of cases whose gold tool appears in the top-k retrieved set.""" + relevant = [c for c in cases if not c.gold_is_none] + if not relevant: + return float("nan") + hits = sum(c.gold_tool in c.retrieved[:k] for c in relevant) + return hits / len(relevant) + +def ndcg_at_k(cases: Sequence[DiscoveryCase], k: int) -> float: + """Binary-relevance NDCG@k; IDCG is 1.0 since there is one gold tool.""" + relevant = [c for c in cases if not c.gold_is_none] + if not relevant: + return float("nan") + total = 0.0 + for c in relevant: + topk = list(c.retrieved[:k]) + if c.gold_tool in topk: + rank = topk.index(c.gold_tool) # 0-indexed + total += 1.0 / math.log2(rank + 2) # DCG; IDCG == 1.0 + return total / len(relevant) + +def conditional_selection_accuracy( + cases: Sequence[DiscoveryCase], k: int +) -> float: + """P(correct selection | gold tool retrieved into top-k). + + Isolates selector quality from retriever quality: the 'searched, + surfaced the right tool, still failed to pick it' failure mode. + """ + eligible = [ + c for c in cases + if not c.gold_is_none and c.gold_tool in c.retrieved[:k] + ] + if not eligible: + return float("nan") + correct = sum(c.selected == c.gold_tool for c in eligible) + return correct / len(eligible) +``` + +#### Distractor-robustness harness (RAG-MCP-style NIAH stress test) + +```python +from collections.abc import Callable, Sequence +import random + +ToolName = str +SelectFn = Callable[[str, Sequence[ToolName]], ToolName | None] + +def distractor_robustness_curve( + intent: str, + gold_tool: ToolName, + distractor_pool: Sequence[ToolName], + select: SelectFn, + sizes: Sequence[int] = (1, 4, 8, 16, 32, 64, 128), + trials: int = 20, + seed: int = 0, +) -> dict[int, float]: + """Selection accuracy as catalog size grows. + + Mirrors the RAG-MCP 'needle-in-a-haystack' MCP stress test: one gold + tool + (N-1) distractors, vary N, report accuracy at each N. + """ + rng = random.Random(seed) + curve: dict[int, float] = {} + for n in sizes: + n_distract = max(0, n - 1) + hits = 0 + for _ in range(trials): + sample = rng.sample(list(distractor_pool), + min(n_distract, len(distractor_pool))) + catalog = sample + [gold_tool] + rng.shuffle(catalog) + hits += (select(intent, catalog) == gold_tool) + curve[n] = hits / trials + return curve +``` + +Empirical anchors this curve should reproduce: a steep monotonic decline absent retrieval pre-filtering — RAG-MCP's full-schema-dump baseline ("Blank Conditioning") bottoms at **13.62%**, and "Actual Match" at 18.20%, versus **43.13%** for retrieval-augmented MCP-RAG [9]; Allen Chan's BFCL-derived study reports gpt-4o calendar accuracy "dropped from 43% with single domain & 4 tools to just 2% with 7 domains and 51 tools," customer-support "from 58%… to just 26%," and "llama-3-3-70b dropped from 21% to 0%" [10] — and a large recovery with progressive disclosure: Anthropic Opus 4 **49%→74%** [8]. + +#### deepdiff-based parameter scorer + +```python +from typing import Any +from deepdiff import DeepDiff + +def parameter_score( + predicted_args: dict[str, Any], + gold_args: dict[str, Any], + *, + ignore_order: bool = True, + significant_digits: int | None = None, +) -> dict[str, float | int]: + """Structural parameter-fill scorer. + + Decouples parameterization failures from selection failures: a tool + can be correctly selected but mis-parameterized (your + 'parameterization' failure class). Granular sub-scores attribute + errors to missing keys vs wrong values vs extras. + """ + diff = DeepDiff( + gold_args, predicted_args, + ignore_order=ignore_order, + significant_digits=significant_digits, + ) + missing = len(diff.get("dictionary_item_removed", [])) + extra = len(diff.get("dictionary_item_added", [])) + changed = len(diff.get("values_changed", {})) + \ + len(diff.get("type_changes", {})) + total_keys = max(len(gold_args), 1) + return { + "exact_match": 1.0 if not diff else 0.0, + "missing_keys": missing, + "extra_keys": extra, # signal for over-parameterization + "changed_values": changed, + "key_recall": 1.0 - missing / total_keys, + } +``` + +#### Inspect AI scorer sketch + +```python +from inspect_ai.scorer import scorer, Score, Target, accuracy, stderr +from inspect_ai.solver import TaskState + +@scorer(metrics=[accuracy(), stderr()]) +def conditional_selection_scorer(k: int = 5): + """CORRECT only if the gold tool was retrieved AND chosen. + Emits metadata so retrieval-miss vs selector-false-negative are + distinguishable in Inspect View.""" + async def score(state: TaskState, target: Target) -> Score: + gold = target.text + retrieved = state.metadata.get("retrieved_tools", [])[:k] + selected = state.metadata.get("selected_tool") + retrieved_ok = gold in retrieved + selected_ok = selected == gold + if not retrieved_ok: + value, kind = "I", "retrieval_miss" + elif not selected_ok: + value, kind = "I", "selector_false_negative" + else: + value, kind = "C", "ok" + return Score(value=value, answer=str(selected), + metadata={"failure_class": kind, + "retrieved_ok": retrieved_ok}) + return score +``` + +### 3. Failure-mode taxonomy for the discovery layer + +Mapping your existing four classes (selection / parameterization / ordering / hallucination) onto discovery-specific extensions, each with a detection signal and scorer type: + +| Failure class | Definition | Detection signal | Scorer type | +|---|---|---|---| +| **Query-planning failure** *(new)* | Model writes a bad search query against the registry | Retrieved set has low overlap with gold; low BM25/embedding score | Log the search query; NDCG@k of *that* query vs an oracle query | +| **Retrieval miss** *(new → selection)* | Right tool exists but not surfaced in top-k | `gold ∉ retrieved[:k]` | recall@k; deterministic set-membership | +| **Selector false-negative** *(new → selection)* | Right tool surfaced but not chosen | `gold ∈ retrieved[:k]` AND `selected ≠ gold` | conditional selection accuracy | +| **Over-selection / over-calling** *(new)* | Too many tools chosen / unnecessary calls | `len(selected) > len(gold)`; extra calls in trace | over-selection rate; trace diff | +| **Namespace / granularity confusion** *(new)* | Confuses similar/overlapping tools (`send-user` vs `send-channel`) | Selected tool is a sibling/near-neighbor of gold | confusion matrix over tool clusters | +| **Tool hallucination** *(your hallucination class)* | Calls a tool that doesn't exist | `selected ∉ registry` | deterministic registry-membership; AST check | +| **Parameterization error** *(your class)* | Correct tool, wrong/missing/extra args | deepdiff non-empty | deepdiff structural scorer | +| **Ordering error** *(your class)* | Correct tools, wrong sequence (nested/dependent) | sequence mismatch vs gold DAG | edit-distance / DAG-topological scorer | +| **Selection error** *(your class, umbrella)* | Wrong tool given correct retrieval | as selector false-negative | conditional selection accuracy | +| **Abstention failure** *(new)* | Acts when no tool applies (or refuses when one does) | `gold_is_none` mismatch | abstention accuracy (MetaTool reliability) | + +**Published taxonomies to cite and build on (2025–2026):** + +- **AgentErrorTaxonomy** ("Where LLM Agents Fail and How They Can Learn From Failures") [16] — a unified failure taxonomy with a two-stage detector; explicitly moves "from descriptive taxonomies to actionable debugging," and catalogs planning brittleness, grounding/tool-use errors, and hallucination-induced cascades. +- **"Characterizing Faults in Agentic AI: A Taxonomy of Types, Symptoms, and Root Causes"** [17] — maps agentic failures to system components (orchestration, evolving internal state, environment feedback), filling the gap left by multi-agent-only analyses. +- **"LLM-based Agents Suffer from Hallucinations: A Survey of Taxonomy, Methods, and Directions"** [18] — taxonomy of agent hallucinations spanning brain/perception/action modules. +- **"Internal Representations as Indicators of Hallucinations in Agent Tool Selection"** (Amazon) [19] — detects tool-calling hallucinations (incorrect tool, malformed parameters, "tool bypass") from internal representations in a single forward pass, "up to 86.4% accuracy," particularly strong at "parameter-level hallucinations and inappropriate tool selections." A concrete *detection signal* for your hallucination and parameterization classes. +- **ToolCert** (Yeon et al., Oct 2025) [37] — adversarial tool-injection certification with Clopper–Pearson bounds on selection accuracy. Use for your distractor-robustness *safety* eval. + +### 4. Synthetic eval-case generation for capability retrieval + +Goal: generate (intent → correct-capability) pairs at scale from your `@command` registry. + +- **Back-translation / inverse generation (tool → synthetic query).** Given a command signature + docstring, prompt an LLM to generate the user intents that *should* route to it. This is the dominant pattern (ToolBench/ToolLLM's instruction generation [24], MTU-Bench's reverse construction from MultiWOZ/SGD intents [30]). The signature is your ground-truth label, so labels are free — but watch for **label leakage**: never include the tool name verbatim in the generated query. Hammer's *function masking* mitigates exactly this by forcing reliance on descriptions over names [22]. +- **Schema-driven generation from tool signatures.** APIGen's three-stage pipeline (format check → actual execution → semantic verification) generated 60k verified entries from 3,673 APIs [20]. RandomWorld inverts the order: sample API-call sequences via type-guided sampling first, then populate the environment [38]. +- **LLM-based user-simulator generation.** τ-bench/τ²-bench's user simulator [4] and IntellAgent's policy-graph-driven synthetic suites generate multi-turn traces; ToolACE uses multi-agent interplay (self-evolution synthesis → 26,507 APIs) with a dual-layer (rule + model) verification system [21]. +- **Distractor injection.** For each (intent, gold tool), inject hard negatives: (a) sibling tools in the same namespace, (b) semantically near tools (embedding nearest neighbors — MassTool's intent-enrichment approach [39]), (c) random tools. The Hammer xLAM-7.5k-Irrelevance set is the canonical "no tool applies" augmentation [22]. +- **Validation / filtering / difficulty calibration.** LLM-as-judge validation (G-Eval/DAG-style), dedup via embedding similarity, and difficulty calibration by binning on retriever score or distractor hardness. MCP-Radar used a strong baseline model (Gemini 2.5 Flash) as a *filter* to ensure tasks genuinely require tool use rather than parametric knowledge "to ensure our benchmark specifically tests tool-use rather than a model's internal knowledge — a common issue of data contamination" [31] — adopt this to avoid trivial cases. + +Registry-driven generation sketch: + +```python +from dataclasses import dataclass +from collections.abc import Sequence + +@dataclass(frozen=True) +class Command: + name: str + description: str + params: dict[str, str] # name -> type + +def make_backtranslation_prompt(cmd: Command, n: int = 5) -> str: + """Generate intents WITHOUT leaking the tool name (function masking).""" + return ( + f"A capability does the following: {cmd.description}\n" + f"It accepts parameters: {list(cmd.params)}\n" + f"Write {n} natural user requests that this capability should " + f"handle. Do NOT mention any function or tool name. Vary phrasing, " + f"specificity, and implied (not explicit) parameter values." + ) + +def inject_distractors( + gold: Command, registry: Sequence[Command], k: int, + embed, namespace_of, +) -> list[str]: + """Hard-negative catalog: siblings + nearest-neighbors + random.""" + siblings = [c.name for c in registry + if namespace_of(c) == namespace_of(gold) and c.name != gold.name] + neighbors = embed.nearest(gold.description, exclude={gold.name}, k=k) + pool = list(dict.fromkeys(siblings + neighbors))[:k - 1] + return pool + [gold.name] +``` + +### 5. Reference implementations / libraries (license-aware) + +You are lock-in-averse; licenses noted. + +- **Inspect AI** (UK AISI + Meridian Labs) — **MIT**, Python ≥3.10 [13]. Your harness backbone: `@scorer`-decorated functions, `TaskState` metadata for retrieval/selection traces, model-graded scorers, `multi_scorer`, MCP tool integration, and `inspect_evals` ships a **BFCL port** (V1/V2/V3 categories, AST matching from the paper's Appendix H) you can fork rather than reimplement [40]. +- **BEIR** — **Apache-2.0** [11]. The standard heterogeneous zero-shot IR benchmark (18 datasets / 9 tasks); its canonical metric is **NDCG@10** ("we… compute nDCG@10 for all datasets"). Use its evaluation interface directly for your retrieval sub-problem. +- **MTEB** — **Apache-2.0** (verify SPDX in repo) [12]. Massive embedding benchmark (8 tasks / 56–58 datasets / 112 languages); its **Retrieval** task is built on BEIR and uses NDCG@10. Use to choose the embedding model for your tool retriever; the HuggingFace MTEB leaderboard is the authoritative source. +- **ToolRet code** (`mangopy/tool-retrieval-benchmark`) — research code; supports embedding + reranker eval in BEIR/MTEB format [1]. The closest existing artifact to what you're building; fork its task format. +- **BFCL harness** (`ShishirPatil/gorilla`) — **Apache-2.0** [3]. `bfcl generate` / `bfcl-eval` PyPI package; AST evaluation scales to thousands of functions; supports vLLM/sglang local serving. +- **τ²-bench / τ³-bench** (`sierra-research/tau2-bench`) — open [4]. `tau2 run` CLI, dual-control env, `pass^k` reliability metric, voice + knowledge-retrieval (BM25) domains. Best for end-task + policy-adherence anchoring. +- **AppWorld** (`StonyBrookNLP/appworld`) — open, with encrypted `.bundle` files + canary string to prevent train leakage [5]; 457 APIs, state-based unit tests, hash-based state diffing. Best stateful-execution anchor. +- **ToolSandbox** (`apple/ToolSandbox`) — **custom Apple license, NOT Apache-2.0** — review terms before redistribution [29]. Stateful, on-policy conversational eval; "open source and proprietary models have a significant performance gap" on state-dependency / canonicalization / insufficient-information tasks. +- **StableToolBench** (`THUNLP-MT`) — open; virtual API server + MirrorAPI simulator for reproducible ToolBench-style eval [28]. Use instead of raw ToolBench. +- **APIGen / xLAM data** (Salesforce) — datasets on HF (`xlam-function-calling-60k`); APIGen pipeline for verifiable generation [20]. **Hammer / xLAM-7.5k-Irrelevance** (MadeAgents) — irrelevance/abstention data + function-masking recipe [22]. +- **RAG-MCP reference impls** (`fintools-ai/rag-mcp`, `memoverflow/rag-mcp`) — open; vector-index-over-tool-metadata pattern with the MCP stress test you can adapt as a distractor-robustness harness [9]. + +**deepdiff** itself is MIT — safe for your parameter scorer. **polyfactory** (MIT) for schema-driven synthetic param generation; **DSPy/GEPA** for optimizing the retrieval query the model writes. + +### 6. Open problems and gotchas + +- **Benchmark contamination & saturation.** BFCL, being old and popular, is contamination-prone and saturating (top models clustered, MMLU-style plateau) [35,41]. Systematic studies show saturation is driven by benchmark age and test-set scale as much as by capability [42]. **Mitigation:** generate your own private, registry-derived eval set; rotate/regenerate it; keep an AppWorld-style canary string to detect leakage into future training [5]. +- **Static-benchmark vs production gap.** Static accuracy on a fixed catalog does not predict production capability-discovery, where the catalog grows, tools are versioned, and descriptions vary in quality. As MarkTechPost notes of Anthropic's Tool Search, "~26 percentage points of accuracy is still retrieval failure on Opus 4… Tool Search assumes the model can write a reasonable search query" [43]. Measure on *your* catalog. +- **Non-determinism / reproducibility.** LLM-judge variance (StableToolBench moved to solvable-pass-rate + an end-to-end trained evaluator to reduce it [28]), sampling temperature, and live-API instability all undermine regression testing. **Mitigation:** prefer deterministic structural scorers (deepdiff, set-membership, AST) over LLM judges wherever the label is structural; reserve LLM-judge for genuinely open-ended cases; report `pass^k`-style reliability, not just mean accuracy [4]. +- **No standardized retrieval-vs-selection decoupling.** This is the field's biggest methodological gap and your biggest opportunity — almost every benchmark pre-annotates candidate tools, hiding retrieval [1]. Your conditional-selection-accuracy metric is genuinely novel contribution territory. +- **Distractor-robustness is under-measured.** Only RAG-MCP and a handful of 2025–2026 papers run the stress test systematically [9]. Make the distractor-robustness curve a first-class, always-reported metric. +- **Progressive-disclosure / tool-search systems need their own evals.** When `defer_loading` is on [8,14], you must evaluate (a) the search query the model writes, (b) whether the right tool is surfaced, (c) the post-retrieval selection — three separate scores. Standard benchmarks evaluate none of these. Note the token-tax stakes: Anthropic's five-server example is "58 tools consuming approximately 55K tokens before the conversation even begins," and tool definitions can "consume 134K tokens before optimization," with an 85% token reduction under Tool Search [8]. +- **MCP-specific gaps.** MCP-RADAR [31], MCP-Atlas [32], and MCPToolBench++ are the only MCP-native benchmarks, all very new. Namespace collisions across servers (`notification-send-*`), schema-bloat token tax, and cross-server orchestration are largely unmeasured. +- **Counterfactual / safety evals.** ToolCert (adversarial tool injection with statistical bounds) [37] and SafeToolBench are the current state of the art. For the discovery layer specifically, test: (a) does injecting a malicious near-duplicate tool divert selection? (b) does the agent abstain when no safe tool applies? (c) does over-selection expose unnecessary attack surface? + +--- + +## Recommendations + +**Stage 1 — Instrument the decoupling (week 1–2).** Add `retrieved_tools` and `selected_tool` to your Inspect `TaskState.metadata` on every run. Ship recall@k, NDCG@k, conditional selection accuracy, and the distractor-robustness curve as standing metrics. *Threshold that changes the plan:* if recall@k is high (>0.9) but conditional selection accuracy is low, your bottleneck is the **selector** (invest in descriptions / few-shot / reasoning); if recall@k is low, your bottleneck is **retrieval** (invest in the embedding model / query planning). + +**Stage 2 — Build the private eval set (week 2–4).** Back-translate intents from your `@command` registry with function-masking (no name leakage), inject sibling + nearest-neighbor + random distractors, and add a Hammer-style "no tool applies" abstention slice (target ≥15% of cases). Validate with an LLM judge + a strong-model filter (MCP-Radar style) to drop trivial/parametric-knowledge cases; dedup by embedding similarity. *Target:* ≥1,000 cases spanning catalog sizes 1→128. + +**Stage 3 — Adopt ToolRet + BEIR framing for retrieval; reuse the BFCL Inspect port for selection (week 4–6).** Score your retriever exactly as ToolRet does (NDCG@10 over your full catalog as the corpus; remember the field-leading embedding model only hit 33.83 there — set realistic targets). For selection AST/structural checks, fork `inspect_evals/bfcl` rather than reimplementing. Use deepdiff for parameters. + +**Stage 4 — Anchor to end-task, don't measure discovery with it (ongoing).** Run τ²-bench and AppWorld quarterly to confirm discovery-layer gains translate to task success and don't introduce collateral damage. Treat their scores as *guardrails*, not *targets*. + +**Stage 5 — Add safety/counterfactual evals before any production catalog expansion.** Implement a ToolCert-style adversarial near-duplicate injection test and report Clopper–Pearson bounds on selection accuracy. *Threshold:* block catalog expansion if adversarial injection drops conditional selection accuracy beyond a pre-agreed margin. + +**What would change these recommendations:** If a standardized retrieval-vs-selection benchmark emerges (watch ToolRet successors and MCP-Atlas), adopt it and reallocate effort from building to running. If your catalog stays under ~20 tools, the distractor problem is minor and you can defer progressive-disclosure evaluation. If Anthropic/MCP ship a native progressive-disclosure eval harness, integrate rather than rebuild. + +--- + +## Caveats + +- **Several headline numbers come from vendor/secondary sources, not peer review.** Anthropic's Tool Search figures (Opus 4: 49%→74%; Opus 4.5: 79.5%→88.1%) are from Anthropic's own engineering blog [8]; treat as directional, not independently replicated. The BFCL "43%→2%" and "58%→26%" figures are from a practitioner analysis (Allen Chan, Medium) summarizing BFCL-derived experiments, not the BFCL paper itself [10]. The "~26 pp still retrieval failure" line is MarkTechPost commentary [43], not an Anthropic statement. +- **ToolRet's "33.83 NDCG@10 for NV-Embed-v1" was verified against the primary ACL Findings 2025 / arXiv source** ("even the best model… achieves an nDCG@10 of only 33.83") [1] — safe to quote. +- **ToolSandbox's exact tool count was not verbatim-confirmed** from the primary source; do not cite a specific number without checking the arXiv PDF. Its license is a bespoke Apple license materially different from BEIR/MTEB's Apache-2.0 [29]. +- **MCP-Atlas (arXiv 2602.00933) and several 2026 papers are very recent** and may not be peer-reviewed; the "Claude Opus 4.5 = 62.3%" figure is from the preprint [32]. +- **Benchmark scores age fast.** All leaderboard numbers (BFCL GLM-4.5 0.778, etc.) are snapshots; re-check before relying on them. +- The author's "~49% baseline tool-selection accuracy at 50+ tools" aligns strikingly with Anthropic's pre-Tool-Search Opus 4 figure (49%) [8] and with AppWorld's GPT-4o normal-task success (~49%) [5], but these measure different things (selection vs end-task); the convergence is coincidental and should not be over-interpreted. + +--- + +## References + +1. Shi Z, Wang Y, Yan L, Ren P, Wang S, Yin D, Ren Z. Retrieval Models Aren't Tool-Savvy: Benchmarking Tool Retrieval for Large Language Models. ACL Findings 2025. [https://aclanthology.org/2025.findings-acl.1258/](https://aclanthology.org/2025.findings-acl.1258/) · arXiv: [https://arxiv.org/abs/2503.01763](https://arxiv.org/abs/2503.01763) · code: [https://github.com/mangopy/tool-retrieval-benchmark](https://github.com/mangopy/tool-retrieval-benchmark) +2. Huang Y, et al. MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use. ICLR 2024. [https://arxiv.org/abs/2310.03128](https://arxiv.org/abs/2310.03128) · code: [https://github.com/HowieHwong/MetaTool](https://github.com/HowieHwong/MetaTool) +3. Patil SG, Mao H, et al. The Berkeley Function Calling Leaderboard (BFCL): From Tool Use to Agentic Evaluation of Large Language Models. ICML 2025. [https://proceedings.mlr.press/v267/patil25a.html](https://proceedings.mlr.press/v267/patil25a.html) · leaderboard (V4): [https://gorilla.cs.berkeley.edu/leaderboard.html](https://gorilla.cs.berkeley.edu/leaderboard.html) · code: [https://github.com/ShishirPatil/gorilla](https://github.com/ShishirPatil/gorilla/tree/main/berkeley-function-call-leaderboard) +4. Barres V, Dong H, Ray S, Si X, Narasimhan K. τ²-Bench: Evaluating Conversational Agents in a Dual-Control Environment. 2025. [https://arxiv.org/abs/2506.07982](https://arxiv.org/abs/2506.07982) · code: [https://github.com/sierra-research/tau2-bench](https://github.com/sierra-research/tau2-bench) +5. Trivedi H, Khot T, et al. AppWorld: A Controllable World of Apps and People for Benchmarking Interactive Coding Agents. ACL 2024. [https://aclanthology.org/2024.acl-long.850/](https://aclanthology.org/2024.acl-long.850/) · site: [https://appworld.dev/](https://appworld.dev/) · code: [https://github.com/StonyBrookNLP/appworld](https://github.com/StonyBrookNLP/appworld) +6. Ye J, et al. ToolHop: A Query-Driven Benchmark for Evaluating Large Language Models in Multi-Hop Tool Use. 2025. [https://arxiv.org/abs/2501.02506](https://arxiv.org/abs/2501.02506) +7. Chen C, et al. ACEBench: Who Wins the Match Point in Tool Usage? 2025. [https://arxiv.org/abs/2501.12851](https://arxiv.org/html/2501.12851) +8. Anthropic. Introducing advanced tool use on the Claude Developer Platform. 2025. [https://www.anthropic.com/engineering/advanced-tool-use](https://www.anthropic.com/engineering/advanced-tool-use) +9. Gan T, Sun Q. RAG-MCP: Mitigating Prompt Bloat in LLM Tool Selection via Retrieval-Augmented Generation. 2025. [https://arxiv.org/abs/2505.03275](https://arxiv.org/abs/2505.03275) +10. Chan A. How Tool Complexity Impacts AI Agents Selection Accuracy. Medium, 2025. [https://achan2013.medium.com/how-tool-complexity-impacts-ai-agents-selection-accuracy-a3b6280ddce5](https://achan2013.medium.com/how-tool-complexity-impacts-ai-agents-selection-accuracy-a3b6280ddce5) +11. Thakur N, Reimers N, Rücklé A, Srivastava A, Gurevych I. BEIR: A Heterogeneous Benchmark for Zero-shot Evaluation of Information Retrieval Models. NeurIPS Datasets & Benchmarks 2021. [https://arxiv.org/abs/2104.08663](https://arxiv.org/abs/2104.08663) · code: [https://github.com/beir-cellar/beir](https://github.com/beir-cellar/beir) +12. Muennighoff N, Tazi N, Magne L, Reimers N. MTEB: Massive Text Embedding Benchmark. 2022. [https://arxiv.org/abs/2210.07316](https://arxiv.org/abs/2210.07316) · code: [https://github.com/embeddings-benchmark/mteb](https://github.com/embeddings-benchmark/mteb) +13. UK AI Security Institute. Inspect AI: A Framework for Large Language Model Evaluations. [https://inspect.aisi.org.uk/](https://inspect.aisi.org.uk/) · code: [https://github.com/UKGovernmentBEIS/inspect_ai](https://github.com/UKGovernmentBEIS/inspect_ai) +14. Shihipar T (Anthropic). MCP Tool Search for Claude Code (announcement, 14 Jan 2026), via atcyrus. [https://www.atcyrus.com/stories/mcp-tool-search-claude-code-context-pollution-guide](https://www.atcyrus.com/stories/mcp-tool-search-claude-code-context-pollution-guide) +15. Synaptic Labs. The Meta-Tool Pattern: Progressive Disclosure for MCP. 2025. [https://blog.synapticlabs.ai/bounded-context-packs-meta-tool-pattern](https://blog.synapticlabs.ai/bounded-context-packs-meta-tool-pattern) +16. Where LLM Agents Fail and How They Can Learn From Failures (AgentErrorTaxonomy / AgentDebug). 2025. [https://arxiv.org/abs/2509.25370](https://arxiv.org/pdf/2509.25370) +17. Characterizing Faults in Agentic AI: A Taxonomy of Types, Symptoms, and Root Causes. 2026. [https://arxiv.org/abs/2603.06847](https://arxiv.org/html/2603.06847v1) +18. Lin X, et al. LLM-based Agents Suffer from Hallucinations: A Survey of Taxonomy, Methods, and Directions. 2025. [https://arxiv.org/abs/2509.18970](https://arxiv.org/abs/2509.18970) +19. Healy K, Srinivasan B, Madathil V, Wu J (Amazon). Internal Representations as Indicators of Hallucinations in Agent Tool Selection. 2026. [https://arxiv.org/abs/2601.05214](https://arxiv.org/pdf/2601.05214) +20. Liu Z, Hoang T, Zhang J, et al. APIGen: Automated Pipeline for Generating Verifiable and Diverse Function-Calling Datasets. NeurIPS 2024. [https://arxiv.org/abs/2406.18518](https://arxiv.org/abs/2406.18518) · site: [https://apigen-pipeline.github.io/](https://apigen-pipeline.github.io/) +21. Liu W, et al. ToolACE: Winning the Points of LLM Function Calling. 2024. [https://arxiv.org/abs/2409.00920](https://arxiv.org/html/2409.00920v2) +22. Lin Q, et al. Hammer: Robust Function-Calling for On-Device Language Models via Function Masking. ICLR 2025. [https://arxiv.org/abs/2410.04587](https://arxiv.org/abs/2410.04587) · code: [https://github.com/MadeAgents/Hammer](https://github.com/MadeAgents/Hammer) +23. Liu J, et al. (NAACL 2025) Boosting LLM Tool-Calling Through Natural and Coherent Dialogue Synthesis (ToolFlow). [https://aclanthology.org/2025.naacl-long.214.pdf](https://aclanthology.org/2025.naacl-long.214.pdf) +24. Qin Y, et al. ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs (ToolBench). ICLR 2024. [https://arxiv.org/abs/2307.16789](https://arxiv.org/abs/2307.16789) · code: [https://github.com/OpenBMB/ToolBench](https://github.com/OpenBMB/ToolBench) +25. Yao S, Shinn N, Razavi P, Narasimhan K. τ-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains. 2024. [https://arxiv.org/abs/2406.12045](https://arxiv.org/abs/2406.12045) +26. Zhong L, Du Z, Zhang X, Hu H, Tang J. ComplexFuncBench: Exploring Multi-Step and Constrained Function Calling under Long-Context Scenario. 2025. [https://arxiv.org/abs/2501.10132](https://github.com/zai-org/ComplexFuncBench) +27. Basu K, et al. NESTFUL: A Benchmark for Evaluating LLMs on Nested Sequences of API Calls. EMNLP 2025. [https://aclanthology.org/2025.emnlp-main.1702/](https://aclanthology.org/2025.emnlp-main.1702/) · arXiv: [https://arxiv.org/abs/2409.03797](https://arxiv.org/abs/2409.03797) +28. Guo Z, et al. StableToolBench: Towards Stable Large-Scale Benchmarking on Tool Learning of Large Language Models. ACL Findings 2024. [https://arxiv.org/abs/2403.07714](https://arxiv.org/abs/2403.07714) · code: [https://github.com/THUNLP-MT/StableToolBench](https://github.com/THUNLP-MT/StableToolBench) +29. Lu J, Holleis T, Zhang Y, et al. (Apple). ToolSandbox: A Stateful, Conversational, Interactive Evaluation Benchmark for LLM Tool Use Capabilities. 2024. [https://arxiv.org/abs/2408.04682](https://arxiv.org/abs/2408.04682) · code: [https://github.com/apple/ToolSandbox](https://github.com/apple/ToolSandbox) +30. Wang P, et al. MTU-Bench: A Multi-granularity Tool-Use Benchmark for Large Language Models. 2024. [https://arxiv.org/abs/2410.11710](https://arxiv.org/abs/2410.11710) · site: [https://mtu-bench-team.github.io/](https://mtu-bench-team.github.io/) +31. Gao X, et al. MCP-RADAR: A Multi-Dimensional Benchmark for Evaluating Tool Use Capabilities in Large Language Models. 2025. [https://arxiv.org/abs/2505.16700](https://arxiv.org/abs/2505.16700) +32. MCP-Atlas: A Large-Scale Benchmark for Tool-Use Competency with Real MCP. 2026. [https://arxiv.org/pdf/2602.00933](https://arxiv.org/pdf/2602.00933) +33. Wang J, et al. GTA: A Benchmark for General Tool Agents (NeurIPS 2024 D&B) & GTA-2 (2026). [https://openreview.net/forum?id=akEt8QAa6V](https://openreview.net/forum?id=akEt8QAa6V) · code: [https://github.com/open-compass/GTA](https://github.com/open-compass/GTA) +34. Li M, et al. API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs. EMNLP 2023. [https://aclanthology.org/2023.emnlp-main.187/](https://aclanthology.org/2023.emnlp-main.187/) +35. iternal.ai. Which LLM to Choose in 2026? Selection Guide + Benchmarks (BFCL saturation discussion). 2026. [https://iternal.ai/llm-selection-guide](https://iternal.ai/llm-selection-guide) +36. LLM Stats. BFCL v2 Benchmark Leaderboard (contamination/bias via live scenarios). [https://llm-stats.com/benchmarks/bfcl-v2](https://llm-stats.com/benchmarks/bfcl-v2) +37. Yeon J, et al. ToolCert: Adversarial Certification of Tool-Selection Accuracy (Clopper–Pearson bounds). 2025. Summarized in Emergent Mind, Tool Selection Accuracy. [https://www.emergentmind.com/topics/tool-selection-accuracy-ts](https://www.emergentmind.com/topics/tool-selection-accuracy-ts) +38. RandomWorld: Procedural Environment Generation for Tool-Use Agents (type-guided sampling). 2026. [https://arxiv.org/html/2601.17829](https://arxiv.org/html/2601.17829) +39. Lin et al. MassTool: search-based user intent modeling for tool retrieval. 2025. Discussed in: Tools Are Under-Documented: Document Expansion Boosts Tool Retrieval. [https://arxiv.org/pdf/2510.22670](https://arxiv.org/pdf/2510.22670) +40. Inspect Evals. BFCL: Berkeley Function-Calling Leaderboard port. [https://ukgovernmentbeis.github.io/inspect_evals/evals/assistants/bfcl/](https://ukgovernmentbeis.github.io/inspect_evals/evals/assistants/bfcl/) +41. Orq.ai. LLM Benchmarks Explained: Significance, Metrics & Challenges (contamination, saturation). 2025. [https://orq.ai/blog/llm-benchmarks](https://orq.ai/blog/llm-benchmarks) +42. When AI Benchmarks Plateau: A Systematic Study of Benchmark Saturation. 2026. [https://arxiv.org/html/2602.16763v1](https://arxiv.org/html/2602.16763v1) +43. MarkTechPost. Hermes Agent Ships Tool Search for MCP: Anthropic Evals Show 49% to 74% Accuracy Gain on Opus 4. 29 May 2026. [https://www.marktechpost.com/2026/05/29/hermes-agent-ships-tool-search-for-mcp-anthropic-evals-show-49-to-74-accuracy-gain-on-opus-4/](https://www.marktechpost.com/2026/05/29/hermes-agent-ships-tool-search-for-mcp-anthropic-evals-show-49-to-74-accuracy-gain-on-opus-4/) +44. Lunar.dev. Dynamic Tool Selection for AI Agents: Solving the Context Management Problem. 2026. [https://www.lunar.dev/post/why-dynamic-tool-discovery-solves-the-context-management-problem](https://www.lunar.dev/post/why-dynamic-tool-discovery-solves-the-context-management-problem) +45. Tian Pan. The Tool Selection Problem: How Agents Choose What to Call When They Have Dozens of Tools. 2026. [https://tianpan.co/blog/2026-04-09-tool-selection-problem-agent-tool-routing-at-scale](https://tianpan.co/blog/2026-04-09-tool-selection-problem-agent-tool-routing-at-scale) \ No newline at end of file diff --git a/misc/docs/ir_04 -- Architecture & Reuse Analysis -- Building ir on the ef + vd Substrate.md b/misc/docs/ir_04 -- Architecture & Reuse Analysis -- Building ir on the ef + vd Substrate.md new file mode 100644 index 0000000..96aa4b7 --- /dev/null +++ b/misc/docs/ir_04 -- Architecture & Reuse Analysis -- Building ir on the ef + vd Substrate.md @@ -0,0 +1,498 @@ +# ir_04 — Architecture & Reuse Analysis: Building `ir` on the `ef` + `vd` Substrate + +> **Purpose.** This document does for `ir` what the user explicitly asked: lay out the +> vision, then map every piece of it against what the ecosystem (`ef`, `vd`, and friends) +> *already provides*, so `ir` is built by **composition, not reinvention**. The headline +> finding is that the lower ~70–80% of `ir`'s stated substrate already exists and is +> mature — most of it in `ef` (indexing/maintenance/retrieval) and `vd` (store + +> metadata filtering). `ir`'s genuine, non-redundant work is a comparatively thin (but +> conceptually hard) band at the top: **multi-surface artifact indexing**, a **selection +> stage**, **corpus-source adapters** for the first concrete corpora, an **agent-callable +> tool surface**, and a **capability-discovery evaluation harness**. +> +> Code is referenced repo-relative under `$PP` (the projects folder): `t/ef/…`, `i/vd/…`, +> etc. Companion research lives alongside this file: `ir_01` (progressive capability +> discovery), `ir_02` (indexing/embedding strategy), `ir_03` (evaluation). + +--- + +## 0. Executive summary + +| Layer of the `ir` vision | Who already provides it | `ir`'s job | +|---|---|---| +| Corpus abstraction (any store as a corpus) | **`ef.corpus`** (`Corpus = MutableMapping`, `as_corpus`) + **`dol`** | Reuse wholesale | +| Change detection / staleness | **`ef`** (`content_hash`, `ChangeDetectingCorpus`, `diagnostics`, `refresh`) | Reuse wholesale; supply per-source signals | +| Corpus maintenance as a repeatable process | **`ef.artifact_graph`** (content-addressed producer DAG, cascade invalidation, config branching) | Reuse wholesale — this *is* the "maintenance contract" | +| Segmentation | **`ef.segmenters`** + **`imbed`** | Reuse | +| Embedding (facade + adapters + wrappers) | **`ef.embedders`** (OpenAI, Voyage, Cohere, SBERT, Gemini, HTTP, hashing) | Reuse | +| Vector store (16 backends) | **`vd`** | Reuse wholesale | +| Metadata filtering (9 Mongo-style ops, `$and/$or/$not/$in/$exists`) | **`vd.filters`** | Reuse wholesale — this is the "hard filter" half of the vision | +| Hybrid retrieval (BM25 + dense + RRF), multi-query | **`vd.search`** | Reuse; extend for scale/large lexical | +| Reranking | **`ef.reranking`** (protocol + cross-encoder) | Reuse; add instruction-steered rerankers | +| Retrieval evaluation (BEIR/MTEB-shaped) | **`ef.evaluation`** | Reuse for IR metrics; **build** capability-discovery eval on top | +| HTTP / agent-callable surface | **`qh`** (+ `ef.service` pattern) | Compose | +| LLM-authored synopsis / problem-class tags | **`oa`** (`prompt_function`) | Compose | +| Corpus sources (skills, packages, GitHub, docs) | **`priv.skills_index`, `projreg`, `hubcap`, `contaix`** | Compose as scope/change-detection adapters | +| **Multi-surface artifact indexing** (one artifact → many heterogeneous filterable+embeddable units) | *nobody — partial in `ef` multi-config* | **BUILD (core seam)** | +| **Selection stage** (distractor-robust commit to a subset) | *nobody* | **BUILD** | +| **Capability-artifact model** (tool / skill / subagent as indexable+executable) | *nobody* | **BUILD** | +| **Agentic "one search tool" + progressive payload disclosure** | *nobody (defer-loading is host-side)* | **BUILD (thin)** | + +**The single most important strategic question this analysis surfaces** (see §8): `ef`'s +own one-line self-description — *"a facade for semantic-embeddings user journeys … `ef` is +**not** RAG — it returns ranked segments; bring your own LLM"* — is **almost verbatim the +`ir` vision's** "retrieval is the core competency … generation/reranking/citation are +layered on top." `ef` and `ir` are positioned on the same spectrum. `ir` must be defined as +a **distinct layer above `ef`**, not a parallel reimplementation of it. The rest of this +document assumes that boundary and draws it precisely. + +--- + +## 1. The vision, restated — and where it already lives + +The `ir` brief makes five load-bearing claims. Each maps onto existing machinery: + +1. **"A uniform retrieval contract across the whole scale spectrum"** — from ad-hoc `find` + over an ephemeral list to a millions-of-docs search engine, same facade. + → This is exactly `ef`'s **progressive-disclosure facade**: the light path + `ingest([...]) → SearchableCorpus` (one line, hashing embedder, in-memory `vd`) and the + heavy path `SourceManager(corpus, …).materialize().search()` share one surface + (`t/ef/ef/source_manager.py`). The store seam (`_open_store`, `t/ef/ef/source_manager.py:956`) + already swaps in-memory ↔ SQLite ↔ pgvector ↔ dedicated vector DB without changing the + caller's contract — precisely the vision's "any backing store, any retrieval strategy, + caller's contract unchanged." + +2. **"Extensible into RAG without being one by default"** — retrieval core; generation + layered on. + → Verbatim `ef`'s stance. `ef.search` returns `SearchHit`s; `ef.retrieve` returns plain + `Segment`s as the RAG plug-in surface (`t/ef/ef/source_manager.py:816`). Generation is + deliberately out of scope. + +3. **"Corpus maintenance as a defined, repeatable process,"** with two pluggable slots — + **scope** (what's in the corpus) and **change detection** (what's stale). + → `ef` already implements both halves of this and more: + - *Change detection*: `content_hash` (SHA-256 over normalized content, text-only for + mappings so metadata edits don't churn — `t/ef/ef/corpus.py:109`) and + `ChangeDetectingCorpus` (fires `added/modified/deleted` events only on real hash + change — `t/ef/ef/corpus.py`). + - *Staleness diagnosis*: four conditions — **orphan / missing / stale / misconfigured** — + computed read-only by `diagnose()` (`t/ef/ef/diagnostics.py`). + - *Repeatable refresh*: four modes — `none / incremental / full / scoped_full` — via a + **pure** `plan_refresh()` → `RefreshPlan` and `refresh_on_change()` handler + (`t/ef/ef/refresh.py`). + - *Cascade invalidation + config branching as one operation*: the **artifact graph** + (§2.3). This is the deepest part of the vision's "maintenance contract," and it is + already built. + What `ef` does **not** prescribe is the *concrete* scope (globs vs API queries vs GitHub) + and *concrete* change signal (mtime vs ETag vs version) per source — exactly the + "abstract slots, concrete-per-source" the vision wants. Those concrete definitions are + where `ir` plugs in existing source packages (§5). + +4. **"Good retrieval is not all embeddings — it is also classic metadata filtering,"** with + hard filters (ownership, package name, domain) constraining the candidate set *before/ + alongside* semantic ranking. + → `vd.filters` is a complete Mongo-style filter language — `$eq/$ne/$gt/$gte/$lt/$lte/ + $in/$nin/$exists` plus `$and/$or/$not` — validated per-backend and pushed down to native + filters where supported (`i/vd/vd/filters.py`, `i/vd/vd/base.py:578`). `ef` already + threads a `filter=` straight through search to `vd` + (`t/ef/ef/source_manager.py:260`). The "hard filter, not fuzzy match" requirement is a + solved problem at the `vd` layer. + +5. **"Deciding what to index is a problem in its own right — not one `ir` should solve + internally; expose the seams."** + → This is the **one place the existing substrate is genuinely insufficient** and is the + heart of `ir`'s new work. See §4. + +**Conclusion of §1:** four of the five pillars are substantially built. The fifth — the +indexing-strategy seam (artifact decomposition into filterable fields + multi-granularity +embeddable surfaces) — is `ir`'s defining contribution, together with the agent-facing +selection/disclosure/eval band that the three `ir_0x` research docs specify. + +--- + +## 2. What `ef` already gives `ir` (the indexing / maintenance / retrieval spine) + +`ef` is a layered facade: **corpus (L0) → segment (L2) → embed (L3) → index in `vd` (L4) → +derive/explore (L5)**, with a cross-cutting content-addressed **artifact graph** and a +**diagnose/refresh** maintenance loop. Reuse verdicts below are "wholesale" unless noted. + +### 2.1 Corpus + change detection (`ef.corpus`, `ef.hashing`) +- `Corpus = MutableMapping[source_id, str | Mapping]`; `as_corpus(None | mapping | dir-path + | iterable)` is the DI seam. A corpus is *any* `dol` store — RAM, filesystem, S3, an API + view. **Reuse wholesale.** +- `content_hash(source, content_keys=…)` — normalized (NFC, `\n`, no BOM) SHA-256; for + mappings hashes `text` only unless `content_keys` opts metadata in. This is the + idempotency primitive every maintenance decision rests on. **Reuse wholesale.** +- `ChangeDetectingCorpus(on_change=…)` wraps any corpus, emitting `ChangeEvent(source_id, + kind, old_hash, new_hash)` and `CorpusDiff`. **Reuse wholesale.** (Out-of-band edits — + a file changed on disk — are caught by `SourceManager.scan()`, which re-hashes the + corpus.) + +### 2.2 Segment / embed facades +- `Segment` TypedDict + `SegmentRecord` (canonical interchange: `text, id, metadata, + parent_id, index, start, end, tokens`; `PROMOTED_METADATA_KEYS` lifts `source, + source_type, tokenizer, token_count, embedding_model, page, license, ingestion_run_id` + into the stored doc). **Reuse wholesale** — this is the data model `ir` records flow + through. +- `Segmenter` protocol + `RecursiveCharacterSegmenter`, `line_segmenter`, `with_overlap`, + `hierarchical`, `materialise`. **Reuse** (and/or `imbed`'s `fixed_step_chunker`). +- `Embedder` protocol (`Iterable[str] -> ndarray(n, dim)`), `as_embedder` DI seam, + `HashingEmbedder` (numpy-only default), adapters (`openai_/voyage_/cohere_/ + sentence_transformers_/gemini_/http_embedder`), wrappers (`Cached/Retrying/Multi/ + Normalizing`). Carries `InputType ∈ {query, document, classification, clustering}` for + asymmetric embedding. **Reuse wholesale.** Note `ir_02`'s recommendation of + *instruction-tuned* embedders and per-query instruction steering maps onto extending the + `InputType`/adapter mechanism, not replacing it. + +### 2.3 The artifact graph — the maintenance engine (`ef.artifact_graph`) +This is the single most valuable reuse for `ir`. Every produced artifact is addressed by its +**recipe**: +``` +artifact_id = H(op_key, op_version, input_ids, params) +``` +so two pipeline configs that share an upstream step share that step's artifact — *segment +once, embed once*, no matter how many configs branch through. Operations: +- `materialize(id)` — lazy backward compute (cache hit if present), recursing into inputs; +- `mark_stale(leaf)` — forward cascade invalidation when a source changes; +- `delete_cascade(id)` — forward delete of an artifact and everything depending only on it; +- `freshness(id) → materialized | stale | unknown`. + +Stores (`store / producers / edges`) are injectable `MutableMapping`s, so the graph persists +to any `dol` backend (SQLite for ≫10⁶ nodes) with **no DB dependency baked into `ef`**. +`ProducerSpec` uses *string* op keys (e.g. `"embed:openai:text-embedding-3-large@1024"`) +resolved through an `ops` registry at materialize time — so the spec is serializable and +branchable. **Reuse wholesale.** The vision's "keep the index (ledger, embeddings, derived +metadata) in sync with a living, mutable corpus" is this graph plus diagnose/refresh. + +### 2.4 Retrieval surface +- `SourceManager` holds corpus + graph + N named configs (each a `(segmenter, embedder)` + pipeline → its own `vd` collection); `SearchableCorpus` is the light wrapper. +- `search(query, *, config, limit, filter) → list[SearchHit(segment, score, source_id)]`; + `retrieve(...) → Segment[]`. Every stored doc carries `source_id, source_hash, + config_hash` provenance (`t/ef/ef/source_manager.py:232`). +- `ef.reranking`: `Reranker` protocol `(query, segments) → scores`, `rerank()`, + `with_reranker(retriever, reranker, fetch_k=50)` decorator, `cross_encoder_reranker`. + **Reuse**; `ir`/`ir_02` add instruction-following rerankers (Voyage rerank-2.5, + Qwen3-Reranker) behind the same protocol. +- `ef.exploration` (`project/cluster/label_clusters/explore`) and `ef.evaluation` + (BEIR/MTEB retrieval metrics, Ragas bridge) — reuse as needed. + +### 2.5 What `ef` does **not** have (relevant to `ir`) +- **Multi-surface indexing of a single structured artifact.** `ef` multi-config gives you + *multiple embedders over the same text* and segment→chunk decomposition. It does **not** + model "one artifact (a package) → a *heterogeneous* set of sub-surfaces (name field, + short description, AI synopsis, per-module text, problem-class list) where some are + **filterable metadata** and some are **separately embedded vectors** with their own + granularity and kind." (§4.) +- **A selection stage.** `ef` returns ranked hits; it never commits to a distractor-robust + subset. (§3, `ir_01`/`ir_03`.) +- **Lexical/hybrid** lives in `vd`, not `ef` (so `ir` reaches for `vd.hybrid_search`). +- **Agent orchestration, answer synthesis, the "one search tool" disclosure protocol** — + deliberately out of scope for `ef`; partly `ir`'s job, partly the host's. + +--- + +## 3. What `vd` already gives `ir` (the store + filter substrate) + +`vd` is a production-grade vector-DB facade. For `ir` it supplies the **entire** storage and +metadata-filtering layer: +- **Contract**: `Document(id, text, vector, metadata)`, `Vector`, `SearchResult(id, text, + score, metadata)`, higher-is-better score normalization across `cosine/dot/l2`. Collections + are `MutableMapping[str, Document] + search()`; adapters implement six raw primitives + (`_write/_read/_drop/_keys/_count/_query`) — `i/vd/vd/base.py`. +- **Metadata filtering**: the 9-operator Mongo-style language with `$and/$or/$not`, validated + per backend (`supported_filter_operators`) and translated to native filters; pure-Python + `matches_filter` is the reference semantics — `i/vd/vd/filters.py`. **This is the vision's + "hard filter, not fuzzy match" requirement, already solved.** +- **Search modes** (`i/vd/vd/search.py`): dense kNN; `hybrid_search` = dense + **BM25** fused + with **Reciprocal Rank Fusion** (native where a backend supports it, client-side fallback + otherwise); `multi_query_search` (`interleave/concatenate/union/best`); + `search_similar_to_document`; standalone `reciprocal_rank_fusion` and `deduplicate_results`. +- **16 backends** via `@register_backend`: `memory` (brute-force reference), `chroma`, + `lancedb`, `sqlite_vec`, `duckdb`, `faiss`, `qdrant`, `weaviate`, `milvus`, `redis`, + `elasticsearch`, `mongodb`, `pgvector`, `pinecone`, `turbopuffer`. Plus `vd.recommend_backend(...)` + to pick by scale/latency/deployment, `check_requirements`/`setup_guide`, JSON/JSONL/dir + import-export, cross-backend `migrate_collection`, analytics (`collection_stats, + find_duplicates, find_outliers`), `chunk_text`/`chunk_documents`/`clean_text`, + `TimeIndexedCollection`, and async wrappers. + +**`vd` gaps relevant to `ir`:** the client-side BM25 fallback is O(N) (fine < ~100k docs; +push lexical into Elasticsearch/pgvector-FTS or a native-hybrid backend at scale); **no +first-class sparse-vector type** (BM25 is post-hoc, not a stored sparse vector); **no +late-interaction/ColBERT** multi-vector-per-doc; **no reranking layer** (that's `ef`'s). Per +`ir_02`, none of these are urgent: document expansion and instruction-tuned embedders + +rerank dominate the accuracy budget; ColBERT was *not* justified on tool-retrieval +benchmarks. + +--- + +## 4. The core new work: the "what do we index?" seam (IndexingStrategy) + +This is where `ir` earns its existence. The vision is precise: **a package is not one +document** — it is a hierarchy of indexable surfaces with two fundamentally different roles: + +- **Structured metadata for *filtering*, not embedding** — `name`, ownership (ours vs. + blessed-third-party), domain/problem-class tags, maturity, dependencies, license. "Ours + vs. third-party" is a *hard filter*; package `name` is a *first-class filterable field*. +- **Embeddable representations at multiple granularities and of multiple *kinds*** — a short + canonical description; a longer AI-authored synopsis; and per-package *sub-surfaces*: + individual modules, distinct functionalities, the **problem classes** a package addresses, + how-to material. Each may warrant its **own** embedded vector so a query matches the *right + part* of a package — and so the "class of solution" goal (find the package to *extend* even + when the exact function is absent) can match the **problem-classes surface** specifically. + +### 4.1 Why `ef`'s model doesn't cover this directly +`ef`'s atom is a `Segment` of one source's `text`, and multi-config means multiple embedders +over that text. The vision needs a different decomposition: **one logical artifact → a set of +*typed units*, each either a filter-field or an embeddable-surface, each with its own metadata +and (for surfaces) its own embedding config.** A package's `name` is not a chunk of text to +embed — it's a filter key. Its "problem-classes" surface is a distinct embedded space from its +"how-to" surface. This is a richer structure than "segment a string into overlapping chunks." + +### 4.2 The seam `ir` should expose +`ir`'s responsibility (per the vision) is **not** to decide what to index, but to expose the +**three-part seam** and ship sensible defaults: + +``` +(a) DECOMPOSE artifact -> { filter_fields: Mapping[str, Filterable], + surfaces: list[Surface] } # Surface = (kind, granularity, text, metadata) +(b) INDEX each surface -> embed (its own ef config) ; each filter_field -> vd metadata +(c) RETRIEVE combine vd metadata-filter (hard) with multi-surface semantic ranking (soft) +``` + +Concretely: +- **`IndexingStrategy` protocol** — `decompose(artifact) -> ArtifactIndexPlan`. The default + strategy treats an artifact as `ef` does today (one text → segments, one embedder); a + *package-aware* strategy emits the filter-fields + multi-surface plan above. This is the + "pluggable extension, not a fork of the core" the vision demands. +- **Surface → `ef` artifact-graph node.** Each surface's embedding is just another + `ProducerSpec` in the existing graph, so multi-surface indexing inherits cascade + invalidation and shared-artifact dedup *for free*. A surface that is itself AI-generated + (the synopsis, the problem-class tags) is an **op in the graph** (`op="synthesize:synopsis@1"`, + backed by `oa.prompt_function`) — so when the source changes, the synopsis is recomputed and + re-embedded by the same `mark_stale` cascade. This is a clean, powerful fit. +- **Filter-fields → `vd` document metadata**, queried with `vd.filters`. "Ours vs. + third-party," `name`, domain tags become `{$eq}`/`{$in}` constraints applied *before/ + alongside* ranking. +- **Cross-surface fusion.** A query may run against several surfaces (description, synopsis, + problem-classes) and fuse via RRF (`vd.reciprocal_rank_fusion`), then rerank + (`ef.with_reranker`). The "match the right part of a package" goal becomes per-surface + retrieval + fusion. + +### 4.3 Defaults vs. overrides (progressive disclosure at the indexing layer) +- **Default (naive corpus):** one surface = the doc text; minimal metadata. Works out of the + box — identical to `ef` today. +- **Override (package-aware):** the rich decomposition, with AI-authored surfaces produced by + graph ops. `ir` ships the *protocol* and a couple of reference strategies; the sophisticated + package indexer is a plug-in. + +This seam is the through-line that unifies all three of `ir`'s first corpora (§5): each is +"a maintained corpus + an `IndexingStrategy` + filter-fields + surfaces + a retrieve/select +pipeline," differing only in concrete decomposition and change signal. + +--- + +## 5. Corpus-source adapters: the first corpora already have sources + +`ir`'s "scope" and "change-detection" slots do not need to be built from scratch for the +first targets — existing packages already enumerate these corpora and produce records with +staleness signals. `ir` wraps them as scope/change-detection adapters and adds only the +index→retrieve→select layer. + +| `ir` corpus | Scope (enumerate) | Change signal | Artifact source | Existing package | +|---|---|---|---|---| +| **A. Capability discovery — skills** | `skills_index.collect_skills()` sweeps `~/.claude/skills` + every `.pth` base's `.claude/skills/` | `refresh=True` re-scan; file mtime | `{name, description, skill_path, parent, base_path}` from SKILL.md frontmatter | **`priv.skills_index`** | +| **A. Capability discovery — MCP tools / subagents** | *no registry yet* — extend the skills_index sweep pattern | n/a yet | tool = typed callable + schema; subagent = Agent Card | **build (thin) atop `priv` pattern**; `ir_01`/`ir_02` | +| **B. Research / dev-context docs** | `projreg.docs.DocStore` nested FS store (`docs//readme|issues|discussions`) | SHA-256 on write (skips unchanged); `fetched_at` | markdown + `{source, fetched_at, sha256, bytes}` | **`projreg`** (+ `contaix` to materialize new docs) | +| **B. Dev artifacts — issues/PRs/discussions/commits** | `projreg.gh_sync` → `gh_cache///…`; or live `hubcap.RepoReader` | `last_sync` + GitHub `.updated_at` | raw JSON + rendered markdown | **`projreg.gh_sync` / `hubcap`** | +| **B. Code context** | `contaix.code_aggregate()` over dir / GitHub / package name | FS mtime / GitHub timestamp | `Mapping[filename, code_str]` → markdown | **`contaix`** | +| **C. Preferred ecosystem — our ~200 packages** | `projreg.ledger.load_ledger()` (`ProjectRecord`, 14 fields) + `priv.dep_graph` | `diff_ledgers(old,new)`; build-config mtime | `ProjectRecord{name, path, description, version, keywords, urls, deps, github_repo, …}` | **`projreg` + `priv.dep_graph`** | +| **C. Preferred ecosystem — blessed third-party** | *curated list* — small authored corpus | manual / version | `{name, problem_class, rationale}` | **build (tiny)** | +| **AI-authored surfaces** (synopsis, problem-class tags) | n/a (generative) | upstream source hash (graph op) | LLM output, cached | **`oa.prompt_function`** (as `ef` graph ops) | +| **External fetch caching / staleness** | n/a | FS mtime | cached bytes | **`graze`** | + +**Notably, `projreg` already contains a working `search.py`** (BM25 + optional `oa` +embeddings cached by SHA-256, returning `SearchHit(score, project, doc_type, snippet)`). It is +a *proto-`ir`* for corpus C. `ir` should **absorb/generalize** that search rather than leave a +second retrieval implementation drifting — treat `projreg.search` as a reference to subsume, +with `projreg` becoming a *source* (ledger + docs + gh_sync) under `ir`'s unified retrieval. + +--- + +## 6. Proposed `ir` shape and package boundary + +``` + ┌─────────────────────────────────────────────────────────┐ + agent ──tool──▶ │ ir.tool (the ONE search-and-select tool; qh-exposable) │ + └───────────────┬─────────────────────────────────────────┘ + │ retrieve() + select() + ┌───────────────────────────┴───────────────────────────┐ + │ ir.select (NEW: distractor-robust commit to subset; │ + │ progressive payload disclosure) │ + └───────────────────────────┬───────────────────────────┘ + │ candidates + ┌───────────────────────────┴───────────────────────────┐ + │ ir.retrieve (compose: vd hard-filter + multi- │ + │ surface dense + BM25 RRF + ef rerank) │ + └───────────────────────────┬───────────────────────────┘ + │ + ┌───────────────────────────┴───────────────────────────┐ + │ ir.index (NEW seam: IndexingStrategy.decompose → │ + │ filter_fields + surfaces; surfaces & AI- │ + │ authored fields are ef artifact-graph ops) │ + └───────────────────────────┬───────────────────────────┘ + │ + ┌────────────────────────────────┴────────────────────────────────┐ + │ REUSED SUBSTRATE │ + │ ef: corpus · change-detect · artifact-graph · segment · embed │ + │ · diagnose/refresh · rerank · evaluate │ + │ vd: store (16 backends) · metadata filters · hybrid · RRF │ + │ sources: priv.skills_index · projreg · hubcap · contaix │ + │ helpers: oa (LLM ops) · graze (cache) · ju (schemas) · qh (HTTP)│ + │ · meshed (DAG wiring) · imbed (chunk/cluster/viz) │ + └──────────────────────────────────────────────────────────────────┘ +``` + +**`ir`'s own modules (the genuinely-new band):** +1. `ir.artifact` — the capability/resource artifact model (tool / skill / subagent / + package / doc), polymorphic: a shared `(name, description)` retrieval key + a typed + **executable/loadable payload** (schema injection, SKILL.md body load, subagent + delegation, package pointer). From `ir_01`/`ir_02`. +2. `ir.index` — the **`IndexingStrategy` seam** (§4): `decompose → {filter_fields, + surfaces}`; default + package-aware strategies; surfaces/AI-fields realized as `ef` + artifact-graph ops. **This is the keystone.** +3. `ir.retrieve` — a thin composer over `vd` (hard filter), multi-surface dense + BM25 + + RRF, and `ef` rerank. Mostly wiring. +4. `ir.select` — **new**: the selection stage. Distractor-robust subset commitment + + progressive disclosure of heavy payloads (defer-load; append-only to protect prompt + cache, per `ir_01`). +5. `ir.sources` — scope/change-detection adapters wrapping `priv.skills_index`, `projreg`, + `hubcap`, `contaix` (§5). +6. `ir.tool` — the single agent-callable surface (`qh.mk_app`-exposable), with the + `ef.service` stateless-handle-registry pattern. +7. `ir.eval` — the capability-discovery harness (§7), building on `ef.evaluation`. + +**Dependency-wise**, `ir` declares `ef` and `vd` (and pulls `oa`/`qh`/`projreg`/`hubcap`/ +`contaix`/`priv` as the source/feature extras). It must **not** re-vendor any of their +internals. + +--- + +## 7. Evaluation: build the capability-discovery harness on `ef.evaluation` + +`ef.evaluation` already gives BEIR/MTEB-shaped retrieval metrics (recall@k, NDCG@k) and a +Ragas bridge. `ir_03` specifies what `ir` must add on top, and it is genuinely new: +- **Stage-decoupled metrics**: retrieval (recall@k/NDCG@k, reuse `ef`) **vs.** *conditional + selection accuracy* `P(correct | gold retrieved)` **vs.** parameter-fill (deepdiff) **vs.** + end-task — never conflated. +- **Distractor-robustness curve**: one gold + N−1 distractors, sweep N ∈ {1,4,…,128}, plot + accuracy. This is the metric that exposes the "43% → 2% as catalog grows" collapse the whole + project exists to prevent. +- **Failure-mode taxonomy** (query-planning miss, retrieval miss, selector false-negative, + over-selection, namespace confusion, hallucination, parameterization, ordering, abstention). +- **Harness**: `ir_03` endorses **Inspect AI** (MIT) as backbone, **deepdiff** for structural + param scoring, **back-translation + function-masking** for synthetic cases, and **ToolRet** + framing for the retrieval slice. These are external; adopt behind `ir.eval` interfaces. + +--- + +## 8. Consolidation, risks, and the `ir`-vs-`ef` boundary + +**(R1) `ir` vs `ef` overlap — the #1 decision.** `ef` and `ir` share positioning almost word +for word. If the boundary is not drawn deliberately, `ir` will reinvent `ef`. The boundary +this analysis recommends: +- `ef` = **substrate**: corpus, content-addressed maintenance, segment/embed, store-wiring, + generic retrieval, generic retrieval-eval. Stays domain-agnostic. +- `ir` = **agentic IR layer**: heterogeneous *capability/resource* artifacts, the + multi-surface **IndexingStrategy** seam, **selection** + progressive disclosure, **source + adapters** for the concrete corpora, the **one-tool** agent surface, and **capability- + discovery eval**. + Anything `ir` builds that turns out to be domain-agnostic (e.g. multi-surface indexing as a + general capability) is a candidate to **push *down* into `ef`** later — but it should + incubate in `ir` first. *Decision needed from the user (see §9).* + +**(R2) `raglab` and `srag` are parallel/legacy.** `raglab` (`a/raglab`) is a CRUD-over-typed- +RAG-resources skeleton that currently only exports a `LazyAccessor`; `srag` (`a/srag`) is an +explicitly experimental LangChain-coupled prototype (`Raglab2.ask`). Both predate the +`ef`+`vd` substrate and the `ir` framing. **Recommendation:** treat them as superseded — mine +any worthwhile ideas, then let `ir` be the definitive substrate rather than maintaining three +RAG-ish efforts. Do not build `ir` on either. *Flagging, not deciding.* + +**(R3) `projreg.search` is a second retrieval implementation.** It works and is the proto-`ir` +for corpus C. Leaving it independent guarantees drift. **Recommendation:** make `projreg` a +*source* (ledger/docs/gh_sync) and route its search through `ir`, deprecating the standalone +scorer once `ir` covers it. + +**(R4) `chromadol` is now redundant for `ir`.** `vd` already wraps Chroma (and 15 others) with +a uniform contract; `ir` should target `vd`, not `chromadol` directly. (`chromadol` remains +fine as a standalone DOL.) + +**(R5) `imbed_data_prep` is an empty placeholder** — not a building block today. + +**(R6) Scale honesty.** `vd`'s client-side BM25 is O(N) (<~100k docs); above that, lean on a +native-hybrid backend or push lexical into ES/pgvector-FTS. `ir` should pick the backend via +`vd.recommend_backend` per corpus and *log* when it falls back to brute-force lexical (no +silent caps). + +**(R7) Foundation-package opportunities to surface (not silently change):** multi-surface +indexing arguably belongs in `ef` long-term; a first-class **sparse-vector** type and a +**reranking** seam arguably belong in `vd`. These are ecosystem improvements to raise with the +user per the "report bugs/improvements in local packages" rule — not to slip in unannounced. + +--- + +## 9. Open decisions for the user + +1. **`ir` ⟂ `ef` boundary (R1).** Confirm `ir` = agentic-IR layer *above* `ef`, with `ef`/`vd` + as hard dependencies and no internal re-vendoring. (Recommended.) Or do you envision `ir` + absorbing parts of `ef`? +2. **Legacy consolidation (R2/R3).** OK to formally mark `raglab`/`srag` superseded and turn + `projreg.search` into an `ir` source rather than a parallel searcher? +3. **First corpus to ship.** The three `ir_0x` docs and the substrate readiness both point at + **capability discovery (skills first)** as the cheapest end-to-end slice (source already + exists in `priv.skills_index`; small N; clean eval via `ir_03`). Confirm that's target #1, + with **preferred-ecosystem (our packages)** — the richest test of the multi-surface seam — + as #2. +4. **Where multi-surface indexing lives.** Incubate the `IndexingStrategy` seam in `ir` + (recommended), with a documented path to graduate it into `ef` if it proves general? +5. **Embedder/reranker defaults.** Adopt `ir_02`'s production stack (instruction-tuned + embedder + Voyage rerank-2.5 / Qwen3-Reranker behind `ef`'s protocols) as `ir` defaults, + keeping the hashing embedder as the zero-dependency light-path default? + +--- + +## 10. References + +**Companion design docs (this folder):** +- `ir_01 — Progressive Capability Discovery for AI Agents — Give the Agent One Search Tool` +- `ir_02 — Indexing & Embedding Strategy for Agentic Capability Artifacts` +- `ir_03 — Evaluating the Capability-Discovery Layer of Agentic AI` + +**Ecosystem code (repo-relative under `$PP`):** +- `t/ef` — Embedding Flow. Key: `ef/corpus.py`, `ef/segments.py`, `ef/segmenters.py`, + `ef/embedders.py`, `ef/embedder_adapters.py`, `ef/artifact_graph.py`, `ef/source_manager.py`, + `ef/diagnostics.py`, `ef/refresh.py`, `ef/reranking.py`, `ef/evaluation.py`, `ef/service.py`. +- `i/vd` — vector-DB facade. Key: `vd/base.py`, `vd/filters.py`, `vd/search.py`, + `vd/providers.py`, `vd/backends/`. +- `t/imbed` — chunking, clustering, planar embeddings, cluster labeling. +- `t/priv` — `priv/skills_index.py`, `priv/dep_graph.py`, `priv/contexts.py`, + `priv/pypi_alignment.py`. +- `projreg` — `read.py` (`ProjectRecord`), `ledger.py`, `docs.py` (`DocStore`), `gh_sync.py`, + `search.py`. +- `t/hubcap` — GitHub facade (`RepoReader`, `repo_slurp.py`). +- `t/contaix` — `code.py` (`code_aggregate`), `web.py`, `markdown.py`. +- `t/oa` — `prompt_function`, embeddings (LLM-authored surfaces). +- `t/graze` — disk cache for external fetches (staleness). +- `i/ju` — schema/contract layer; `i/qh` — HTTP/agent surface; `i/meshed` — DAG composition. + +**External work (from `ir_01`–`ir_03`, for the agentic-IR layer):** +- ToolRet — tool-retrieval benchmark, arXiv 2503.01763. +- Tool-DE (LLM document expansion) — arXiv 2510.22670. +- RAG-MCP — arXiv 2505.03275. ScaleMCP (content-hash CRUD sync) — arXiv 2505.06416. +- Re-Invoke (synthetic queries) — arXiv 2408.01875. +- Anthropic Tool Search Tool (2025-11). Inspect AI (MIT). BEIR / MTEB. deepdiff. + +--- + +*Authored as the architecture/reuse analysis for `ir`. The operative instruction throughout: +**compose `ef` + `vd` + existing source packages; build only the agentic-IR band on top.*** diff --git a/misc/docs/ir_idea.md b/misc/docs/ir_idea.md new file mode 100644 index 0000000..b9812bb --- /dev/null +++ b/misc/docs/ir_idea.md @@ -0,0 +1,43 @@ +## `ir` — An Information Retrieval Substrate for Agentic Systems + +### General context and vision + +`ir` is a general-purpose **information retrieval substrate**: a single, coherent abstraction for "find the relevant things in this corpus" that scales across the entire spectrum of retrieval needs. At one end, it serves as a lightweight, on-the-fly, ad hoc `find`-like function over a small or ephemeral collection. At the other, it operates as a full search engine over corpora of many millions of documents. It is designed to be **extensible into a RAG system** without being one by default — retrieval is the core competency, and generation, reranking, citation, and answer synthesis are layered capabilities composed on top rather than baked in. + +The design goal is **architectural extensibility through a uniform retrieval contract**. Whatever the corpus and whatever the scale, the same facade applies: a swappable pipeline of indexing, retrieval, selection, and corpus-maintenance components behind a stable, declaratively-configured interface. `ir` can be pointed at *any* corpus, with *any* backing store (in-memory, SQLite, pgvector, a dedicated vector DB), and *any* retrieval strategy (lexical, dense, hybrid, late-interaction, reranked, agentic), without the caller's contract changing. New corpus types and retrieval strategies are added by composition, not rewriting — progressive disclosure at the architecture level: simple retrieval simple, sophisticated retrieval possible. + +A first-class concern is **corpus maintenance as a defined, repeatable process** — keeping the index (ledger, embeddings, derived metadata) in sync with a living, mutable corpus. Generalized, the maintenance contract has two pluggable definitions: + +- **Scope** — what the corpus *is*: how a refresh process enumerates documents in scope (paths, globs, queries against an external system, API endpoints). +- **Change detection** — what counts as *stale*: modified-date, content hash, version, ETag, or source-specific signals — driving incremental reindexing, invalidation, and additions/renames/deletions. + +These are abstract slots the system exposes; concrete definitions are determined per source (and an agent can be tasked with figuring out the right scope and change-detection definitions for a new source). + +### Current target: specialized search tools over maintained corpora, exposed to AI agents + +The immediate focus is the **retrieval-for-orchestration layer** — using `ir` to give AI agents *specialized, maintained search tools* they can call to find exactly what they need during a task, rather than carrying everything in context. This breaks into several related corpora: + +**1. Capability discovery and selection.** The progressive-disclosure problem: an agent facing a large catalog of MCP tools, skills, and subagents suffers context saturation and degraded selection accuracy when everything is loaded eagerly. `ir` provides the single meta-capability — a search-and-select tool — through which the agent discovers and dynamically binds only the relevant subset. This spans **retrieval** (ranked candidates from a heterogeneous, multi-type index) and **selection** (committing to a precise, distractor-robust subset), over artifacts that differ in kind: a skill is a document, a tool is a typed callable with a schema, a subagent is a delegatable context-bearing capability. + +**2. Knowledge and development-context retrieval.** The same substrate pointed at the corpora that accumulate around real project work: + +- **Research and supporting documents** — the deep-research reports and knowledge files produced during development, living untidily across `docs/`, `misc/docs/`, and shared locations, often duplicated, symlinked, or reused across projects. `ir` makes this dispersed, overlapping set searchable as a coherent corpus and keeps its ledger and embeddings current as documents are added, renamed, edited, or moved — becoming a **research-report retrieval tool** an agent calls to surface relevant prior research for a given subject. +- **Development artifacts** — GitHub issues, discussions, PRs, and commit messages, which carry useful context to pull and analyze during agentic code development. The same maintenance contract applies, with source-specific scope and change-detection definitions. + +**3. Preferred-ecosystem discovery.** A search tool over the curated set of tools we *prefer to use*, so that when an agent needs to accomplish something it reaches first for sanctioned solutions rather than arbitrary ones. This corpus has two kinds of members: + +- **Preferred third-party tools** — opinionated choices for specific problem classes (e.g., scikit-learn for ML, not TensorFlow). Retrieval here answers "what's our blessed tool for *this kind* of task?" +- **Our own packages** — the 200+ packages we maintain. Here retrieval serves two distinct goals: the usual "what functionality exists?" (so agents prioritize our own tooling), *and* a **generative/extensional** goal — surfacing the right general *class* of solution so that, even when the exact functionality is absent, the agent is directed to the appropriate package to *add* it, growing the ecosystem coherently rather than scattering new code arbitrarily. + +The unifying claim is that all of these are the same problem — a maintained corpus, an index kept in sync, a retrieval-and-selection pipeline, and a clean agent-callable surface — differing only in concrete scope, change-detection signal, artifact representation, and index configuration. `ir` is the substrate that makes them one system; these corpora are its first concrete instantiations. + +### Special attention: the ecosystem target and the "what do we index?" problem + +The preferred-ecosystem target deserves separate scrutiny because it sharpens two questions that run through the whole project but are most acute here: **what do we embed**, and the recognition that **good retrieval is not all embeddings — it is also classic metadata filtering**. The ultimate goal of `ir` is to find the *appropriate resource for a specific need*; semantic similarity is only one signal toward that goal, and for richly-structured artifacts like packages it is often not the dominant one. + +A single package is not one document — it is a hierarchy of indexable surfaces, and several distinct concerns coexist: + +- **Structured metadata for filtering**, not embedding — package name, ownership (ours vs. third-party), domain/problem-class tags, maturity, dependencies, license. The package name in particular wants to be a first-class filterable field, and "ours vs. preferred-third-party" is a hard filter, not a fuzzy match. Much of the precision in "find the appropriate resource" comes from constraining the candidate set this way *before* or *alongside* semantic ranking. +- **Embeddable representations at multiple granularities and of multiple kinds** — a short canonical description; a longer AI-authored synopsis written by studying the package; and per-package *sub-surfaces*: individual modules, distinct functionalities, the problem classes a package addresses, how-to material. Each of these may warrant its own embedded representation so that a query can match the *right part* of a package rather than the package as an undifferentiated blob — and so that the "class of solution" goal (point 3 above) can match against the problem-classes surface specifically. + +The crucial architectural stance is that **deciding what to index is a problem in its own right, and not one `ir` should solve internally.** Producing the synopsis, deciding the sub-surfaces, generating problem-class tags, choosing what becomes filterable metadata versus embedded text — these are upstream *indexing-strategy* concerns, frequently themselves AI-assisted, that vary per corpus and will evolve. `ir`'s responsibility is to expose the **seams** to plug these concerns in: a clean separation between (a) how an artifact is decomposed into filterable fields and embeddable units, (b) how those units are indexed, and (c) how retrieval combines metadata filtering with semantic ranking over them. `ir` should ship **sensible defaults** so it works out of the box on a naive corpus, while leaving every one of these decisions overridable — so that the sophisticated, package-aware indexing strategy this target ultimately needs is a pluggable extension, not a fork of the core. \ No newline at end of file diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..6a3eb65 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,160 @@ +[build-system] +requires = [ + "hatchling", +] +build-backend = "hatchling.build" + +[project] +name = "ir" +version = "0.0.1" +description = "Information Retrieval" +readme = "README.md" +requires-python = ">=3.10" +keywords = [] +authors = [] +dependencies = [ + "numpy", + "dol", + "ef", + "vd", +] + +[project.license] +text = "mit" + +[project.urls] +Homepage = "https://github.com/i2mint/ir" +Repository = "https://github.com/i2mint/ir" +Documentation = "https://i2mint.github.io/ir" + +[project.optional-dependencies] +dev = [ + "pytest>=7.0", + "pytest-cov>=4.0", + "ruff>=0.1.0", +] +docs = [ + "sphinx>=6.0", + "sphinx-rtd-theme>=1.0", +] + +[tool.ruff] +line-length = 88 +target-version = "py310" +exclude = [ + "**/*.ipynb", + ".git", + ".venv", + "build", + "dist", + "tests", + "examples", + "scrap", +] + +[tool.ruff.lint] +select = [ + "D100", +] +ignore = [ + "D203", + "E501", + "B905", +] + +[tool.ruff.lint.pydocstyle] +convention = "google" + +[tool.ruff.lint.per-file-ignores] +"**/tests/*" = [ + "D", +] +"**/examples/*" = [ + "D", +] +"**/scrap/*" = [ + "D", +] + +[tool.pytest.ini_options] +minversion = "6.0" +testpaths = [ + "tests", +] +doctest_optionflags = [ + "NORMALIZE_WHITESPACE", + "ELLIPSIS", +] + +[tool.wads.ci] +project_name = "" + +[tool.wads.ci.commands] +pre_test = [] +test = [] +post_test = [] +lint = [] +format = [] + +[tool.wads.ci.env] +required_envvars = [] +test_envvars = [] +extra_envvars = [] + +[tool.wads.ci.env.defaults] + +[tool.wads.ci.quality.ruff] +enabled = true + +[tool.wads.ci.quality.black] +enabled = false + +[tool.wads.ci.quality.mypy] +enabled = false + +[tool.wads.ci.testing] +enabled = true +python_versions = [ + "3.10", + "3.12", +] +pytest_args = [ + "-v", + "--tb=short", +] +coverage_enabled = true +coverage_threshold = 0 +coverage_report_format = [ + "term", + "xml", +] +exclude_paths = [ + "examples", + "scrap", +] +test_on_windows = true + +[tool.wads.ci.metrics] +enabled = true +config_path = ".github/umpyre-config.yml" +storage_branch = "code-metrics" +python_version = "3.10" +force_run = false + +[tool.wads.ci.build] +sdist = true +wheel = true + +[tool.wads.ci.publish] +enabled = true +skip_ci_marker = "[skip ci]" +publish_marker = "[publish]" + +[tool.wads.ci.docs] +enabled = true +builder = "epythet" +ignore_paths = [ + "tests/", + "scrap/", + "examples/", +] diff --git a/tests/test_maintenance.py b/tests/test_maintenance.py new file mode 100644 index 0000000..aaadcd7 --- /dev/null +++ b/tests/test_maintenance.py @@ -0,0 +1,41 @@ +"""Regression tests for incremental-maintenance edge cases.""" + +import ir +from ir.base import storage_key +from ir.store import CorpusStore + + +def test_strategy_change_triggers_reindex(): + # Same content, different strategy -> must re-decompose (not skipped). + text = "\n\n".join(f"paragraph {i} with several words here" for i in range(30)) + docs = {"d": {"text": text}} + store = CorpusStore.memory() + + ir.build( + ir.CorpusSource.from_mapping(docs, name="m", strategy=ir.WholeText()), + store=store, + embedder="light", + ) + assert len(store) == 1 # WholeText -> one surface + + ir.build( + ir.CorpusSource.from_mapping( + docs, name="m", strategy=ir.Chunked(chunk_size=120, overlap=20) + ), + store=store, + embedder="light", + ) + assert len(store) > 1 # Chunked -> many surfaces, despite unchanged content + assert "Chunked" in store.get_ledger_entry(storage_key("d"))["strategy_id"] + + +def test_from_skills_fetcher_injection(): + fake = [ + {"name": "alpha", "description": "manage alpha widgets", "parent": "pkgA"}, + {"name": "beta", "description": "manage beta gadgets", "parent": "pkgB"}, + ] + src = ir.CorpusSource.from_skills(fetcher=lambda: fake) + assert set(src.scope) == {"alpha", "beta"} + corpus = ir.build(src, store=CorpusStore.memory(), embedder="light") + assert len(corpus) == 2 + assert ir.search(corpus, "alpha widgets", k=1)[0].metadata["name"] == "alpha" diff --git a/tests/test_pipeline.py b/tests/test_pipeline.py new file mode 100644 index 0000000..27e7732 --- /dev/null +++ b/tests/test_pipeline.py @@ -0,0 +1,95 @@ +"""End-to-end pipeline tests using the light (numpy-only) hashing embedder. + +These are hermetic: no model download, no API keys, no network. They exercise +build/search, incremental maintenance, deletion, metadata filtering, surface +selection, and local persistence. +""" + +import ir +from ir.base import storage_key +from ir.store import CorpusStore + +DOCS = { + "git": "git version control branching merge commit repository clone", + "cooking": "recipe oven bake flour sugar cake dessert kitchen knife", + "astronomy": "telescope galaxy nebula star planet orbit cosmos comet", +} + + +def _src(docs=DOCS, name="t", **kw): + return ir.CorpusSource.from_mapping(docs, name=name, strategy=ir.WholeText(), **kw) + + +def test_build_and_rank(): + corpus = ir.build(_src(), store=CorpusStore.memory(), embedder="light") + assert len(corpus) == 3 + assert ( + ir.search(corpus, "merge a git branch in the repository", k=1)[0].artifact_id + == "git" + ) + assert ir.search(corpus, "bake a cake in the oven", k=1)[0].artifact_id == "cooking" + + +def test_idempotent_rebuild_is_stable(): + store = CorpusStore.memory() + ir.build(_src(), store=store, embedder="light") + v1 = store.get_ledger_entry(storage_key("git"))["version"] + ir.build(_src(), store=store, embedder="light") + assert len(store) == 3 + assert store.get_ledger_entry(storage_key("git"))["version"] == v1 + + +def test_incremental_edit_updates_version(): + store = CorpusStore.memory() + ir.build(_src(), store=store, embedder="light") + v1 = store.get_ledger_entry(storage_key("git"))["version"] + edited = dict(DOCS, git=DOCS["git"] + " rebase stash tag") + ir.build(_src(edited), store=store, embedder="light") + assert len(store) == 3 + assert store.get_ledger_entry(storage_key("git"))["version"] != v1 + + +def test_full_refresh_prunes_deletions(): + store = CorpusStore.memory() + ir.build(_src(), store=store, embedder="light") + smaller = {k: v for k, v in DOCS.items() if k != "astronomy"} + ir.build(_src(smaller), store=store, embedder="light") + assert len(store) == 2 + assert store.get_ledger_entry(storage_key("astronomy")) is None + + +def test_metadata_filter(): + src = _src( + {k: {"text": v} for k, v in DOCS.items()}, + name="tf", + metadata_of=lambda aid, raw: {"kind": "tech" if aid == "git" else "other"}, + ) + corpus = ir.build(src, store=CorpusStore.memory(), embedder="light") + hits = ir.search(corpus, "anything at all", k=5, filter={"kind": "tech"}) + assert [h.artifact_id for h in hits] == ["git"] + + +def test_surface_dedupe_per_artifact(): + # One long doc -> many chunk surfaces; per_artifact collapses to one hit. + text = "\n\n".join( + f"deployment paragraph {i} server systemd caddy" for i in range(20) + ) + src = ir.CorpusSource.from_mapping( + {"big": {"text": text}}, + name="tc", + strategy=ir.Chunked(chunk_size=120, overlap=20), + ) + corpus = ir.build(src, store=CorpusStore.memory(), embedder="light") + assert len(corpus) > 1 # multiple chunk records + hits = ir.search(corpus, "deploy to the server", k=5) + assert len(hits) == 1 and hits[0].artifact_id == "big" + + +def test_local_persistence_roundtrip(tmp_path, monkeypatch): + monkeypatch.setenv("IR_DATA_DIR", str(tmp_path / "data")) + monkeypatch.setenv("IR_CACHE_DIR", str(tmp_path / "cache")) + monkeypatch.setenv("IR_CONFIG_DIR", str(tmp_path / "config")) + ir.build(_src(name="persist"), embedder="light") # default = local store + reopened = ir.open_corpus("persist") + assert len(reopened) == 3 + assert ir.search(reopened, "merge a git branch", k=1)[0].artifact_id == "git" diff --git a/tests/test_sources_local.py b/tests/test_sources_local.py new file mode 100644 index 0000000..7c0cd03 --- /dev/null +++ b/tests/test_sources_local.py @@ -0,0 +1,59 @@ +"""Source-adapter tests against the real local ecosystem. + +These need ``priv`` and the projects folder (``$PP`` / ``$PTH_FILEPATH``), so +they are skipped where unavailable (e.g. CI). They use the light embedder to +stay fast — they validate the *source plumbing* (scope, metadata, ids), not +semantic quality. +""" + +import os + +import pytest + +import ir +from ir.store import CorpusStore + +_has_priv = False +try: + import priv.skills_index # noqa: F401 + + _has_priv = True +except Exception: + pass + +_has_pp = bool(os.environ.get("PP") or os.environ.get("PTH_FILEPATH")) + +skills_only = pytest.mark.skipif(not _has_priv, reason="priv not importable") +ecosystem = pytest.mark.skipif( + not (_has_priv and _has_pp), reason="priv/$PP not available" +) + + +@skills_only +def test_skills_source_scope_and_build(): + src = ir.CorpusSource.from_skills() + assert len(src.scope) > 20 # the ecosystem has many skills + sample = next(iter(src.scope.values())) + assert "name" in sample and "description" in sample + corpus = ir.build(src, store=CorpusStore.memory(), embedder="light") + assert len(corpus) == len(src.scope) # Skill -> one surface per skill + hits = ir.search(corpus, "anything", k=3) + assert hits and hits[0].metadata.get("name") + + +@ecosystem +def test_md_reports_source_excludes_allcaps(): + src = ir.CorpusSource.from_md_reports() + # No ALL-CAPS report filenames (README/CLAUDE/MEMORY/SKILL). + for meta in (src.metadata_of(k, v) for k, v in list(src.scope.items())[:200]): + fn = meta.get("filename", "") + assert not fn.isupper() or not fn.endswith(".md") + + +@ecosystem +def test_packages_source_scope_shape(): + src = ir.CorpusSource.from_packages() + assert len(src.scope) > 50 + rec = src.scope.get("dol") or next(iter(src.scope.values())) + assert rec["owner"] == "ours" + assert "name" in rec and "path" in rec diff --git a/tests/test_store.py b/tests/test_store.py new file mode 100644 index 0000000..fe2ad0b --- /dev/null +++ b/tests/test_store.py @@ -0,0 +1,66 @@ +"""Unit tests for the CorpusStore repository layer (in-memory).""" + +import numpy as np + +from ir.base import Record +from ir.store import CorpusStore + + +def _rec(rid="r1", aid="a1", kind="document", idx=0, vec=(1.0, 0.0, 0.0)): + return Record( + id=rid, + artifact_id=aid, + surface_kind=kind, + surface_index=idx, + text="some text", + vector=np.asarray(vec, dtype=np.float32), + metadata={"owner": "ours"}, + ) + + +def test_record_make_id_is_deterministic(): + assert Record.make_id("a", "k", 0) == Record.make_id("a", "k", 0) + assert Record.make_id("a", "k", 0) != Record.make_id("a", "k", 1) + + +def test_put_get_delete_roundtrip(): + store = CorpusStore.memory() + rec = _rec() + store.put_record(rec) + assert len(store) == 1 + got = store.get_record(rec.id) + assert got.artifact_id == "a1" + assert got.metadata["owner"] == "ours" + np.testing.assert_allclose(got.vector, rec.vector) + store.delete_record(rec.id) + assert len(store) == 0 + + +def test_matrix_is_l2_normalized_and_aligned(): + store = CorpusStore.memory() + store.put_record(_rec(rid="r1", vec=(3.0, 0.0, 0.0))) + store.put_record(_rec(rid="r2", vec=(0.0, 4.0, 0.0))) + ids, mat, metas = store.matrix() + assert mat.shape == (2, 3) + np.testing.assert_allclose(np.linalg.norm(mat, axis=1), [1.0, 1.0], atol=1e-6) + assert len(ids) == len(metas) == 2 + + +def test_matrix_cache_invalidated_on_write(): + store = CorpusStore.memory() + store.put_record(_rec(rid="r1")) + assert store.matrix()[1].shape == (1, 3) + store.put_record(_rec(rid="r2")) + assert store.matrix()[1].shape == (2, 3) + + +def test_ledger_and_config(): + store = CorpusStore.memory() + store.set_ledger_entry( + "k", {"artifact_id": "a", "version": "v1", "record_ids": ["r1"]} + ) + assert store.get_ledger_entry("k")["version"] == "v1" + store.set_config({"embedder_id": "hashing__dim512"}) + assert store.get_config()["embedder_id"] == "hashing__dim512" + store.delete_ledger_entry("k") + assert store.get_ledger_entry("k") is None diff --git a/tests/test_strategy.py b/tests/test_strategy.py new file mode 100644 index 0000000..e0dc9d2 --- /dev/null +++ b/tests/test_strategy.py @@ -0,0 +1,73 @@ +"""Unit tests for indexing strategies (artifact -> filter fields + surfaces).""" + +from ir.strategy import Chunked, Package, Skill, WholeText, _split + + +def test_wholetext_str_and_mapping(): + plan = WholeText().decompose("a", "hello world", {"topic": "x"}) + assert len(plan.surfaces) == 1 + assert plan.surfaces[0].kind == "document" + assert plan.surfaces[0].text == "hello world" + assert plan.filter_fields == {"topic": "x"} + + plan2 = WholeText().decompose("b", {"text": "body here", "extra": 1}, {}) + assert plan2.surfaces[0].text == "body here" + + +def test_wholetext_empty_yields_no_surface(): + plan = WholeText().decompose("a", " ", {}) + assert plan.surfaces == [] + + +def test_chunked_produces_multiple_chunks(): + text = "\n\n".join( + f"Paragraph number {i} with some filler words." for i in range(40) + ) + plan = Chunked(chunk_size=200, overlap=20).decompose("doc", text, {"p": "proj"}) + assert len(plan.surfaces) > 1 + assert all(s.kind == "chunk" for s in plan.surfaces) + assert plan.surfaces[0].metadata["chunk_index"] == 0 + assert plan.filter_fields["p"] == "proj" + + +def test_skill_embeds_name_and_description_only(): + raw = {"name": "deploy", "description": "Push apps to the server.", "parent": "tw"} + plan = Skill().decompose("deploy", raw, {}) + assert len(plan.surfaces) == 1 + assert plan.surfaces[0].kind == "capability" + assert "deploy" in plan.surfaces[0].text and "server" in plan.surfaces[0].text + assert plan.filter_fields["name"] == "deploy" + assert plan.filter_fields["parent"] == "tw" + + +def test_package_description_plus_readme_chunks_and_filter_fields(): + raw = { + "name": "vd", + "description": "Facade over vector databases.", + "readme": "\n\n".join(f"Section {i} text body." for i in range(30)), + "owner": "ours", + "deps": ["dol", "numpy"], + } + plan = Package(chunk_size=150, overlap=20).decompose("vd", raw, {}) + kinds = {s.kind for s in plan.surfaces} + assert "description" in kinds and "readme_chunk" in kinds + assert plan.filter_fields["owner"] == "ours" + assert plan.filter_fields["name"] == "vd" + assert plan.filter_fields["has_readme"] is True + assert plan.filter_fields["deps"] == ["dol", "numpy"] + + +def test_split_packs_to_chunk_size_not_per_paragraph(): + # 60 short paragraphs (~25 chars each) ~ 1500 chars total. + text = "\n\n".join(f"short paragraph numbered {i}" for i in range(60)) + chunks = _split(text, chunk_size=300, overlap=30) + # Packed: far fewer chunks than paragraphs, each near the target size. + assert len(chunks) < 15 + assert all(len(c) <= 300 + 30 for c in chunks) + + +def test_split_hard_splits_oversized_paragraph(): + text = "x" * 1000 + chunks = _split(text, chunk_size=200, overlap=20) + assert len(chunks) >= 5 + assert all(len(c) <= 200 for c in chunks)