diff --git a/ir/__init__.py b/ir/__init__.py index 45af6e1..61ff3cc 100644 --- a/ir/__init__.py +++ b/ir/__init__.py @@ -27,6 +27,7 @@ from __future__ import annotations from . import embed as _embed # noqa: F401 (sets USE_TF=0 before transformers) +from . import registry from .base import Artifact, IndexPlan, Record, SearchHit, Surface from .index import Corpus, build, open_corpus from .retrieve import search as _search @@ -51,8 +52,19 @@ "build", "open_corpus", "search", + "register", + "corpora", + "build_corpus", ] +register = registry.register +corpora = registry.registered + + +def build_corpus(name, **kwargs): + """Build (or update) a registered/preset corpus by name; returns a Corpus.""" + return build(registry.source_for(name), **kwargs) + def search(corpus, query, **kwargs): """Search a :class:`~ir.index.Corpus`, or a corpus *name* (reopened lazily).""" diff --git a/ir/__main__.py b/ir/__main__.py new file mode 100644 index 0000000..9d73db7 --- /dev/null +++ b/ir/__main__.py @@ -0,0 +1,18 @@ +"""``python -m ir`` / ``ir`` CLI entry point (argh dispatch).""" + +from __future__ import annotations + +from .cli import COMMANDS + + +def main(): + """Dispatch the ``ir`` command-line interface.""" + import argh + + parser = argh.ArghParser() + argh.add_commands(parser, COMMANDS) + parser.dispatch() + + +if __name__ == "__main__": + main() diff --git a/ir/cli.py b/ir/cli.py new file mode 100644 index 0000000..14faaa3 --- /dev/null +++ b/ir/cli.py @@ -0,0 +1,83 @@ +"""Command-line surface for ``ir`` (argh-dispatched). + +Commands operate on **named** corpora from the registry (see +:mod:`ir.registry`):: + + ir build skills # build/update the skills preset corpus + ir search skills "deploy app" # query it + ir ls # list corpora + record counts + ir info packages # config + stats for a corpus + ir register notes files --root ~/notes --pattern '.*\\.md$' + ir rm notes # unregister (keeps built data) +""" + +from __future__ import annotations + +from . import registry +from .index import build as _build +from .index import open_corpus + + +def ls(): + """List registered corpora with their kind, embedder, and record count.""" + entries = registry.registered() + if not entries: + return "No corpora registered. Try: ir build skills" + lines = [] + for name, e in entries.items(): + try: + count = len(open_corpus(name)) + except Exception: + count = 0 + lines.append( + f"{name:18} {e['kind']:10} {e.get('embedder', 'default'):10} records={count}" + ) + return "\n".join(lines) + + +def register(name, kind, *, root=None, pattern=None, embedder="default"): + """Register a named corpus. kind: skills | packages | reports | files.""" + params = {} + if root: + params["root"] = root + if pattern: + params["pattern"] = pattern + registry.register(name, kind, embedder=embedder, **params) + return f"registered {name!r} (kind={kind}, embedder={embedder})" + + +def build(name, *, embedder=None, full=True): + """Build or incrementally update a registered (or preset) corpus.""" + source = registry.source_for(name) + corpus = _build(source, embedder=embedder, full=full) + return f"built {name!r}: {len(corpus)} records (embedder {corpus.embedder_id})" + + +def search(name, query, *, k=10): + """Search a built corpus and print the top-k hits.""" + corpus = open_corpus(name) + if len(corpus) == 0: + return f"corpus {name!r} is empty; build it first: ir build {name}" + lines = [] + for h in corpus.search(query, k=k): + # artifact_id is the unique label across corpora (skill name[@parent], + # package name, or a report's relative path). + lines.append(f"{h.score:+.3f} {h.artifact_id} [{h.surface_kind}]") + return "\n".join(lines) or "(no matches)" + + +def info(name): + """Show a corpus's stored config and stats.""" + corpus = open_corpus(name) + cfg = corpus.store.get_config() + reg = registry.get(name) + return f"name: {name}\nregistered: {reg}\nrecords: {len(corpus)}\nconfig: {cfg}" + + +def rm(name): + """Unregister a corpus (does not delete its built data).""" + registry.unregister(name) + return f"unregistered {name!r}" + + +COMMANDS = [ls, register, build, search, info, rm] diff --git a/ir/registry.py b/ir/registry.py new file mode 100644 index 0000000..ca6bda6 --- /dev/null +++ b/ir/registry.py @@ -0,0 +1,99 @@ +"""Named-corpus registry — persistent, reusable corpus definitions. + +A registry entry records *how* to (re)build a corpus — its ``kind`` (a source +preset), parameters, and embedder spec — so a corpus becomes a stable name you +can build once and query across sessions. The registry is a single JSON file +under the config dir (``~/.config/ir/corpora.json``). + +Presets map to :class:`~ir.sources.CorpusSource` constructors: + +- ``skills`` → :meth:`CorpusSource.from_skills` +- ``packages`` → :meth:`CorpusSource.from_packages` +- ``reports`` → :meth:`CorpusSource.from_md_reports` +- ``files`` → :meth:`CorpusSource.from_files` (needs ``root``; optional + ``pattern``) + +Unregistered preset names (``skills``/``packages``/``reports``) are +auto-registered with defaults on first use, so ``ir build skills`` just works. +""" + +from __future__ import annotations + +import json +from typing import Any + +from .config import registry_path +from .sources import CorpusSource + +PRESETS = ("skills", "packages", "reports") + + +def _load() -> dict[str, Any]: + path = registry_path() + if path.exists(): + return json.loads(path.read_text(encoding="utf-8")) + return {} + + +def _save(entries: dict[str, Any]) -> None: + registry_path().write_text(json.dumps(entries, indent=2), encoding="utf-8") + + +def register(name: str, kind: str, *, embedder: str = "default", **params) -> dict: + """Register (or overwrite) a named corpus definition.""" + if kind not in PRESETS and kind != "files": + raise ValueError( + f"Unknown corpus kind {kind!r}; use one of {PRESETS} or 'files'." + ) + entries = _load() + entries[name] = {"kind": kind, "embedder": embedder, "params": params} + _save(entries) + return entries[name] + + +def registered() -> dict[str, Any]: + """All registered corpus definitions, keyed by name.""" + return _load() + + +def get(name: str) -> dict | None: + """The registry entry for *name*, or ``None``.""" + return _load().get(name) + + +def unregister(name: str) -> None: + """Remove *name* from the registry (does not delete built data).""" + entries = _load() + entries.pop(name, None) + _save(entries) + + +def source_from_entry(name: str, entry: dict) -> CorpusSource: + """Reconstruct a :class:`CorpusSource` from a registry entry.""" + kind = entry["kind"] + params = dict(entry.get("params", {})) + embedder = entry.get("embedder", "default") + if kind == "skills": + return CorpusSource.from_skills(name=name, embedder=embedder) + if kind == "packages": + return CorpusSource.from_packages(name=name, embedder=embedder) + if kind == "reports": + return CorpusSource.from_md_reports(name=name, embedder=embedder) + if kind == "files": + root = params.pop("root") + return CorpusSource.from_files(root, name=name, embedder=embedder, **params) + raise ValueError(f"Unknown corpus kind {kind!r}.") + + +def source_for(name: str) -> CorpusSource: + """Resolve *name* to a source, auto-registering a preset if needed.""" + entry = get(name) + if entry is None: + if name in PRESETS: + entry = register(name, name) + else: + raise KeyError( + f"Corpus {name!r} is not registered. Register it with " + f"`ir register`, or use a preset name: {PRESETS}." + ) + return source_from_entry(name, entry) diff --git a/pyproject.toml b/pyproject.toml index 6a3eb65..0f9bca0 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -6,7 +6,7 @@ build-backend = "hatchling.build" [project] name = "ir" -version = "0.0.1" +version = "0.1.0" description = "Information Retrieval" readme = "README.md" requires-python = ">=3.10" @@ -17,8 +17,12 @@ dependencies = [ "dol", "ef", "vd", + "argh", ] +[project.scripts] +ir = "ir.__main__:main" + [project.license] text = "mit" diff --git a/tests/test_registry.py b/tests/test_registry.py new file mode 100644 index 0000000..c505d7d --- /dev/null +++ b/tests/test_registry.py @@ -0,0 +1,72 @@ +"""Tests for the named-corpus registry, facade, and CLI (hermetic).""" + +import ir +from ir import cli, registry + + +def _isolate(tmp_path, monkeypatch): + monkeypatch.setenv("IR_CONFIG_DIR", str(tmp_path / "config")) + monkeypatch.setenv("IR_DATA_DIR", str(tmp_path / "data")) + monkeypatch.setenv("IR_CACHE_DIR", str(tmp_path / "cache")) + + +def test_register_get_unregister(tmp_path, monkeypatch): + _isolate(tmp_path, monkeypatch) + registry.register("foo", "files", root="/some/dir", pattern=r".*\.md$") + assert registry.get("foo")["kind"] == "files" + assert "foo" in registry.registered() + registry.unregister("foo") + assert registry.get("foo") is None + + +def test_register_rejects_unknown_kind(tmp_path, monkeypatch): + _isolate(tmp_path, monkeypatch) + try: + registry.register("bad", "nonsense") + except ValueError: + return + raise AssertionError("expected ValueError for unknown kind") + + +def test_files_corpus_build_search_roundtrip(tmp_path, monkeypatch): + _isolate(tmp_path, monkeypatch) + docs = tmp_path / "docs" + docs.mkdir() + (docs / "deploy.md").write_text("How to deploy the app to the server with systemd.") + (docs / "baking.md").write_text( + "Recipe to bake a cake in the oven with flour sugar." + ) + + ir.register("notes", "files", root=str(docs), pattern=r".*\.md$") + corpus = ir.build_corpus("notes", embedder="light") + assert len(corpus) >= 2 + + hits = ir.search("notes", "bake a cake in the oven", k=1) + assert hits[0].artifact_id == "baking.md" + + +def test_source_for_auto_registers_preset(tmp_path, monkeypatch): + _isolate(tmp_path, monkeypatch) + # A preset name resolves even if not explicitly registered (auto-registered). + # Use the registry layer; building 'skills'/'packages' needs priv/$PP, so we + # only assert the registration happens, not the build. + try: + registry.source_for("reports") + except Exception: + pass # building the source may need $PP; registration is what we check + assert registry.get("reports") is not None + assert registry.get("reports")["kind"] == "reports" + + +def test_cli_commands_return_strings(tmp_path, monkeypatch): + _isolate(tmp_path, monkeypatch) + docs = tmp_path / "docs" + docs.mkdir() + (docs / "a.md").write_text("alpha widgets and gadgets for testing the cli output.") + cli.register("notes", "files", root=str(docs), pattern=r".*\.md$") + cli.build("notes", embedder="light") + assert "notes" in cli.ls() + assert "alpha" in cli.search("notes", "widgets", k=3) or "a.md" in cli.search( + "notes", "widgets", k=3 + ) + assert "records:" in cli.info("notes")