Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions ir/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@
from __future__ import annotations

from . import embed as _embed # noqa: F401 (sets USE_TF=0 before transformers)
from . import registry
from .base import Artifact, IndexPlan, Record, SearchHit, Surface
from .index import Corpus, build, open_corpus
from .retrieve import search as _search
Expand All @@ -51,8 +52,19 @@
"build",
"open_corpus",
"search",
"register",
"corpora",
"build_corpus",
]

register = registry.register
corpora = registry.registered


def build_corpus(name, **kwargs):
"""Build (or update) a registered/preset corpus by name; returns a Corpus."""
return build(registry.source_for(name), **kwargs)


def search(corpus, query, **kwargs):
"""Search a :class:`~ir.index.Corpus`, or a corpus *name* (reopened lazily)."""
Expand Down
18 changes: 18 additions & 0 deletions ir/__main__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
"""``python -m ir`` / ``ir`` CLI entry point (argh dispatch)."""

from __future__ import annotations

from .cli import COMMANDS


def main():
"""Dispatch the ``ir`` command-line interface."""
import argh

parser = argh.ArghParser()
argh.add_commands(parser, COMMANDS)
parser.dispatch()


if __name__ == "__main__":
main()
83 changes: 83 additions & 0 deletions ir/cli.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,83 @@
"""Command-line surface for ``ir`` (argh-dispatched).

Commands operate on **named** corpora from the registry (see
:mod:`ir.registry`)::

ir build skills # build/update the skills preset corpus
ir search skills "deploy app" # query it
ir ls # list corpora + record counts
ir info packages # config + stats for a corpus
ir register notes files --root ~/notes --pattern '.*\\.md$'
ir rm notes # unregister (keeps built data)
"""

from __future__ import annotations

from . import registry
from .index import build as _build
from .index import open_corpus


def ls():
"""List registered corpora with their kind, embedder, and record count."""
entries = registry.registered()
if not entries:
return "No corpora registered. Try: ir build skills"
lines = []
for name, e in entries.items():
try:
count = len(open_corpus(name))
except Exception:
count = 0
lines.append(
f"{name:18} {e['kind']:10} {e.get('embedder', 'default'):10} records={count}"
)
return "\n".join(lines)


def register(name, kind, *, root=None, pattern=None, embedder="default"):
"""Register a named corpus. kind: skills | packages | reports | files."""
params = {}
if root:
params["root"] = root
if pattern:
params["pattern"] = pattern
registry.register(name, kind, embedder=embedder, **params)
return f"registered {name!r} (kind={kind}, embedder={embedder})"


def build(name, *, embedder=None, full=True):
"""Build or incrementally update a registered (or preset) corpus."""
source = registry.source_for(name)
corpus = _build(source, embedder=embedder, full=full)
return f"built {name!r}: {len(corpus)} records (embedder {corpus.embedder_id})"


def search(name, query, *, k=10):
"""Search a built corpus and print the top-k hits."""
corpus = open_corpus(name)
if len(corpus) == 0:
return f"corpus {name!r} is empty; build it first: ir build {name}"
lines = []
for h in corpus.search(query, k=k):
# artifact_id is the unique label across corpora (skill name[@parent],
# package name, or a report's relative path).
lines.append(f"{h.score:+.3f} {h.artifact_id} [{h.surface_kind}]")
return "\n".join(lines) or "(no matches)"


def info(name):
"""Show a corpus's stored config and stats."""
corpus = open_corpus(name)
cfg = corpus.store.get_config()
reg = registry.get(name)
return f"name: {name}\nregistered: {reg}\nrecords: {len(corpus)}\nconfig: {cfg}"


def rm(name):
"""Unregister a corpus (does not delete its built data)."""
registry.unregister(name)
return f"unregistered {name!r}"


COMMANDS = [ls, register, build, search, info, rm]
99 changes: 99 additions & 0 deletions ir/registry.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,99 @@
"""Named-corpus registry — persistent, reusable corpus definitions.

A registry entry records *how* to (re)build a corpus — its ``kind`` (a source
preset), parameters, and embedder spec — so a corpus becomes a stable name you
can build once and query across sessions. The registry is a single JSON file
under the config dir (``~/.config/ir/corpora.json``).

Presets map to :class:`~ir.sources.CorpusSource` constructors:

- ``skills`` → :meth:`CorpusSource.from_skills`
- ``packages`` → :meth:`CorpusSource.from_packages`
- ``reports`` → :meth:`CorpusSource.from_md_reports`
- ``files`` → :meth:`CorpusSource.from_files` (needs ``root``; optional
``pattern``)

Unregistered preset names (``skills``/``packages``/``reports``) are
auto-registered with defaults on first use, so ``ir build skills`` just works.
"""

from __future__ import annotations

import json
from typing import Any

from .config import registry_path
from .sources import CorpusSource

PRESETS = ("skills", "packages", "reports")


def _load() -> dict[str, Any]:
path = registry_path()
if path.exists():
return json.loads(path.read_text(encoding="utf-8"))
return {}


def _save(entries: dict[str, Any]) -> None:
registry_path().write_text(json.dumps(entries, indent=2), encoding="utf-8")


def register(name: str, kind: str, *, embedder: str = "default", **params) -> dict:
"""Register (or overwrite) a named corpus definition."""
if kind not in PRESETS and kind != "files":
raise ValueError(
f"Unknown corpus kind {kind!r}; use one of {PRESETS} or 'files'."
)
entries = _load()
entries[name] = {"kind": kind, "embedder": embedder, "params": params}
_save(entries)
return entries[name]


def registered() -> dict[str, Any]:
"""All registered corpus definitions, keyed by name."""
return _load()


def get(name: str) -> dict | None:
"""The registry entry for *name*, or ``None``."""
return _load().get(name)


def unregister(name: str) -> None:
"""Remove *name* from the registry (does not delete built data)."""
entries = _load()
entries.pop(name, None)
_save(entries)


def source_from_entry(name: str, entry: dict) -> CorpusSource:
"""Reconstruct a :class:`CorpusSource` from a registry entry."""
kind = entry["kind"]
params = dict(entry.get("params", {}))
embedder = entry.get("embedder", "default")
if kind == "skills":
return CorpusSource.from_skills(name=name, embedder=embedder)
if kind == "packages":
return CorpusSource.from_packages(name=name, embedder=embedder)
if kind == "reports":
return CorpusSource.from_md_reports(name=name, embedder=embedder)
if kind == "files":
root = params.pop("root")
return CorpusSource.from_files(root, name=name, embedder=embedder, **params)
raise ValueError(f"Unknown corpus kind {kind!r}.")


def source_for(name: str) -> CorpusSource:
"""Resolve *name* to a source, auto-registering a preset if needed."""
entry = get(name)
if entry is None:
if name in PRESETS:
entry = register(name, name)
else:
raise KeyError(
f"Corpus {name!r} is not registered. Register it with "
f"`ir register`, or use a preset name: {PRESETS}."
)
return source_from_entry(name, entry)
6 changes: 5 additions & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ build-backend = "hatchling.build"

[project]
name = "ir"
version = "0.0.1"
version = "0.1.0"
description = "Information Retrieval"
readme = "README.md"
requires-python = ">=3.10"
Expand All @@ -17,8 +17,12 @@ dependencies = [
"dol",
"ef",
"vd",
"argh",
]

[project.scripts]
ir = "ir.__main__:main"

[project.license]
text = "mit"

Expand Down
72 changes: 72 additions & 0 deletions tests/test_registry.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,72 @@
"""Tests for the named-corpus registry, facade, and CLI (hermetic)."""

import ir
from ir import cli, registry


def _isolate(tmp_path, monkeypatch):
monkeypatch.setenv("IR_CONFIG_DIR", str(tmp_path / "config"))
monkeypatch.setenv("IR_DATA_DIR", str(tmp_path / "data"))
monkeypatch.setenv("IR_CACHE_DIR", str(tmp_path / "cache"))


def test_register_get_unregister(tmp_path, monkeypatch):
_isolate(tmp_path, monkeypatch)
registry.register("foo", "files", root="/some/dir", pattern=r".*\.md$")
assert registry.get("foo")["kind"] == "files"
assert "foo" in registry.registered()
registry.unregister("foo")
assert registry.get("foo") is None


def test_register_rejects_unknown_kind(tmp_path, monkeypatch):
_isolate(tmp_path, monkeypatch)
try:
registry.register("bad", "nonsense")
except ValueError:
return
raise AssertionError("expected ValueError for unknown kind")


def test_files_corpus_build_search_roundtrip(tmp_path, monkeypatch):
_isolate(tmp_path, monkeypatch)
docs = tmp_path / "docs"
docs.mkdir()
(docs / "deploy.md").write_text("How to deploy the app to the server with systemd.")
(docs / "baking.md").write_text(
"Recipe to bake a cake in the oven with flour sugar."
)

ir.register("notes", "files", root=str(docs), pattern=r".*\.md$")
corpus = ir.build_corpus("notes", embedder="light")
assert len(corpus) >= 2

hits = ir.search("notes", "bake a cake in the oven", k=1)
assert hits[0].artifact_id == "baking.md"


def test_source_for_auto_registers_preset(tmp_path, monkeypatch):
_isolate(tmp_path, monkeypatch)
# A preset name resolves even if not explicitly registered (auto-registered).
# Use the registry layer; building 'skills'/'packages' needs priv/$PP, so we
# only assert the registration happens, not the build.
try:
registry.source_for("reports")
except Exception:
pass # building the source may need $PP; registration is what we check
assert registry.get("reports") is not None
assert registry.get("reports")["kind"] == "reports"


def test_cli_commands_return_strings(tmp_path, monkeypatch):
_isolate(tmp_path, monkeypatch)
docs = tmp_path / "docs"
docs.mkdir()
(docs / "a.md").write_text("alpha widgets and gadgets for testing the cli output.")
cli.register("notes", "files", root=str(docs), pattern=r".*\.md$")
cli.build("notes", embedder="light")
assert "notes" in cli.ls()
assert "alpha" in cli.search("notes", "widgets", k=3) or "a.md" in cli.search(
"notes", "widgets", k=3
)
assert "records:" in cli.info("notes")
Loading