Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion ir/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -75,5 +75,7 @@ def search(corpus, query, **kwargs):

# The evaluation harness is reachable as ``ir.eval`` (its ``ef`` imports are
# lazy, so this does not weigh down ``import ir``). Kept out of ``__all__`` so a
# star-import does not shadow the ``eval`` builtin.
# star-import does not shadow the ``eval`` builtin. ``ir.eval_gen`` is the
# build-time case generator (its ``oa`` import is lazy too).
from . import eval # noqa: E402,F401 (submodule attribute: ir.eval)
from . import eval_gen # noqa: E402,F401 (submodule attribute: ir.eval_gen)
33 changes: 32 additions & 1 deletion ir/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@
ir info packages # config + stats for a corpus
ir register notes files --root ~/notes --pattern '.*\\.md$'
ir rm notes # unregister (keeps built data)
ir eval-gen skills skills_eval.jsonl --k 5 # generate cases (needs oa/LLM)
ir eval skills skills_eval.jsonl --mode hybrid # score retrieval on a case file
"""

Expand Down Expand Up @@ -113,4 +114,34 @@ def eval(name, cases, *, mode="hybrid", k=10):
return out


COMMANDS = [ls, register, build, search, info, rm, eval]
def eval_gen(name, out, *, k=5, abstention_frac=0.15, max_artifacts=None):
"""Generate an eval-case file for a corpus by back-translation (needs oa/LLM).

Writes a DiscoveryCase JSONL set (gold cases + an abstention slice) for the
registered corpus *name* to *out*, stamping a corpus-signature into the
header so the frozen file can be checked against the live corpus later. This
command calls an LLM via oa; scoring it afterwards (`ir eval`) is offline.
"""
from .eval import save_cases
from .eval_gen import build_eval_set, corpus_signature

source = registry.source_for(name)
kwargs = {}
if max_artifacts is not None:
kwargs["max_artifacts"] = int(max_artifacts)
cases = build_eval_set(
source, k=k, abstention_frac=abstention_frac, corpus_name=name, **kwargs
)
save_cases(
cases,
out,
meta={"corpus": name, "corpus_signature": corpus_signature(source), "k": k},
)
n_gold = sum(not c.gold_is_none for c in cases)
return (
f"wrote {len(cases)} cases ({n_gold} gold, {len(cases) - n_gold} abstention) "
f"to {out!r}"
)


COMMANDS = [ls, register, build, search, info, rm, eval, eval_gen]
Loading
Loading