diff --git a/.claude/hooks/observation-recorder.py b/.claude/hooks/observation-recorder.py new file mode 100755 index 000000000..b544c0f76 --- /dev/null +++ b/.claude/hooks/observation-recorder.py @@ -0,0 +1,93 @@ +#!/usr/bin/env python3 +"""Stamp a knowledge source the moment a tool that reads it is called. + +Runs on PostToolUse. The point is that observation recording must not depend on +the assistant choosing to record: an assistant that cannot notice its context +has gone stale is exactly the one that will forget to write down when it last +looked. This watches what actually happened instead. + +Mapping is by tool name, deliberately coarse. A calendar tool means the calendar +was observed; which calendar, and whether the assistant then used the result +correctly, are different questions this does not pretend to answer. + +Silent and exit 0 throughout. A vault that cannot record an observation is no +worse off than one with no ledger at all, and nothing here is worth interrupting +a tool call for. +""" +from __future__ import annotations + +import json +import os +import sys +from pathlib import Path + +# Substring match against the tool name, first hit wins. Ordered so that more +# specific patterns precede general ones. +TOOL_SOURCES: tuple[tuple[str, str], ...] = ( + ("calendar_get", "calendar"), + ("calendar_search", "calendar"), + ("calendar_", "calendar"), + ("apple-mail", "email"), + ("apple_mail", "email"), + ("gmail", "email"), + ("search_meetings", "meetings"), + ("get_meeting", "meetings"), + ("granola", "meetings"), + ("wispr", "meetings"), + ("list_tasks", "tasks"), + ("update_task", "tasks"), + ("create_task", "tasks"), + ("get_week_progress", "week_priorities"), + ("get_week_priorities", "week_priorities"), + ("get_quarterly_goals", "quarter_goals"), + ("get_goal_status", "quarter_goals"), + ("pipedrive", "pipeline"), + ("lookup_person", "people"), + ("build_people_index", "people"), + ("list_companies", "accounts"), + ("refresh_company", "accounts"), +) + + +def _source_for(tool_name: str) -> str | None: + lowered = (tool_name or "").lower() + for needle, source in TOOL_SOURCES: + if needle in lowered: + return source + return None + + +def main() -> int: + try: + payload = json.load(sys.stdin) + except (json.JSONDecodeError, ValueError): + return 0 + if not isinstance(payload, dict): + return 0 + + tool_name = "" + for key in ("tool_name", "toolName", "name", "tool"): + value = payload.get(key) + if isinstance(value, str) and value: + tool_name = value + break + + source = _source_for(tool_name) + if source is None: + return 0 + + vault = Path(os.environ.get("CLAUDE_PROJECT_DIR") or os.getcwd()) + sys.path.insert(0, str(vault)) + try: + from core.utils.freshness import observe + except Exception: # noqa: BLE001 - a vault without the module is not a fault + return 0 + try: + observe(vault, source) + except Exception: # noqa: BLE001 + return 0 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/.gitignore b/.gitignore index 64c5ac0ac..37cd4c6fd 100644 --- a/.gitignore +++ b/.gitignore @@ -111,6 +111,7 @@ System/my-customizations.md !.claude/hooks/session-clock.sh !.claude/hooks/correction-capture.sh !.claude/hooks/correction-capture.py +!.claude/hooks/observation-recorder.py !.claude/hooks/dex-core-orientation.sh !.claude/hooks/ensure-mcp-user-scope.cjs !.claude/hooks/connection-health-checker.cjs diff --git a/System/knowledge-half-life.example.yaml b/System/knowledge-half-life.example.yaml new file mode 100644 index 000000000..2fd8e65ad --- /dev/null +++ b/System/knowledge-half-life.example.yaml @@ -0,0 +1,50 @@ +# How long an observation stays trustworthy, per source. +# +# Context carries no freshness. Everything an assistant holds presents with +# equal authority whether it was observed a minute ago or yesterday, and in a +# long session that is how a stale calendar read, an unchecked inbox, or +# yesterday's date get used as though fresh. +# +# This file makes the decay assumption explicit and per-vault, because +# volatility is personal: one person's pipeline moves weekly, another's task +# list moves hourly. Shipped as a seed, so your tuning is never overwritten by +# an update. +# +# Durations accept s, m, h, d. `never` means the observation does not decay -- +# use it for things that are superseded rather than aged out. + +sources: + # World state, cheap to re-observe, expensive to be wrong about. + clock: {half_life: 5m, note: "re-stated every prompt by the session-clock hook"} + calendar: {half_life: 30m} + email: {half_life: 30m} + ci: {half_life: 10m, note: "pull request and build state"} + meetings: {half_life: 1h, note: "whether new captures exist"} + transcripts: {half_life: never, note: "a finalised capture is a record of what was said; superseded by a later meeting, never aged out"} + tasks: {half_life: 4h} + timesheet: {half_life: 8h} + + # Vault content: changes on a human cadence. + week_priorities: {half_life: 2d} + people: {half_life: 14d} + accounts: {half_life: 7d, note: "deal state moves faster than the person behind it"} + pipeline: {half_life: 2d} + quarter_goals: {half_life: 14d} + pillars: {half_life: 90d} + + # Not observations. These are superseded, never stale. + user_decisions: {half_life: never} + user_corrections: {half_life: never} + +# What a given output must have freshly observed before it is written. +# +# This is the half that can actually be enforced. An assistant cannot reliably +# audit its own memory for staleness, but "this artefact requires a calendar +# read newer than its half-life" is checkable from the outside. +artefacts: + daily-plan: {requires_fresh: [clock, calendar, email, tasks, week_priorities]} + daily-review: {requires_fresh: [clock, calendar, email, tasks, meetings]} + week-plan: {requires_fresh: [clock, calendar, tasks, week_priorities, quarter_goals]} + week-review: {requires_fresh: [clock, tasks, week_priorities, quarter_goals]} + meeting-prep: {requires_fresh: [clock, calendar, people, accounts]} + pipeline-sync: {requires_fresh: [clock, pipeline, accounts]} diff --git a/System/knowledge-half-life.yaml b/System/knowledge-half-life.yaml new file mode 100644 index 000000000..2fd8e65ad --- /dev/null +++ b/System/knowledge-half-life.yaml @@ -0,0 +1,50 @@ +# How long an observation stays trustworthy, per source. +# +# Context carries no freshness. Everything an assistant holds presents with +# equal authority whether it was observed a minute ago or yesterday, and in a +# long session that is how a stale calendar read, an unchecked inbox, or +# yesterday's date get used as though fresh. +# +# This file makes the decay assumption explicit and per-vault, because +# volatility is personal: one person's pipeline moves weekly, another's task +# list moves hourly. Shipped as a seed, so your tuning is never overwritten by +# an update. +# +# Durations accept s, m, h, d. `never` means the observation does not decay -- +# use it for things that are superseded rather than aged out. + +sources: + # World state, cheap to re-observe, expensive to be wrong about. + clock: {half_life: 5m, note: "re-stated every prompt by the session-clock hook"} + calendar: {half_life: 30m} + email: {half_life: 30m} + ci: {half_life: 10m, note: "pull request and build state"} + meetings: {half_life: 1h, note: "whether new captures exist"} + transcripts: {half_life: never, note: "a finalised capture is a record of what was said; superseded by a later meeting, never aged out"} + tasks: {half_life: 4h} + timesheet: {half_life: 8h} + + # Vault content: changes on a human cadence. + week_priorities: {half_life: 2d} + people: {half_life: 14d} + accounts: {half_life: 7d, note: "deal state moves faster than the person behind it"} + pipeline: {half_life: 2d} + quarter_goals: {half_life: 14d} + pillars: {half_life: 90d} + + # Not observations. These are superseded, never stale. + user_decisions: {half_life: never} + user_corrections: {half_life: never} + +# What a given output must have freshly observed before it is written. +# +# This is the half that can actually be enforced. An assistant cannot reliably +# audit its own memory for staleness, but "this artefact requires a calendar +# read newer than its half-life" is checkable from the outside. +artefacts: + daily-plan: {requires_fresh: [clock, calendar, email, tasks, week_priorities]} + daily-review: {requires_fresh: [clock, calendar, email, tasks, meetings]} + week-plan: {requires_fresh: [clock, calendar, tasks, week_priorities, quarter_goals]} + week-review: {requires_fresh: [clock, tasks, week_priorities, quarter_goals]} + meeting-prep: {requires_fresh: [clock, calendar, people, accounts]} + pipeline-sync: {requires_fresh: [clock, pipeline, accounts]} diff --git a/core/portable_contract.py b/core/portable_contract.py index 13f34b70f..620a2efff 100644 --- a/core/portable_contract.py +++ b/core/portable_contract.py @@ -225,6 +225,9 @@ def _r(rule_id: str, path: str, kind: str, ownership: str, note: str = "") -> Ru _r("seed-pillars-live", "System/pillars.yaml", "file", "seed", "shipped empty; user pillar registry — never overwritten"), _r("seed-pillars-example", "System/pillars.example.yaml", "file", "seed"), + _r("seed-half-life-live", "System/knowledge-half-life.yaml", "file", "seed", + "per-vault decay assumptions; user tuning is never overwritten"), + _r("seed-half-life-example", "System/knowledge-half-life.example.yaml", "file", "seed"), _r("seed-trusted-mcps-example", "System/trusted-mcps.example.yaml", "file", "seed"), _r("seed-mcp-example", "System/.mcp.json.example", "file", "seed"), _r("seed-env-example", "env.example", "file", "seed"), diff --git a/core/tests/test_freshness.py b/core/tests/test_freshness.py new file mode 100644 index 000000000..0e637a65b --- /dev/null +++ b/core/tests/test_freshness.py @@ -0,0 +1,158 @@ +"""Freshness must be a fact about the ledger, never an opinion about the assistant. + +The failure these guard against: a long session where a calendar read from four +hours ago and one from a minute ago carry identical authority, so the stale one +gets used and nothing says otherwise. +""" +from __future__ import annotations + +import time + +import pytest + +from core.utils import freshness + +CONFIG = """ +sources: + clock: {half_life: 5m} + calendar: {half_life: 30m} + email: {half_life: 30m} + tasks: {half_life: 4h} + week_priorities: {half_life: 2d} + user_corrections: {half_life: never} +artefacts: + daily-plan: {requires_fresh: [clock, calendar, email, tasks, week_priorities]} +""" + + +def _vault(tmp_path, config: str = CONFIG): + (tmp_path / "System").mkdir(parents=True, exist_ok=True) + (tmp_path / "System" / "knowledge-half-life.yaml").write_text(config, encoding="utf-8") + return tmp_path + + +@pytest.mark.parametrize( + ("text", "seconds"), + [("45s", 45), ("30m", 1800), ("4h", 14400), ("2d", 172800), ("1.5h", 5400)], +) +def test_durations_parse(text, seconds): + assert freshness.parse_duration(text) == seconds + + +def test_never_is_not_a_duration_but_a_declaration(): + assert freshness.parse_duration("never") is None + + +def test_a_typo_in_a_half_life_fails_loudly(tmp_path): + """Silently becoming "fresh forever" is the worst possible reading of a typo.""" + with pytest.raises(ValueError): + freshness.parse_duration("30 minutes") + + vault = _vault(tmp_path, "sources:\n calendar: {half_life: soon}\n") + with pytest.raises(freshness.HalfLifeUnavailable): + freshness.load_config(vault) + + +def test_a_missing_config_is_unavailable_not_all_fresh(tmp_path): + """Not being able to judge freshness must not read as everything being fine.""" + with pytest.raises(freshness.HalfLifeUnavailable): + freshness.load_config(tmp_path) + + +def test_an_artefact_requiring_an_unknown_source_refuses_to_load(tmp_path): + """Such a contract could never be satisfied, and would fail invisibly.""" + vault = _vault( + tmp_path, + "sources:\n calendar: {half_life: 30m}\nartefacts:\n x: {requires_fresh: [calendar, ghost]}\n", + ) + with pytest.raises(freshness.HalfLifeUnavailable, match="ghost"): + freshness.load_config(vault) + + +def test_a_source_never_observed_is_not_fresh(tmp_path): + vault = _vault(tmp_path) + config = freshness.load_config(vault) + + assert freshness.is_fresh(config, vault, "calendar") is False + assert freshness.age_seconds(vault, "calendar") is None + + +def test_observation_makes_a_source_fresh_and_time_takes_it_away(tmp_path): + vault = _vault(tmp_path) + config = freshness.load_config(vault) + now = time.time() + + freshness.observe(vault, "calendar", at=now) + assert freshness.is_fresh(config, vault, "calendar", now=now + 60) is True + + # 30-minute half-life: 31 minutes later it is not. + assert freshness.is_fresh(config, vault, "calendar", now=now + 1860) is False + + +def test_a_source_that_does_not_decay_is_always_fresh(tmp_path): + """Corrections and decisions are superseded, not aged out.""" + vault = _vault(tmp_path) + config = freshness.load_config(vault) + + assert freshness.is_fresh(config, vault, "user_corrections", now=time.time() + 10**7) is True + + +def test_an_unknown_source_gets_the_cautious_answer(tmp_path): + vault = _vault(tmp_path) + config = freshness.load_config(vault) + + assert freshness.is_fresh(config, vault, "astrology") is False + + +def test_missing_for_names_exactly_what_the_artefact_lacks(tmp_path): + vault = _vault(tmp_path) + config = freshness.load_config(vault) + now = time.time() + + freshness.observe(vault, "calendar", at=now) + freshness.observe(vault, "tasks", at=now) + freshness.observe(vault, "email", at=now - 4 * 3600) # stale + + missing = freshness.missing_for(config, vault, "daily-plan", now=now) + + assert set(missing) == {"clock", "email", "week_priorities"} + + +def test_an_artefact_with_no_contract_requires_nothing(tmp_path): + """Not every output needs a contract; inventing one is worse than having none.""" + vault = _vault(tmp_path) + config = freshness.load_config(vault) + + assert freshness.missing_for(config, vault, "some-other-skill") == () + + +def test_report_distinguishes_stale_from_never_observed(tmp_path): + """"Old" and "never looked" call for different responses from a reader.""" + vault = _vault(tmp_path) + config = freshness.load_config(vault) + now = time.time() + freshness.observe(vault, "email", at=now - 4 * 3600) + + rows = {r["source"]: r["state"] for r in freshness.report(config, vault, "daily-plan", now=now)["sources"]} + + assert rows["email"] == "STALE" + assert rows["calendar"] == "NOT OBSERVED" + + +def test_a_corrupt_ledger_reads_as_no_observations(tmp_path): + """A damaged ledger must not make everything look freshly observed.""" + vault = _vault(tmp_path) + ledger = vault / "System" / ".dex" + ledger.mkdir(parents=True, exist_ok=True) + (ledger / "observations.json").write_text("{not json", encoding="utf-8") + + assert freshness.read_ledger(vault) == {} + + +def test_observing_never_raises_even_when_the_ledger_cannot_be_written(tmp_path): + """Losing an observation is survivable. Failing a tool call over one is not.""" + vault = _vault(tmp_path) + (vault / "System" / ".dex").mkdir(parents=True, exist_ok=True) + (vault / "System" / ".dex" / "observations.json").mkdir() + + freshness.observe(vault, "calendar") # must not raise diff --git a/core/tests/test_learning_routing.py b/core/tests/test_learning_routing.py new file mode 100644 index 000000000..b56e548c4 --- /dev/null +++ b/core/tests/test_learning_routing.py @@ -0,0 +1,195 @@ +"""Captured learnings are worth nothing until something routes them. + +These cover the mechanical half only: parsing, clustering, the trigger, and +recording an outcome. Applying an edit stays in the skill, because it must be +shown and confirmed first. +""" +from __future__ import annotations + +from datetime import date, timedelta +from pathlib import Path + +import pytest + +from core.utils import learning_routing as routing + +TODAY = date(2026, 8, 19) + + +def _day_file(vault: Path, day: str, body: str) -> Path: + folder = vault / routing.LEARNINGS_RELATIVE + folder.mkdir(parents=True, exist_ok=True) + path = folder / f"{day}.md" + path.write_text(f"# Session Learnings - {day}\n\n---\n\n{body}", encoding="utf-8") + return path + + +ENTRY = """## 09:15 - Correction + +**What was said:** + +> stop over inferring from timesheet codes + +**Why it matters:** it invents facts about the day. +**Status:** pending + +--- + +## 11:40 - Correction + +**What was said:** + +> the daily-plan skill skipped step 5.8 entirely + +**Why it matters:** an unrun step looks identical to an empty one. +**Status:** pending + +--- +""" + + +def test_entries_are_parsed_with_their_location(tmp_path): + path = _day_file(tmp_path, "2026-08-18", ENTRY) + + entries = routing.parse_file(path) + + assert len(entries) == 2 + assert entries[0].time == "09:15" + assert "timesheet codes" in entries[0].body + assert entries[0].day == date(2026, 8, 18) + assert all(e.is_pending for e in entries) + + +def test_a_file_that_is_not_a_day_is_ignored(tmp_path): + folder = tmp_path / routing.LEARNINGS_RELATIVE + folder.mkdir(parents=True) + readme = folder / "README.md" + readme.write_text("## 09:00 - not a learning\n**Status:** pending\n", encoding="utf-8") + + assert routing.parse_file(readme) == [] + + +def test_a_malformed_file_yields_nothing_rather_than_raising(tmp_path): + path = _day_file(tmp_path, "2026-08-18", "no entries here at all\n") + + assert routing.parse_file(path) == [] + + +def test_already_routed_entries_are_not_pending(tmp_path): + _day_file( + tmp_path, + "2026-08-18", + "## 09:15 - Done one\n\n**Status:** implemented (2026-08-18 — CLAUDE-custom.md)\n\n---\n", + ) + + entries = routing.read_all(tmp_path) + + assert len(entries) == 1 + assert routing.pending(entries) == [] + + +def test_clusters_group_by_destination_not_by_wording(tmp_path): + _day_file(tmp_path, "2026-08-18", ENTRY) + + clusters = routing.cluster(routing.read_all(tmp_path)) + kinds = {c.kind for c in clusters} + + assert "behavioural" in kinds + assert "skill-defect" in kinds + for c in clusters: + assert c.destination, "every cluster must name where it goes" + + +def test_a_cluster_of_several_entries_becomes_one_edit(tmp_path): + """Josh's point: eight related entries are one rule, not eight edits.""" + many = "".join( + f"## 0{n}:00 - Correction\n\n> stop assuming, always verify first\n\n**Status:** pending\n\n---\n\n" + for n in range(1, 5) + ) + _day_file(tmp_path, "2026-08-18", many) + + clusters = routing.cluster(routing.read_all(tmp_path)) + + assert len(clusters) == 1 + assert len(clusters[0].entries) == 4 + + +def test_the_trigger_fires_on_volume(tmp_path): + body = "".join( + f"## 09:{n:02d} - Correction\n\n> stop doing that\n\n**Status:** pending\n\n---\n\n" + for n in range(12) + ) + _day_file(tmp_path, "2026-08-19", body) + + due, reason = routing.should_review(routing.read_all(tmp_path), today=TODAY) + + assert due is True + assert "pending" in reason + + +def test_the_trigger_fires_on_age_even_for_a_single_entry(tmp_path): + """A count-only trigger never fires on a slow, steady leak.""" + old = (TODAY - timedelta(days=30)).isoformat() + _day_file(tmp_path, old, "## 09:00 - Correction\n\n> stop that\n\n**Status:** pending\n\n---\n") + + due, reason = routing.should_review(routing.read_all(tmp_path), today=TODAY) + + assert due is True + assert "days old" in reason + + +def test_the_trigger_stays_quiet_on_a_small_recent_backlog(tmp_path): + """A hook that fires into a healthy vault is noise.""" + _day_file(tmp_path, "2026-08-19", "## 09:00 - Correction\n\n> stop that\n\n**Status:** pending\n\n---\n") + + due, _ = routing.should_review(routing.read_all(tmp_path), today=TODAY) + + assert due is False + + +def test_the_trigger_is_silent_on_an_empty_vault(tmp_path): + due, reason = routing.should_review(routing.read_all(tmp_path), today=TODAY) + + assert due is False + assert reason == "nothing pending" + + +def test_recording_an_outcome_says_where_it_went(tmp_path): + path = _day_file(tmp_path, "2026-08-18", ENTRY) + entry = routing.parse_file(path)[0] + + assert routing.set_status(entry, routing.IMPLEMENTED, "CLAUDE-custom.md", today=TODAY) is True + + text = path.read_text(encoding="utf-8") + assert "**Status:** implemented (2026-08-19 — CLAUDE-custom.md)" in text + # A falling count must mean something was installed, not that it aged out. + assert routing.pending(routing.parse_file(path)) != routing.parse_file(path) + + +def test_dropping_an_entry_records_why(tmp_path): + path = _day_file(tmp_path, "2026-08-18", ENTRY) + entry = routing.parse_file(path)[0] + + routing.set_status(entry, routing.DROPPED, "already covered by an existing rule", today=TODAY) + + assert "dropped (2026-08-19 — already covered" in path.read_text(encoding="utf-8") + + +def test_two_entries_in_one_file_never_have_their_statuses_crossed(tmp_path): + path = _day_file(tmp_path, "2026-08-18", ENTRY) + second = routing.parse_file(path)[1] + + routing.set_status(second, routing.IMPLEMENTED, "process-meetings SKILL.md", today=TODAY) + + reparsed = routing.parse_file(path) + assert reparsed[0].is_pending, "the first entry must be untouched" + assert reparsed[1].status == routing.IMPLEMENTED + + +def test_an_invented_status_is_refused(tmp_path): + """Without a third state, stale entries either linger or get quietly deleted.""" + path = _day_file(tmp_path, "2026-08-18", ENTRY) + entry = routing.parse_file(path)[0] + + with pytest.raises(ValueError): + routing.set_status(entry, "done", "somewhere") diff --git a/core/utils/freshness.py b/core/utils/freshness.py new file mode 100644 index 000000000..5d7c450a3 --- /dev/null +++ b/core/utils/freshness.py @@ -0,0 +1,203 @@ +"""Whether what the assistant knows is still true, per source. + +The problem this addresses: context carries no freshness marker. A calendar read +from four hours ago and one from a minute ago sit side by side with identical +authority, so in a long session stale observations get used as though fresh. + +Two deliberate boundaries, because they decide what this module can honestly do: + +**Observations are recorded mechanically, not self-reported.** An assistant that +cannot notice its context is stale also cannot be trusted to record when it last +looked. The companion PostToolUse hook stamps a source the moment a tool that +reads it is called, so the ledger reflects what happened rather than what the +assistant believes happened. + +**Staleness is a fact about the ledger, not about the assistant.** This module +answers "was `calendar` observed within its half-life", which is checkable. It +does not answer "does the assistant hold a stale belief", which is not. + +The useful consequence is the artefact contract: `missing_for("daily-plan")` +returns the sources a daily plan requires and does not have fresh, and that is +enforceable from outside the assistant. +""" +from __future__ import annotations + +import json +import re +import time +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +CONFIG_RELATIVE = Path("System") / "knowledge-half-life.yaml" +LEDGER_RELATIVE = Path("System") / ".dex" / "observations.json" + +NEVER = "never" +_DURATION = re.compile(r"^\s*(?P\d+(?:\.\d+)?)\s*(?P[smhd])\s*$", re.IGNORECASE) +_UNIT_SECONDS = {"s": 1, "m": 60, "h": 3600, "d": 86400} + + +class HalfLifeUnavailable(RuntimeError): + """The configuration could not be read, so nothing can be judged stale. + + Deliberately distinct from "everything is fresh". A caller that cannot read + the config knows nothing about freshness and must say so rather than + reporting a clean result. + """ + + +def parse_duration(value: Any) -> float | None: + """Seconds for a duration string, or None for ``never``. + + Raises ValueError on anything else, because a typo in a half-life should + fail loudly rather than silently become "fresh forever". + """ + if isinstance(value, str) and value.strip().lower() == NEVER: + return None + match = _DURATION.match(str(value)) + if not match: + raise ValueError(f"not a duration: {value!r} (expected e.g. 30m, 4h, 2d, or never)") + return float(match.group("value")) * _UNIT_SECONDS[match.group("unit").lower()] + + +@dataclass(frozen=True) +class Source: + name: str + half_life_seconds: float | None # None means it never decays + + @property + def decays(self) -> bool: + return self.half_life_seconds is not None + + +@dataclass(frozen=True) +class Config: + sources: dict[str, Source] + artefacts: dict[str, tuple[str, ...]] + + def source(self, name: str) -> Source | None: + return self.sources.get(name) + + +def load_config(vault_root: Path) -> Config: + """Read the half-life declaration. Raises when it cannot be trusted.""" + path = vault_root / CONFIG_RELATIVE + try: + import yaml + + raw = yaml.safe_load(path.read_text(encoding="utf-8")) or {} + except FileNotFoundError as error: + raise HalfLifeUnavailable(f"no half-life config at {CONFIG_RELATIVE}") from error + except Exception as error: # noqa: BLE001 - surfaced as one honest failure + raise HalfLifeUnavailable(f"half-life config unreadable: {error}") from error + + if not isinstance(raw, dict): + raise HalfLifeUnavailable("half-life config is not a mapping") + + sources: dict[str, Source] = {} + for name, body in (raw.get("sources") or {}).items(): + if not isinstance(body, dict) or "half_life" not in body: + raise HalfLifeUnavailable(f"source {name!r} has no half_life") + try: + seconds = parse_duration(body["half_life"]) + except ValueError as error: + raise HalfLifeUnavailable(f"source {name!r}: {error}") from error + sources[str(name)] = Source(str(name), seconds) + + artefacts: dict[str, tuple[str, ...]] = {} + for name, body in (raw.get("artefacts") or {}).items(): + required = (body or {}).get("requires_fresh") or [] + unknown = [s for s in required if s not in sources] + if unknown: + # A contract naming a source that does not exist would silently + # never be satisfiable, which is worse than refusing to load. + raise HalfLifeUnavailable(f"artefact {name!r} requires unknown source(s): {unknown}") + artefacts[str(name)] = tuple(str(s) for s in required) + + return Config(sources=sources, artefacts=artefacts) + + +def _ledger_path(vault_root: Path) -> Path: + return vault_root / LEDGER_RELATIVE + + +def read_ledger(vault_root: Path) -> dict[str, float]: + """Last observation time per source. A missing ledger is empty, not an error.""" + try: + data = json.loads(_ledger_path(vault_root).read_text(encoding="utf-8")) + except (FileNotFoundError, json.JSONDecodeError, OSError): + return {} + if not isinstance(data, dict): + return {} + return {str(k): float(v) for k, v in data.items() if isinstance(v, (int, float))} + + +def observe(vault_root: Path, source: str, *, at: float | None = None) -> None: + """Record that a source was just read. Written atomically.""" + import os + + path = _ledger_path(vault_root) + ledger = read_ledger(vault_root) + ledger[source] = time.time() if at is None else at + try: + path.parent.mkdir(parents=True, exist_ok=True) + tmp = path.with_suffix(".tmp") + tmp.write_text(json.dumps(ledger, indent=2, sort_keys=True), encoding="utf-8") + os.replace(tmp, path) + except OSError: + # Losing one observation is survivable; failing a prompt over it is not. + return + + +def age_seconds(vault_root: Path, source: str, *, now: float | None = None) -> float | None: + """How long since a source was observed, or None if it never has been.""" + seen = read_ledger(vault_root).get(source) + if seen is None: + return None + return (time.time() if now is None else now) - seen + + +def is_fresh(config: Config, vault_root: Path, source: str, *, now: float | None = None) -> bool: + """Whether a source counts as freshly observed. + + An unknown source is not fresh: a caller asking about something the config + does not describe should get the cautious answer, not a confident one. + """ + declared = config.source(source) + if declared is None: + return False + if not declared.decays: + return True + age = age_seconds(vault_root, source, now=now) + if age is None: + return False + return age <= (declared.half_life_seconds or 0) + + +def missing_for(config: Config, vault_root: Path, artefact: str, *, now: float | None = None) -> tuple[str, ...]: + """Sources this artefact requires fresh and does not have. + + An unknown artefact returns nothing rather than raising: not every output + needs a contract, and inventing a requirement would be worse than having none. + """ + required = config.artefacts.get(artefact) + if not required: + return () + return tuple(s for s in required if not is_fresh(config, vault_root, s, now=now)) + + +def report(config: Config, vault_root: Path, artefact: str, *, now: float | None = None) -> dict[str, Any]: + """Everything a Sources block needs, in one call.""" + required = config.artefacts.get(artefact, ()) + rows = [] + for name in required: + age = age_seconds(vault_root, name, now=now) + rows.append( + { + "source": name, + "observed_seconds_ago": age, + "fresh": is_fresh(config, vault_root, name, now=now), + "state": "NOT OBSERVED" if age is None else "fresh" if is_fresh(config, vault_root, name, now=now) else "STALE", + } + ) + return {"artefact": artefact, "sources": rows, "missing": list(missing_for(config, vault_root, artefact, now=now))} diff --git a/core/utils/learning_routing.py b/core/utils/learning_routing.py new file mode 100644 index 000000000..49223b7fb --- /dev/null +++ b/core/utils/learning_routing.py @@ -0,0 +1,256 @@ +"""Read captured learnings, cluster them, and propose where each should go. + +Written as a starting point for the routing step described in #503 by +@joshm-simril, against the shape @davekilleen specified there. The routing table +below is Josh's, not mine. + +**What this module deliberately does not do: apply anything.** The requirement +that Dex never silently rewrites its own instructions means the edit must be +shown and confirmed before it lands, and confirmation belongs in the skill that +can hold a conversation. Everything here is analysis and bookkeeping: parse the +entries, group the ones that share a cause, propose a destination, and record +the outcome once a human has decided. + +The split matters. Clustering and destination-proposal are mechanical and +testable; deciding whether a proposed edit is right is judgement. Mixing them +would put an unreviewable decision inside a function that looks like a helper. +""" +from __future__ import annotations + +import re +from dataclasses import dataclass, field +from datetime import date, datetime +from pathlib import Path + +LEARNINGS_RELATIVE = Path("System") / "Session_Learnings" + +PENDING = "pending" +IMPLEMENTED = "implemented" +DROPPED = "dropped" + +_HEADING = re.compile(r"^##\s+\[?(?P