diff --git a/README.md b/README.md index c208197..c8d7701 100644 --- a/README.md +++ b/README.md @@ -198,7 +198,7 @@ Primary REST: `POST /v1/proof` → `ProofReceipt` with claim `controlled_proof_n ### Hermes (agentic operator) 1. Install package (`pip install -e '.[dev]'`). -2. Copy skill: `cp -R skills/hermes-continuity-forge ~/.hermes/skills/` (or your Hermes skills path). +2. Copy skill: `cp -R skills/hermes-continuity-forge "${HERMES_HOME:-$HOME/.hermes}/skills"/` (or your Hermes skills path). 3. Wire MCP stdio to `.venv/bin/continuity-forge-mcp` — see [`docs/hermes/mcp.example.json`](docs/hermes/mcp.example.json). 4. Read [`docs/hermes/README.md`](docs/hermes/README.md). @@ -277,8 +277,8 @@ See `skills/scriptwriting/SKILL.md` and `skills/scriptwriting/references/continu Install both: ```bash -cp -R skills/scriptwriting ~/.hermes/skills/ -cp -R skills/hermes-continuity-forge ~/.hermes/skills/ +cp -R skills/scriptwriting "${HERMES_HOME:-$HOME/.hermes}/skills"/ +cp -R skills/hermes-continuity-forge "${HERMES_HOME:-$HOME/.hermes}/skills"/ ``` ## Symbolic Cinematic Layer: kubrick Skill @@ -309,6 +309,6 @@ See: Install: ```bash -cp -R skills/kubrick ~/.hermes/skills/ -cp -R skills/hermes-continuity-forge ~/.hermes/skills/ +bash skills/kubrick/install.sh --dry-run # review, then --apply +cp -R skills/hermes-continuity-forge "${HERMES_HOME:-$HOME/.hermes}/skills"/ ``` diff --git a/apps/mcp/src/continuity_forge_mcp/server.py b/apps/mcp/src/continuity_forge_mcp/server.py index 6b7a563..761c6ca 100644 --- a/apps/mcp/src/continuity_forge_mcp/server.py +++ b/apps/mcp/src/continuity_forge_mcp/server.py @@ -518,36 +518,46 @@ def queue_generation( idempotency_key: str = "mcp-queue-generation", rationale: str = "PROPOSED generation (no canon write)", seed: str = "0", + expected_state_hash: str | None = None, ) -> dict[str, Any]: """Generate a PROPOSED mock media candidate for a shot (no canon mutation). MutationEnvelope fields are validated for audit consistency; output remains - PROPOSED and never becomes film canon. + PROPOSED and never becomes film canon. Pass the reviewed project state_hash; + omission binds to a fresh server snapshot. The store lock covers generation + and persistence within this runtime (not across independent processes). """ - MutationEnvelope.from_parts( + runtime = _rt() + project = runtime.project_store.get_project(document_key) + if project is None or not project.shot_contracts: + raise ValueError("project or shot contracts not found") + envelope = MutationEnvelope.from_parts( actor_id=actor_id, authorization_scope=authorization_scope, idempotency_key=idempotency_key, rationale=rationale, - ) - project = _rt().project_store.get_project(document_key) - if project is None or not project.shot_contracts: - raise ValueError("project or shot contracts not found") - contract = next( - ( - c - for c in project.shot_contracts.get("contracts") or [] - if str(c.get("shot_id")) == shot_id + expected_state_hash=( + expected_state_hash if expected_state_hash is not None else project.state_hash ), - None, ) - if contract is None: - raise ValueError("shot not found") - candidate = _rt().gateway.generate_for_shot(contract, seed=seed) - artifact_store = _rt().artifact_store - if artifact_store is not None: - artifact_store.put(candidate) - return candidate.model_dump(mode="json") + with runtime.project_store.candidate_write(document_key, envelope): + contract = next( + ( + c + for c in project.shot_contracts.get("contracts") or [] + if str(c.get("shot_id")) == shot_id + ), + None, + ) + if contract is None: + raise ValueError("shot not found") + candidate = runtime.gateway.generate_for_shot(contract, seed=seed) + artifact_store = runtime.artifact_store + if artifact_store is not None: + # Recheck before persistence, including re-entrant provider callbacks. + with runtime.project_store.candidate_write(document_key, envelope): + artifact_store.put(candidate) + return candidate.model_dump(mode="json") @mcp.tool() @@ -561,42 +571,52 @@ def run_shot_repair_loop( seed: str = "0", max_attempts: int = 3, fail_first: bool = False, + expected_state_hash: str | None = None, ) -> dict[str, Any]: """Run generate→validate→repair loop for one shot (mock worker). MutationEnvelope fields are validated for audit consistency; accepted - candidates remain PROPOSED. + candidates remain PROPOSED. Pass the reviewed project state_hash; omission + binds to a fresh server snapshot. The store lock covers the loop and + persistence within this runtime (not across independent processes). """ - MutationEnvelope.from_parts( + runtime = _rt() + project = runtime.project_store.get_project(document_key) + if project is None or not project.shot_contracts: + raise ValueError("project or shot contracts not found") + envelope = MutationEnvelope.from_parts( actor_id=actor_id, authorization_scope=authorization_scope, idempotency_key=idempotency_key, rationale=rationale, - ) - project = _rt().project_store.get_project(document_key) - if project is None or not project.shot_contracts: - raise ValueError("project or shot contracts not found") - contract = next( - ( - c - for c in project.shot_contracts.get("contracts") or [] - if str(c.get("shot_id")) == shot_id + expected_state_hash=( + expected_state_hash if expected_state_hash is not None else project.state_hash ), - None, - ) - if contract is None: - raise ValueError("shot not found") - result = run_repair_loop( - contract, - gateway=_rt().gateway, - seed=seed, - max_attempts=max_attempts, - fail_first=fail_first, ) - artifact_store = _rt().artifact_store - if result.accepted_candidate is not None and artifact_store is not None: - artifact_store.put(result.accepted_candidate) - return result.model_dump(mode="json") + with runtime.project_store.candidate_write(document_key, envelope): + contract = next( + ( + c + for c in project.shot_contracts.get("contracts") or [] + if str(c.get("shot_id")) == shot_id + ), + None, + ) + if contract is None: + raise ValueError("shot not found") + result = run_repair_loop( + contract, + gateway=runtime.gateway, + seed=seed, + max_attempts=max_attempts, + fail_first=fail_first, + ) + artifact_store = runtime.artifact_store + if result.accepted_candidate is not None and artifact_store is not None: + # Recheck before persistence, including re-entrant provider callbacks. + with runtime.project_store.candidate_write(document_key, envelope): + artifact_store.put(result.accepted_candidate) + return result.model_dump(mode="json") def main() -> None: diff --git a/docs/hermes/README.md b/docs/hermes/README.md index b082200..664559e 100644 --- a/docs/hermes/README.md +++ b/docs/hermes/README.md @@ -16,8 +16,8 @@ Copy (or symlink) the skill into Hermes’ skills directory: ```bash # from continuity-forge repo root -mkdir -p ~/.hermes/skills # adjust if your Hermes install uses another path -cp -R skills/hermes-continuity-forge ~/.hermes/skills/ +mkdir -p "${HERMES_HOME:-$HOME/.hermes}/skills" # adjust if your Hermes install uses another path +cp -R skills/hermes-continuity-forge "${HERMES_HOME:-$HOME/.hermes}/skills"/ # or project-local, if Hermes supports workspace skills: # cp -R skills/hermes-continuity-forge /path/to/workspace/.hermes/skills/ ``` @@ -47,7 +47,7 @@ In addition to the operator skill (`hermes-continuity-forge`), the repo ships `s **Kubrick works completely on its own inside Hermes** (no continuity-forge required). The skill is distributed by simple directory copy: ```bash -pip install continuity-forge[kubrick-helpers] +python -m pip install -e '.[kubrick-helpers]' # Python 3.12+, repo root ``` This provides `kubrick_helpers` Python package + `kubrick-retrieve` / `kubrick-evolve` CLIs. @@ -55,9 +55,9 @@ This provides `kubrick_helpers` Python package + `kubrick-retrieve` / `kubrick-e Install it the same way: ```bash -cp -R skills/kubrick ~/.hermes/skills/ +bash skills/kubrick/install.sh --dry-run # review, then --apply # or -cp -R skills/kubrick ~/.hermes/skills/creative/ +bash skills/kubrick/install.sh creative --dry-run # review, then --apply ``` See `skills/kubrick/README.md` and `skills/kubrick/SKILL.md` for details on its executable retrieval helper and self-evolution system. @@ -168,8 +168,8 @@ See `skills/kubrick/references/continuity-forge-integration.md` for exact handof Install: ```bash -cp -R skills/kubrick ~/.hermes/skills/ -cp -R skills/hermes-continuity-forge ~/.hermes/skills/ +bash skills/kubrick/install.sh --dry-run # review, then --apply +cp -R skills/hermes-continuity-forge "${HERMES_HOME:-$HOME/.hermes}/skills"/ ``` ## Symbolic Cinematic Layer: kubrick Skill @@ -207,5 +207,5 @@ See skills/kubrick/references/continuity-forge-integration.md for exact handoff Both skills are designed to be used together. The kubrick skill produces high-quality, drift-resistant creative material with sophisticated symbolic layer. The operator skill ensures it is governed by the deterministic kernel. Install: -cp -R skills/kubrick ~/.hermes/skills/ -cp -R skills/hermes-continuity-forge ~/.hermes/skills/ +bash skills/kubrick/install.sh --dry-run # review, then --apply +cp -R skills/hermes-continuity-forge "${HERMES_HOME:-$HOME/.hermes}/skills"/ diff --git a/packages/kubrick_helpers/src/kubrick_helpers/evolution.py b/packages/kubrick_helpers/src/kubrick_helpers/evolution.py index 7f654ce..a15161f 100644 --- a/packages/kubrick_helpers/src/kubrick_helpers/evolution.py +++ b/packages/kubrick_helpers/src/kubrick_helpers/evolution.py @@ -13,186 +13,384 @@ """ import argparse +import hashlib +import copy import json import os import glob -from datetime import datetime +from datetime import datetime, timezone from collections import defaultdict -from typing import Dict +from pathlib import Path +import tempfile +import uuid try: import yaml except ImportError: yaml = None + def load_json(path): with open(path, "r") as f: return json.load(f) + def save_json(path, data): os.makedirs(os.path.dirname(path), exist_ok=True) with open(path, "w") as f: json.dump(data, f, indent=2) + def load_all_sidecars(patterns_dir: str): sidecars = {} for root, _, files in os.walk(patterns_dir): for fname in files: if fname.endswith(".json"): path = os.path.join(root, fname) + if Path(path).is_symlink(): + raise ValueError("sidecar symlink is not allowed") data = load_json(path) sidecars[data.get("pattern_id", fname)] = {"path": path, "data": data} return sidecars -def aggregate_usage(receipts_dir, outcomes_dir): - usage = defaultdict(lambda: {"uses": 0, "total_score": 0.0, "success_signals": 0, "failure_signals": 0, "projects": []}) - for path in glob.glob(os.path.join(receipts_dir, "*.json")) + glob.glob(os.path.join(receipts_dir, "*.yaml")): - try: +def aggregate_usage(receipts_dir, outcomes_dir, consumed=None): + """Content-address events, not filenames; never silently ignore malformed evidence.""" + usage = defaultdict(lambda: {"events": {}}) + seen = dict(consumed or {}) + for directory, kind in ((receipts_dir, "retrieval"), (outcomes_dir, "outcome")): + for path in sorted( + glob.glob(os.path.join(directory, "*.json")) + + glob.glob(os.path.join(directory, "*.yaml")) + ): if path.endswith(".json"): - receipt = load_json(path) + record = load_json(path) else: if yaml is None: - continue - receipt = yaml.safe_load(open(path)) - rec = receipt.get("retrieval_receipt", receipt) - ranked = rec.get("ranked_patterns", []) - for item in ranked[:3]: - pid = item.get("pattern_id") - if pid: - usage[pid]["uses"] += 1 - usage[pid]["total_score"] += item.get("total_score", 0.5) - usage[pid]["projects"].append(rec.get("request_hash", "unknown")) - except Exception as e: - print(f"Warning: could not parse {path}: {e}") - - for path in glob.glob(os.path.join(outcomes_dir, "*.json")): - try: - outcome = load_json(path) - pid = outcome.get("pattern_id") - if pid: - signal = outcome.get("outcome", "neutral") - if signal == "success": - usage[pid]["success_signals"] += 1 - elif signal in ("failure", "debt", "collision", "revision_broken"): - usage[pid]["failure_signals"] += 1 - usage[pid]["projects"].append(outcome.get("project", "unknown")) - except Exception as e: - print(f"Warning: could not parse outcome {path}: {e}") - - return usage + raise ValueError("PyYAML is required to read YAML evidence") + with open(path) as stream: + record = yaml.safe_load(stream) + rec = record.get("retrieval_receipt", record) if kind == "retrieval" else record + identity = rec.get("event_id") or ( + rec.get("request_hash") + if kind == "retrieval" + else [rec.get("project"), rec.get("pattern_id")] + ) + if not identity or identity == [None, rec.get("pattern_id")]: + raise ValueError("evidence requires event_id or request_hash/project identity") + event_id = hashlib.sha256( + json.dumps([kind, identity], sort_keys=True).encode() + ).hexdigest() + content = {k: v for k, v in rec.items() if k not in ("timestamp", "metadata", "date")} + digest = hashlib.sha256( + json.dumps(content, sort_keys=True, allow_nan=False).encode() + ).hexdigest() + if event_id in seen and seen[event_id] != digest: + raise ValueError("conflicting evidence for stable identity") + seen[event_id] = digest + if kind == "retrieval": + for item in rec.get("ranked_patterns", [])[:3]: + pid = item["pattern_id"] + score = item.get("total_score", 0.5) + if type(score) not in (int, float): + raise ValueError("retrieval score must be numeric, not coerced") + if not 0 <= score <= 1: + raise ValueError("retrieval score must be finite and in [0, 1]") + usage[pid]["events"][event_id] = { + "evidence_hash": digest, + "uses": 1, + "total_score": score, + "projects": [rec.get("request_hash", "unknown")], + } + else: + pid = rec["pattern_id"] + signal = rec.get("outcome", "neutral") + if signal not in ( + "success", + "failure", + "debt", + "collision", + "revision_broken", + "neutral", + ): + raise ValueError("unknown outcome signal") + usage[pid]["events"][event_id] = { + "evidence_hash": digest, + "success_signals": int(signal == "success"), + "failure_signals": int( + signal in ("failure", "debt", "collision", "revision_broken") + ), + "projects": [rec.get("project", "unknown")], + } + return dict(usage) + def evolve_sidecar(sidecar_info, usage_stats): - data = sidecar_info["data"] - pid = data["pattern_id"] - stats = usage_stats.get(pid, {}) - uses = stats.get("uses", 0) - if uses == 0: + data = copy.deepcopy(sidecar_info["data"]) + stats = usage_stats.get(data["pattern_id"], {}) + if not stats: return None - - avg_score = stats["total_score"] / uses if uses > 0 else 0.5 - success = stats.get("success_signals", 0) - failure = stats.get("failure_signals", 0) - total_signals = success + failure or 1 - success_rate = success / total_signals - - current_conf = data.get("confidence", 0.7) + # Legacy aggregate callers get exact-window dedupe. File-based callers use event IDs. + incoming = stats.get("events") + if incoming is None: + key = hashlib.sha256( + json.dumps(stats, sort_keys=True, allow_nan=False).encode() + ).hexdigest() + incoming = {key: stats} + events = data.get("evolution_events", {}) + for key in set(incoming) & set(events): + if incoming[key] != events[key]: + raise ValueError("conflicting consumed evidence for stable identity") + if not set(incoming) - set(events): + return None + events.update(incoming) + uses = sum(e.get("uses", 0) for e in events.values()) + if not uses: + return None # outcomes are applied once there is actual retrieval evidence + avg_score = sum(e.get("total_score", 0) for e in events.values()) / uses + success = sum(e.get("success_signals", 0) for e in events.values()) + failure = sum(e.get("failure_signals", 0) for e in events.values()) + success_rate = success / (success + failure) if success + failure else 0.5 + current = data.get("confidence", 0.7) + baseline = data.get("evolution_baseline_confidence", current) delta = 0.0 - if uses >= 3: - delta += (avg_score - 0.6) * 0.15 - delta += (success_rate - 0.5) * 0.25 - + delta += (avg_score - 0.6) * 0.15 + (success_rate - 0.5) * 0.25 if failure > success and uses >= 2: delta -= 0.1 - - new_conf = max(0.3, min(0.98, round(current_conf + delta, 4))) - - if "usage_history" not in data: - data["usage_history"] = [] - data["usage_history"].append({ - "date": datetime.utcnow().isoformat() + "Z", - "uses_in_window": uses, - "avg_retrieval_score": round(avg_score, 4), - "success_rate": round(success_rate, 4), - "confidence_before": current_conf, - "confidence_after": new_conf, - "delta": round(delta, 4), - "source_projects": list(set(stats.get("projects", [])))[:5] - }) - - data["confidence"] = new_conf - data["version"] = data.get("version", "0.6.0") - parts = data["version"].split(".") - if len(parts) >= 3: + confidence = max(0.3, min(0.98, round(baseline + delta, 4))) + now = datetime.now(timezone.utc).isoformat() + data.setdefault("usage_history", []).append( + { + "date": now, + "uses_in_window": uses, + "avg_retrieval_score": round(avg_score, 4), + "success_rate": round(success_rate, 4), + "confidence_before": current, + "confidence_after": confidence, + "delta": round(confidence - current, 4), + "source_projects": sorted({p for e in events.values() for p in e.get("projects", [])})[ + :5 + ], + } + ) + data["evolution_events"] = events + data["evolution_baseline_confidence"] = baseline + data["confidence"] = confidence + parts = data.get("version", "0.6.0").split(".") + if len(parts) == 3: parts[2] = str(int(parts[2]) + 1) - data["version"] = ".".join(parts) - - data["last_evolved"] = datetime.utcnow().isoformat() + "Z" + data["version"] = ".".join(parts) + data["last_evolved"] = now return data -def update_index_from_usage(index_path, usage_stats, sidecars): - if yaml is None: - return None - try: - with open(index_path, "r") as f: - index = yaml.safe_load(f) - except Exception: - return None +def update_index_from_usage(index_path, usage_stats, sidecars, dry_run=False): + if yaml is None: + raise ValueError("PyYAML is required for index evolution") + with open(index_path) as stream: + index = yaml.safe_load(stream) changed = False - for problem, data in index.get("by_dramatic_problem", {}).items(): - current_patterns = data.get("patterns", []) + for data in index.get("by_dramatic_problem", {}).values(): + current = data.get("patterns", []) scored = [] - for pid in current_patterns: + for pid in current: if pid in usage_stats and pid in sidecars: - score = usage_stats[pid].get("success_signals", 0) + (sidecars[pid]["data"].get("confidence", 0.7) * 2) - scored.append((pid, score)) - if scored: - scored.sort(key=lambda x: x[1], reverse=True) - new_order = [p[0] for p in scored] - if new_order != current_patterns: - data["patterns"] = new_order - changed = True - - if changed: - with open(index_path, "w") as f: - yaml.safe_dump(index, f, sort_keys=False) - return "corpus-index.yaml updated with performance-based ordering" - return None - -def run_evolution(receipts_dir: str, outcomes_dir: str, patterns_dir: str = None, index_path: str = None) -> Dict: - """Programmatic interface. Returns evolution receipt.""" - if patterns_dir is None: - raise ValueError("patterns_dir is required for evolution") - - sidecars = load_all_sidecars(patterns_dir) - usage = aggregate_usage(receipts_dir, outcomes_dir) - - evolved = [] - for pid, info in sidecars.items(): - updated = evolve_sidecar(info, usage) - if updated: - save_json(info["path"], updated) - evolved.append(pid) - - index_msg = None - if index_path: - index_msg = update_index_from_usage(index_path, usage, sidecars) - - evolution_receipt = { - "evolution_receipt": { - "timestamp": datetime.utcnow().isoformat() + "Z", + info = sidecars[pid]["data"] + success = sum( + e.get("success_signals", 0) for e in info.get("evolution_events", {}).values() + ) + scored.append((pid, success + info.get("confidence", 0.7) * 2)) + scored.sort(key=lambda pair: pair[1], reverse=True) + ranked = iter(pid for pid, _ in scored) + scored_ids = {pid for pid, _ in scored} + order = [next(ranked) if pid in scored_ids else pid for pid in current] + if order != current: + data["patterns"] = order + changed = True + if not changed: + return None + text = yaml.safe_dump(index, sort_keys=False) + if not dry_run: + Path(index_path).write_text(text) + return text + + +def run_evolution( + receipts_dir, + outcomes_dir, + patterns_dir=None, + index_path=None, + dry_run=True, + expected_plan_hash=None, +): + """Plan by default. Explicit apply stages all files and rolls back caught I/O failures.""" + if patterns_dir is None or not Path(patterns_dir).is_dir(): + raise ValueError("patterns_dir must be an existing project-owned corpus directory") + for directory in (receipts_dir, outcomes_dir): + if not Path(directory).is_dir(): + raise ValueError("evidence directories must exist (empty is allowed)") + root = Path(patterns_dir).resolve() + lock = root / ".evolution.lock" + recovery_required = False + if not dry_run: + lock.mkdir() # single-writer guard; stale lock requires human recovery + + def input_hash(): + paths = list(root.rglob("*.json")) + for directory in (receipts_dir, outcomes_dir): + paths.extend(Path(directory).glob("*.json")) + paths.extend(Path(directory).glob("*.yaml")) + if index_path: + paths.append(Path(index_path)) + paths.append(Path(__file__)) + items = [ + (str(path.absolute()), hashlib.sha256(path.read_bytes()).hexdigest()) + for path in sorted(set(paths)) + ] + return hashlib.sha256(json.dumps(items).encode()).hexdigest() + + try: + plan_hash = input_hash() + if expected_plan_hash is not None and expected_plan_hash != plan_hash: + raise ValueError("stale evolution plan; review a new dry-run") + sidecars = load_all_sidecars(root) + consumed = {} + for info in sidecars.values(): + for event_id, event in info["data"].get("evolution_events", {}).items(): + digest = event.get("evidence_hash") + if digest is None: # legacy aggregate events have no evidence hash + continue + if event_id in consumed and consumed[event_id] != digest: + raise ValueError("conflicting consumed evidence for stable identity") + consumed[event_id] = digest + usage = aggregate_usage(receipts_dir, outcomes_dir, consumed) + if set(usage) - set(sidecars): + raise ValueError( + "unknown pattern in evidence: " + ", ".join(sorted(set(usage) - set(sidecars))) + ) + from jsonschema import Draft7Validator + + schema_path = Path(__file__).with_name("symbolic-narrative-pattern.schema.json") + if not schema_path.exists(): + schema_path = ( + Path(__file__).resolve().parents[1] + / "schemas/symbolic-narrative-pattern.schema.json" + ) + validator = Draft7Validator(load_json(schema_path)) + changes = {} + evolved = [] + for pid, info in sidecars.items(): + updated = evolve_sidecar(info, usage) + if updated: + validator.validate(updated) + info["data"] = updated + changes[Path(info["path"])] = json.dumps( + updated, indent=2, allow_nan=False + ).encode() + evolved.append(pid) + index_text = ( + update_index_from_usage(index_path, usage, sidecars, dry_run=True) + if index_path + else None + ) + if index_text is not None: + changes[Path(index_path)] = index_text.encode() + receipt = { + "timestamp": datetime.now(timezone.utc).isoformat(), "patterns_evolved": evolved, - "index_update": index_msg, - "usage_window": { - "receipts_scanned": len(glob.glob(os.path.join(receipts_dir, "*"))), - "outcomes_scanned": len(glob.glob(os.path.join(outcomes_dir, "*"))) - } + "dry_run": dry_run, + "plan_hash": plan_hash, + "index_update": index_text is not None, + "changes": [ + { + "path": str(path.resolve()), + "before_sha256": hashlib.sha256(path.read_bytes()).hexdigest(), + "after_sha256": hashlib.sha256(value).hexdigest(), + } + for path, value in changes.items() + ], } - } + result = {"evolution_receipt": receipt} + if input_hash() != plan_hash: + raise ValueError("stale inputs changed during planning") + if changes and not dry_run: + receipt_path = root.parent / ("evolution-" + uuid.uuid4().hex + ".json") + receipt["receipt_path"] = str(receipt_path) + # Include exact prior contents for manual recovery after a process/host crash. + receipt["before_contents"] = {str(path.resolve()): path.read_text() for path in changes} + changes[receipt_path] = json.dumps(result, indent=2, allow_nan=False).encode() + originals = {path: path.read_bytes() if path.exists() else None for path in changes} + modes = {path: path.stat().st_mode & 0o7777 for path in changes if path.exists()} + staged = {} + applied = [] + try: + for path, value in changes.items(): + with tempfile.NamedTemporaryFile(dir=path.parent, delete=False) as stream: + staged[path] = Path(stream.name) + stream.write(value) + stream.flush() + if path in modes: + os.chmod(stream.name, modes[path]) + os.fsync(stream.fileno()) + if input_hash() != plan_hash: + raise ValueError("stale inputs changed during staging") + # Persist provenance before touching corpus; success is returned only after read-back. + order = [receipt_path] + [p for p in changes if p != receipt_path] + for path in order: + os.replace(staged[path], path) + applied.append(path) + for path, value in changes.items(): + if path.read_bytes() != value: + raise OSError("evolution read-back mismatch") + except Exception as apply_error: + rollback_errors = [] + for path in reversed(applied): + if path == receipt_path: + continue # retain provenance until every corpus restore is verified + backup = None + original = originals[path] + try: + if original is None: + path.unlink() + else: + try: + with tempfile.NamedTemporaryFile( + dir=path.parent, delete=False + ) as stream: + backup = Path(stream.name) + stream.write(original) + stream.flush() + os.chmod(stream.name, modes[path]) + os.replace(backup, path) + if path.read_bytes() != original: + raise OSError("rollback read-back mismatch") + finally: + if backup is not None: + backup.unlink(missing_ok=True) + except Exception as restore_error: # noqa: BLE001 - report all failures + rollback_errors.append(f"{path}: {restore_error!r}") + if not rollback_errors and receipt_path in applied: + try: + receipt_path.unlink() + except Exception as restore_error: # noqa: BLE001 - report all failures + rollback_errors.append(f"{receipt_path}: {restore_error!r}") + if rollback_errors: + recovery_required = True + raise RuntimeError( + f"evolution apply failed: {apply_error!r}; rollback failed: " + + "; ".join(rollback_errors) + + f"; manual recovery required using {receipt_path}; lock retained at {lock}" + ) from apply_error + raise + finally: + for temp in staged.values(): + temp.unlink(missing_ok=True) + return result + finally: + if not dry_run and not recovery_required: + lock.rmdir() - return evolution_receipt def main(): parser = argparse.ArgumentParser() @@ -200,21 +398,27 @@ def main(): parser.add_argument("--outcomes-dir", required=True) parser.add_argument("--patterns-dir", required=True) parser.add_argument("--index", help="Optional path to corpus-index.yaml to update") + parser.add_argument( + "--expected-plan-hash", help="Reject apply if reviewed dry-run inputs changed" + ) + mode = parser.add_mutually_exclusive_group() + mode.add_argument("--dry-run", action="store_true", help="Plan only (default); zero writes") + mode.add_argument( + "--apply", action="store_true", help="Explicitly apply reviewed local corpus changes" + ) args = parser.parse_args() receipt = run_evolution( receipts_dir=args.receipts_dir, outcomes_dir=args.outcomes_dir, patterns_dir=args.patterns_dir, - index_path=args.index + index_path=args.index, + dry_run=not args.apply, + expected_plan_hash=args.expected_plan_hash, ) print(json.dumps(receipt, indent=2)) - if receipt["evolution_receipt"]["patterns_evolved"]: - print(f"\nEvolved {len(receipt['evolution_receipt']['patterns_evolved'])} patterns.") - else: - print("\nNo patterns met evolution thresholds in this window.") if __name__ == "__main__": main() diff --git a/packages/kubrick_helpers/src/kubrick_helpers/symbolic-narrative-pattern.schema.json b/packages/kubrick_helpers/src/kubrick_helpers/symbolic-narrative-pattern.schema.json new file mode 100644 index 0000000..d071bd8 --- /dev/null +++ b/packages/kubrick_helpers/src/kubrick_helpers/symbolic-narrative-pattern.schema.json @@ -0,0 +1,199 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema#", + "title": "SymbolicNarrativePattern", + "type": "object", + "required": [ + "pattern_id", + "title", + "domain", + "source_tier", + "dramatic_operations", + "cinematic_affordances", + "confidence" + ], + "properties": { + "pattern_id": { + "type": "string" + }, + "title": { + "type": "string" + }, + "domain": { + "type": "string", + "enum": [ + "semiotic", + "myth-folklore", + "ritual-liminal", + "dream-unconscious", + "alchemical", + "spatial-geometric", + "sound-rhythmic", + "cinematic", + "historical-esoteric", + "contemporary" + ] + }, + "source_tier": { + "type": "string", + "enum": [ + "PRIMARY", + "EARLY_COMMENTARY", + "SCHOLARLY", + "PRACTITIONER", + "COMPARATIVE", + "POPULAR", + "INTERNET" + ] + }, + "source_refs": { + "type": "array", + "items": { + "type": "object", + "properties": { + "author": { + "type": "string" + }, + "title": { + "type": "string" + }, + "date": { + "type": "string" + }, + "cultural_context": { + "type": "string" + } + } + } + }, + "dramatic_operations": { + "type": "array", + "items": { + "type": "string" + } + }, + "transformation_grammars": { + "type": "array", + "items": { + "type": "string" + } + }, + "cinematic_affordances": { + "type": "array", + "items": { + "type": "string" + } + }, + "applicable_genres": { + "type": "array", + "items": { + "type": "string" + } + }, + "applicable_formats": { + "type": "array", + "items": { + "type": "string" + } + }, + "cultural_scope": { + "type": "array", + "items": { + "type": "string" + } + }, + "misuse_risks": { + "type": "array", + "items": { + "type": "string" + } + }, + "mutation_requirements": { + "type": "object" + }, + "confidence": { + "type": "number", + "minimum": 0, + "maximum": 1 + }, + "retrieval_score": { + "type": "object", + "properties": { + "dramatic_fit": { + "type": "number" + }, + "character_fit": { + "type": "number" + }, + "cultural_fit": { + "type": "number" + }, + "cinematic_fit": { + "type": "number" + }, + "source_quality": { + "type": "number" + }, + "mutation_potential": { + "type": "number" + }, + "continuity_compatibility": { + "type": "number" + }, + "clich\u00e9_risk": { + "type": "number" + } + } + }, + "version": { + "type": "string" + }, + "last_evolved": { + "type": "string" + }, + "evolution_baseline_confidence": { + "type": "number", + "minimum": 0, + "maximum": 1 + }, + "usage_history": { + "type": "array", + "items": { + "type": "object" + } + }, + "evolution_events": { + "type": "object", + "additionalProperties": { + "type": "object", + "properties": { + "evidence_hash": { + "type": "string" + }, + "uses": { + "type": "integer", + "minimum": 0 + }, + "total_score": { + "type": "number", + "minimum": 0 + }, + "success_signals": { + "type": "integer", + "minimum": 0 + }, + "failure_signals": { + "type": "integer", + "minimum": 0 + }, + "projects": { + "type": "array", + "items": { + "type": "string" + } + } + }, + "additionalProperties": false + } + } + } +} diff --git a/packages/operator/src/continuity_forge_operator/store.py b/packages/operator/src/continuity_forge_operator/store.py index 3800ca7..4ed5e06 100644 --- a/packages/operator/src/continuity_forge_operator/store.py +++ b/packages/operator/src/continuity_forge_operator/store.py @@ -2,6 +2,8 @@ from __future__ import annotations +from collections.abc import Iterator +from contextlib import contextmanager from datetime import timedelta from threading import RLock from uuid import UUID @@ -131,6 +133,18 @@ def _check_expected_project_state(self, document_key: str, envelope: MutationEnv "expected_state_hash conflict: does not match current project state_hash" ) + @contextmanager + def candidate_write(self, document_key: str, envelope: MutationEnvelope) -> Iterator[None]: + """Validate candidate state and serialize writes with this store's project commits. + + This is the in-process store lock, not a distributed write lease. + """ + with self._lock: + if document_key not in self._projects: + raise OperatorError("unknown project") + self._check_expected_project_state(document_key, envelope) + yield + def ingest_script( self, *, diff --git a/pyproject.toml b/pyproject.toml index 0c476ca..396d73b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -17,6 +17,9 @@ dependencies = [ [project.optional-dependencies] dev = [ + # Full test collection exercises the optional Kubrick helpers. + "pyyaml>=6.0,<7", + "jsonschema>=4.0,<5", "pytest>=8.3,<9", "pytest-cov>=5,<7", "ruff>=0.6,<1", @@ -39,7 +42,7 @@ openai = [ http = [ "httpx>=0.27,<1", ] -kubrick-helpers = ["pyyaml>=6.0"] +kubrick-helpers = ["pyyaml>=6.0", "jsonschema>=4.0"] production = [ "temporalio>=1.7,<2", "psycopg[binary]>=3.1,<4", diff --git a/skills/hermes-continuity-forge/SKILL.md b/skills/hermes-continuity-forge/SKILL.md index 9a245f9..aeb051c 100644 --- a/skills/hermes-continuity-forge/SKILL.md +++ b/skills/hermes-continuity-forge/SKILL.md @@ -28,14 +28,14 @@ Continuity Forge is a **deterministic cinematic-production kernel**. You call it - `idempotency_key` (unique per intent) - `rationale` (human-readable why) - `expected_state_hash` when continuing prior project state -6. **Write lease** before mutating a project (`acquire_write_lease` → work → `release_write_lease`). +6. **Write lease** before canon/project or approval writes; release in finally after successful acquisition. Read-side compilation needs none. Candidate artifact writes (`queue_generation`, `run_shot_repair_loop`) validate project expected-state hashes under the runtime store lock but do not require a project lease and never promote canon. 7. **No unbounded director loop.** Work shot-by-shot or pipeline-by-pipeline with validation. If a user asks you to “just generate the whole movie in chat,” refuse and route through **breakdown** (structure + continuity) or shot contracts + proof/repair tools. ## Prefer MCP -Assume MCP server `continuity-forge` is configured (`continuity-forge-mcp`). Tool catalog: `references/mcp-tools.md`. +Confirm MCP server `continuity-forge` is configured (`continuity-forge-mcp`). Bootstrap from repo `docs/SETUP.md` (Python 3.12+, editable checkout), then `docs/hermes/README.md` and `docs/hermes/mcp.example.json` for stdio registration. A pip package does not install the skill. Never invent host tool names or credentials. Tool catalog: `references/mcp-tools.md`. REST fallback when MCP is unavailable (same host as UI): @@ -63,7 +63,7 @@ REST fallback when MCP is unavailable (same host as UI): **Goal:** User pastes/imports a script → machine-readable **shot-by-shot breakdown with continuity** for connectors or review. 1. Obtain Fountain/FDX source (user paste). -2. Call **`build_breakdown`** (MCP) or REST `POST /v1/breakdown` with `title`, `text`, `document_key`, `format`. +2. Call **`build_breakdown`** (MCP) or REST `POST /v1/breakdown` with `title`, `document_key`, `format`. MCP uses `source`; REST uses `text`. 3. Optionally **`build_breakdown_markdown`** for a human-readable export. 4. Return a short summary to the human: - claim (`shot_breakdown_with_continuity_not_production_film`) @@ -104,8 +104,8 @@ On lease conflict: report holder/expiry; do not force; ask human or wait. 1. Ensure project exists (Workflow C). 2. `list_shot_summaries` or project shot contracts → pick `shot_id`. -3. `queue_generation` **or** `run_shot_repair_loop`. -4. Present candidate as **PROPOSED**. +3. `queue_generation` **or** `run_shot_repair_loop` with explicit actor_id, authorization_scope (`generation:preview` / `generation:repair`), unique-per-intent idempotency_key and rationale. Do not rely on static server defaults; see complete recipes in `references/workflows.md`. +4. Check returned candidate shot ID, authority **PROPOSED**, status and validation results. Read project status again and confirm its state_hash is unchanged. Candidate artifacts have no public MCP read-back endpoint: distinguish returned candidate verification from persisted-artifact verification; do not claim persistence without a store read-back route. 5. Approvals only with lease + envelope and explicit human instruction. --- @@ -122,7 +122,7 @@ On lease conflict: report holder/expiry; do not force; ask human or wait. 1. Hold lease as actor. 2. `POST /v1/approvals/request` then `decide` with new idempotency keys. -3. Never auto-grant without explicit human instruction. +3. Never auto-grant without explicit human instruction. Read `GET /v1/projects/{key}/approvals`, match exact approval_id/status, then release the acquired lease in finally; failed read-back means unverified, not success. ## Communication style diff --git a/skills/hermes-continuity-forge/references/workflows.md b/skills/hermes-continuity-forge/references/workflows.md index 6cd676b..554ff35 100644 --- a/skills/hermes-continuity-forge/references/workflows.md +++ b/skills/hermes-continuity-forge/references/workflows.md @@ -60,38 +60,74 @@ Summarize JSON: claim, hashes, shots[].status / attempts / repair_actions. ## 3. Ingest under lease (MCP) -```text -1. acquire_write_lease(document_key="{{DOC}}", holder="{{ACTOR}}", ttl_seconds=600) -2. ingest_script( - source="{{SOURCE}}", - document_key="{{DOC}}", - actor_id="{{ACTOR}}", - authorization_scope="kernel:pipeline", - idempotency_key="ingest-{{DOC}}-{{unique}}", - rationale="Hermes operator ingest", - title="…", - format="fountain" - ) -3. get_project_status("{{DOC}}") -4. release_write_lease("{{DOC}}", "{{ACTOR}}") +Use the exact try/finally ingest recipe in the writing skills' integration guide. +This complete sequence uses freshly read project state (not a pipeline shots hash): + +```python +acquire_write_lease(document_key=DOC, holder=ACTOR, ttl_seconds=600) +try: + prior = get_project_status(document_key=DOC) + result = ingest_script( + source=SOURCE, + document_key=DOC, + actor_id=ACTOR, + authorization_scope="kernel:pipeline", + idempotency_key=INTENT, + rationale="Apply user-approved screenplay revision", + expected_state_hash=prior["state_hash"] if prior else None, + ) + status = get_project_status(document_key=DOC) + assert status["state_hash"] == result["project"]["state_hash"] +finally: + release_write_lease(document_key=DOC, holder=ACTOR) ``` ---- - -## 4. Repair one shot (MCP) - -```text -1. list_shot_summaries(source=…) OR use stored project contracts after ingest -2. run_shot_repair_loop( - document_key="{{DOC}}", - shot_id="{{SHOT}}", - seed="hermes-1", - max_attempts=3, - fail_first=false - ) -3. Report authority=PROPOSED, status, attempts, repair plan +## 4. Candidate generation / repair (MCP) + +Resolve SHOT from the project's stored contracts. ACTOR is stable; INTENT and +REPAIR_INTENT are distinct, fresh per logical operation (reused only on retries). +These calls may write candidate artifacts but do not acquire a canon lease. +Pass the freshly read project `state_hash` as `expected_state_hash` (not a shot or +pipeline hash). The server validates it before generation and persistence while +holding the runtime project-store lock. Omission binds to a server-read snapshot, +not to an earlier client review. On conflict, re-read and review before retrying. +The server constructs command_schema_version; do not pass that argument. +This lock coordinates a single store instance; it is not a distributed lease +across independently hydrated filesystem/Postgres runtimes. + +```python +before = get_project_status(document_key=DOC) +candidate = queue_generation( + document_key=DOC, + shot_id=SHOT, + actor_id=ACTOR, + authorization_scope="generation:preview", + idempotency_key=INTENT, + rationale="User requested mock preview", + expected_state_hash=before["state_hash"], + seed="hermes-1", +) +repair = run_shot_repair_loop( + document_key=DOC, + shot_id=SHOT, + actor_id=ACTOR, + authorization_scope="generation:repair", + idempotency_key=REPAIR_INTENT, + rationale="User requested bounded mock repair", + expected_state_hash=before["state_hash"], + seed="hermes-1", + max_attempts=3, + fail_first=False, +) +after = get_project_status(document_key=DOC) +assert before["state_hash"] == after["state_hash"] ``` +Verify returned shot IDs, candidate authority=PROPOSED, validation/status and actual +attempts; do not claim an accepted_candidate exists if repair failed. There is no +public MCP candidate fetch tool. Return-payload checks plus unchanged canon are +not read-back of persisted candidate storage; label that limit explicitly. + --- ## 5. Drift audit (MCP) @@ -111,7 +147,8 @@ Summarize JSON: claim, hashes, shots[].status / attempts / repair_actions. 1. acquire lease as {{ACTOR}} 2. POST /v1/approvals/request { document_key, kind, actor_id, idempotency_key, rationale } 3. POST /v1/approvals/decide { approval_id, status: granted|denied, … } -4. release lease +4. GET /v1/projects/{key}/approvals — match exact approval_id and status +5. release the acquired lease in finally (including errors) ``` Only when the human explicitly requests grant/deny for the kind in scope. diff --git a/skills/kubrick/QUICKSTART.md b/skills/kubrick/QUICKSTART.md index b755bc0..bfedd1a 100644 --- a/skills/kubrick/QUICKSTART.md +++ b/skills/kubrick/QUICKSTART.md @@ -1,57 +1,8 @@ -# Kubrick Quickstart - -Kubrick is a **standalone Hermes skill** for precise symbolic narrative engineering. It works without Continuity Forge. - -## 1. Install - -```bash -# From inside this folder -./install.sh # → ~/.hermes/skills/kubrick -./install.sh creative # → ~/.hermes/skills/creative/kubrick -``` - -Or manually: -```bash -cp -R . ~/.hermes/skills/kubrick -``` - -Restart Hermes after installing. - -## 2. Basic Usage (Retrieval) - -Give it a brief: - -```bash -python scripts/retrieve_symbolic_patterns.py --brief evals/retrieval/inputs/sample_melodrama_lowbudget.yaml -``` - -It outputs a `retrieval_receipt` with ranked symbolic patterns, scores, and provenance. - -## 3. Evolution (Self-improvement) - -After using the skill on real projects, drop receipts/outcomes into `references/usage/` and run: - -```bash -python scripts/evolve_from_use.py -``` - -This updates pattern confidence and ordering based on actual results. - -## 4. Triggers (in Hermes) - -- "develop screenplay" -- "kubrick style" -- "symbolic narrative" -- "motif engineering" -- "handoff to continuity forge" -- "diagnose script" - -See `SKILL.md` for the full list and detailed procedures. - -## Next Steps - -- Read `README.md` for full capabilities and distribution notes. -- Explore `references/patterns/` and `evals/` for examples. -- See `examples/minimal-retrieval-example.zip` for a tiny input + expected pair. - -**This skill runs completely independently inside Hermes.** +# Kubrick quickstart + +1. Read [README.md](README.md) and resolve the embedded variant/source commit. +2. Preview the profile-aware install, then explicitly apply if authorized. +3. Load SKILL.md and references/standalone-procedure.md for creative work. +4. For optional retrieval/evolution, follow references/evolution-safety.md with + explicit project-owned paths. Never run evolution against installed defaults. +5. For explicitly requested canon handoff, use references/continuity-forge-integration.md. diff --git a/skills/kubrick/README.md b/skills/kubrick/README.md index dd3f6ea..333824e 100644 --- a/skills/kubrick/README.md +++ b/skills/kubrick/README.md @@ -1,121 +1,55 @@ -# Kubrick — Symbolic Cinematic Narrative Engineering System +# Kubrick — embedded Continuity Forge variant -![Kubrick — Symbolic Cinematic Narrative Engineering](assets/kubrick-hero.svg) +Creative narrative work is standalone. Load SKILL.md and +[standalone-procedure](references/standalone-procedure.md); Forge is optional for +explicit canon handoff, not a prerequisite to write or diagnose a scene. -**The primary skill for precise, motif-driven, geometrically rigorous cinematic storytelling.** +## Install safely -kubrick is the evolved replacement for earlier narrative engineering tools. It delivers production-ready scripts, scene contracts, and symbolic architecture that resist generic AI writing while creating latent, powerful visual and thematic systems. - -## Quick Install (Standalone in Hermes) - -```bash -# From inside this directory -./install.sh # installs to ~/.hermes/skills/kubrick -./install.sh creative # installs to ~/.hermes/skills/creative/kubrick -``` - -Or manually: -```bash -cp -R . ~/.hermes/skills/kubrick -# or categorized -cp -R . ~/.hermes/skills/creative/kubrick -``` - -**Kubrick works completely on its own** — no continuity-forge installation required. - -After install, restart Hermes and use natural language triggers (see SKILL.md). - -## What Makes It Different - -- **Observed first, meaning second**: Every motif begins with concrete, observable form before any interpretation. -- **Mandatory mutation**: No motif recurs identically unless stagnation is the dramatic point. -- **Three-channel symbolism**: Diegetic (objects/behavior), Dramaturgical (structure/choice), Cinematic (framing/geometry/rhythm) — power comes from crossing channels without explanation. -- **Provenance-linked Symbolic Narrative Pattern System**: Full `SymbolicNarrativePattern` schema, Narrative Affordance Registry, Transformation Grammar Registry, and 10+ domain packs grounded in PRIMARY/SCHOLARLY sources. -- **Executable Retrieval**: `scripts/retrieve_symbolic_patterns.py` provides deterministic, scored retrieval with exclusions, saturation awareness, and `NOT_COMPUTABLE` fallback. -- **Self-Evolution from Use**: The skill improves itself. Retrievals are auto-logged; project outcomes adjust pattern confidence, usage history, and index ordering via `scripts/evolve_from_use.py`. -- **Forge-native**: Produces clean `symbolic_architecture` and `cinematic_encoding` ready for Continuity Forge ledger and shot contracts. - -## Key New Capabilities (0.7.x) - -- Machine-readable pattern sidecars (`references/patterns/`) -- Deterministic retrieval with score decomposition and receipt emission -- Autonomous evolution engine that learns from real project/Forge usage -- Full support for project symbolic ledger, revision diffing, cultural review gates, and production feasibility - -## Distribution & Installation - -**Kubrick is a Hermes skill, not a Python package.** - -It is **not** distributed via PyPI or included in the `continuity-forge` wheel. The core package only ships the production kernel (IR, compiler, ledger, harness, etc.). Skills live as self-contained directories. - -### Recommended installation - -From the continuity-forge repo root: +From the continuity-forge checkout, record `git rev-parse HEAD`, then: ```bash -mkdir -p ~/.hermes/skills -cp -R skills/kubrick ~/.hermes/skills/ -cp -R skills/hermes-continuity-forge ~/.hermes/skills/ # strongly recommended companion +bash skills/kubrick/install.sh --dry-run +# Review active-profile target and backup root before opting in: +bash skills/kubrick/install.sh --apply ``` -Categorized layout (if your Hermes setup uses `creative/`): - -```bash -cp -R skills/kubrick ~/.hermes/skills/creative/ -``` - -You can also symlink for development: -```bash -ln -s "$(pwd)/skills/kubrick" ~/.hermes/skills/kubrick -``` - -### Why directory-only distribution? - -- Skills contain markdown, schemas, examples, and small helper scripts that Hermes loads directly. -- **Kubrick works completely on its own inside Hermes**. No continuity-forge installation is required. -- Copy the skill directory and the scripts run self-contained with no external dependencies. - -**Note for Continuity Forge users**: An optional `kubrick-helpers` Python package exists inside the continuity-forge repo (for CLIs and imports when using the full Python package). It is **not required** to use this skill in Hermes. - -See `QUICKSTART.md` for the fastest way to get started. - -See `docs/hermes/README.md` for general Hermes + Continuity Forge integration guidance. - - -## Quick Start - -A minimal working example (brief + expected receipt) is included in `examples/minimal-retrieval-example.zip`. - -Load `kubrick`. - -**Retrieval example:** -```bash -python scripts/retrieve_symbolic_patterns.py --brief my-brief.yaml -``` - -**Evolution (after use):** -```bash -# After projects, drop outcomes in references/usage/outcomes/ -python scripts/evolve_from_use.py -``` - -See SKILL.md for full procedures, prompts, and evolution workflow. - -## Core Artifacts - -- symbolic_intent contract -- motif_registry (observed_form + lifecycle) -- cinematic_encoding (relational + shot recurrence) -- symbolic_architecture (Forge handoff) -- retrieval_receipt -- evolution_receipt - -## Companion - -Use with `hermes-continuity-forge` for the full symbolic-to-production pipeline with memory and revision safety. - -## Version - -0.8.0 (Executable Retrieval + Self-Evolution) - -See CHANGELOG.md for details. +`HERMES_HOME` selects the profile. `--target` selects an explicit destination; +`creative` selects categorized installation. From another cwd, use the absolute +installer path, never `cp -R .`. Python-only Windows hosts can invoke +`scripts/install_skill.py`. This installs the skill directory, not a package. +See [source ownership, backups and evolution safety](references/evolution-safety.md). +Do not install this variant alongside another `kubrick` trigger or assume personal, +organizational and embedded releases are identical. No license is changed here. + +## Optional Python helpers + +The continuity-forge wheel includes kubrick_helpers and its CLIs, **not** SKILL.md +or the narrative corpus. Python 3.12+ editable install from repo root: +`python -m pip install -e '.[kubrick-helpers]'`. +The standalone evolution script requires `pyyaml` and `jsonschema` in a reviewed +venv. Creative use needs neither. Retrieval emits candidate ranking receipts; +persist real receipts explicitly in project-owned directories. + +## Evolution + +No unattended auto-evolution is enabled. Copy the pattern corpus/index to a +project directory and use the complete plan/apply recipe in +[references/evolution-safety.md](references/evolution-safety.md). +Default mode and `--dry-run` are zero-write. `--apply` validates changed sidecars, +deduplicates immutable evidence, recomputes cumulative scores from a fixed +baseline, preserves unused index entries, and writes a unique recovery receipt. +Caught I/O failures roll back; multi-file changes are not crash-atomic. + +## Forge handoff + +Use [the MCP handoff recipe](references/continuity-forge-integration.md). +There is no Forge ingest CLI. Symbolic packets remain proposed attachments unless +the actual kernel schema and returned persisted state demonstrate support. + +## Verification + +From a Python 3.12+ checkout with `.[dev,kubrick-helpers]`: +`python -m pytest tests/test_kubrick_evolution_audit.py tests/test_kubrick_install_audit.py tests/test_authored_skill_recipes.py tests/test_kubrick_esoteric.py -q`. +All installer/evolution writes use disposable directories; recipe tests use a +fresh local runtime and mock provider, not live services. diff --git a/skills/kubrick/SKILL.md b/skills/kubrick/SKILL.md index 952d5ab..bf2695a 100644 --- a/skills/kubrick/SKILL.md +++ b/skills/kubrick/SKILL.md @@ -74,23 +74,16 @@ A motif becomes powerful when it crosses channels without being explicitly ident ## Prerequisites -- Continuity Forge installed and in PATH: - ```bash - pip install -e '.[dev]' # from continuity-forge repo - continuity-forge --help - ``` -- (Recommended) `continuity-forge-mcp` configured in Hermes for tool use. -- Optional: `humanizer` for final voice. - -Env for Forge (pass to any MCP/terminal calls): -```bash -export CF_STORE_ROOT="$HOME/.local/share/continuity-forge" -# export CF_PROVIDER=mock -``` +Creative use is standalone: load this directory's SKILL.md and the relevant references. +Read `references/standalone-procedure.md` before routing, drafting or scoring. +For optional Forge handoff only, use a Python 3.12+ full editable checkout and the +operator skill; setup and MCP registration are in repo `docs/SETUP.md` and +`docs/hermes/README.md`. Package installation does not install skill directories. +Do not configure credentials, provider calls or a live store merely to write a scene. ## Request Routing & Modes -Same as base (DEVELOP, DRAFT, DIAGNOSE, REVISE, POLISH, CONTINUITY, PRODUCTION, ADAPT) plus symbolic-specific routing. +Read `references/standalone-procedure.md` for mode routing, ordered phases, artifact selection and the cited diagnosis rubric; load the named local narrative references for the selected mode. When the goal is production use with Forge, prefer: - DEVELOP → handoff to Forge ingest/compile @@ -113,7 +106,7 @@ When the goal is production use with Forge, prefer: ## Core Workflow (Phases) -1–11. (Intake → Premise → Characters → World → Theme → Macrostructure → Sequences/Beats → Scene Engine → Dialogue/Prose → Continuity Ledger → Revision) — same as base, now augmented with symbolic tracking. +Read `references/standalone-procedure.md` for mode routing, ordered phases, artifact selection and the cited diagnosis rubric; load the named local narrative references for the selected mode. **Module 5B — Symbolic Dramaturgy and Cinematic Encoding** (new primary module, integrated throughout): @@ -125,7 +118,7 @@ Before or alongside scene work: - Record `tradition_boundaries` and `correspondence_map` (private/hidden where appropriate). - Translate to `cinematic_encoding`: composition_patterns (relational, not cliché), geometric_patterns, blocking_patterns, camera_patterns, shot_recurrence (with mutation ledger), edit_cadence, sonic_motifs, production_design_states. -**12. Handoff to Continuity Forge (critical phase)** +**12. Handoff to Continuity Forge (optional, explicit handoff)** After foundations or scene contracts (now including symbolic architecture) are approved: - Use Forge to materialize canonical state: @@ -156,11 +149,11 @@ A–L (see `references/anti-slop-patterns.md`) plus M–W for symbolic work: - **Gate V (Mystery by Obscurity)**: Ambiguity from withheld causal information rather than open relation. - **Gate W (Premature Closure)**: Explicit confirmation of the "correct" interpretation. -Additional Forge gate (carried over): Gate M (Forge Bypass) — generating changes without Forge ledger update. +Additional Forge gate (carried over): Gate F-CANON (Forge Bypass) — generating changes without Forge ledger update. ## Output Selection Logic -Same as base. Preferred handoff artifacts: +Read `references/standalone-procedure.md` for mode routing, ordered phases, artifact selection and the cited diagnosis rubric; load the named local narrative references for the selected mode. - Structured project brief (matches Forge intake) - Scene contracts (feed `build_shot_contracts`) - Approved canon list (for mutation envelopes) @@ -171,7 +164,7 @@ Same as base. Preferred handoff artifacts: **Handoff rules**: - Creative development (this skill) produces **PROPOSED** or draft material (including symbolic proposals). -- Forge ingestion makes it canonical. +- Only schema-validated deterministic output committed through Forge is canonical; narrative/model proposals are not promoted merely by attachment. - Use leases + full mutation contract (`actor_id`, `authorization_scope`, `idempotency_key`, `rationale`) for any write path. - Always surface Forge receipts/hashes in responses. - Claim policy: material generated here is for development; final identity lives in Forge. @@ -185,11 +178,11 @@ See the companion skill `hermes-continuity-forge` for operator details (leases, ## Format-Specific Routing -Same as base, with the addition that Forge's shot contracts and ledger are format-aware. Symbolic density and cinematic encoding expectations scale with format (features support richer recurrence and geometric layering; shorts demand extreme compression and precision). +Read `references/standalone-procedure.md` for mode routing, ordered phases, artifact selection and the cited diagnosis rubric; load the named local narrative references for the selected mode. ## Diagnosis Rubric -Same 1-5 rubric. When Forge is in play, also score "Forge alignment". When symbolic work active, additionally score: +Use the anchored 1–5 rubric in `references/standalone-procedure.md`. When Forge is in play, also score "Forge alignment". When symbolic work active, additionally score: - Motif mutation and lifecycle fidelity - Channel crossing without explanation - Relational composition (vs. cliché) @@ -220,42 +213,14 @@ Same 1-5 rubric. When Forge is in play, also score "Forge alignment". When symbo -## Evolution from Use (Self-Improving Corpus) - -Kubrick is designed to improve itself through repeated application. - -**Core Mechanism** -- Every retrieval automatically logs a receipt to `references/usage/receipts/`. -- After Forge handoff, revision, or project review, record outcomes in `references/usage/outcomes/`. -- Run the evolution engine: `python scripts/evolve_from_use.py` - -**What Evolves** -- Pattern `confidence` is raised for patterns that repeatedly deliver clean results (low debt, successful mutations, no collisions). -- `usage_history` is appended to sidecars with performance data. -- `corpus-index.yaml` re-orders suggestions based on observed success. -- Weak or overused patterns have confidence lowered and may be flagged for deprecation or mutation rule changes. - -**How to Feed It** -1. After a project or significant sequence: - ```bash - # record outcome - echo '{"pattern_id": "alchemical_nigredo_putrefaction", "project": "my-film-042", "outcome": "success", "signals": ["clean revision", "Forge accepted"]}' > references/usage/outcomes/$(date +%s).json - ``` -2. Run evolution: - ```bash - python scripts/evolve_from_use.py - ``` -3. The engine produces `references/evolution/evolution-*.json` receipts. - -**Integration with Ledger** -Project symbolic ledgers can be copied to `references/usage/ledgers/` for richer signals (saturation trends, debt accumulation, revision diff results). - -**Governance** -- Evolution only adjusts confidence and history. Structural changes (new patterns, new grammars) still require human review. -- All changes are timestamped and accompanied by an evolution receipt. -- You can disable auto-logging or run evolution in dry-run mode. +## Evolution from Use (explicit, local corpus maintenance) -This turns every real use of the skill (especially when paired with Continuity Forge) into training data that sharpens future retrieval. +Read `references/evolution-safety.md` before running evolution. It is plan-only by +default, requires explicit project-owned corpus/evidence paths, validates changed +sidecars, preserves unused index entries, and consumes stable receipt identities +once. Use `--apply` only after reviewing the plan. This is heuristic ranking, +not measured calibration or proof of artistic success. No automatic mutation is +triggered by loading the skill. **Key Commands / Behaviors** - "Develop this premise with strong symbolic architecture and motif lifecycle" → DEVELOP + symbolic_intent + motif_registry (observed first) + cinematic_encoding. diff --git a/skills/kubrick/install.sh b/skills/kubrick/install.sh index 96c23a8..91d9320 100755 --- a/skills/kubrick/install.sh +++ b/skills/kubrick/install.sh @@ -1,42 +1,5 @@ #!/usr/bin/env bash -# Kubrick — Hermes Skill Installer -# Usage: -# ./install.sh # installs to ~/.hermes/skills/kubrick -# ./install.sh creative # installs to ~/.hermes/skills/creative/kubrick - -set -e - -TARGET_BASE="${HOME}/.hermes/skills" -SUBDIR="" - -if [[ "$1" == "creative" || "$1" == "categorized" ]]; then - SUBDIR="creative/" -fi - -DEST="${TARGET_BASE}/${SUBDIR}kubrick" - -echo "Installing Kubrick to: ${DEST}" -mkdir -p "$(dirname "${DEST}")" - -if [ -d "${DEST}" ]; then - echo "Existing installation found. Backing up to ${DEST}.bak" - rm -rf "${DEST}.bak" - mv "${DEST}" "${DEST}.bak" -fi - -cp -R . "${DEST}" - -# Make scripts executable -chmod +x "${DEST}/scripts/"*.py 2>/dev/null || true - -echo "" -echo "✅ Kubrick installed successfully." -echo "" -echo "Location: ${DEST}" -echo "" -echo "Next steps:" -echo " 1. Restart Hermes or reload skills." -echo " 2. Try triggers like: 'develop screenplay', 'kubrick style', 'symbolic narrative'" -echo "" -echo "The skill works completely standalone inside Hermes." -echo "Optional: pair it with the hermes-continuity-forge skill for full production handoff." +# Profile-aware embedded Kubrick installer. Plan by default; --apply installs. +set -euo pipefail +SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)" +exec "${PYTHON:-python3}" "${SCRIPT_DIR}/scripts/install_skill.py" "$@" diff --git a/skills/kubrick/references/continuity-forge-integration.md b/skills/kubrick/references/continuity-forge-integration.md index 04cae9d..fea0829 100644 --- a/skills/kubrick/references/continuity-forge-integration.md +++ b/skills/kubrick/references/continuity-forge-integration.md @@ -1,122 +1,57 @@ -# Continuity Forge Integration - -This skill is the **creative / structural** layer. Continuity Forge is the **deterministic kernel** that owns canonical state. - -## When to Handoff - -- After premise, characters, structure, or scene contracts are approved by the user. -- Before or instead of writing full prose pages when the goal is production use. -- On any material change to canon (new approved scenes, character traits that affect continuity, structure revisions). - -## Recommended Handoff Flow - -1. Produce clean artifacts from this skill (project brief, scene contracts, character bibles, approved canon list, symbolic_architecture and cinematic_encoding when applicable). -2. Ingest via Forge (prefer MCP or CLI with proper mutation envelope): - - `acquire_write_lease` - - `ingest_script` (or `compile_script` + ingest) - - `build_ledger` / `build_shot_contracts` -3. Capture receipt (hashes, document_key, shot IDs). -4. Reference Forge state in future work (use `get_project_status`, `inspect_scene`, etc. for grounding). - -## CLI Examples - -```bash -# Basic compile from fountain or structured text -continuity-forge compile path/to/outline.fountain --out out/ - -# Or from a scene contract / brief you produced here -continuity-forge ingest --document-key myfilm --source structured-outline.md +# Optional Continuity Forge handoff + +Creative work is standalone. For an explicitly requested canon handoff, load +`hermes-continuity-forge`, resolve the authorized document and actor, and confirm +MCP registration through repo `docs/hermes/README.md` (Python 3.12+; `docs/SETUP.md`). +No `ingest` subcommand exists in the Forge CLI. The CLI `compile` command is a +local deterministic parse/export, not a canonical project write. + +Use `terminal(command="continuity-forge compile path/to/script.fountain --out path/to/ir.json")` +only for a reviewed output file. Compile Fountain/FDX screenplay source, not a +JSON scene contract, outline prose or symbolic packet. Preserve source as immutable input. + +## MCP sequence + +This Python-shaped recipe names MCP tools, not a new client library. Substitute +`DOC`, `ACTOR`, `SOURCE` and `INTENT` from the approved request; `INTENT` is a fresh +unique key per logical write, reused only for a retry of that identical intent. +Never import server persistence into an operator client. The server constructs +command_schema_version through MutationEnvelope; it is not an ingest_script argument. + +```python +acquire_write_lease(document_key=DOC, holder=ACTOR, ttl_seconds=600) +try: + prior = get_project_status(document_key=DOC) + result = ingest_script( + source=SOURCE, document_key=DOC, actor_id=ACTOR, + authorization_scope="kernel:pipeline", idempotency_key=INTENT, + rationale="Apply user-approved screenplay revision", title="Reviewed script", + format="fountain", revision="0.1.0", + expected_state_hash=prior["state_hash"] if prior else None, + ) + status = get_project_status(document_key=DOC) + assert status is not None + assert status["state_hash"] == result["project"]["state_hash"] +finally: + release_write_lease(document_key=DOC, holder=ACTOR) ``` -## MCP Tools (via companion `hermes-continuity-forge` skill) - -Typical tools you will call after creative work: -- `compile_script` -- `ingest_script` (with mutation contract) -- `build_ledger` -- `build_shot_contracts` -- `get_project_status` -- `audit_drift` - -Always include: -- `document_key` -- `actor_id` (e.g. hermes-kubrick-) -- Full mutation envelope when writing - -## Mutation Contract Requirements (when ingesting changes) +Acquire must succeed before entering the try/finally; never release someone else's +lease. Stop on conflicts, stale hashes, failed schema validation or incomplete +read-back. Keep the full result, document key, run ID, scene/shot IDs and hashes. +Future revisions use fresh project `state_hash`, not a pipeline shots hash. -From the operator skill: -- actor_id -- authorization_scope -- idempotency_key -- rationale -- expected_state_hash (when updating existing) +## Narrative and symbolic packets -This skill produces the *rationale* and *content*. The operator skill (or direct MCP call) supplies the envelope. +Briefs, character bibles, scene contracts, symbolic_architecture and cinematic_encoding +are **PROPOSED attachments for review**, not guaranteed compiler inputs or supported +IR fields. No automatic motif/geometry field mapping is promised. Check the actual +kernel models and round-trip returned schemas before claiming a constraint was +stored. The deterministic kernel alone owns approved identity, ledger, IR and shot +contracts; a narrative skill or model cannot promote attachments into canon. -## Boundaries +## Verification without live writes -- This skill may generate **PROPOSED** narrative material and scene contracts (including symbolic proposals). -- Forge owns the canonical ledger, IR, and shot contracts. -- Never claim "this is now in the film" until you have a Forge receipt with committed status. -- For drift or contradictions discovered here: run local CONTINUITY pass, then cross-validate with Forge `audit_drift`. - -## Recommended Pairing - -Load both skills: -- `kubrick` for creative development, diagnosis, anti-slop, voice, structure, symbolic dramaturgy and cinematic encoding. -- `hermes-continuity-forge` for leases, ingestion, proof, shot repair, approvals. - -See main repo `docs/hermes/README.md` and the companion skill for full operator rules. - -## Export Mapping: symbolic_architecture → Continuity Forge IR / Shot Contracts - -When symbolic work is approved, the following mapping should be used when handoff artifacts are prepared for Forge (via MCP, CLI ingest, or structured brief). - -### Core Mapping Table - -symbolic_architecture: - governing_tension: → thematic_tension in project brief or scene contract - symbolic_intent: → scene_contract.symbolic_intent (dramatic_function, intended_payoff, prohibited_explanation) - motif_registry: → continuity_ledger.motifs[] and/or shot_contracts.recurrence[] - - motif_id + observed_form → ledger entry with state, recurrence_plan, provenance - - transformations → mutation history in ledger - - candidate_functions → narrative functions attached to motif - archetypal_functions: → character or role fields in scene contracts (as observable behaviors only) - correspondence_map: → shot_contracts.geometry or production_notes (only if provenance is PRIMARY/SCHOLARLY) - tradition_boundaries: → metadata on any symbolic element (do not allow unsupported equivalence) - recurrence_plan / inversion_plan / convergence_plan / residue_plan: - → shot_contracts.recurrence, mutation, convergence, residue fields - cinematic_encoding: - composition_patterns: → shot_contracts.composition (e.g., symmetry, negative_space, occlusion) - geometric_patterns: → shot_contracts.geometry (circle, grid, spiral, broken symmetry, etc.) - blocking_patterns: → shot_contracts.blocking (power_center, threshold, prohibited_zone, formation) - camera_patterns: → shot_contracts.camera (height, distance, movement, lens_behavior) - shot_recurrence: → explicit recurrence in shot contracts with mutation required - edit_cadence: → editing notes in shot contracts (rhythm, ellipsis, match cuts) - sonic_motifs: → sound design parameters (acousmatic, bridges, silence, residue) - production_design_states: → production design states tied to motifs - -### Handoff Rules -- Only include symbolic_architecture when it has passed quality gates and has provenance. -- For every motif or pattern exported, include at minimum: observed_form, mutation history, provenance reference, and intended dramatic function. -- Never export "meaning" — export the structural instructions (what must recur, how it must mutate, which cinematic parameters carry the charge). -- Forge will treat these as constraints on the Production IR and shot contracts, not as interpretive notes. - -Example (simplified): -```yaml -handoff: - symbolic_architecture: - motif_registry: - - motif_id: "red_carpet" - observed_form: "geometric red pattern underfoot" - recurrence_plan: "must mutate ownership, scale, or context each appearance" - provenance: "PRIMARY via film analysis (Kubrick Shining)" - cinematic_encoding: - geometric_patterns: ["broken symmetry", "repeating grid with contamination"] - blocking_patterns: ["power_center moves along the pattern"] - forge_target: - ledger.motifs: ["red_carpet"] - shot_contracts.geometry: ["grid", "contamination"] - shot_contracts.recurrence: {mutation_required: true} -``` +From a full checkout, use `terminal(command="python -m pytest tests/test_authored_skill_recipes.py tests/contract/test_mcp.py -q")`. +Tests bind this exact recipe to real MCP signatures with a fresh in-memory runtime +and mock provider; no external service, credentials or live approval is involved. diff --git a/skills/kubrick/references/evolution-safety.md b/skills/kubrick/references/evolution-safety.md new file mode 100644 index 0000000..af6cb91 --- /dev/null +++ b/skills/kubrick/references/evolution-safety.md @@ -0,0 +1,82 @@ +# Evolution and installation safety + +## Source ownership + +This is the **embedded continuity-forge Kubrick variant**, maintained with +scrimshawlife-ctrl/continuity-forge. It is not interchangeable with standalone +personal or Zero-State-LLC/Kubrick releases. Record `git rev-parse HEAD` from the +chosen checkout before installation. Keep only one `kubrick` trigger installed in +the intended profile. Migrating variants requires an explicit human choice and +backup of project data; do not merge histories or infer license changes. +Scriptwriting remains a compatibility procedural variant; prefer Kubrick for new +symbolic work, but do not silently replace an existing scriptwriting installation. + +## Installation + +Resolve the absolute skill directory (from skill_view or checkout), not caller cwd. +Use `terminal(command='bash skills/kubrick/install.sh --dry-run')` from repo root; +review source, target and backup root, then the same command with `--apply`. +The installer resolves `${HERMES_HOME:-$HOME/.hermes}`, supports `--target`, stages +only this skill and keeps unique target-keyed backups outside skill discovery. +Windows without Bash: use `python skills/kubrick/scripts/install_skill.py --dry-run`. +The installer validates syntax/content, not creative quality. To restore, stop +users of the target, review the exact reported backup and copy it back to that +same target; never choose a backup from a different target hash. Keep old backups. + +## Evolution prerequisites + +Creative use needs no package. Evolution requires Python 3.12+, PyYAML and jsonschema: +`terminal(command="python -m pip install pyyaml jsonschema")` in a reviewed venv; +or full-checkout `python -m pip install -e '.[kubrick-helpers]'`. +Embedded script and packaged `kubrick-evolve` share tested behavior and schema. +Do not write usage into an installed skill. Copy `references/patterns/` and +`references/corpus-index.yaml` to a project-owned corpus and create separate +receipts/outcomes directories with write_file or approved filesystem tools. + +Use `terminal(command='python skills/kubrick/scripts/evolve_from_use.py --receipts-dir PROJECT/receipts --outcomes-dir PROJECT/outcomes --patterns-dir PROJECT/patterns --index PROJECT/corpus-index.yaml --dry-run')`. +Replace PROJECT with the actual reviewed absolute project directory. Dry-run and +omitted mode both create **zero files** (including no receipts/locks). Review the +JSON change list and hashes; replace `--dry-run` with +`--apply --expected-plan-hash REVIEWED_HASH` (the returned plan_hash) to mutate that +explicit corpus. The packaged `kubrick-evolve` accepts exactly the same arguments. + +## Evidence contract + +Persist real retrieval output explicitly at a project path; don't invent successes. +The embedded retrieval helper defaults to no logging; opt in with `--log-dir PROJECT/receipts`. +Retrieval JSON/YAML contains `retrieval_receipt` or the unwrapped record, with +`request_hash` (or `event_id`) and `ranked_patterns` (top three consumed). +Each item has `pattern_id` and finite numeric `total_score` in [0,1]. +Outcomes have `pattern_id`, `project`, `outcome` (success, failure, debt, collision, +revision_broken, neutral), and preferably immutable `event_id`. +Without event_id an outcome identity is project+pattern: corrections conflict, +not fresh evidence. Renames, copies, timestamp and metadata changes do not count +again. A stable identity with altered scored contents is rejected, even after +prior consumption. Do not mint a new ID to disguise a correction; review/rebuild +from the original baseline with a curated evidence set instead. + +Late outcomes recompute confidence from a fixed baseline plus all consumed events, +not from already-adjusted confidence. Identical/subset evidence is a no-op; +no new history/version or receipt is written. Legacy aggregate-only Python callers +get exact-window dedupe only; use file evidence for overlapping windows. +Pre-upgrade histories cannot prove consumed event IDs: restore a reviewed pristine +corpus and replay curated evidence once, rather than using already-inflated legacy +confidence as a baseline. Unknown patterns, invalid schema/JSON/YAML or nonfinite +numbers stop the operation, not partial success. + +## Apply / recovery / verification + +Apply uses a single-corpus lock, validates and stages all changed sidecars/index, +then replaces individual files atomically with rollback for caught failures. +It is **not crash-atomic across files**. A unique evolution-UUID.json beside the +patterns directory is persisted before corpus replacement and includes prior +contents and before/after hashes. Its existence alone is not success: compare all +after hashes with current files. A process/host crash may leave partial application +and a stale lock; stop writers, inspect the receipt, restore all before contents +or finish a reviewed recovery, then remove the stale lock. Do not auto-clear it. + +Only confidence, history, evolution event/baseline state, version, timestamp and +index ordering change; no new patterns or structural edits are inferred. Read-back +must match the planned hashes. Run the identical apply again: no files or mtimes +should change. Regression command from checkout: +`terminal(command="python -m pytest tests/test_kubrick_evolution_audit.py tests/test_kubrick_install_audit.py -q")`. diff --git a/skills/kubrick/references/standalone-procedure.md b/skills/kubrick/references/standalone-procedure.md new file mode 100644 index 0000000..630046a --- /dev/null +++ b/skills/kubrick/references/standalone-procedure.md @@ -0,0 +1,34 @@ +# Standalone narrative procedure + +No Forge, unseen base skill, or Python package is needed for creative work. +Load only the reference relevant to the selected mode, then produce the smallest +artifact that answers the request. Do not invent approvals or imply draft notes are canon. + +| Mode | Input and completion artifact | Required local reference | +|---|---|---| +| DEVELOP | Audience, format, premise, constraints → logline, dramatic question, character want/obstacle/stakes, causal beats for approval | story-structure.md | +| DRAFT | Approved foundations → scenes with objective, opposition, turn, irreversible result | scene-engineering.md | +| DIAGNOSE | Supplied pages → cited defects, rubric, prioritized repairs; no unsolicited rewrite | anti-slop-patterns.md | +| REVISE | Locked material + approved changes → revised pages and change log | continuity.md | +| POLISH | Stable structure → behavior-led dialogue with distinct voices; preserve facts | character-and-dialogue.md | +| CONTINUITY | Supplied facts → contradictions with scene citations, unresolved questions | continuity.md | +| PRODUCTION | Approved scenes → proposed brief, scene contracts, constraints and provenance | scene-engineering.md | +| ADAPT | Source + target format → retained dramatic core, compression/expansion plan, sample | format-specific-guidance.md | + +1. Intake: identify format, audience, scope, supplied source, locks and requested output. Ask only for blocking omissions. +2. Develop: premise → character wants/opposition → world rules → thematic conflict → macrostructure → causal sequences. Get foundation approval before extensive pages. +3. Scene pass: each scene changes a relationship, available choice, knowledge or material state. State entry/exit facts and payoff obligations. +4. Dialogue pass: remove explanations already legible in behavior; differentiate voices by tactics and omissions. +5. Continuity pass: compare all revised facts and chronology against supplied locks; record unresolved contradictions rather than silently replacing them. +6. Diagnose: score causality, agency, stakes, scene turns, voice, economy and continuity on 1–5 (1 broken/absent; 2 major repair; 3 workable with repair; 4 strong; 5 consistently supported). Cite one specific scene/line per score. Use N/A where source is insufficient. Read anti-slop-patterns.md and report every applicable gate pass/fail. +7. Deliver the requested artifact, changed facts, remaining defects and next approval needed. A draft remains PROPOSED. If Forge handoff is requested, load continuity-forge-integration.md and the operator skill; compile screenplay source, not arbitrary JSON briefs. + +Shorts prioritize one pressure line and compressed turns; features support layered +payoffs; pilots establish a repeatable episode engine; podcasts require audible +clarity; video essays require argument/evidence. Use format-specific-guidance.md +for the chosen format, not claimed automatic format awareness in the compiler. + +## Verification +A small standalone smoke task: develop a one-scene short about returning a lost key. +Deliver a logline, character want/obstacle/stakes, scene entry/turn/exit contract, +and cited rubric. No tools, Forge writes or corpus evolution are implied. diff --git a/skills/kubrick/schemas/symbolic-narrative-pattern.schema.json b/skills/kubrick/schemas/symbolic-narrative-pattern.schema.json index 4367a1a..d071bd8 100644 --- a/skills/kubrick/schemas/symbolic-narrative-pattern.schema.json +++ b/skills/kubrick/schemas/symbolic-narrative-pattern.schema.json @@ -2,44 +2,197 @@ "$schema": "http://json-schema.org/draft-07/schema#", "title": "SymbolicNarrativePattern", "type": "object", - "required": ["pattern_id", "title", "domain", "source_tier", "dramatic_operations", "cinematic_affordances", "confidence"], + "required": [ + "pattern_id", + "title", + "domain", + "source_tier", + "dramatic_operations", + "cinematic_affordances", + "confidence" + ], "properties": { - "pattern_id": { "type": "string" }, - "title": { "type": "string" }, - "domain": { "type": "string", "enum": ["semiotic", "myth-folklore", "ritual-liminal", "dream-unconscious", "alchemical", "spatial-geometric", "sound-rhythmic", "cinematic", "historical-esoteric", "contemporary"] }, - "source_tier": { "type": "string", "enum": ["PRIMARY", "EARLY_COMMENTARY", "SCHOLARLY", "PRACTITIONER", "COMPARATIVE", "POPULAR", "INTERNET"] }, + "pattern_id": { + "type": "string" + }, + "title": { + "type": "string" + }, + "domain": { + "type": "string", + "enum": [ + "semiotic", + "myth-folklore", + "ritual-liminal", + "dream-unconscious", + "alchemical", + "spatial-geometric", + "sound-rhythmic", + "cinematic", + "historical-esoteric", + "contemporary" + ] + }, + "source_tier": { + "type": "string", + "enum": [ + "PRIMARY", + "EARLY_COMMENTARY", + "SCHOLARLY", + "PRACTITIONER", + "COMPARATIVE", + "POPULAR", + "INTERNET" + ] + }, "source_refs": { "type": "array", "items": { "type": "object", "properties": { - "author": { "type": "string" }, - "title": { "type": "string" }, - "date": { "type": "string" }, - "cultural_context": { "type": "string" } + "author": { + "type": "string" + }, + "title": { + "type": "string" + }, + "date": { + "type": "string" + }, + "cultural_context": { + "type": "string" + } } } }, - "dramatic_operations": { "type": "array", "items": { "type": "string" } }, - "transformation_grammars": { "type": "array", "items": { "type": "string" } }, - "cinematic_affordances": { "type": "array", "items": { "type": "string" } }, - "applicable_genres": { "type": "array", "items": { "type": "string" } }, - "applicable_formats": { "type": "array", "items": { "type": "string" } }, - "cultural_scope": { "type": "array", "items": { "type": "string" } }, - "misuse_risks": { "type": "array", "items": { "type": "string" } }, - "mutation_requirements": { "type": "object" }, - "confidence": { "type": "number", "minimum": 0, "maximum": 1 }, + "dramatic_operations": { + "type": "array", + "items": { + "type": "string" + } + }, + "transformation_grammars": { + "type": "array", + "items": { + "type": "string" + } + }, + "cinematic_affordances": { + "type": "array", + "items": { + "type": "string" + } + }, + "applicable_genres": { + "type": "array", + "items": { + "type": "string" + } + }, + "applicable_formats": { + "type": "array", + "items": { + "type": "string" + } + }, + "cultural_scope": { + "type": "array", + "items": { + "type": "string" + } + }, + "misuse_risks": { + "type": "array", + "items": { + "type": "string" + } + }, + "mutation_requirements": { + "type": "object" + }, + "confidence": { + "type": "number", + "minimum": 0, + "maximum": 1 + }, "retrieval_score": { "type": "object", "properties": { - "dramatic_fit": { "type": "number" }, - "character_fit": { "type": "number" }, - "cultural_fit": { "type": "number" }, - "cinematic_fit": { "type": "number" }, - "source_quality": { "type": "number" }, - "mutation_potential": { "type": "number" }, - "continuity_compatibility": { "type": "number" }, - "cliché_risk": { "type": "number" } + "dramatic_fit": { + "type": "number" + }, + "character_fit": { + "type": "number" + }, + "cultural_fit": { + "type": "number" + }, + "cinematic_fit": { + "type": "number" + }, + "source_quality": { + "type": "number" + }, + "mutation_potential": { + "type": "number" + }, + "continuity_compatibility": { + "type": "number" + }, + "clich\u00e9_risk": { + "type": "number" + } + } + }, + "version": { + "type": "string" + }, + "last_evolved": { + "type": "string" + }, + "evolution_baseline_confidence": { + "type": "number", + "minimum": 0, + "maximum": 1 + }, + "usage_history": { + "type": "array", + "items": { + "type": "object" + } + }, + "evolution_events": { + "type": "object", + "additionalProperties": { + "type": "object", + "properties": { + "evidence_hash": { + "type": "string" + }, + "uses": { + "type": "integer", + "minimum": 0 + }, + "total_score": { + "type": "number", + "minimum": 0 + }, + "success_signals": { + "type": "integer", + "minimum": 0 + }, + "failure_signals": { + "type": "integer", + "minimum": 0 + }, + "projects": { + "type": "array", + "items": { + "type": "string" + } + } + }, + "additionalProperties": false } } } diff --git a/skills/kubrick/scripts/evolve_from_use.py b/skills/kubrick/scripts/evolve_from_use.py index 328d661..a15161f 100755 --- a/skills/kubrick/scripts/evolve_from_use.py +++ b/skills/kubrick/scripts/evolve_from_use.py @@ -1,47 +1,34 @@ #!/usr/bin/env python3 """ -Kubrick Self-Evolution Engine +Kubrick Self-Evolution Engine (importable + CLI) -This script is self-contained. It works completely independently when the -skill is installed in Hermes (cp -R skills/kubrick ~/.hermes/skills/). +Usage (CLI): + kubrick-evolve --receipts-dir ... --outcomes-dir ... -It does not require continuity-forge or any optional extras. +Usage (import): + from kubrick_helpers.evolution import run_evolution + receipt = run_evolution(receipts_dir=..., outcomes_dir=...) -Evolves the symbolic corpus according to actual use. - -Usage: - python scripts/evolve_from_use.py --receipts-dir references/usage/receipts --outcomes-dir references/usage/outcomes - -It: -- Aggregates retrieval receipts -- Incorporates project outcomes (success/failure signals from Forge/revision/ledger) -- Adjusts confidence, adds usage_history to sidecars -- Updates corpus-index with observed performance -- Emits an evolution_receipt -- Never mutates without producing a receipt and provenance note - -Run periodically or after significant project activity. +Emits evolution receipt and mutates sidecars in place when paths provided. """ import argparse +import hashlib +import copy import json import os import glob -from datetime import datetime +from datetime import datetime, timezone from collections import defaultdict +from pathlib import Path +import tempfile +import uuid try: import yaml except ImportError: yaml = None -SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) -SKILL_ROOT = os.path.abspath(os.path.join(SCRIPT_DIR, "..")) -PATTERNS_DIR = os.path.join(SKILL_ROOT, "references", "patterns") -USAGE_RECEIPTS = os.path.join(SKILL_ROOT, "references", "usage", "receipts") -USAGE_OUTCOMES = os.path.join(SKILL_ROOT, "references", "usage", "outcomes") -EVOLUTION_DIR = os.path.join(SKILL_ROOT, "references", "evolution") - def load_json(path): with open(path, "r") as f: @@ -54,188 +41,383 @@ def save_json(path, data): json.dump(data, f, indent=2) -def find_sidecar(pattern_id): - for root, _, files in os.walk(PATTERNS_DIR): - for f in files: - if f == f"{pattern_id}.json": - return os.path.join(root, f) - return None - - -def load_all_sidecars(): +def load_all_sidecars(patterns_dir: str): sidecars = {} - for root, _, files in os.walk(PATTERNS_DIR): + for root, _, files in os.walk(patterns_dir): for fname in files: if fname.endswith(".json"): path = os.path.join(root, fname) + if Path(path).is_symlink(): + raise ValueError("sidecar symlink is not allowed") data = load_json(path) sidecars[data.get("pattern_id", fname)] = {"path": path, "data": data} return sidecars -def aggregate_usage(receipts_dir, outcomes_dir): - usage = defaultdict(lambda: {"uses": 0, "total_score": 0.0, "success_signals": 0, "failure_signals": 0, "projects": []}) - - for path in glob.glob(os.path.join(receipts_dir, "*.json")) + glob.glob(os.path.join(receipts_dir, "*.yaml")): - try: +def aggregate_usage(receipts_dir, outcomes_dir, consumed=None): + """Content-address events, not filenames; never silently ignore malformed evidence.""" + usage = defaultdict(lambda: {"events": {}}) + seen = dict(consumed or {}) + for directory, kind in ((receipts_dir, "retrieval"), (outcomes_dir, "outcome")): + for path in sorted( + glob.glob(os.path.join(directory, "*.json")) + + glob.glob(os.path.join(directory, "*.yaml")) + ): if path.endswith(".json"): - receipt = load_json(path) + record = load_json(path) else: if yaml is None: - continue - receipt = yaml.safe_load(open(path)) - rec = receipt.get("retrieval_receipt", receipt) - ranked = rec.get("ranked_patterns", []) - for item in ranked[:3]: - pid = item.get("pattern_id") - if pid: - usage[pid]["uses"] += 1 - usage[pid]["total_score"] += item.get("total_score", 0.5) - usage[pid]["projects"].append(rec.get("request_hash", "unknown")) - except Exception as e: - print(f"Warning: could not parse {path}: {e}") - - for path in glob.glob(os.path.join(outcomes_dir, "*.json")): - try: - outcome = load_json(path) - pid = outcome.get("pattern_id") - if pid: - signal = outcome.get("outcome", "neutral") - if signal == "success": - usage[pid]["success_signals"] += 1 - elif signal in ("failure", "debt", "collision", "revision_broken"): - usage[pid]["failure_signals"] += 1 - usage[pid]["projects"].append(outcome.get("project", "unknown")) - except Exception as e: - print(f"Warning: could not parse outcome {path}: {e}") - - return usage + raise ValueError("PyYAML is required to read YAML evidence") + with open(path) as stream: + record = yaml.safe_load(stream) + rec = record.get("retrieval_receipt", record) if kind == "retrieval" else record + identity = rec.get("event_id") or ( + rec.get("request_hash") + if kind == "retrieval" + else [rec.get("project"), rec.get("pattern_id")] + ) + if not identity or identity == [None, rec.get("pattern_id")]: + raise ValueError("evidence requires event_id or request_hash/project identity") + event_id = hashlib.sha256( + json.dumps([kind, identity], sort_keys=True).encode() + ).hexdigest() + content = {k: v for k, v in rec.items() if k not in ("timestamp", "metadata", "date")} + digest = hashlib.sha256( + json.dumps(content, sort_keys=True, allow_nan=False).encode() + ).hexdigest() + if event_id in seen and seen[event_id] != digest: + raise ValueError("conflicting evidence for stable identity") + seen[event_id] = digest + if kind == "retrieval": + for item in rec.get("ranked_patterns", [])[:3]: + pid = item["pattern_id"] + score = item.get("total_score", 0.5) + if type(score) not in (int, float): + raise ValueError("retrieval score must be numeric, not coerced") + if not 0 <= score <= 1: + raise ValueError("retrieval score must be finite and in [0, 1]") + usage[pid]["events"][event_id] = { + "evidence_hash": digest, + "uses": 1, + "total_score": score, + "projects": [rec.get("request_hash", "unknown")], + } + else: + pid = rec["pattern_id"] + signal = rec.get("outcome", "neutral") + if signal not in ( + "success", + "failure", + "debt", + "collision", + "revision_broken", + "neutral", + ): + raise ValueError("unknown outcome signal") + usage[pid]["events"][event_id] = { + "evidence_hash": digest, + "success_signals": int(signal == "success"), + "failure_signals": int( + signal in ("failure", "debt", "collision", "revision_broken") + ), + "projects": [rec.get("project", "unknown")], + } + return dict(usage) def evolve_sidecar(sidecar_info, usage_stats): - data = sidecar_info["data"] - pid = data["pattern_id"] - stats = usage_stats.get(pid, {}) - uses = stats.get("uses", 0) - if uses == 0: + data = copy.deepcopy(sidecar_info["data"]) + stats = usage_stats.get(data["pattern_id"], {}) + if not stats: return None - - avg_score = stats["total_score"] / uses if uses > 0 else 0.5 - success = stats.get("success_signals", 0) - failure = stats.get("failure_signals", 0) - total_signals = success + failure or 1 - success_rate = success / total_signals - - current_conf = data.get("confidence", 0.7) + # Legacy aggregate callers get exact-window dedupe. File-based callers use event IDs. + incoming = stats.get("events") + if incoming is None: + key = hashlib.sha256( + json.dumps(stats, sort_keys=True, allow_nan=False).encode() + ).hexdigest() + incoming = {key: stats} + events = data.get("evolution_events", {}) + for key in set(incoming) & set(events): + if incoming[key] != events[key]: + raise ValueError("conflicting consumed evidence for stable identity") + if not set(incoming) - set(events): + return None + events.update(incoming) + uses = sum(e.get("uses", 0) for e in events.values()) + if not uses: + return None # outcomes are applied once there is actual retrieval evidence + avg_score = sum(e.get("total_score", 0) for e in events.values()) / uses + success = sum(e.get("success_signals", 0) for e in events.values()) + failure = sum(e.get("failure_signals", 0) for e in events.values()) + success_rate = success / (success + failure) if success + failure else 0.5 + current = data.get("confidence", 0.7) + baseline = data.get("evolution_baseline_confidence", current) delta = 0.0 - if uses >= 3: - delta += (avg_score - 0.6) * 0.15 - delta += (success_rate - 0.5) * 0.25 - + delta += (avg_score - 0.6) * 0.15 + (success_rate - 0.5) * 0.25 if failure > success and uses >= 2: delta -= 0.1 - - new_conf = max(0.3, min(0.98, round(current_conf + delta, 4))) - - if "usage_history" not in data: - data["usage_history"] = [] - data["usage_history"].append({ - "date": datetime.utcnow().isoformat() + "Z", - "uses_in_window": uses, - "avg_retrieval_score": round(avg_score, 4), - "success_rate": round(success_rate, 4), - "confidence_before": current_conf, - "confidence_after": new_conf, - "delta": round(delta, 4), - "source_projects": list(set(stats.get("projects", [])))[:5] - }) - - data["confidence"] = new_conf - data["version"] = data.get("version", "0.6.0") - parts = data["version"].split(".") - if len(parts) >= 3: + confidence = max(0.3, min(0.98, round(baseline + delta, 4))) + now = datetime.now(timezone.utc).isoformat() + data.setdefault("usage_history", []).append( + { + "date": now, + "uses_in_window": uses, + "avg_retrieval_score": round(avg_score, 4), + "success_rate": round(success_rate, 4), + "confidence_before": current, + "confidence_after": confidence, + "delta": round(confidence - current, 4), + "source_projects": sorted({p for e in events.values() for p in e.get("projects", [])})[ + :5 + ], + } + ) + data["evolution_events"] = events + data["evolution_baseline_confidence"] = baseline + data["confidence"] = confidence + parts = data.get("version", "0.6.0").split(".") + if len(parts) == 3: parts[2] = str(int(parts[2]) + 1) - data["version"] = ".".join(parts) - - data["last_evolved"] = datetime.utcnow().isoformat() + "Z" + data["version"] = ".".join(parts) + data["last_evolved"] = now return data -def update_index_from_usage(index_path, usage_stats, sidecars): +def update_index_from_usage(index_path, usage_stats, sidecars, dry_run=False): if yaml is None: - return None - try: - with open(index_path, "r") as f: - index = yaml.safe_load(f) - except Exception: - return None - + raise ValueError("PyYAML is required for index evolution") + with open(index_path) as stream: + index = yaml.safe_load(stream) changed = False - for problem, data in index.get("by_dramatic_problem", {}).items(): - current_patterns = data.get("patterns", []) + for data in index.get("by_dramatic_problem", {}).values(): + current = data.get("patterns", []) scored = [] - for pid in current_patterns: + for pid in current: if pid in usage_stats and pid in sidecars: - score = usage_stats[pid].get("success_signals", 0) + (sidecars[pid]["data"].get("confidence", 0.7) * 2) - scored.append((pid, score)) - if scored: - scored.sort(key=lambda x: x[1], reverse=True) - new_order = [p[0] for p in scored] - if new_order != current_patterns: - data["patterns"] = new_order - changed = True - - if changed: - with open(index_path, "w") as f: - yaml.safe_dump(index, f, sort_keys=False) - return "corpus-index.yaml updated with performance-based ordering" - return None + info = sidecars[pid]["data"] + success = sum( + e.get("success_signals", 0) for e in info.get("evolution_events", {}).values() + ) + scored.append((pid, success + info.get("confidence", 0.7) * 2)) + scored.sort(key=lambda pair: pair[1], reverse=True) + ranked = iter(pid for pid, _ in scored) + scored_ids = {pid for pid, _ in scored} + order = [next(ranked) if pid in scored_ids else pid for pid in current] + if order != current: + data["patterns"] = order + changed = True + if not changed: + return None + text = yaml.safe_dump(index, sort_keys=False) + if not dry_run: + Path(index_path).write_text(text) + return text + + +def run_evolution( + receipts_dir, + outcomes_dir, + patterns_dir=None, + index_path=None, + dry_run=True, + expected_plan_hash=None, +): + """Plan by default. Explicit apply stages all files and rolls back caught I/O failures.""" + if patterns_dir is None or not Path(patterns_dir).is_dir(): + raise ValueError("patterns_dir must be an existing project-owned corpus directory") + for directory in (receipts_dir, outcomes_dir): + if not Path(directory).is_dir(): + raise ValueError("evidence directories must exist (empty is allowed)") + root = Path(patterns_dir).resolve() + lock = root / ".evolution.lock" + recovery_required = False + if not dry_run: + lock.mkdir() # single-writer guard; stale lock requires human recovery + + def input_hash(): + paths = list(root.rglob("*.json")) + for directory in (receipts_dir, outcomes_dir): + paths.extend(Path(directory).glob("*.json")) + paths.extend(Path(directory).glob("*.yaml")) + if index_path: + paths.append(Path(index_path)) + paths.append(Path(__file__)) + items = [ + (str(path.absolute()), hashlib.sha256(path.read_bytes()).hexdigest()) + for path in sorted(set(paths)) + ] + return hashlib.sha256(json.dumps(items).encode()).hexdigest() + + try: + plan_hash = input_hash() + if expected_plan_hash is not None and expected_plan_hash != plan_hash: + raise ValueError("stale evolution plan; review a new dry-run") + sidecars = load_all_sidecars(root) + consumed = {} + for info in sidecars.values(): + for event_id, event in info["data"].get("evolution_events", {}).items(): + digest = event.get("evidence_hash") + if digest is None: # legacy aggregate events have no evidence hash + continue + if event_id in consumed and consumed[event_id] != digest: + raise ValueError("conflicting consumed evidence for stable identity") + consumed[event_id] = digest + usage = aggregate_usage(receipts_dir, outcomes_dir, consumed) + if set(usage) - set(sidecars): + raise ValueError( + "unknown pattern in evidence: " + ", ".join(sorted(set(usage) - set(sidecars))) + ) + from jsonschema import Draft7Validator + + schema_path = Path(__file__).with_name("symbolic-narrative-pattern.schema.json") + if not schema_path.exists(): + schema_path = ( + Path(__file__).resolve().parents[1] + / "schemas/symbolic-narrative-pattern.schema.json" + ) + validator = Draft7Validator(load_json(schema_path)) + changes = {} + evolved = [] + for pid, info in sidecars.items(): + updated = evolve_sidecar(info, usage) + if updated: + validator.validate(updated) + info["data"] = updated + changes[Path(info["path"])] = json.dumps( + updated, indent=2, allow_nan=False + ).encode() + evolved.append(pid) + index_text = ( + update_index_from_usage(index_path, usage, sidecars, dry_run=True) + if index_path + else None + ) + if index_text is not None: + changes[Path(index_path)] = index_text.encode() + receipt = { + "timestamp": datetime.now(timezone.utc).isoformat(), + "patterns_evolved": evolved, + "dry_run": dry_run, + "plan_hash": plan_hash, + "index_update": index_text is not None, + "changes": [ + { + "path": str(path.resolve()), + "before_sha256": hashlib.sha256(path.read_bytes()).hexdigest(), + "after_sha256": hashlib.sha256(value).hexdigest(), + } + for path, value in changes.items() + ], + } + result = {"evolution_receipt": receipt} + if input_hash() != plan_hash: + raise ValueError("stale inputs changed during planning") + if changes and not dry_run: + receipt_path = root.parent / ("evolution-" + uuid.uuid4().hex + ".json") + receipt["receipt_path"] = str(receipt_path) + # Include exact prior contents for manual recovery after a process/host crash. + receipt["before_contents"] = {str(path.resolve()): path.read_text() for path in changes} + changes[receipt_path] = json.dumps(result, indent=2, allow_nan=False).encode() + originals = {path: path.read_bytes() if path.exists() else None for path in changes} + modes = {path: path.stat().st_mode & 0o7777 for path in changes if path.exists()} + staged = {} + applied = [] + try: + for path, value in changes.items(): + with tempfile.NamedTemporaryFile(dir=path.parent, delete=False) as stream: + staged[path] = Path(stream.name) + stream.write(value) + stream.flush() + if path in modes: + os.chmod(stream.name, modes[path]) + os.fsync(stream.fileno()) + if input_hash() != plan_hash: + raise ValueError("stale inputs changed during staging") + # Persist provenance before touching corpus; success is returned only after read-back. + order = [receipt_path] + [p for p in changes if p != receipt_path] + for path in order: + os.replace(staged[path], path) + applied.append(path) + for path, value in changes.items(): + if path.read_bytes() != value: + raise OSError("evolution read-back mismatch") + except Exception as apply_error: + rollback_errors = [] + for path in reversed(applied): + if path == receipt_path: + continue # retain provenance until every corpus restore is verified + backup = None + original = originals[path] + try: + if original is None: + path.unlink() + else: + try: + with tempfile.NamedTemporaryFile( + dir=path.parent, delete=False + ) as stream: + backup = Path(stream.name) + stream.write(original) + stream.flush() + os.chmod(stream.name, modes[path]) + os.replace(backup, path) + if path.read_bytes() != original: + raise OSError("rollback read-back mismatch") + finally: + if backup is not None: + backup.unlink(missing_ok=True) + except Exception as restore_error: # noqa: BLE001 - report all failures + rollback_errors.append(f"{path}: {restore_error!r}") + if not rollback_errors and receipt_path in applied: + try: + receipt_path.unlink() + except Exception as restore_error: # noqa: BLE001 - report all failures + rollback_errors.append(f"{receipt_path}: {restore_error!r}") + if rollback_errors: + recovery_required = True + raise RuntimeError( + f"evolution apply failed: {apply_error!r}; rollback failed: " + + "; ".join(rollback_errors) + + f"; manual recovery required using {receipt_path}; lock retained at {lock}" + ) from apply_error + raise + finally: + for temp in staged.values(): + temp.unlink(missing_ok=True) + return result + finally: + if not dry_run and not recovery_required: + lock.rmdir() def main(): parser = argparse.ArgumentParser() - parser.add_argument("--receipts-dir", default=USAGE_RECEIPTS) - parser.add_argument("--outcomes-dir", default=USAGE_OUTCOMES) + parser.add_argument("--receipts-dir", required=True) + parser.add_argument("--outcomes-dir", required=True) + parser.add_argument("--patterns-dir", required=True) + parser.add_argument("--index", help="Optional path to corpus-index.yaml to update") + parser.add_argument( + "--expected-plan-hash", help="Reject apply if reviewed dry-run inputs changed" + ) + mode = parser.add_mutually_exclusive_group() + mode.add_argument("--dry-run", action="store_true", help="Plan only (default); zero writes") + mode.add_argument( + "--apply", action="store_true", help="Explicitly apply reviewed local corpus changes" + ) args = parser.parse_args() - sidecars = load_all_sidecars() - usage = aggregate_usage(args.receipts_dir, args.outcomes_dir) - - evolved = [] - for pid, info in sidecars.items(): - updated = evolve_sidecar(info, usage) - if updated: - save_json(info["path"], updated) - evolved.append(pid) - - index_msg = update_index_from_usage( - os.path.join(SKILL_ROOT, "references", "corpus-index.yaml"), - usage, - sidecars + receipt = run_evolution( + receipts_dir=args.receipts_dir, + outcomes_dir=args.outcomes_dir, + patterns_dir=args.patterns_dir, + index_path=args.index, + dry_run=not args.apply, + expected_plan_hash=args.expected_plan_hash, ) - evolution_receipt = { - "evolution_receipt": { - "timestamp": datetime.utcnow().isoformat() + "Z", - "patterns_evolved": evolved, - "index_update": index_msg, - "usage_window": { - "receipts_scanned": len(glob.glob(os.path.join(args.receipts_dir, "*"))), - "outcomes_scanned": len(glob.glob(os.path.join(args.outcomes_dir, "*"))) - } - } - } - - os.makedirs(EVOLUTION_DIR, exist_ok=True) - receipt_path = os.path.join(EVOLUTION_DIR, f"evolution-{datetime.utcnow().strftime('%Y%m%d-%H%M')}.json") - save_json(receipt_path, evolution_receipt) - - print(json.dumps(evolution_receipt, indent=2)) - if evolved: - print(f"\nEvolved {len(evolved)} patterns. Sidecars updated in place with usage_history.") - else: - print("\nNo patterns met evolution thresholds in this window.") + print(json.dumps(receipt, indent=2)) if __name__ == "__main__": diff --git a/skills/kubrick/scripts/install_skill.py b/skills/kubrick/scripts/install_skill.py new file mode 100644 index 0000000..5ba5026 --- /dev/null +++ b/skills/kubrick/scripts/install_skill.py @@ -0,0 +1,93 @@ +#!/usr/bin/env python3 +"""Install only this embedded skill, with target-scoped backups outside discovery.""" + +import argparse +import hashlib +import json +import os +from pathlib import Path +import shutil +import tempfile +import uuid + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("category", nargs="?", choices=["creative", "categorized"]) + parser.add_argument("--target", type=Path) + mode = parser.add_mutually_exclusive_group() + mode.add_argument("--dry-run", action="store_true", help="Plan only (default)") + mode.add_argument("--apply", action="store_true", help="Install after reviewing target") + args = parser.parse_args() + source = Path(__file__).resolve().parents[1] + home = Path(os.environ.get("HERMES_HOME", str(Path.home() / ".hermes"))).resolve() + target = ( + args.target or home / "skills" / ("creative/kubrick" if args.category else "kubrick") + ).resolve() + if target == source or target in source.parents or source in target.parents: + parser.error("source and target must not overlap") + if target.exists() and not target.is_dir(): + parser.error("target must be a directory") + key = hashlib.sha256(str(target).encode()).hexdigest()[:20] + backup_root = home / "receipts/kubrick-backups" / key + if target == backup_root or target in backup_root.parents or backup_root in target.parents: + parser.error("target and backup root must not overlap") + # Fail before creating any directories if source integrity is broken. + for required in ("SKILL.md", "references/corpus-index.yaml", "scripts/evolve_from_use.py"): + if not (source / required).is_file(): + parser.error(f"missing skill source: {required}") + for path in (source / "scripts").glob("*.py"): + compile(path.read_text(), str(path), "exec") # syntax validation, no execution + plan = { + "source": str(source), + "target": str(target), + "backup_root": str(backup_root), + "dry_run": not args.apply, + "source_variant": "scrimshawlife-ctrl/continuity-forge:skills/kubrick", + } + if not args.apply: + print(json.dumps(plan)) + return + target.parent.mkdir(parents=True, exist_ok=True) + stage = Path(tempfile.mkdtemp(prefix=".kubrick-stage-", dir=target.parent)) + backup = None + activated = False + backup_complete = False + try: + shutil.copytree( + source, + stage, + dirs_exist_ok=True, + ignore=shutil.ignore_patterns("__pycache__", "*.pyc", ".git"), + ) + for required in ("SKILL.md", "references/corpus-index.yaml", "scripts/evolve_from_use.py"): + if (stage / required).read_bytes() != (source / required).read_bytes(): + raise OSError(f"staging verification failed: {required}") + if target.exists(): + backup_root.mkdir(parents=True, exist_ok=True) + backup = backup_root / uuid.uuid4().hex + # Copy backup first: the old install remains intact if copying fails. + shutil.copytree(target, backup) + backup_complete = True + shutil.rmtree(target) + os.replace(stage, target) + activated = True + if (target / "SKILL.md").read_bytes() != (source / "SKILL.md").read_bytes(): + raise OSError("installed skill verification failed") + except Exception: + if backup_complete and backup is not None and backup.is_dir(): + if target.exists(): + shutil.rmtree(target) + shutil.copytree(backup, target) + elif activated: + shutil.rmtree(target) + raise + finally: + if stage.exists(): + shutil.rmtree(stage) + plan.update(installed=True, backup=str(backup) if backup else None) + print(json.dumps(plan)) + + +if __name__ == "__main__": + main() diff --git a/skills/kubrick/scripts/retrieve_symbolic_patterns.py b/skills/kubrick/scripts/retrieve_symbolic_patterns.py index 030b36a..9092a1a 100755 --- a/skills/kubrick/scripts/retrieve_symbolic_patterns.py +++ b/skills/kubrick/scripts/retrieve_symbolic_patterns.py @@ -359,7 +359,7 @@ def log_receipt(receipt: Dict, log_dir: str = None) -> str: return path -def run_retrieval(brief: Dict) -> Dict: +def run_retrieval(brief: Dict, log_dir: str = None) -> Dict: patterns_db = load_all_patterns() ranked_patterns, rejected_patterns = rank_patterns(brief, patterns_db) @@ -374,13 +374,15 @@ def run_retrieval(brief: Dict) -> Dict: receipt = build_receipt( brief, ranked_patterns, rejected_patterns, esoteric_selection=esoteric_selection ) - receipt["retrieval_receipt"]["logged_to"] = log_receipt(receipt) + if log_dir is not None: + receipt["retrieval_receipt"]["logged_to"] = log_receipt(receipt, log_dir) return receipt def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("--brief", type=str, help="Path to YAML/JSON brief") + parser.add_argument("--log-dir", help="Explicit project-owned receipt directory (default: no logging)") args = parser.parse_args() if args.brief: @@ -389,7 +391,7 @@ def main() -> None: else: brief = yaml.safe_load(sys.stdin) or {} - receipt = run_retrieval(brief or {}) + receipt = run_retrieval(brief or {}, log_dir=args.log_dir) print(yaml.dump(receipt, sort_keys=False, default_flow_style=False)) if receipt["retrieval_receipt"]["status"] == "NOT_COMPUTABLE": sys.exit(1) diff --git a/skills/scriptwriting/SKILL.md b/skills/scriptwriting/SKILL.md index ca69fe8..fffe8b8 100644 --- a/skills/scriptwriting/SKILL.md +++ b/skills/scriptwriting/SKILL.md @@ -49,23 +49,16 @@ This skill **does not** own canon, run the deterministic kernel, or claim produc ## Prerequisites -- Continuity Forge installed and in PATH: - ```bash - pip install -e '.[dev]' # from continuity-forge repo - continuity-forge --help - ``` -- (Recommended) `continuity-forge-mcp` configured in Hermes for tool use. -- Optional: `humanizer` for final voice. - -Env for Forge (pass to any MCP/terminal calls): -```bash -export CF_STORE_ROOT="$HOME/.local/share/continuity-forge" -# export CF_PROVIDER=mock -``` +Creative use is standalone: load this directory's SKILL.md and the relevant references. +Read `references/standalone-procedure.md` before routing, drafting or scoring. +For optional Forge handoff only, use a Python 3.12+ full editable checkout and the +operator skill; setup and MCP registration are in repo `docs/SETUP.md` and +`docs/hermes/README.md`. Package installation does not install skill directories. +Do not configure credentials, provider calls or a live store merely to write a scene. ## Request Routing & Modes -Same as base (DEVELOP, DRAFT, DIAGNOSE, REVISE, POLISH, CONTINUITY, PRODUCTION, ADAPT). +Read `references/standalone-procedure.md` for mode routing, ordered phases, artifact selection and the cited diagnosis rubric; load the named local narrative references for the selected mode. When the goal is production use with Forge, prefer: - DEVELOP → handoff to Forge ingest/compile @@ -73,15 +66,15 @@ When the goal is production use with Forge, prefer: ## Core Operating Principles -(unchanged from base — Structure Before Pages, Drama Is Change Under Pressure, Behavior Before Explanation, Causality, Compression, Specificity, Approved Material Is Canon). +Read `references/standalone-procedure.md` for mode routing, ordered phases, artifact selection and the cited diagnosis rubric; load the named local narrative references for the selected mode. **Forge-specific addition**: Once material is ingested to Forge under a lease + mutation contract, the Forge ledger + IR becomes the source of truth. Chat memory or local artifacts are proposals only until committed via Forge. ## Core Workflow (Phases) -1–11. (Intake → Premise → Characters → World → Theme → Macrostructure → Sequences/Beats → Scene Engine → Dialogue/Prose → Continuity Ledger → Revision) — same as base. +Read `references/standalone-procedure.md` for mode routing, ordered phases, artifact selection and the cited diagnosis rubric; load the named local narrative references for the selected mode. -**12. Handoff to Continuity Forge (new critical phase)** +**12. Handoff to Continuity Forge (optional, explicit handoff)** After foundations or scene contracts are approved: @@ -97,12 +90,12 @@ See `references/continuity-forge-integration.md` for exact commands and mutation ## Anti-Slop Quality Gates -Same A–L as base. Additionally: -- **Gate M (Forge Bypass)**: Generating or committing narrative changes without updating the Forge ledger/IR. Always hand off material changes. +Read `references/standalone-procedure.md` for mode routing, ordered phases, artifact selection and the cited diagnosis rubric; load the named local narrative references for the selected mode. +- **Gate F-CANON (Forge Bypass)**: Generating or committing narrative changes without updating the Forge ledger/IR. Always hand off material changes. ## Output Selection Logic -Same as base. Preferred handoff artifacts: +Read `references/standalone-procedure.md` for mode routing, ordered phases, artifact selection and the cited diagnosis rubric; load the named local narrative references for the selected mode. - Structured project brief (matches Forge intake) - Scene contracts (feed `build_shot_contracts`) - Approved canon list (for mutation envelopes) @@ -112,7 +105,7 @@ Same as base. Preferred handoff artifacts: **Handoff rules**: - Creative development (this skill) produces **PROPOSED** or draft material. -- Forge ingestion makes it canonical. +- Only schema-validated deterministic output committed through Forge is canonical; narrative/model proposals are not promoted merely by attachment. - Use leases + full mutation contract (`actor_id`, `authorization_scope`, `idempotency_key`, `rationale`) for any write path. - Always surface Forge receipts/hashes in responses. - Claim policy: material generated here is for development; final identity lives in Forge. @@ -126,11 +119,11 @@ See the companion skill `hermes-continuity-forge` for operator details (leases, ## Format-Specific Routing -Same as base, with the addition that Forge's shot contracts and ledger are format-aware (features, pilots, shorts have different expectations for scene/shot density). +Read `references/standalone-procedure.md` for mode routing, ordered phases, artifact selection and the cited diagnosis rubric; load the named local narrative references for the selected mode. ## Diagnosis Rubric -Same 1-5 rubric. When Forge is in play, also score "Forge alignment" (does the output produce clean ingestable material?). +Use the anchored 1–5 rubric in `references/standalone-procedure.md`. When Forge is in play, also score "Forge alignment" (does the output produce clean ingestable material?). ## Validation Requirements @@ -159,7 +152,7 @@ Then hand off: "compile this to Continuity Forge and ingest under lease". - `references/format-specific-guidance.md` - `references/anti-slop-patterns.md` - **`references/continuity-forge-integration.md`** (new — handoff commands, MCP patterns) -- `schemas/` +- `references/schemas/` - `templates/` - `evals/` diff --git a/skills/scriptwriting/references/continuity-forge-integration.md b/skills/scriptwriting/references/continuity-forge-integration.md index d96439d..7addc8a 100644 --- a/skills/scriptwriting/references/continuity-forge-integration.md +++ b/skills/scriptwriting/references/continuity-forge-integration.md @@ -1,70 +1,62 @@ -# Continuity Forge Integration - -This skill is the **creative / structural** layer. Continuity Forge is the **deterministic kernel** that owns canonical state. - -## When to Handoff - -- After premise, characters, structure, or scene contracts are approved by the user. -- Before or instead of writing full prose pages when the goal is production use. -- On any material change to canon (new approved scenes, character traits that affect continuity, structure revisions). - -## Recommended Handoff Flow - -1. Produce clean artifacts from this skill (project brief, scene contracts, character bibles, approved canon list). -2. Ingest via Forge (prefer MCP or CLI with proper mutation envelope): - - `acquire_write_lease` - - `ingest_script` (or `compile_script` + ingest) - - `build_ledger` / `build_shot_contracts` -3. Capture receipt (hashes, document_key, shot IDs). -4. Reference Forge state in future work (use `get_project_status`, `inspect_scene`, etc. for grounding). - -## CLI Examples - -```bash -# Basic compile from fountain or structured text -continuity-forge compile path/to/outline.fountain --out out/ - -# Or from a scene contract / brief you produced here -continuity-forge ingest --document-key myfilm --source structured-outline.md +# Optional Continuity Forge handoff + +Creative work is standalone. For an explicitly requested canon handoff, load +`hermes-continuity-forge`, resolve the authorized document and actor, and confirm +MCP registration through repo `docs/hermes/README.md` (Python 3.12+; `docs/SETUP.md`). +No `ingest` subcommand exists in the Forge CLI. The CLI `compile` command is a +local deterministic parse/export, not a canonical project write. + +Use `terminal(command="continuity-forge compile path/to/script.fountain --out path/to/ir.json")` +only for a reviewed output file. Compile Fountain/FDX screenplay source, not a +JSON scene contract, outline prose or symbolic packet. Preserve source as immutable input. + +## MCP sequence + +This Python-shaped recipe names MCP tools, not a new client library. Substitute +`DOC`, `ACTOR`, `SOURCE` and `INTENT` from the approved request; `INTENT` is a fresh +unique key per logical write, reused only for a retry of that identical intent. +Never import server persistence into an operator client. The server constructs +command_schema_version through MutationEnvelope; it is not an ingest_script argument. + +```python +acquire_write_lease(document_key=DOC, holder=ACTOR, ttl_seconds=600) +try: + prior = get_project_status(document_key=DOC) + result = ingest_script( + source=SOURCE, + document_key=DOC, + actor_id=ACTOR, + authorization_scope="kernel:pipeline", + idempotency_key=INTENT, + rationale="Apply user-approved screenplay revision", + title="Reviewed script", + format="fountain", + revision="0.1.0", + expected_state_hash=prior["state_hash"] if prior else None, + ) + status = get_project_status(document_key=DOC) + assert status is not None + assert status["state_hash"] == result["project"]["state_hash"] +finally: + release_write_lease(document_key=DOC, holder=ACTOR) ``` -## MCP Tools (via companion `hermes-continuity-forge` skill) - -Typical tools you will call after creative work: -- `compile_script` -- `ingest_script` (with mutation contract) -- `build_ledger` -- `build_shot_contracts` -- `get_project_status` -- `audit_drift` - -Always include: -- `document_key` -- `actor_id` (e.g. hermes-scriptwriting-) -- Full mutation envelope when writing - -## Mutation Contract Requirements (when ingesting changes) - -From the operator skill: -- actor_id -- authorization_scope -- idempotency_key -- rationale -- expected_state_hash (when updating existing) - -This skill produces the *rationale* and *content*. The operator skill (or direct MCP call) supplies the envelope. - -## Boundaries +Acquire must succeed before entering the try/finally; never release someone else's +lease. Stop on conflicts, stale hashes, failed schema validation or incomplete +read-back. Keep the full result, document key, run ID, scene/shot IDs and hashes. +Future revisions use fresh project `state_hash`, not a pipeline shots hash. -- This skill may generate **PROPOSED** narrative material and scene contracts. -- Forge owns the canonical ledger, IR, and shot contracts. -- Never claim "this is now in the film" until you have a Forge receipt with committed status. -- For drift or contradictions discovered here: run local CONTINUITY pass, then cross-validate with Forge `audit_drift`. +## Narrative and symbolic packets -## Recommended Pairing +Briefs, character bibles, scene contracts, symbolic_architecture and cinematic_encoding +are **PROPOSED attachments for review**, not guaranteed compiler inputs or supported +IR fields. No automatic motif/geometry field mapping is promised. Check the actual +kernel models and round-trip returned schemas before claiming a constraint was +stored. The deterministic kernel alone owns approved identity, ledger, IR and shot +contracts; a narrative skill or model cannot promote attachments into canon. -Load both skills: -- `scriptwriting` for creative development, diagnosis, anti-slop, voice, structure. -- `hermes-continuity-forge` for leases, ingestion, proof, shot repair, approvals. +## Verification without live writes -See main repo `docs/hermes/README.md` and the companion skill for full operator rules. +From a full checkout, use `terminal(command="python -m pytest tests/test_authored_skill_recipes.py tests/contract/test_mcp.py -q")`. +Tests bind this exact recipe to real MCP signatures with a fresh in-memory runtime +and mock provider; no external service, credentials or live approval is involved. diff --git a/skills/scriptwriting/references/standalone-procedure.md b/skills/scriptwriting/references/standalone-procedure.md new file mode 100644 index 0000000..630046a --- /dev/null +++ b/skills/scriptwriting/references/standalone-procedure.md @@ -0,0 +1,34 @@ +# Standalone narrative procedure + +No Forge, unseen base skill, or Python package is needed for creative work. +Load only the reference relevant to the selected mode, then produce the smallest +artifact that answers the request. Do not invent approvals or imply draft notes are canon. + +| Mode | Input and completion artifact | Required local reference | +|---|---|---| +| DEVELOP | Audience, format, premise, constraints → logline, dramatic question, character want/obstacle/stakes, causal beats for approval | story-structure.md | +| DRAFT | Approved foundations → scenes with objective, opposition, turn, irreversible result | scene-engineering.md | +| DIAGNOSE | Supplied pages → cited defects, rubric, prioritized repairs; no unsolicited rewrite | anti-slop-patterns.md | +| REVISE | Locked material + approved changes → revised pages and change log | continuity.md | +| POLISH | Stable structure → behavior-led dialogue with distinct voices; preserve facts | character-and-dialogue.md | +| CONTINUITY | Supplied facts → contradictions with scene citations, unresolved questions | continuity.md | +| PRODUCTION | Approved scenes → proposed brief, scene contracts, constraints and provenance | scene-engineering.md | +| ADAPT | Source + target format → retained dramatic core, compression/expansion plan, sample | format-specific-guidance.md | + +1. Intake: identify format, audience, scope, supplied source, locks and requested output. Ask only for blocking omissions. +2. Develop: premise → character wants/opposition → world rules → thematic conflict → macrostructure → causal sequences. Get foundation approval before extensive pages. +3. Scene pass: each scene changes a relationship, available choice, knowledge or material state. State entry/exit facts and payoff obligations. +4. Dialogue pass: remove explanations already legible in behavior; differentiate voices by tactics and omissions. +5. Continuity pass: compare all revised facts and chronology against supplied locks; record unresolved contradictions rather than silently replacing them. +6. Diagnose: score causality, agency, stakes, scene turns, voice, economy and continuity on 1–5 (1 broken/absent; 2 major repair; 3 workable with repair; 4 strong; 5 consistently supported). Cite one specific scene/line per score. Use N/A where source is insufficient. Read anti-slop-patterns.md and report every applicable gate pass/fail. +7. Deliver the requested artifact, changed facts, remaining defects and next approval needed. A draft remains PROPOSED. If Forge handoff is requested, load continuity-forge-integration.md and the operator skill; compile screenplay source, not arbitrary JSON briefs. + +Shorts prioritize one pressure line and compressed turns; features support layered +payoffs; pilots establish a repeatable episode engine; podcasts require audible +clarity; video essays require argument/evidence. Use format-specific-guidance.md +for the chosen format, not claimed automatic format awareness in the compiler. + +## Verification +A small standalone smoke task: develop a one-scene short about returning a lost key. +Deliver a logline, character want/obstacle/stakes, scene entry/turn/exit contract, +and cited rubric. No tools, Forge writes or corpus evolution are implied. diff --git a/tests/test_authored_skill_recipes.py b/tests/test_authored_skill_recipes.py new file mode 100644 index 0000000..0366ec6 --- /dev/null +++ b/tests/test_authored_skill_recipes.py @@ -0,0 +1,91 @@ +"""Authored playbook regression checks; never contacts live services.""" + +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] + + +def test_writing_skills_have_standalone_procedure_and_real_ingest_route(): + for name in ("kubrick", "scriptwriting"): + skill = ROOT / "skills" / name + body = (skill / "SKILL.md").read_text().lower() + assert "same as base" not in body + assert "same 1-5 rubric" not in body + assert (skill / "references/standalone-procedure.md").is_file() + integration = (skill / "references/continuity-forge-integration.md").read_text() + assert "continuity-forge ingest " not in integration + assert "finally:" in integration + assert "expected_state_hash=" in integration + + +def test_operator_quick_path_separates_mcp_and_rest(): + body = (ROOT / "skills/hermes-continuity-forge/SKILL.md").read_text() + assert "MCP uses `source`; REST uses `text`" in body + workflows = (ROOT / "skills/hermes-continuity-forge/references/workflows.md").read_text() + assert 'authorization_scope="generation:repair"' in workflows + assert 'authorization_scope="generation:preview"' in workflows + assert "finally:" in workflows + + +def test_exact_mcp_recipes_execute_in_isolated_runtime(monkeypatch, tmp_path): + import ast + import inspect + import re + + from continuity_forge_auth import AuthService + from continuity_forge_harness import RunStore + from continuity_forge_mcp import server + from continuity_forge_operator import ProjectStore + from continuity_forge_providers import ArtifactStore, ProviderGateway + from continuity_forge_runtime.factory import RuntimeContext + + runs = RunStore() + runtime = RuntimeContext( + runs, + ProjectStore(run_store=runs), + ProviderGateway(), + AuthService(), + ArtifactStore(tmp_path / "artifacts"), + "test", + ) + monkeypatch.setattr(server, "_rt", lambda: runtime) + names = [ + "acquire_write_lease", + "release_write_lease", + "get_project_status", + "ingest_script", + "queue_generation", + "run_shot_repair_loop", + ] + scope = {name: getattr(server, name) for name in names} + scope.update( + DOC="recipe-test", + ACTOR="recipe-actor", + INTENT="ingest-1", + REPAIR_INTENT="repair-1", + SOURCE="INT. ROOM - DAY\n\nMara returns a key.\n", + ) + guide = ROOT / "skills/kubrick/references/continuity-forge-integration.md" + snippets = re.findall(r"```python\n(.*?)```", guide.read_text(), re.DOTALL) + workflows = ROOT / "skills/hermes-continuity-forge/references/workflows.md" + snippets += re.findall(r"```python\n(.*?)```", workflows.read_text(), re.DOTALL) + for index, snippet in enumerate(snippets): + for node in ast.walk(ast.parse(snippet)): + if ( + isinstance(node, ast.Call) + and isinstance(node.func, ast.Name) + and node.func.id in names + ): + inspect.signature(scope[node.func.id]).bind(**{k.arg: None for k in node.keywords}) + scope["INTENT"] = f"recipe-{index}" + if "queue_generation" in snippet: + project = runtime.project_store.get_project(scope["DOC"]) + scope["SHOT"] = project.shot_contracts["contracts"][0]["shot_id"] + # Execute repository-owned recipe text only, never fetched/user-supplied code. + exec(compile(snippet, str(guide), "exec"), scope) # noqa: S102 — trusted repo recipes only + assert scope["candidate"]["authority"] == "PROPOSED" + assert scope["repair"]["status"] == "accepted_proposed" + assert ( + server.acquire_write_lease("recipe-test", "different-actor")["holder"] == "different-actor" + ) + server.release_write_lease("recipe-test", "different-actor") diff --git a/tests/test_candidate_state_guard.py b/tests/test_candidate_state_guard.py new file mode 100644 index 0000000..8715f80 --- /dev/null +++ b/tests/test_candidate_state_guard.py @@ -0,0 +1,155 @@ +"""Candidate writes bind to project state without promoting canon.""" + +import pytest +from continuity_forge_auth import AuthService +from continuity_forge_harness import RunStore +from continuity_forge_mcp import server +from continuity_forge_operator import MutationEnvelope, OperatorError, ProjectStore +from continuity_forge_providers import ArtifactStore, ProviderGateway +from continuity_forge_runtime.factory import RuntimeContext + + +@pytest.fixture +def candidate_runtime(monkeypatch, tmp_path): + runs = RunStore() + store = ProjectStore(run_store=runs) + runtime = RuntimeContext( + runs, store, ProviderGateway(), AuthService(), ArtifactStore(tmp_path / "artifacts"), "test" + ) + monkeypatch.setattr(server, "_rt", lambda: runtime) + project, _ = store.ingest_script( + document_key="film", + title="Film", + text="INT. ROOM - DAY\n\nMara returns a key.\n", + revision="1", + format="fountain", + require_lease=False, + envelope=MutationEnvelope.from_parts( + actor_id="test", + authorization_scope="kernel:pipeline", + idempotency_key="initial", + rationale="test fixture", + ), + ) + return runtime, project, tmp_path / "artifacts" + + +@pytest.mark.parametrize("tool", [server.queue_generation, server.run_shot_repair_loop]) +def test_stale_candidate_hash_rejected_before_provider_or_persistence( + candidate_runtime, tool, monkeypatch +): + runtime, project, artifacts = candidate_runtime + + def forbidden(*args, **kwargs): + pytest.fail("provider called with stale project state") + + monkeypatch.setattr(runtime.gateway, "generate_for_shot", forbidden) + with pytest.raises(OperatorError, match="expected_state_hash conflict"): + tool( + "film", project.shot_contracts["contracts"][0]["shot_id"], expected_state_hash="0" * 64 + ) + assert not list(artifacts.rglob("*.json")) + + +@pytest.mark.parametrize("tool", [server.queue_generation, server.run_shot_repair_loop]) +def test_revision_between_snapshot_and_guard_is_rejected(candidate_runtime, tool, monkeypatch): + runtime, project, artifacts = candidate_runtime + get_project = runtime.project_store.get_project + generate = runtime.gateway.generate_for_shot + reads = 0 + + def racing_read(key): + nonlocal reads + snapshot = get_project(key) + reads += 1 + if reads == 1: + reingest(runtime, project) + return snapshot + + def forbidden(*args, **kwargs): + pytest.fail("generation preceded current-state validation") + + monkeypatch.setattr(runtime.project_store, "get_project", racing_read) + monkeypatch.setattr(runtime.gateway, "generate_for_shot", forbidden) + with pytest.raises(OperatorError, match="expected_state_hash conflict"): + tool("film", project.shot_contracts["contracts"][0]["shot_id"]) + assert not list(artifacts.rglob("*.json")) + monkeypatch.setattr(runtime.gateway, "generate_for_shot", generate) + + +@pytest.mark.parametrize("tool", [server.queue_generation, server.run_shot_repair_loop]) +@pytest.mark.parametrize("explicit", [False, True]) +def test_candidate_guard_holds_through_persistence(candidate_runtime, tool, monkeypatch, explicit): + from concurrent.futures import ThreadPoolExecutor + from threading import Event + + runtime, project, artifacts = candidate_runtime + attempted = Event() + completed = Event() + put = runtime.artifact_store.put + future = None + + def revise(): + attempted.set() + result = reingest(runtime, project) + completed.set() + return result + + with ThreadPoolExecutor(max_workers=1) as pool: + + def checked_put(candidate): + nonlocal future + future = pool.submit(revise) + assert attempted.wait(5) + assert not completed.wait(0.1), "revision committed before candidate persistence" + assert runtime.project_store.get_project("film").state_hash == project.state_hash + return put(candidate) + + monkeypatch.setattr(runtime.artifact_store, "put", checked_put) + kwargs = {"expected_state_hash": project.state_hash} if explicit else {} + result = tool("film", project.shot_contracts["contracts"][0]["shot_id"], **kwargs) + assert future is not None + revised, _ = future.result(timeout=5) + assert revised.state_hash != project.state_hash + assert result.get("authority", result.get("status")) in {"PROPOSED", "accepted_proposed"} + candidate = result.get("accepted_candidate", result) + assert ArtifactStore(artifacts).get(candidate["content_hash"]) == candidate + + +def reingest(runtime, project): + return runtime.project_store.ingest_script( + document_key="film", + title="Film", + text="INT. ROOM - DAY\n\nMara drops a key.\n", + revision="2", + format="fountain", + require_lease=False, + envelope=MutationEnvelope.from_parts( + actor_id="test", + authorization_scope="kernel:pipeline", + idempotency_key="revision", + rationale="concurrent revision", + expected_state_hash=project.state_hash, + ), + ) + + +@pytest.mark.parametrize("tool", [server.queue_generation, server.run_shot_repair_loop]) +def test_reentrant_revision_during_generation_cannot_persist_stale_candidate( + candidate_runtime, tool, monkeypatch +): + runtime, project, artifacts = candidate_runtime + generate = runtime.gateway.generate_for_shot + revised = False + + def revise_then_generate(*args, **kwargs): + nonlocal revised + if not revised: + revised = True + reingest(runtime, project) + return generate(*args, **kwargs) + + monkeypatch.setattr(runtime.gateway, "generate_for_shot", revise_then_generate) + with pytest.raises(OperatorError, match="expected_state_hash conflict"): + tool("film", project.shot_contracts["contracts"][0]["shot_id"]) + assert not list(artifacts.rglob("*.json")) diff --git a/tests/test_kubrick_esoteric.py b/tests/test_kubrick_esoteric.py index 5c0e6d5..bd949f1 100644 --- a/tests/test_kubrick_esoteric.py +++ b/tests/test_kubrick_esoteric.py @@ -6,7 +6,6 @@ from kubrick_helpers.esoteric import requested, select - INDEX = { "activation": {"explicit_terms": ["alchemy", "hidden symbolism"]}, "selection_policy": { @@ -41,9 +40,7 @@ "misuse_risks": ["random surrealism"], }, }, - "problem_routes": { - "identity_breakdown": ["alchemical_nigredo", "choronzon_drift"] - }, + "problem_routes": {"identity_breakdown": ["alchemical_nigredo", "choronzon_drift"]}, } @@ -66,7 +63,9 @@ def test_missing_observable_evidence_fails_closed(): ) assert result["status"] == "NOT_COMPUTABLE" assert result["selections"] == [] - assert any(item["reason"] == "observable evidence missing" for item in result["rejected_concepts"]) + assert any( + item["reason"] == "observable evidence missing" for item in result["rejected_concepts"] + ) def test_selection_is_bounded_proposed_and_evidence_grounded(): diff --git a/tests/test_kubrick_evolution_audit.py b/tests/test_kubrick_evolution_audit.py new file mode 100644 index 0000000..ca09394 --- /dev/null +++ b/tests/test_kubrick_evolution_audit.py @@ -0,0 +1,479 @@ +"""Regression coverage for embedded and packaged Kubrick evidence evolution.""" + +import importlib.util +import json +from pathlib import Path + +import pytest +from jsonschema import ValidationError +from yaml import YAMLError + +ROOT = Path(__file__).resolve().parents[1] +ENGINES = [ + ROOT / "skills/kubrick/scripts/evolve_from_use.py", + ROOT / "packages/kubrick_helpers/src/kubrick_helpers/evolution.py", +] + + +@pytest.fixture(params=ENGINES, ids=["embedded", "package"]) +def engine(request): + spec = importlib.util.spec_from_file_location("evolution_audit", request.param) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_repeated_evidence_does_not_inflate_confidence(engine): + info = {"data": {"pattern_id": "p", "confidence": 0.7}} + usage = { + "p": { + "uses": 3, + "total_score": 2.4, + "success_signals": 1, + "failure_signals": 0, + "projects": ["film"], + } + } + first = engine.evolve_sidecar(info, usage) + info["data"] = first + before = json.dumps(first, sort_keys=True) + assert engine.evolve_sidecar(info, usage) is None + assert json.dumps(info["data"], sort_keys=True) == before + + +def tree(root): + return { + str(p.relative_to(root)): (p.read_bytes(), p.stat().st_mtime_ns) + for p in root.rglob("*") + if p.is_file() + } + + +def corpus(root): + for name in ("patterns", "receipts", "outcomes"): + (root / name).mkdir() + (root / "patterns/p.json").write_text( + json.dumps( + { + "pattern_id": "p", + "confidence": 0.7, + "title": "Test pattern", + "domain": "cinematic", + "source_tier": "PRIMARY", + "dramatic_operations": [], + "cinematic_affordances": [], + } + ) + ) + for i in range(3): + (root / f"receipts/{i}.json").write_text( + json.dumps( + { + "request_hash": str(i), + "ranked_patterns": [{"pattern_id": "p", "total_score": 0.8}], + } + ) + ) + (root / "outcomes/ok.json").write_text( + json.dumps({"pattern_id": "p", "outcome": "success", "project": "film"}) + ) + (root / "index.yaml").write_text( + "by_dramatic_problem:\n test:\n patterns: [unused, p, missing]\n" + ) + + +def test_cli_dry_run_is_zero_write(engine, tmp_path): + import subprocess + import sys + + corpus(tmp_path) + before = tree(tmp_path) + result = subprocess.run( + [ + sys.executable, + engine.__file__, + "--receipts-dir", + str(tmp_path / "receipts"), + "--outcomes-dir", + str(tmp_path / "outcomes"), + "--patterns-dir", + str(tmp_path / "patterns"), + "--index", + str(tmp_path / "index.yaml"), + "--dry-run", + ], + capture_output=True, + text=True, + check=False, + ) + assert result.returncode == 0, result.stderr + assert json.loads(result.stdout)["evolution_receipt"]["dry_run"] is True + assert tree(tmp_path) == before + + +def test_duplicate_files_and_overlapping_windows_are_consumed_once(engine, tmp_path): + corpus(tmp_path) + usage = engine.aggregate_usage(tmp_path / "receipts", tmp_path / "outcomes") + info = {"data": {"pattern_id": "p", "confidence": 0.7}} + first = engine.evolve_sidecar(info, usage) + info["data"] = first + (tmp_path / "receipts/copy.json").write_bytes((tmp_path / "receipts/0.json").read_bytes()) + duplicated = engine.aggregate_usage(tmp_path / "receipts", tmp_path / "outcomes") + assert duplicated == usage + (tmp_path / "outcomes/new.json").write_text( + json.dumps({"pattern_id": "p", "outcome": "failure", "project": "other"}) + ) + second = engine.evolve_sidecar( + info, engine.aggregate_usage(tmp_path / "receipts", tmp_path / "outcomes") + ) + assert second["confidence"] == 0.73 # baseline .7 + retrieval .03, balanced outcomes + assert engine.evolve_sidecar({"data": second}, usage) is None # old subset is not new evidence + + +def test_apply_has_unique_receipt_and_repeat_is_noop(engine, tmp_path): + corpus(tmp_path) + kwargs = { + "receipts_dir": tmp_path / "receipts", + "outcomes_dir": tmp_path / "outcomes", + "patterns_dir": tmp_path / "patterns", + "index_path": tmp_path / "index.yaml", + "dry_run": False, + } + result = engine.run_evolution(**kwargs)["evolution_receipt"] + assert Path(result["receipt_path"]).is_file() + before = tree(tmp_path) + assert engine.run_evolution(**kwargs)["evolution_receipt"]["patterns_evolved"] == [] + assert tree(tmp_path) == before + + +def test_invalid_index_prevents_partial_apply(engine, tmp_path): + corpus(tmp_path) + (tmp_path / "index.yaml").write_text("by_dramatic_problem: [") + before = tree(tmp_path) + with pytest.raises((YAMLError, ValidationError)): + engine.run_evolution( + tmp_path / "receipts", + tmp_path / "outcomes", + tmp_path / "patterns", + tmp_path / "index.yaml", + dry_run=False, + ) + assert tree(tmp_path) == before + + +def test_same_identity_changed_evidence_is_rejected(engine, tmp_path): + corpus(tmp_path) + usage = engine.aggregate_usage(tmp_path / "receipts", tmp_path / "outcomes") + evolved = engine.evolve_sidecar({"data": {"pattern_id": "p", "confidence": 0.7}}, usage) + path = tmp_path / "receipts/0.json" + rec = json.loads(path.read_text()) + rec["ranked_patterns"][0]["total_score"] = 0.4 + path.write_text(json.dumps(rec)) + with pytest.raises(ValueError, match="conflicting"): + engine.evolve_sidecar( + {"data": evolved}, engine.aggregate_usage(tmp_path / "receipts", tmp_path / "outcomes") + ) + + +@pytest.mark.parametrize("recipients", [[{"pattern_id": "q", "total_score": 0.8}], []]) +def test_consumed_identity_cannot_move_to_disjoint_patterns(engine, tmp_path, recipients): + corpus(tmp_path) + q = json.loads((tmp_path / "patterns/p.json").read_text()) + q["pattern_id"] = "q" + (tmp_path / "patterns/q.json").write_text(json.dumps(q)) + args = (tmp_path / "receipts", tmp_path / "outcomes", tmp_path / "patterns") + engine.run_evolution(*args, dry_run=False) + path = tmp_path / "receipts/0.json" + record = json.loads(path.read_text()) + record["ranked_patterns"] = recipients + path.write_text(json.dumps(record)) + before = tree(tmp_path) + with pytest.raises(ValueError, match="conflicting.*stable identity"): + engine.run_evolution(*args, dry_run=False) + assert tree(tmp_path) == before + assert not (tmp_path / "patterns/.evolution.lock").exists() + + +def test_timestamp_change_is_not_new_evidence(engine, tmp_path): + corpus(tmp_path) + usage = engine.aggregate_usage(tmp_path / "receipts", tmp_path / "outcomes") + path = tmp_path / "receipts/0.json" + rec = json.loads(path.read_text()) + rec["timestamp"] = "new timestamp" + path.write_text(json.dumps(rec)) + assert engine.aggregate_usage(tmp_path / "receipts", tmp_path / "outcomes") == usage + + +def test_unknown_pattern_fails_closed(engine, tmp_path): + corpus(tmp_path) + (tmp_path / "patterns/p.json").unlink() + with pytest.raises(ValueError, match="unknown pattern"): + engine.run_evolution(tmp_path / "receipts", tmp_path / "outcomes", tmp_path / "patterns") + + +def test_schema_invalid_sidecar_is_rejected_before_apply(engine, tmp_path): + corpus(tmp_path) + path = tmp_path / "patterns/p.json" + data = json.loads(path.read_text()) + data["title"] = 42 + path.write_text(json.dumps(data)) + before = tree(tmp_path) + with pytest.raises((YAMLError, ValidationError)): + engine.run_evolution( + tmp_path / "receipts", tmp_path / "outcomes", tmp_path / "patterns", dry_run=False + ) + assert tree(tmp_path) == before + + +@pytest.mark.parametrize( + "bad", ["{", '{"request_hash":"bad","ranked_patterns":[{"pattern_id":"p","total_score":NaN}]}'] +) +def test_malformed_evidence_prevents_all_writes(engine, tmp_path, bad): + corpus(tmp_path) + (tmp_path / "receipts/bad.json").write_text(bad) + before = tree(tmp_path) + with pytest.raises((ValueError, TypeError)): + engine.run_evolution( + tmp_path / "receipts", tmp_path / "outcomes", tmp_path / "patterns", dry_run=False + ) + assert tree(tmp_path) == before + + +def test_changed_consumed_outcome_is_conflict(engine, tmp_path): + corpus(tmp_path) + first = engine.evolve_sidecar( + {"data": {"pattern_id": "p", "confidence": 0.7}}, + engine.aggregate_usage(tmp_path / "receipts", tmp_path / "outcomes"), + ) + path = tmp_path / "outcomes/ok.json" + data = json.loads(path.read_text()) + data["outcome"] = "failure" + path.write_text(json.dumps(data)) + with pytest.raises(ValueError, match="conflicting"): + engine.evolve_sidecar( + {"data": first}, engine.aggregate_usage(tmp_path / "receipts", tmp_path / "outcomes") + ) + + +def test_caught_replace_failure_rolls_back_every_file(engine, tmp_path, monkeypatch): + corpus(tmp_path) + q = json.loads((tmp_path / "patterns/p.json").read_text()) + q["pattern_id"] = "q" + (tmp_path / "patterns/q.json").write_text(json.dumps(q)) + (tmp_path / "receipts/q.json").write_text( + json.dumps( + {"request_hash": "q", "ranked_patterns": [{"pattern_id": "q", "total_score": 0.8}]} + ) + ) + before = {p: value[0] for p, value in tree(tmp_path).items()} + replace = engine.os.replace + calls = 0 + + def fail_third(source, target): + nonlocal calls + calls += 1 + if calls == 3: + raise OSError("injected replacement failure") + return replace(source, target) + + monkeypatch.setattr(engine.os, "replace", fail_third) + with pytest.raises(OSError, match="injected"): + engine.run_evolution( + tmp_path / "receipts", + tmp_path / "outcomes", + tmp_path / "patterns", + tmp_path / "index.yaml", + dry_run=False, + ) + assert {p: value[0] for p, value in tree(tmp_path).items()} == before + assert not (tmp_path / "patterns/.evolution.lock").exists() + + +@pytest.mark.parametrize("pattern_count", [2, 3]) +def test_double_fault_retains_recovery_guard_and_attempts_all_restores( + engine, tmp_path, monkeypatch, pattern_count +): + corpus(tmp_path) + template = json.loads((tmp_path / "patterns/p.json").read_text()) + for pid in ["q", "r"][: pattern_count - 1]: + (tmp_path / f"patterns/{pid}.json").write_text(json.dumps(dict(template, pattern_id=pid))) + (tmp_path / f"receipts/{pid}.json").write_text( + json.dumps( + { + "request_hash": pid, + "ranked_patterns": [{"pattern_id": pid, "total_score": 0.8}], + } + ) + ) + before = {p: p.read_bytes() for p in (tmp_path / "patterns").glob("*.json")} + replace = engine.os.replace + targets = [] + + def double_fault(source, target): + targets.append(Path(target)) + if len(targets) == pattern_count + 1: + raise OSError("injected apply failure") + if len(targets) == pattern_count + 2: + raise OSError("injected restore failure") + return replace(source, target) + + monkeypatch.setattr(engine.os, "replace", double_fault) + with pytest.raises(Exception) as caught: + engine.run_evolution( + tmp_path / "receipts", tmp_path / "outcomes", tmp_path / "patterns", dry_run=False + ) + message = str(caught.value) + assert "injected apply failure" in message + assert "injected restore failure" in message + receipt_path = targets[0] + assert str(receipt_path) in message + assert receipt_path.is_file() + lock = tmp_path / "patterns/.evolution.lock" + assert lock.is_dir() + receipt = json.loads(receipt_path.read_text())["evolution_receipt"] + assert receipt["before_contents"] == {str(p): b.decode() for p, b in before.items()} + activated = targets[1:pattern_count] + assert targets[pattern_count + 1 :] == list(reversed(activated)) + failed_restore = activated[-1] + for path, contents in before.items(): + assert (path.read_bytes() == contents) == (path != failed_restore) + assert set((tmp_path / "patterns").iterdir()) == set(before) | {lock} + with pytest.raises(FileExistsError): + engine.run_evolution( + tmp_path / "receipts", tmp_path / "outcomes", tmp_path / "patterns", dry_run=False + ) + + +@pytest.mark.skipif(__import__("os").name != "posix", reason="POSIX mode contract") +@pytest.mark.parametrize("rollback", [False, True]) +def test_evolution_preserves_staged_and_rollback_modes(engine, tmp_path, monkeypatch, rollback): + import stat + + corpus(tmp_path) + q = json.loads((tmp_path / "patterns/p.json").read_text()) + q["pattern_id"] = "q" + (tmp_path / "patterns/q.json").write_text(json.dumps(q)) + (tmp_path / "receipts/q.json").write_text( + json.dumps( + {"request_hash": "q", "ranked_patterns": [{"pattern_id": "q", "total_score": 0.1}]} + ) + ) + (tmp_path / "index.yaml").write_text("by_dramatic_problem:\n test:\n patterns: [q, p]\n") + paths = [tmp_path / "patterns/p.json", tmp_path / "index.yaml"] + modes = dict(zip(paths, [0o640, 0o664], strict=True)) + for path, mode in modes.items(): + path.chmod(mode) + before = {path: path.read_bytes() for path in paths} + replace = engine.os.replace + installed = [] + + def inspect_replace(source, target): + if target in modes: + assert stat.S_IMODE(Path(source).stat().st_mode) == modes[target] + installed.append(target) + return replace(source, target) + + monkeypatch.setattr(engine.os, "replace", inspect_replace) + read_bytes = Path.read_bytes + failed = False + + def fail_verification(path): + nonlocal failed + if rollback and len(installed) == 2 and not failed and path == paths[0]: + failed = True + raise OSError("injected read-back failure") + return read_bytes(path) + + monkeypatch.setattr(Path, "read_bytes", fail_verification) + args = (tmp_path / "receipts", tmp_path / "outcomes", tmp_path / "patterns", paths[1]) + if rollback: + with pytest.raises(OSError, match="injected read-back"): + engine.run_evolution(*args, dry_run=False) + assert {path: path.read_bytes() for path in paths} == before + assert installed == paths + list(reversed(paths)) + else: + engine.run_evolution(*args, dry_run=False) + assert installed == paths + assert {path: stat.S_IMODE(path.stat().st_mode) for path in paths} == modes + + +def test_embedded_package_code_and_schema_parity(): + assert ENGINES[0].read_bytes() == ENGINES[1].read_bytes() + assert ( + ROOT / "skills/kubrick/schemas/symbolic-narrative-pattern.schema.json" + ).read_bytes() == ENGINES[1].with_name("symbolic-narrative-pattern.schema.json").read_bytes() + + +@pytest.mark.parametrize("score", [True, "0.8"]) +def test_scores_do_not_coerce_invalid_types(engine, tmp_path, score): + corpus(tmp_path) + (tmp_path / "receipts/0.json").write_text( + json.dumps( + {"request_hash": "0", "ranked_patterns": [{"pattern_id": "p", "total_score": score}]} + ) + ) + with pytest.raises(ValueError): + engine.aggregate_usage(tmp_path / "receipts", tmp_path / "outcomes") + + +def test_sidecar_symlink_cannot_escape_write_root(engine, tmp_path): + corpus(tmp_path) + outside = tmp_path / "outside.json" + path = tmp_path / "patterns/p.json" + path.rename(outside) + path.symlink_to(outside) + before = tree(tmp_path) + with pytest.raises(ValueError, match="symlink"): + engine.run_evolution( + tmp_path / "receipts", tmp_path / "outcomes", tmp_path / "patterns", dry_run=False + ) + assert tree(tmp_path) == before + + +def test_reviewed_plan_rejects_changed_evidence(engine, tmp_path): + corpus(tmp_path) + args = ( + tmp_path / "receipts", + tmp_path / "outcomes", + tmp_path / "patterns", + tmp_path / "index.yaml", + ) + plan = engine.run_evolution(*args)["evolution_receipt"] + assert "plan_hash" in plan + (tmp_path / "receipts/0.json").write_text( + json.dumps( + {"request_hash": "new", "ranked_patterns": [{"pattern_id": "p", "total_score": 0.8}]} + ) + ) + before = tree(tmp_path) + with pytest.raises(ValueError, match="stale"): + engine.run_evolution(*args, dry_run=False, expected_plan_hash=plan["plan_hash"]) + assert tree(tmp_path) == before + + +def test_retrieval_does_not_log_to_installed_skill_by_default(tmp_path, monkeypatch): + spec = importlib.util.spec_from_file_location( + "retrieval_audit", ROOT / "skills/kubrick/scripts/retrieve_symbolic_patterns.py" + ) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + monkeypatch.setattr(module, "SKILL_ROOT", str(tmp_path)) + module.run_retrieval({"dramatic_problem": "identity"}) + assert not list(tmp_path.iterdir()) + + +def test_index_preserves_unused_and_missing_sidecar_entries(engine, tmp_path): + import yaml + + index = tmp_path / "index.yaml" + index.write_text("by_dramatic_problem:\n test:\n patterns: [unused, p, missing]\n") + engine.update_index_from_usage( + str(index), {"p": {"success_signals": 1}}, {"p": {"data": {"confidence": 0.8}}} + ) + assert sorted(yaml.safe_load(index.read_text())["by_dramatic_problem"]["test"]["patterns"]) == [ + "missing", + "p", + "unused", + ] diff --git a/tests/test_kubrick_install_audit.py b/tests/test_kubrick_install_audit.py new file mode 100644 index 0000000..8ddbd5a --- /dev/null +++ b/tests/test_kubrick_install_audit.py @@ -0,0 +1,89 @@ +"""Installer exercises run only in isolated HOME/profile directories.""" + +import json +import os +import subprocess +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +INSTALLER = ROOT / "skills/kubrick/install.sh" + + +def install(tmp_path, *args): + env = dict(os.environ, HOME=str(tmp_path / "home"), HERMES_HOME=str(tmp_path / "profile")) + return subprocess.run( + ["bash", str(INSTALLER), *args], + cwd=tmp_path, + env=env, + capture_output=True, + text=True, + check=False, + ) + + +def test_install_plan_is_zero_write_and_profile_aware(tmp_path): + result = install(tmp_path, "--dry-run") + assert result.returncode == 0, result.stderr + assert not list(tmp_path.iterdir()) + plan = json.loads(result.stdout) + assert plan["target"] == str(tmp_path / "profile/skills/kubrick") + assert plan["source"] == str(ROOT / "skills/kubrick") + + +def test_install_from_unrelated_cwd_preserves_unique_backups(tmp_path): + result = install(tmp_path, "--apply") + assert result.returncode == 0, result.stderr + target = tmp_path / "profile/skills/kubrick" + assert (target / "SKILL.md").is_file() + assert not (tmp_path / "home/.hermes").exists() + (target / "sentinel").write_text("user data") + assert install(tmp_path, "--apply").returncode == 0 + backups = list((tmp_path / "profile/receipts/kubrick-backups").rglob("sentinel")) + assert len(backups) == 1 + assert backups[0].read_text() == "user data" + assert not list((tmp_path / "profile/skills").glob("*.bak*")) + assert install(tmp_path, "--apply").returncode == 0 + assert backups[0].read_text() == "user data" + + +def test_install_rejects_profile_root_before_writes(tmp_path): + result = install(tmp_path, "--target", str(tmp_path / "profile"), "--dry-run") + assert result.returncode != 0 + assert not list(tmp_path.iterdir()) + + +def test_failed_backup_copy_does_not_replace_existing_install(tmp_path, monkeypatch): + import importlib.util + import sys + + spec = importlib.util.spec_from_file_location( + "install_audit", INSTALLER.parent / "scripts/install_skill.py" + ) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + profile = tmp_path / "profile" + target = profile / "skills/kubrick" + target.mkdir(parents=True) + (target / "sentinel").write_text("keep") + monkeypatch.setenv("HERMES_HOME", str(profile)) + monkeypatch.setattr(sys, "argv", ["installer", "--apply"]) + copytree = module.shutil.copytree + + def failing_copy(source, destination, *args, **kwargs): + if Path(source) == target: + Path(destination).mkdir(parents=True) + raise OSError("backup interrupted") + return copytree(source, destination, *args, **kwargs) + + monkeypatch.setattr(module.shutil, "copytree", failing_copy) + import pytest + + with pytest.raises(OSError, match="backup interrupted"): + module.main() + assert (target / "sentinel").read_text() == "keep" + + +def test_install_refuses_source_ancestry(tmp_path): + result = install(tmp_path, "--target", str(ROOT / "skills"), "--dry-run") + assert result.returncode != 0