From f91ea0625d795b402d7b3b5a946cfad33ac9c041 Mon Sep 17 00:00:00 2001 From: Sean Phan Date: Fri, 18 Sep 2026 18:33:16 +0000 Subject: [PATCH 01/12] feat: add evidence-refined ask and api-only ingest receipts --- CLAUDE.md | 22 + README.md | 89 ++- src/av/cli/ask.py | 14 +- src/av/cli/config_cmd.py | 17 + src/av/cli/ingest.py | 16 + src/av/core/config.py | 65 +- src/av/db/models.py | 1 + src/av/db/repository.py | 70 +++ src/av/pipeline/cascade.py | 17 +- src/av/pipeline/ingest.py | 116 +++- src/av/pipeline/transcript_sidecar.py | 131 ++++ src/av/providers/base.py | 8 + src/av/providers/openai.py | 178 ++++-- src/av/providers/usage.py | 98 +++ src/av/search/inspection.py | 435 +++++++++++++ src/av/search/rag.py | 261 +++++++- src/av/search/refine.py | 689 +++++++++++++++++++++ src/av/search/semantic.py | 7 + src/av/search/usage.py | 61 ++ tests/test_ask_refinement.py | 843 ++++++++++++++++++++++++++ tests/test_config.py | 21 +- tests/test_ingest_usage.py | 181 ++++++ tests/test_provider.py | 79 ++- tests/test_transcript_sidecar.py | 223 +++++++ 24 files changed, 3538 insertions(+), 104 deletions(-) create mode 100644 src/av/pipeline/transcript_sidecar.py create mode 100644 src/av/providers/usage.py create mode 100644 src/av/search/inspection.py create mode 100644 src/av/search/refine.py create mode 100644 src/av/search/usage.py create mode 100644 tests/test_ask_refinement.py create mode 100644 tests/test_ingest_usage.py create mode 100644 tests/test_transcript_sidecar.py diff --git a/CLAUDE.md b/CLAUDE.md index ece838b..95e04ae 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -111,6 +111,7 @@ av ingest /folder/ # batch directory av ingest video.mp4 --force # re-ingest av ingest "https://youtu.be/..." # YouTube URL av ingest video.mp4 --dense-vision # structured dense captions +av ingest video.mp4 --transcript-json transcript.json # validated external transcript; skips built-in ASR ``` ```json {"status": "complete", "video_id": "uuid", "filename": "video.mp4", "duration_sec": 120.5, "artifacts_count": 42, "elapsed_sec": 15.3} @@ -127,6 +128,12 @@ On partial failure: `{"status": "complete_with_warnings", ..., "warnings": ["Tra {"answer": "...", "citations": [{"video_id": "uuid", "start_sec": 120.0, "source_type": "transcript", "text": "...", "score": 0.91}], "confidence": 0.85} ``` +With `AV_TYPESAFE_API_KEY` (or `TYPESAFE_API_KEY`) configured, `av ask` +automatically runs Jev source relevance, configurable bounded scene grouping, answer +synthesis, and a separate answer-support Noul. `--no-refine` preserves legacy RAG +for one request. Refined responses retain answer/citations/confidence and add route, +evidence, refinement, warning, inspected-window, and stage-usage metadata. + ### `av list` / `av info ` / `av transcript ` / `av export` / `av open ` See `av --help` for details. @@ -180,12 +187,27 @@ When a capability is unavailable (e.g. Anthropic has no Whisper), the pipeline s |----------|---------|-------------| | `AV_API_KEY` | (none) | API key (overrides config.json) | | `AV_API_BASE_URL` | `https://api.openai.com/v1` | API endpoint | +| `AV_API_TIMEOUT_SEC` | `120` | Per-attempt API timeout | +| `AV_API_MAX_RETRIES` | `1` | Explicit retry count; SDK retries stay disabled | +| `AV_ALLOW_OAUTH_FALLBACK` | `false` | Allow reading local OAuth caches only when explicitly enabled | +| `AV_ALLOW_CODEX_FALLBACK` | `false` | Allow spawning Codex for caption fallback only when explicitly enabled | | `AV_PROVIDER` | (none) | Provider name | | `AV_TRANSCRIBE_MODEL` | `whisper-1` | Transcription model | | `AV_VISION_MODEL` | `gpt-4-1` | Vision/caption model | | `AV_EMBED_MODEL` | `text-embedding-3-small` | Embedding model | | `AV_CHAT_MODEL` | `gpt-4-1` | Chat/RAG model | | `AV_DB_PATH` | `~/.config/av/av.db` | Database location | +| `AV_TYPESAFE_API_KEY` | (none) | Jev/System One key; enables ask refinement by default | +| `AV_TYPESAFE_ENDPOINT` | `https://api.typesafe.ai/v1/systemone` | Explicit System One endpoint | +| `AV_TYPESAFE_MODEL` | `jev-latest` | System One model | +| `AV_REFINE_RELEVANCE_MIN` | `0.5` | Minimum source-relevance Noul probability | +| `AV_REFINE_SUPPORT_MIN` | `0.5` | Minimum answer-support Noul probability | +| `AV_REFINE_MAX_SCENES` | `8` | Maximum merged scenes sent to synthesis | +| `AV_REFINE_BATCH_SIZE` | `10` | Sources per System One relevance request | +| `AV_REFINE_CONTEXT_EVENTS` | `3` | Maximum temporal events on each side of a hit | +| `AV_STRONG_VISION_API_BASE_URL` | (none) | Explicit OpenAI-compatible sampled-frame endpoint | +| `AV_STRONG_VISION_API_KEY` | (none) | Key for the stronger sampled-frame endpoint | +| `AV_STRONG_VISION_MODEL` | (none) | Explicit stronger vision model; never selected implicitly | | `DEEPSEEK_API_KEY` | (none) | Key for a self-hosted DeepSeek-V4.1-Flash server | | `SGLANG_API_KEY` | (none) | Alias for the same, matching SGLang's own naming | diff --git a/README.md b/README.md index eb31bf9..586df67 100644 --- a/README.md +++ b/README.md @@ -30,6 +30,63 @@ av search "person with red bag" av ask "what happened at 2:30?" ``` +### Refined `av ask` (optional) + +Configure a TypeSafe System One key to make Jev refinement automatic for `av ask`: + +```bash +export AV_TYPESAFE_API_KEY="..." # TYPESAFE_API_KEY is also accepted +av ask "when does the person enter the room?" +av ask "when does the person enter the room?" --no-refine # legacy RAG for this call +``` + +The refined path uses Jev only for typed decisions: it filters source relevance, +uses a configurable bounded temporal neighborhood to form local scenes, merges +overlapping same-video scenes, and ranks them by `relevance probability × retrieval +score`. A separate Jev Noul checks whether the answer is supported; source relevance +is not treated as answer correctness. + +If Jev is unavailable, `av ask` visibly warns and falls back to raw retrieval. If +Jev validly rejects every hit, the result is empty instead of restoring rejected +hits. Refined JSON includes `route`, `evidence_status`, `refinement`, `warnings`, +`inspected_windows`, and per-stage token usage when providers report it. Unknown +usage remains `null`. + +Missing, malformed, or out-of-range System One probabilities are treated as a +refinement failure: `av` reports the fallback and does not invent a confidence. + +See the [AV ask refinement cookbook](cookbook/README.md) for a reproducible recipe, +offline cost arithmetic, and receipt provenance. + +FTS5 remains the first retrieval stage. An unscoped query with zero FTS matches does +not scan the video archive or invoke sampled-frame inspection. + +### API-only reproducible ingest + +Use an explicit OpenAI-compatible endpoint/key and keep both fallback flags disabled: + +```bash +export AV_API_BASE_URL="https://your-provider.example/v1" +export AV_API_KEY="..." +export AV_VISION_MODEL="your-cheap-vision-model" +export AV_ALLOW_OAUTH_FALLBACK="false" +export AV_ALLOW_CODEX_FALLBACK="false" + +av ingest video.mp4 --dense-vision --max-frames 120 --no-embed +``` + +To import a timestamped transcript produced by a separate public ASR script, pass a +validated sidecar instead of running built-in ASR: + +```bash +av ingest video.mp4 --dense-vision --transcript-json transcript.json +``` + +The sidecar contains `segments` with `start_sec`, `end_sec`, and `text`, plus optional +`model` and public `provenance`. It is validated against the probed video duration +before database changes or API calls. AV records the import as local work with zero +provider requests; external ASR usage or cost is not attributed to this ingest. + ### Surveillance Detection ```bash @@ -242,17 +299,47 @@ Env vars always override config.json: ```bash export AV_API_KEY="sk-..." export AV_API_BASE_URL="https://api.openai.com/v1" # or any OpenAI-compatible endpoint +export AV_API_TIMEOUT_SEC="120" +export AV_API_MAX_RETRIES="1" +export AV_ALLOW_OAUTH_FALLBACK="false" # never read local auth caches unless explicitly enabled +export AV_ALLOW_CODEX_FALLBACK="false" # never spawn Codex unless explicitly enabled export AV_TRANSCRIBE_MODEL="whisper" export AV_VISION_MODEL="gpt-4-1" export AV_EMBED_MODEL="text-embedding-3-small" export AV_CHAT_MODEL="gpt-4-1" +# Optional Jev/System One refinement (automatic when a key is present) +export AV_TYPESAFE_API_KEY="..." # TYPESAFE_API_KEY also works +export AV_TYPESAFE_ENDPOINT="https://api.typesafe.ai/v1/systemone" +export AV_TYPESAFE_MODEL="jev-latest" +export AV_REFINE_RELEVANCE_MIN="0.5" +export AV_REFINE_SUPPORT_MIN="0.5" +export AV_REFINE_MAX_SCENES="8" +export AV_REFINE_BATCH_SIZE="10" +export AV_REFINE_CONTEXT_EVENTS="3" + +# Optional bounded sampled-frame fallback after an unsupported answer +export AV_STRONG_VISION_API_BASE_URL="https://your-explicit-endpoint.example/v1" +export AV_STRONG_VISION_API_KEY="..." +export AV_STRONG_VISION_MODEL="your-explicit-model" +export AV_INSPECTION_MAX_WINDOWS="2" +export AV_INSPECTION_MAX_SECONDS="120" +export AV_INSPECTION_MAX_FRAMES="12" +export AV_INSPECTION_MAX_ATTEMPTS="1" +export AV_INSPECTION_DENSE_PASS="false" + # Self-hosted DeepSeek-V4.1-Flash via SGLang export AV_PROVIDER="deepseek" export AV_API_BASE_URL="http://your-sglang-host:30000/v1" export DEEPSEEK_API_KEY="..." # only if your server requires one ``` +API requests use the configured timeout and explicit retry limit. Ingestion JSON +includes `stage_usage` for transcription, captioning, caption summarization, and +embeddings, plus the effective frame/request settings. Request failures are counted; +token totals become `null` with a completeness flag when any provider omits usage. +No dollar total is inferred. + ## Requirements - Python 3.11+ @@ -267,7 +354,7 @@ export DEEPSEEK_API_KEY="..." # only if your server requires one | `av config show` | Show current configuration | | `av ingest ` | Ingest video file(s) into the index | | `av search ` | Full-text + semantic search | -| `av ask ` | RAG Q&A with citations | +| `av ask ` | RAG Q&A; automatically refines with Jev when configured | | `av list` | List all indexed videos | | `av info ` | Detailed video metadata | | `av transcript ` | Output transcript (VTT/SRT/text) | diff --git a/src/av/cli/ask.py b/src/av/cli/ask.py index f2d0e72..eedd678 100644 --- a/src/av/cli/ask.py +++ b/src/av/cli/ask.py @@ -20,13 +20,25 @@ def ask_cmd( video_id: str = typer.Option(None, "--video-id", "-v", help="Restrict to specific video"), top_k: int = typer.Option(DEFAULT_TOP_K, "--top-k", "-k", help="Context chunks"), db: str = typer.Option(None, "--db", help="Database path override"), + no_refine: bool = typer.Option( + False, + "--no-refine", + help="Skip configured Jev relevance and scene refinement", + ), ) -> None: """Ask a question about indexed video content (RAG).""" config = get_config(db_path=Path(db) if db else None) repo = Repository(config.db_path) try: - result = ask(question, repo, config, video_id=video_id, top_k=top_k) + result = ask( + question, + repo, + config, + video_id=video_id, + top_k=top_k, + refine=not no_refine, + ) output_json(result) except Exception as e: error(str(e)) diff --git a/src/av/cli/config_cmd.py b/src/av/cli/config_cmd.py index 489ff1d..f64451d 100644 --- a/src/av/cli/config_cmd.py +++ b/src/av/cli/config_cmd.py @@ -57,10 +57,26 @@ def config_show() -> None: "api_base_url": config.api_base_url, "api_key": "***" if config.api_key else "(not set)", "openai_api_key": "***" if config.openai_api_key else "(not set)", + "api_timeout_sec": config.api_timeout_sec, + "api_max_retries": config.api_max_retries, + "allow_oauth_fallback": config.allow_oauth_fallback, + "allow_codex_fallback": config.allow_codex_fallback, "transcribe_model": config.transcribe_model or "(disabled)", "vision_model": config.vision_model, "embed_model": config.embed_model or "(disabled)", "chat_model": config.chat_model, + "typesafe_api_key": "***" if config.typesafe_api_key else "(not set)", + "typesafe_endpoint": config.typesafe_endpoint, + "typesafe_model": config.typesafe_model, + "refine_enabled": config.refine_enabled, + "refine_relevance_min": config.refine_relevance_min, + "refine_support_min": config.refine_support_min, + "refine_max_scenes": config.refine_max_scenes, + "refine_batch_size": config.refine_batch_size, + "refine_context_events": config.refine_context_events, + "strong_vision_api_base_url": config.strong_vision_api_base_url or "(not set)", + "strong_vision_api_key": "***" if config.strong_vision_api_key else "(not set)", + "strong_vision_model": config.strong_vision_model or "(not set)", "db_path": str(config.db_path), }) @@ -96,6 +112,7 @@ def config_setup() -> None: # Resolve API key based on provider if provider_key == "openai-oauth": + config_data["allow_oauth_fallback"] = True token = _openclaw_oauth_token() or _codex_oauth_token() if token: _console.print(" [green]✓[/green] Found Codex OAuth token") diff --git a/src/av/cli/ingest.py b/src/av/cli/ingest.py index cc34566..87612f6 100644 --- a/src/av/cli/ingest.py +++ b/src/av/cli/ingest.py @@ -34,10 +34,15 @@ def ingest( help="Topic for captioning: security, traffic, warehouse, retail, meeting, general, or a custom description", ), frame_captions: bool = typer.Option(False, "--frame-captions", help="Legacy per-frame captioning (old --captions behavior)"), + transcript_json: Path | None = typer.Option( + None, "--transcript-json", exists=True, file_okay=True, dir_okay=False, + readable=True, help="Import timestamped transcript JSON for one video instead of running ASR", + ), db: str = typer.Option(None, "--db", help="Database path override"), ) -> None: """Ingest video file(s) into the av index.""" videos = [] + transcript_path = transcript_json.expanduser().resolve() if transcript_json else None if is_url(path): progress("Detected URL input. Downloading video first...") @@ -49,12 +54,19 @@ def ingest( videos = [downloaded] else: target = Path(path).expanduser().resolve() + if transcript_path is not None and target.is_dir(): + error("--transcript-json requires one concrete video file or URL, not a directory.") + raise typer.Exit(1) videos = discover_videos(target) if not videos: error(f"No video files found at: {path}") raise typer.Exit(1) + if transcript_path is not None and len(videos) != 1: + error("--transcript-json requires exactly one video.") + raise typer.Exit(1) + config = get_config(db_path=Path(db) if db else None) repo = Repository(config.db_path) @@ -79,6 +91,7 @@ def ingest( dense_output_dir=Path(dense_output_dir).expanduser().resolve() if dense_output_dir else None, topic=topic, frame_captions=frame_captions, + transcript_json=transcript_path, ) results.append(result) except AVError as e: @@ -91,3 +104,6 @@ def ingest( output_json(results[0]) else: output_json({"results": results, "total": len(results)}) + + if transcript_path is not None and any(result.get("status") == "error" for result in results): + raise typer.Exit(1) diff --git a/src/av/core/config.py b/src/av/core/config.py index 0a0fcef..b0b9407 100644 --- a/src/av/core/config.py +++ b/src/av/core/config.py @@ -53,6 +53,10 @@ class AVConfig(BaseSettings): api_base_url: str = Field(default="https://api.openai.com/v1") api_key: str = Field(default="") openai_api_key: str = Field(default="") + api_timeout_sec: float = Field(default=120.0, gt=0) + api_max_retries: int = Field(default=1, ge=0, le=3) + allow_oauth_fallback: bool = Field(default=False) + allow_codex_fallback: bool = Field(default=False) # Models transcribe_model: str = Field(default=DEFAULT_TRANSCRIBE_MODEL) @@ -60,6 +64,30 @@ class AVConfig(BaseSettings): embed_model: str = Field(default=DEFAULT_EMBED_MODEL) chat_model: str = Field(default=DEFAULT_CHAT_MODEL) + # Optional System One query refinement. A credential enables refinement by + # default; callers can still opt out per request. + typesafe_api_key: str = Field(default="") + typesafe_endpoint: str = Field(default="https://api.typesafe.ai/v1/systemone") + typesafe_model: str = Field(default="jev-latest") + typesafe_timeout_sec: float = Field(default=30.0, gt=0) + typesafe_max_retries: int = Field(default=1, ge=0, le=3) + refine_enabled: bool = Field(default=True) + refine_relevance_min: float = Field(default=0.5, ge=0, le=1) + refine_support_min: float = Field(default=0.5, ge=0, le=1) + refine_max_scenes: int = Field(default=8, ge=1, le=50) + refine_batch_size: int = Field(default=10, ge=1, le=50) + refine_context_events: int = Field(default=3, ge=0, le=12) + + # Optional stronger sampled-frame inspection after an unsupported answer. + strong_vision_api_base_url: str = Field(default="") + strong_vision_api_key: str = Field(default="") + strong_vision_model: str = Field(default="") + inspection_max_windows: int = Field(default=2, ge=0, le=8) + inspection_max_seconds: float = Field(default=120.0, ge=0) + inspection_max_frames: int = Field(default=12, ge=0, le=128) + inspection_max_attempts: int = Field(default=1, ge=1, le=2) + inspection_dense_pass: bool = Field(default=False) + # Database db_path: Path = Field(default=DEFAULT_DB_PATH) @@ -80,13 +108,37 @@ def get_config(db_path: Path | None = None) -> AVConfig: "api_base_url", "api_key", "openai_api_key", + "api_timeout_sec", + "api_max_retries", + "allow_oauth_fallback", + "allow_codex_fallback", "transcribe_model", "vision_model", "embed_model", "chat_model", + "typesafe_api_key", + "typesafe_endpoint", + "typesafe_model", + "typesafe_timeout_sec", + "typesafe_max_retries", + "refine_enabled", + "refine_relevance_min", + "refine_support_min", + "refine_max_scenes", + "refine_batch_size", + "refine_context_events", + "strong_vision_api_base_url", + "strong_vision_api_key", + "strong_vision_model", + "inspection_max_windows", + "inspection_max_seconds", + "inspection_max_frames", + "inspection_max_attempts", + "inspection_dense_pass", ): env_name = f"AV_{key.upper()}" - if key in file_data and env_name not in os.environ: + alias_is_set = key == "typesafe_api_key" and "TYPESAFE_API_KEY" in os.environ + if key in file_data and env_name not in os.environ and not alias_is_set: init_kwargs[key] = file_data[key] config = AVConfig(**init_kwargs) @@ -95,6 +147,11 @@ def get_config(db_path: Path | None = None) -> AVConfig: if not config.openai_api_key: config.openai_api_key = os.environ.get("OPENAI_API_KEY", "") + if "AV_TYPESAFE_API_KEY" not in os.environ: + config.typesafe_api_key = os.environ.get("TYPESAFE_API_KEY", config.typesafe_api_key) + if "AV_TYPESAFE_MODEL" not in os.environ: + config.typesafe_model = os.environ.get("TYPESAFE_DEFAULT_MODEL", config.typesafe_model) + if db_path is not None: config.db_path = db_path return config @@ -115,7 +172,7 @@ def get_openai_config(config: AVConfig) -> AVConfig | None: key = (config.openai_api_key or "").strip() # Fallback: Codex OAuth tokens (same mechanism as _resolve_api_key in openai.py) - if not key: + if not key and config.allow_oauth_fallback: from av.providers.openai import _codex_oauth_token, _openclaw_oauth_token key = _openclaw_oauth_token() or _codex_oauth_token() or "" @@ -126,6 +183,10 @@ def get_openai_config(config: AVConfig) -> AVConfig | None: provider="openai", api_base_url="https://api.openai.com/v1", api_key=key, + api_timeout_sec=config.api_timeout_sec, + api_max_retries=config.api_max_retries, + allow_oauth_fallback=False, + allow_codex_fallback=config.allow_codex_fallback, transcribe_model="whisper-1", embed_model="text-embedding-3-small", vision_model=config.vision_model, diff --git a/src/av/db/models.py b/src/av/db/models.py index 2c2e0d5..9e8e31a 100644 --- a/src/av/db/models.py +++ b/src/av/db/models.py @@ -49,6 +49,7 @@ class SearchResult(BaseModel): video_id: str filename: str timestamp_sec: float + end_sec: float | None = None timestamp_formatted: str source_type: str text: str diff --git a/src/av/db/repository.py b/src/av/db/repository.py index 2180d0a..d2395a7 100644 --- a/src/av/db/repository.py +++ b/src/av/db/repository.py @@ -303,6 +303,7 @@ def search_fts( video_id=r["video_id"], filename=r["filename"], timestamp_sec=r["start_sec"], + end_sec=r["end_sec"], timestamp_formatted=_fmt_timestamp(r["start_sec"]), source_type=r["type"], text=r["text"], @@ -311,6 +312,75 @@ def search_fts( ) return results + def get_refinement_window( + self, + video_id: str, + center_sec: float, + *, + before: int = 6, + after: int = 6, + ) -> list[ArtifactRecord]: + """Return a bounded run of query-time artifacts around one hit. + + Long whole-video summaries and reports are deliberately excluded: they are + useful retrieval hints, but they are not fixed-time scene chunks and must not + cause scene refinement to widen into an archive scan. + """ + allowed = ("transcript", "caption", "scene", "dense_caption") + marks = ",".join("?" for _ in allowed) + prev_rows = self.conn.execute( + f"""SELECT * FROM artifacts + WHERE video_id = ? AND type IN ({marks}) AND text != '' AND start_sec <= ? + ORDER BY start_sec DESC, id DESC + LIMIT ?""", + (video_id, *allowed, center_sec, max(before + 1, 1)), + ).fetchall() + next_rows = self.conn.execute( + f"""SELECT * FROM artifacts + WHERE video_id = ? AND type IN ({marks}) AND text != '' AND start_sec > ? + ORDER BY start_sec ASC, id ASC + LIMIT ?""", + (video_id, *allowed, center_sec, max(after, 0)), + ).fetchall() + rows = list(reversed(prev_rows)) + list(next_rows) + seen: set[str] = set() + out: list[ArtifactRecord] = [] + for row in rows: + if row["id"] in seen: + continue + seen.add(row["id"]) + out.append(ArtifactRecord(**dict(row))) + return out + + def get_artifacts_overlapping( + self, + video_id: str, + start_sec: float, + end_sec: float, + *, + limit: int = 50, + ) -> list[ArtifactRecord]: + """Return bounded scene evidence from one video only.""" + rows = self.conn.execute( + """SELECT * FROM artifacts + WHERE video_id = ? + AND type IN ('transcript', 'caption', 'scene', 'dense_caption') + AND text != '' + AND ( + (COALESCE(end_sec, start_sec) > start_sec + AND start_sec < ? + AND end_sec > ?) + OR + (COALESCE(end_sec, start_sec) <= start_sec + AND start_sec >= ? + AND start_sec <= ?) + ) + ORDER BY start_sec, id + LIMIT ?""", + (video_id, end_sec, start_sec, start_sec, end_sec, limit), + ).fetchall() + return [ArtifactRecord(**dict(row)) for row in rows] + def get_embeddings_for_artifacts(self, artifact_ids: list[str]) -> dict[str, list[float]]: """Load embedding vectors for a set of artifact IDs.""" if not artifact_ids: diff --git a/src/av/pipeline/cascade.py b/src/av/pipeline/cascade.py index 2f225e0..71b284f 100644 --- a/src/av/pipeline/cascade.py +++ b/src/av/pipeline/cascade.py @@ -128,6 +128,7 @@ def run_cascade( topic: str = "general", chunk_duration_sec: int = DEFAULT_CHUNK_DURATION_SEC, frames_per_chunk: int = DEFAULT_FRAMES_PER_CHUNK, + usage_receipt: dict | None = None, ) -> tuple[list[ArtifactRecord], list[ArtifactRecord], list[ArtifactRecord]]: """Run the three-layer captioning cascade. @@ -138,10 +139,12 @@ def run_cascade( layer1: list[ArtifactRecord] = [] layer2: list[ArtifactRecord] = [] temp_dirs: list[Path] = [] + captioner: OpenAICaptioner | None = None + llm: OpenAILLM | None = None + num_chunks = max(1, math.ceil(duration_sec / chunk_duration_sec)) try: # Compute chunks - num_chunks = max(1, math.ceil(duration_sec / chunk_duration_sec)) captioner = OpenAICaptioner(config) meta_base = { @@ -249,5 +252,17 @@ def run_cascade( finally: for d in temp_dirs: shutil.rmtree(d, ignore_errors=True) + if usage_receipt is not None: + usage_receipt.update({ + "caption": captioner.usage.snapshot() if captioner else None, + "caption_summary": llm.usage.snapshot() if llm else None, + "settings": { + "concurrency": 1, + "chunk_duration_sec": chunk_duration_sec, + "frames_per_chunk": frames_per_chunk, + "chunk_count": num_chunks, + "maximum_frame_requests": num_chunks * frames_per_chunk, + }, + }) return layer0, layer1, layer2 diff --git a/src/av/pipeline/ingest.py b/src/av/pipeline/ingest.py index 625462b..2e83dc9 100644 --- a/src/av/pipeline/ingest.py +++ b/src/av/pipeline/ingest.py @@ -20,7 +20,9 @@ from av.pipeline.chunker import chunk_artifacts from av.pipeline.dense_caption import export_dense_outputs, render_dense_prompt from av.pipeline.ffmpeg import extract_audio, extract_frames, get_video_info +from av.pipeline.transcript_sidecar import load_transcript_sidecar from av.providers.openai import OpenAICaptioner, OpenAIEmbedder, OpenAITranscriber +from av.providers.usage import ProviderUsage from av.utils.hashing import file_hash from av.utils.principles import load_principles @@ -106,6 +108,7 @@ def ingest_video( dense_output_dir: Path | None = None, topic: str = "general", frame_captions: bool = False, + transcript_json: Path | None = None, ) -> dict: """Ingest a single video file. Returns JSON-serializable result dict.""" start_time = time.time() @@ -116,10 +119,22 @@ def ingest_video( if not path.is_file(): raise IngestError(f"Not a file: {path}") - # Step 1: Hash + idempotency check + # Step 1: Hash and validate all local inputs before mutating the database. fhash = file_hash(path) + print(f" Probing: {path.name}...", file=sys.stderr) + meta = get_video_info(path) + transcript_sidecar = ( + load_transcript_sidecar(transcript_json, duration_sec=meta.duration_sec) + if transcript_json is not None + else None + ) + existing = repo.get_video_by_hash(fhash) if existing and not force: + if transcript_sidecar is not None: + raise IngestError( + "Video is already ingested; use --force to apply an explicit transcript sidecar." + ) print(f" Skipping (already ingested): {path.name}", file=sys.stderr) return { "status": "skipped", @@ -130,11 +145,6 @@ def ingest_video( if existing and force: print(f" Re-ingesting (--force): {path.name}", file=sys.stderr) - repo.delete_video(existing.id) - - # Step 2: Extract metadata - print(f" Probing: {path.name}...", file=sys.stderr) - meta = get_video_info(path) video_id = str(uuid.uuid4()) ingest_config = { @@ -145,6 +155,16 @@ def ingest_video( "force": force, "dense_vision": dense_vision, "principles_path": str(principles_path) if principles_path else None, + "transcript_sidecar": transcript_sidecar is not None, + "provider": config.provider, + "transcribe_model": config.transcribe_model, + "vision_model": config.vision_model, + "embed_model": config.embed_model, + "chat_model": config.chat_model, + "api_timeout_sec": config.api_timeout_sec, + "api_max_retries": config.api_max_retries, + "allow_oauth_fallback": config.allow_oauth_fallback, + "allow_codex_fallback": config.allow_codex_fallback, } video = VideoRecord( @@ -172,8 +192,12 @@ def ingest_video( "would_caption": captions, "would_embed": not no_embed, "would_dense_vision": dense_vision, + "would_import_transcript_sidecar": transcript_sidecar is not None, } + if existing and force: + repo.delete_video(existing.id) + # Step 3: Long video warning if meta.duration_sec > LONG_VIDEO_WARN_MINUTES * 60: mins = meta.duration_sec / 60 @@ -189,6 +213,14 @@ def ingest_video( audio_path: Path | None = None frames_dir: Path | None = None warnings: list[str] = [] + transcription_usage = ProviderUsage() + caption_usage = ProviderUsage() + caption_summary_usage = ProviderUsage() + embedding_usage = ProviderUsage() + cascade_settings: dict = {} + frame_caption_frame_count = 0 + dense_frame_count = 0 + transcription_source = "sidecar" if transcript_sidecar is not None else "disabled" try: transcript_artifacts: list[ArtifactRecord] = [] @@ -202,7 +234,30 @@ def ingest_video( # Step 4: Extract audio and transcribe (best-effort) transcribe_cfg = oai_config or config can_transcribe = bool(transcribe_cfg.transcribe_model) - if can_transcribe: + if transcript_sidecar is not None: + sidecar_meta: dict = {} + if transcript_sidecar.model is not None: + sidecar_meta["model"] = transcript_sidecar.model + if transcript_sidecar.provenance is not None: + sidecar_meta["provenance"] = transcript_sidecar.provenance + transcript_artifacts = [ + ArtifactRecord( + id=str(uuid.uuid4()), + video_id=video_id, + type="transcript", + start_sec=segment.start_sec, + end_sec=segment.end_sec, + text=segment.text, + meta_json=json.dumps(sidecar_meta) if sidecar_meta else None, + ) + for segment in transcript_sidecar.segments + ] + if transcript_artifacts: + repo.insert_artifacts_batch(transcript_artifacts) + artifacts_count += len(transcript_artifacts) + elif can_transcribe: + transcription_source = "provider" + transcriber: OpenAITranscriber | None = None try: print(f" Extracting audio...", file=sys.stderr) audio_path = extract_audio(path) @@ -240,6 +295,9 @@ def ingest_video( msg = f"Transcription skipped: {e}" warnings.append(msg) print(f" Warning: {msg}", file=sys.stderr) + finally: + if transcriber is not None: + transcription_usage.merge(transcriber.usage.snapshot()) else: msg = f"Transcription disabled (provider={config.provider or 'current'})." warnings.append(msg) @@ -248,6 +306,7 @@ def ingest_video( # Step 5: Captions — cascade (default) or legacy per-frame cascade_artifacts: list[ArtifactRecord] = [] if captions: + cascade_receipt: dict = {} try: l0, l1, l2 = run_cascade( path, @@ -255,6 +314,7 @@ def ingest_video( config, meta.duration_sec, topic=topic, + usage_receipt=cascade_receipt, ) cascade_artifacts = l0 + l1 + l2 if cascade_artifacts: @@ -266,8 +326,13 @@ def ingest_video( msg = f"Cascade captioning skipped: {e}" warnings.append(msg) print(f" Warning: {msg}", file=sys.stderr) + finally: + caption_usage.merge(cascade_receipt.get("caption")) + caption_summary_usage.merge(cascade_receipt.get("caption_summary")) + cascade_settings = cascade_receipt.get("settings") or {} if frame_captions: + captioner: OpenAICaptioner | None = None try: print( f" Frame captioning enabled (fps={fps_sample}, max={max_frames}). This uses the vision API.", @@ -277,6 +342,7 @@ def ingest_video( frames_dir = frames[0][0].parent if frames else None if frames: + frame_caption_frame_count = len(frames) captioner = OpenAICaptioner(config) frame_paths = [f[0] for f in frames] timestamps = [f[1] for f in frames] @@ -303,9 +369,13 @@ def ingest_video( msg = f"Frame captioning skipped: {e}" warnings.append(msg) print(f" Warning: {msg}", file=sys.stderr) + finally: + if captioner is not None: + caption_usage.merge(captioner.usage.snapshot()) # Step 5b: Dense visual captions (best-effort) if dense_vision: + captioner = None try: if not frames_dir: frames = extract_frames(path, fps_sample=fps_sample, max_frames=max_frames) @@ -321,6 +391,7 @@ def ingest_video( prompt = render_dense_prompt(template_path, principles) if frames: + dense_frame_count = len(frames) print(f" Dense vision captioning enabled for {len(frames)} frame(s)...", file=sys.stderr) captioner = OpenAICaptioner(config) frame_paths = [f[0] for f in frames] @@ -363,6 +434,9 @@ def ingest_video( msg = f"Dense vision skipped: {e}" warnings.append(msg) print(f" Warning: {msg}", file=sys.stderr) + finally: + if captioner is not None: + caption_usage.merge(captioner.usage.snapshot()) # Step 6: Embeddings (best-effort) embed_cfg = oai_config or config @@ -373,6 +447,7 @@ def ingest_video( print(f" {msg}", file=sys.stderr) if not no_embed and can_embed: + embedder: OpenAIEmbedder | None = None try: all_artifacts = transcript_artifacts + caption_artifacts + dense_artifacts if all_artifacts: @@ -395,6 +470,9 @@ def ingest_video( msg = f"Embeddings skipped: {e}" warnings.append(msg) print(f" Warning: {msg}", file=sys.stderr) + finally: + if embedder is not None: + embedding_usage.merge(embedder.usage.snapshot()) # Step 7: Mark complete repo.update_video_status(video_id, "complete") @@ -416,6 +494,30 @@ def ingest_video( "duration_sec": round(meta.duration_sec, 2), "artifacts_count": artifacts_count, "elapsed_sec": round(elapsed, 2), + "stage_usage": { + "transcription": transcription_usage.snapshot(), + "caption": caption_usage.snapshot(), + "caption_summary": caption_summary_usage.snapshot(), + "embedding": embedding_usage.snapshot(), + }, + "ingest_settings": { + "provider": config.provider or "openai-compatible", + "transcribe_model": config.transcribe_model or None, + "vision_model": config.vision_model or None, + "embed_model": config.embed_model or None, + "chat_model": config.chat_model or None, + "api_timeout_sec": config.api_timeout_sec, + "api_max_retries": config.api_max_retries, + "allow_oauth_fallback": config.allow_oauth_fallback, + "allow_codex_fallback": config.allow_codex_fallback, + "transcription_source": transcription_source, + "caption_concurrency": 1, + "fps_sample": fps_sample, + "max_frames": max_frames, + "frame_caption_frames": frame_caption_frame_count, + "dense_caption_frames": dense_frame_count, + "cascade": cascade_settings, + }, } if warnings: out["warnings"] = warnings diff --git a/src/av/pipeline/transcript_sidecar.py b/src/av/pipeline/transcript_sidecar.py new file mode 100644 index 0000000..89578ef --- /dev/null +++ b/src/av/pipeline/transcript_sidecar.py @@ -0,0 +1,131 @@ +"""Validate timestamped transcript sidecars before importing transcript artifacts.""" + +from __future__ import annotations + +from dataclasses import dataclass +import json +import math +from pathlib import Path + +from av.core.exceptions import IngestError + + +class TranscriptSidecarError(IngestError): + """An explicitly supplied transcript sidecar is invalid.""" + + +@dataclass(frozen=True) +class TranscriptSidecarSegment: + start_sec: float + end_sec: float + text: str + + +@dataclass(frozen=True) +class TranscriptSidecar: + segments: tuple[TranscriptSidecarSegment, ...] + model: str | None = None + provenance: dict | None = None + + +def _finite_number(value: object) -> bool: + if isinstance(value, bool) or not isinstance(value, (int, float)): + return False + try: + return math.isfinite(value) + except OverflowError: + return False + + +def _reject_constant(_value: str) -> None: + raise TranscriptSidecarError("Transcript sidecar must contain finite JSON numbers.") + + +def _unique_object(pairs: list[tuple[str, object]]) -> dict: + result = {} + for key, value in pairs: + if key in result: + raise TranscriptSidecarError("Transcript sidecar contains duplicate object keys.") + result[key] = value + return result + + +def _validate_json_numbers(value: object) -> None: + if isinstance(value, float) and not math.isfinite(value): + raise TranscriptSidecarError("Transcript sidecar must contain finite JSON numbers.") + if isinstance(value, dict): + for child in value.values(): + _validate_json_numbers(child) + elif isinstance(value, list): + for child in value: + _validate_json_numbers(child) + + +def load_transcript_sidecar(path: Path, *, duration_sec: float) -> TranscriptSidecar: + """Read one video's transcript without rewriting its times, order, or text. + + Accept a segment list, or an object with required segments and optional + model and provenance fields. Empty lists represent no speech. Segment + timestamps are in seconds relative to the media start and must fall entirely + within its probed duration. No other fields are accepted for import. + """ + if not _finite_number(duration_sec) or duration_sec <= 0: + raise TranscriptSidecarError("Video duration must be a positive finite number.") + + try: + raw = json.loads( + path.read_text(encoding="utf-8"), + parse_constant=_reject_constant, + object_pairs_hook=_unique_object, + ) + except (OSError, UnicodeError): + raise TranscriptSidecarError("Could not read transcript sidecar as UTF-8 JSON.") from None + except (ValueError, RecursionError): + raise TranscriptSidecarError("Transcript sidecar is not valid JSON.") from None + + try: + _validate_json_numbers(raw) + except RecursionError: + raise TranscriptSidecarError("Transcript sidecar JSON is nested too deeply.") from None + + model = None + provenance = None + if isinstance(raw, dict): + if "segments" not in raw or set(raw) - {"segments", "model", "provenance"}: + raise TranscriptSidecarError( + "Transcript sidecar object requires segments and permits only model and provenance." + ) + segments = raw["segments"] + if "model" in raw: + model = raw["model"] + if not isinstance(model, str) or not model.strip(): + raise TranscriptSidecarError("Transcript sidecar model must be a nonempty string.") + if "provenance" in raw: + provenance = raw["provenance"] + if not isinstance(provenance, dict): + raise TranscriptSidecarError("Transcript sidecar provenance must be a JSON object.") + elif isinstance(raw, list): + segments = raw + else: + raise TranscriptSidecarError("Transcript sidecar root must be a segment list or an object.") + + if not isinstance(segments, list): + raise TranscriptSidecarError("Transcript sidecar segments must be a list.") + + validated = [] + for index, segment in enumerate(segments): + label = f"Transcript segment {index + 1}" + if not isinstance(segment, dict) or set(segment) != {"start_sec", "end_sec", "text"}: + raise TranscriptSidecarError(f"{label} requires only start_sec, end_sec, and text.") + start = segment["start_sec"] + end = segment["end_sec"] + if not _finite_number(start) or not _finite_number(end): + raise TranscriptSidecarError(f"{label} timestamps must be finite numbers, excluding booleans.") + if not 0 <= start < end <= duration_sec: + raise TranscriptSidecarError(f"{label} must satisfy 0 <= start_sec < end_sec <= video duration.") + text = segment["text"] + if not isinstance(text, str) or not text.strip(): + raise TranscriptSidecarError(f"{label} text must be a nonempty string.") + validated.append(TranscriptSidecarSegment(start_sec=start, end_sec=end, text=text)) + + return TranscriptSidecar(segments=tuple(validated), model=model, provenance=provenance) diff --git a/src/av/providers/base.py b/src/av/providers/base.py index 2e0ff7a..309e0ac 100644 --- a/src/av/providers/base.py +++ b/src/av/providers/base.py @@ -29,6 +29,14 @@ class ChunkCaption: frame_count: int +@dataclass +class CompletionResult: + text: str + input_tokens: int | None = None + output_tokens: int | None = None + usage: dict | None = None + + class TranscriberProvider(ABC): @abstractmethod def transcribe(self, audio_path: Path) -> list[TranscriptSegment]: diff --git a/src/av/providers/openai.py b/src/av/providers/openai.py index f65028f..c31e9da 100644 --- a/src/av/providers/openai.py +++ b/src/av/providers/openai.py @@ -6,6 +6,7 @@ import json import subprocess import sys +import time from pathlib import Path from openai import OpenAI @@ -16,11 +17,13 @@ Caption, CaptionerProvider, ChunkCaption, + CompletionResult, EmbedderProvider, LLMProvider, TranscriberProvider, TranscriptSegment, ) +from av.providers.usage import ProviderUsage def _token_from_codex_auth_file(auth_path: Path) -> str | None: @@ -97,9 +100,10 @@ def _resolve_api_key(config: AVConfig) -> str: return deepseek_key(config) # Prefer OpenClaw auth-profile OAuth (often fresher), then Codex CLI cache. - oauth = _openclaw_oauth_token() or _codex_oauth_token() - if oauth: - return oauth + if config.allow_oauth_fallback: + oauth = _openclaw_oauth_token() or _codex_oauth_token() + if oauth: + return oauth # Final fallback maintains previous explicit failure behavior. return configured or "no-key" @@ -109,6 +113,9 @@ def _client(config: AVConfig) -> OpenAI: kwargs: dict = { "base_url": config.api_base_url, "api_key": _resolve_api_key(config), + "timeout": config.api_timeout_sec, + # Retries are explicit below so receipts count every attempted request. + "max_retries": 0, } # Anthropic's OpenAI-compatible endpoint requires anthropic-version header if config.provider == "anthropic": @@ -116,6 +123,25 @@ def _client(config: AVConfig) -> OpenAI: return OpenAI(**kwargs) +def _call_with_retries(config: AVConfig, usage: ProviderUsage, operation): + last_error: Exception | None = None + for attempt in range(config.api_max_retries + 1): + try: + response = operation() + except Exception as exc: + usage.record_failure() + last_error = exc + if attempt < config.api_max_retries: + time.sleep(min(0.25 * (2**attempt), 1.0)) + continue + raise + usage.record_success(getattr(response, "usage", None)) + return response + if last_error is not None: + raise last_error + raise RuntimeError("provider request did not run") + + def _extract_codex_answer(stdout: str) -> str: lines = [ln.rstrip() for ln in stdout.splitlines()] @@ -171,16 +197,20 @@ class OpenAITranscriber(TranscriberProvider): def __init__(self, config: AVConfig): self.config = config self.client = _client(config) + self.usage = ProviderUsage() def transcribe(self, audio_path: Path) -> list[TranscriptSegment]: try: - with open(audio_path, "rb") as f: - response = self.client.audio.transcriptions.create( - model=self.config.transcribe_model, - file=f, - response_format="verbose_json", - timestamp_granularities=["segment"], - ) + def operation(): + with open(audio_path, "rb") as f: + return self.client.audio.transcriptions.create( + model=self.config.transcribe_model, + file=f, + response_format="verbose_json", + timestamp_granularities=["segment"], + ) + + response = _call_with_retries(self.config, self.usage, operation) except Exception as e: raise APIError(f"Transcription failed: {e}", provider="openai") from e @@ -203,6 +233,7 @@ class OpenAICaptioner(CaptionerProvider): def __init__(self, config: AVConfig): self.config = config self.client = _client(config) + self.usage = ProviderUsage() def caption_frames( self, frame_paths: list[Path], timestamps: list[float], prompt: str | None = None @@ -216,30 +247,36 @@ def caption_frames( ext = fp.suffix.lstrip(".").lower() if ext == "jpg": ext = "jpeg" - response = self.client.chat.completions.create( - model=self.config.vision_model, - messages=[ - { - "role": "user", - "content": [ - { - "type": "text", - "text": prompt or "Describe this video frame in one detailed sentence. Focus on actions, objects, and scene context.", - }, - { - "type": "image_url", - "image_url": {"url": f"data:image/{ext};base64,{img_data}"}, - }, - ], - } - ], - max_tokens=200, + response = _call_with_retries( + self.config, + self.usage, + lambda: self.client.chat.completions.create( + model=self.config.vision_model, + messages=[ + { + "role": "user", + "content": [ + { + "type": "text", + "text": prompt or "Describe this video frame in one detailed sentence. Focus on actions, objects, and scene context.", + }, + { + "type": "image_url", + "image_url": {"url": f"data:image/{ext};base64,{img_data}"}, + }, + ], + } + ], + max_tokens=200, + ), ) text = response.choices[0].message.content or "" captions.append(Caption(timestamp_sec=ts, text=text.strip(), frame_path=str(fp))) except Exception as e: err = str(e) - if "Missing scopes: model.request" in err or "model_not_found" in err: + if self.config.allow_codex_fallback and ( + "Missing scopes: model.request" in err or "model_not_found" in err + ): try: fallback = _codex_cli_caption( prompt or "Describe this frame in one concise sentence with concrete actions and key objects.", @@ -272,15 +309,21 @@ def caption_chunk( }) try: - response = self.client.chat.completions.create( - model=self.config.vision_model, - messages=[{"role": "user", "content": content}], - max_tokens=500, + response = _call_with_retries( + self.config, + self.usage, + lambda: self.client.chat.completions.create( + model=self.config.vision_model, + messages=[{"role": "user", "content": content}], + max_tokens=500, + ), ) return (response.choices[0].message.content or "").strip() except Exception as e: err = str(e) - if "Missing scopes: model.request" in err or "model_not_found" in err: + if self.config.allow_codex_fallback and ( + "Missing scopes: model.request" in err or "model_not_found" in err + ): return _codex_cli_caption(prompt, frame_paths) raise @@ -290,14 +333,19 @@ def __init__(self, config: AVConfig): self.config = config self.client = _client(config) self._dim: int | None = None + self.usage = ProviderUsage() def embed(self, texts: list[str]) -> list[list[float]]: if not texts: return [] try: - response = self.client.embeddings.create( - model=self.config.embed_model, - input=texts, + response = _call_with_retries( + self.config, + self.usage, + lambda: self.client.embeddings.create( + model=self.config.embed_model, + input=texts, + ), ) vecs = [item.embedding for item in response.data] if vecs and self._dim is None: @@ -322,35 +370,53 @@ class OpenAILLM(LLMProvider): def __init__(self, config: AVConfig): self.config = config self.client = _client(config) + self.usage = ProviderUsage() def complete(self, prompt: str, context: str) -> str: + return self.complete_with_usage(prompt, context).text + + def complete_with_usage(self, prompt: str, context: str) -> CompletionResult: try: - response = self.client.chat.completions.create( - model=self.config.chat_model, - messages=[ - { - "role": "system", - "content": VIDEO_QA_SYSTEM_PROMPT, - }, - { - "role": "user", - "content": f"Context from video analysis:\n\n{context}\n\nQuestion: {prompt}", - }, - ], + response = _call_with_retries( + self.config, + self.usage, + lambda: self.client.chat.completions.create( + model=self.config.chat_model, + messages=[ + { + "role": "system", + "content": VIDEO_QA_SYSTEM_PROMPT, + }, + { + "role": "user", + "content": f"Context from video analysis:\n\n{context}\n\nQuestion: {prompt}", + }, + ], + ), + ) + usage = getattr(response, "usage", None) + return CompletionResult( + text=(response.choices[0].message.content or "").strip(), + input_tokens=getattr(usage, "prompt_tokens", None) if usage else None, + output_tokens=getattr(usage, "completion_tokens", None) if usage else None, + usage=self.usage.snapshot(), ) - return (response.choices[0].message.content or "").strip() except Exception as e: raise APIError(f"Chat completion failed: {e}", provider="openai") from e def summarize(self, system_prompt: str, user_content: str) -> str: """Generic system/user LLM call for cascade summarization.""" try: - response = self.client.chat.completions.create( - model=self.config.chat_model, - messages=[ - {"role": "system", "content": system_prompt}, - {"role": "user", "content": user_content}, - ], + response = _call_with_retries( + self.config, + self.usage, + lambda: self.client.chat.completions.create( + model=self.config.chat_model, + messages=[ + {"role": "system", "content": system_prompt}, + {"role": "user", "content": user_content}, + ], + ), ) return (response.choices[0].message.content or "").strip() except Exception as e: diff --git a/src/av/providers/usage.py b/src/av/providers/usage.py new file mode 100644 index 0000000..1aea924 --- /dev/null +++ b/src/av/providers/usage.py @@ -0,0 +1,98 @@ +"""Actual request and token receipts for provider-backed ingest stages.""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import Any + + +def _value(source: Any, *names: str) -> Any: + for name in names: + if isinstance(source, dict) and name in source: + return source[name] + value = getattr(source, name, None) + if value is not None: + return value + return None + + +def _cached_tokens(usage: Any) -> int | None: + direct = _value(usage, "cached_input_tokens", "cached_tokens") + if isinstance(direct, int) and direct >= 0: + return direct + details = _value(usage, "prompt_tokens_details", "input_tokens_details") + cached = _value(details, "cached_tokens") + return cached if isinstance(cached, int) and cached >= 0 else None + + +@dataclass +class ProviderUsage: + requests: int = 0 + successful_requests: int = 0 + failed_requests: int = 0 + input_tokens: int | None = None + output_tokens: int | None = None + cached_input_tokens: int | None = None + input_tokens_complete: bool = True + output_tokens_complete: bool = True + cached_input_tokens_complete: bool = True + + def _add(self, key: str, complete_key: str, value: Any) -> None: + if not getattr(self, complete_key): + setattr(self, key, None) + return + if isinstance(value, bool) or not isinstance(value, int) or value < 0: + setattr(self, key, None) + setattr(self, complete_key, False) + return + setattr(self, key, (getattr(self, key) or 0) + value) + + def record_success(self, usage: Any) -> None: + self.requests += 1 + self.successful_requests += 1 + self._add("input_tokens", "input_tokens_complete", _value(usage, "prompt_tokens", "input_tokens")) + self._add("output_tokens", "output_tokens_complete", _value(usage, "completion_tokens", "output_tokens")) + self._add("cached_input_tokens", "cached_input_tokens_complete", _cached_tokens(usage)) + + def record_failure(self) -> None: + self.requests += 1 + self.failed_requests += 1 + for key, complete_key in ( + ("input_tokens", "input_tokens_complete"), + ("output_tokens", "output_tokens_complete"), + ("cached_input_tokens", "cached_input_tokens_complete"), + ): + setattr(self, key, None) + setattr(self, complete_key, False) + + def merge(self, receipt: dict | None) -> None: + if not isinstance(receipt, dict) or not receipt: + return + self.requests += int(receipt.get("requests") or 0) + self.successful_requests += int(receipt.get("successful_requests") or 0) + self.failed_requests += int(receipt.get("failed_requests") or 0) + for key, complete_key in ( + ("input_tokens", "input_tokens_complete"), + ("output_tokens", "output_tokens_complete"), + ("cached_input_tokens", "cached_input_tokens_complete"), + ): + if not getattr(self, complete_key) or receipt.get(complete_key) is False: + setattr(self, key, None) + setattr(self, complete_key, False) + continue + value = receipt.get(key) + if isinstance(value, int): + setattr(self, key, (getattr(self, key) or 0) + value) + + def snapshot(self) -> dict: + return { + "requests": self.requests, + "successful_requests": self.successful_requests, + "failed_requests": self.failed_requests, + "input_tokens": self.input_tokens, + "output_tokens": self.output_tokens, + "cached_input_tokens": self.cached_input_tokens, + "input_tokens_complete": self.input_tokens_complete, + "output_tokens_complete": self.output_tokens_complete, + "cached_input_tokens_complete": self.cached_input_tokens_complete, + } diff --git a/src/av/search/inspection.py b/src/av/search/inspection.py new file mode 100644 index 0000000..acac1e4 --- /dev/null +++ b/src/av/search/inspection.py @@ -0,0 +1,435 @@ +"""Bounded sampled-frame inspection for unsupported refined answers. + +The inspection transport is intentionally separate from ordinary providers: it +uses only the endpoint, model, and optional credential configured for this stage. +It has no OAuth lookup, CLI fallback, SDK retry, or seed retry. +""" + +from __future__ import annotations + +import base64 +import json +import math +import tempfile +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +import requests + +from av.bench.frames import sample_at +from av.core.config import AVConfig +from av.db.repository import Repository, _fmt_timestamp +from av.search.usage import new_usage, record_usage + + +@dataclass(frozen=True) +class InspectionWindow: + video_id: str + filename: str + video_path: Path + requested_start_sec: float + requested_end_sec: float + start_sec: float + end_sec: float + truncated: bool = False + + +@dataclass +class VisionResponse: + ok: bool + text: str = "" + input_tokens: int | None = None + output_tokens: int | None = None + attempted: bool = True + + +class ExplicitVisionClient: + """One-attempt OpenAI-compatible image request with explicit credentials.""" + + def __init__(self, config: AVConfig, *, session: requests.Session | None = None) -> None: + base = config.strong_vision_api_base_url.rstrip("/") + self.endpoint = base if base.endswith("/chat/completions") else f"{base}/chat/completions" + self.api_key = config.strong_vision_api_key + self.model = config.strong_vision_model + self.timeout = config.api_timeout_sec + self.session = session or requests.Session() + + @staticmethod + def _image_url(path: Path) -> str: + extension = path.suffix.lstrip(".").lower() or "jpeg" + if extension == "jpg": + extension = "jpeg" + encoded = base64.b64encode(path.read_bytes()).decode() + return f"data:image/{extension};base64,{encoded}" + + def ask(self, images: list[Path], prompt: str) -> VisionResponse: + try: + content: list[dict] = [{"type": "text", "text": prompt}] + content.extend( + {"type": "image_url", "image_url": {"url": self._image_url(image)}} + for image in images + ) + except OSError: + return VisionResponse(ok=False, attempted=False) + headers = {"Content-Type": "application/json"} + if self.api_key: + headers["Authorization"] = f"Bearer {self.api_key}" + payload = { + "model": self.model, + "messages": [{"role": "user", "content": content}], + "max_tokens": 1024, + "temperature": 0, + } + try: + response = self.session.post( + self.endpoint, + headers=headers, + json=payload, + timeout=self.timeout, + ) + except requests.RequestException: + return VisionResponse(ok=False) + if not response.ok: + return VisionResponse(ok=False) + try: + data = response.json() + except ValueError: + return VisionResponse(ok=False) + if not isinstance(data, dict): + return VisionResponse(ok=False) + choices = data.get("choices") + if not isinstance(choices, list) or not choices or not isinstance(choices[0], dict): + return VisionResponse(ok=False) + message = choices[0].get("message") + text = message.get("content") if isinstance(message, dict) else None + if not isinstance(text, str): + return VisionResponse(ok=False) + usage = data.get("usage") if isinstance(data.get("usage"), dict) else {} + input_tokens = usage.get("prompt_tokens") + output_tokens = usage.get("completion_tokens") + return VisionResponse( + ok=True, + text=text.strip(), + input_tokens=input_tokens if isinstance(input_tokens, int) and input_tokens >= 0 else None, + output_tokens=output_tokens if isinstance(output_tokens, int) and output_tokens >= 0 else None, + ) + + +def _select_windows( + results: list[dict], + repo: Repository, + config: AVConfig, +) -> tuple[list[InspectionWindow], list[str]]: + warnings: list[str] = [] + if config.inspection_max_windows <= 0 or config.inspection_max_seconds <= 0: + return [], ["Sampled-frame inspection budget is exhausted."] + selected: list[InspectionWindow] = [] + seconds_left = config.inspection_max_seconds + for result in results: + if len(selected) >= config.inspection_max_windows or seconds_left <= 0: + break + if result.get("evidence_scope") == "broad": + warnings.append("Broad summary/report evidence is not eligible for sampled-frame inspection.") + continue + try: + video = repo.get_video(str(result.get("video_id") or "")) + except Exception: + warnings.append("A selected scene has no indexed video record for sampled-frame inspection.") + continue + path = Path(video.file_path) + if not path.is_file(): + warnings.append(f"Media is unavailable for sampled-frame inspection: {video.filename}.") + continue + requested_start = float(result.get("timestamp_sec") or 0) + raw_end = result.get("end_sec") + requested_end = float(raw_end) if isinstance(raw_end, (int, float)) else requested_start + start = max(0.0, min(requested_start, video.duration_sec)) + end = max(start, min(requested_end, video.duration_sec)) + if end <= start: + end = min(video.duration_sec, start + 1.0) + actual_end = min(end, start + seconds_left) + if actual_end <= start: + continue + selected.append( + InspectionWindow( + video_id=video.id, + filename=video.filename, + video_path=path, + requested_start_sec=requested_start, + requested_end_sec=requested_end, + start_sec=start, + end_sec=actual_end, + truncated=(start != requested_start or actual_end != requested_end), + ) + ) + seconds_left -= actual_end - start + return selected, warnings + + +def _window_key(window: InspectionWindow) -> str: + return f"{window.video_id}:{window.start_sec:.3f}:{window.end_sec:.3f}" + + +def sample_window_timestamps( + windows: list[InspectionWindow], + max_frames: int, +) -> dict[str, list[float]]: + """Allocate a hard frame budget; two frames cover both ends when affordable.""" + if not windows or max_frames <= 0: + return {} + counts = [0] * len(windows) + remaining = max_frames + for index in range(len(windows)): + if remaining <= 0: + break + counts[index] += 1 + remaining -= 1 + for index in range(len(windows)): + if remaining <= 0: + break + counts[index] += 1 + remaining -= 1 + cursor = 0 + while remaining > 0: + counts[cursor % len(windows)] += 1 + cursor += 1 + remaining -= 1 + + out: dict[str, list[float]] = {} + for window, count in zip(windows, counts): + if count <= 0: + timestamps: list[float] = [] + elif count == 1: + timestamps = [round((window.start_sec + window.end_sec) / 2, 3)] + else: + last = max(window.start_sec, window.end_sec - 0.001) + step = (last - window.start_sec) / (count - 1) + timestamps = [round(window.start_sec + step * index, 3) for index in range(count)] + out[_window_key(window)] = timestamps + return out + + +def _strip_code_fence(text: str) -> str: + cleaned = text.strip() + if cleaned.startswith("```"): + cleaned = cleaned.split("\n", 1)[1] if "\n" in cleaned else cleaned[3:] + if cleaned.endswith("```"): + cleaned = cleaned[:-3] + return cleaned.strip() + + +def _parse_result( + text: str, + windows: list[InspectionWindow], + sampled_by_video: dict[str, list[float]], +) -> tuple[str, list[dict]] | None: + try: + data = json.loads(_strip_code_fence(text)) + except (TypeError, ValueError): + return None + if not isinstance(data, dict) or data.get("supported") is not True: + return None + answer = data.get("answer") + evidence = data.get("evidence") + if not isinstance(answer, str) or not answer.strip() or not isinstance(evidence, list): + return None + allowed: dict[str, list[tuple[float, float, str]]] = {} + for window in windows: + allowed.setdefault(window.video_id, []).append( + (window.start_sec, window.end_sec, window.filename) + ) + citations: list[dict] = [] + for item in evidence: + if not isinstance(item, dict): + return None + video_id = item.get("video_id") + timestamp = item.get("timestamp_sec") + description = item.get("description") + if ( + not isinstance(video_id, str) + or isinstance(timestamp, bool) + or not isinstance(timestamp, (int, float)) + or not math.isfinite(float(timestamp)) + or not isinstance(description, str) + or not description.strip() + ): + return None + matches = [ + entry + for entry in allowed.get(video_id, []) + if entry[0] <= float(timestamp) <= entry[1] + ] + successful = sampled_by_video.get(video_id, []) + if not matches or not any(abs(float(timestamp) - value) <= 0.01 for value in successful): + return None + citations.append({ + "video_id": video_id, + "filename": matches[0][2], + "start_sec": float(timestamp), + "end_sec": float(timestamp), + "source_type": "sampled_frame_inspection", + "text": description.strip(), + "score": None, + }) + if not citations: + return None + return answer.strip(), citations + + +def _attempt_budgets(config: AVConfig, window_count: int) -> list[int]: + total = config.inspection_max_frames + if total <= 0 or window_count <= 0: + return [] + minimum_end_coverage = window_count * 2 + if ( + config.inspection_dense_pass + and config.inspection_max_attempts > 1 + and total >= minimum_end_coverage * 2 + ): + first = max(minimum_end_coverage, total // 3) + return [first, total - first][: config.inspection_max_attempts] + return [total] + + +def inspect_with_stronger_vision( + question: str, + initial_answer: str, + results: list[dict], + repo: Repository, + config: AVConfig, + *, + provider_factory=ExplicitVisionClient, +) -> dict: + usage = new_usage() + warnings: list[str] = [] + if not config.strong_vision_api_base_url or not config.strong_vision_model: + return { + "status": "not_configured", + "answer": None, + "citations": [], + "windows": [], + "usage": usage, + "warnings": ["Stronger sampled-frame inspection is not configured."], + } + windows, selection_warnings = _select_windows(results, repo, config) + warnings.extend(selection_warnings) + if not windows: + return { + "status": "unavailable", + "answer": None, + "citations": [], + "windows": [], + "usage": usage, + "warnings": warnings or ["No bounded media window was available for sampled-frame inspection."], + } + budgets = _attempt_budgets(config, len(windows)) + if not budgets: + return { + "status": "budget_exhausted", + "answer": None, + "citations": [], + "windows": [], + "usage": usage, + "warnings": warnings + ["Sampled-frame inspection frame budget is exhausted."], + } + + try: + provider = provider_factory(config) + except Exception: + return { + "status": "unavailable", + "answer": None, + "citations": [], + "windows": [], + "usage": usage, + "warnings": warnings + ["The stronger sampled-frame inspection provider could not be initialized."], + } + inspected: list[dict] = [] + for attempt, budget in enumerate(budgets, 1): + plan = sample_window_timestamps(windows, budget) + frame_paths: list[Path] = [] + frame_labels: list[str] = [] + sampled_by_video: dict[str, list[float]] = {} + with tempfile.TemporaryDirectory(prefix="av_ask_inspect_") as temp_dir: + root = Path(temp_dir) + for index, window in enumerate(windows): + requested = plan.get(_window_key(window), []) + frame_set = sample_at( + window.video_path, + requested, + out_dir=root / f"video_{index}", + ) + successful = list(frame_set.timestamps) + sampled_by_video.setdefault(window.video_id, []).extend(successful) + for path, timestamp in zip(frame_set.paths, successful): + frame_paths.append(path) + frame_labels.append( + f"image {len(frame_paths)}: video_id={window.video_id}, " + f"absolute_time={_fmt_timestamp(timestamp)} ({timestamp:.3f}s)" + ) + inspected.append({ + "video_id": window.video_id, + "filename": window.filename, + "requested_start_sec": window.requested_start_sec, + "requested_end_sec": window.requested_end_sec, + "start_sec": window.start_sec, + "end_sec": window.end_sec, + "truncated": window.truncated, + "requested_timestamps": requested, + "sampled_timestamps": successful, + "all_requested_frames_extracted": len(successful) == len(requested), + "attempt": attempt, + }) + if not frame_paths: + warnings.append("Frame extraction returned no usable sampled frames.") + continue + prompt = ( + "Inspect only the supplied independent sampled frames. They do not provide native video or audio, " + "and they do not establish what happened between sampled timestamps. Answer only when the visible " + "frames support it. Return strict JSON with keys: supported (boolean), answer (string), evidence " + "(array of objects with video_id, timestamp_sec, description). Every evidence timestamp must be one " + "of the successfully sampled absolute timestamps listed below. If evidence is insufficient, set " + "supported=false and evidence=[].\n\n" + f"Question: {question}\nInitial answer to verify or replace: {initial_answer}\n\n" + "Successful frame mapping:\n" + "\n".join(frame_labels) + ) + try: + response = provider.ask(frame_paths, prompt) + except Exception: + record_usage(usage, None, requests=1, ambiguous_attempts=True) + warnings.append("The stronger sampled-frame inspection provider was unavailable.") + continue + if getattr(response, "attempted", True): + record_usage( + usage, + { + "input_tokens": response.input_tokens, + "output_tokens": response.output_tokens, + }, + requests=1, + ) + if not response.ok: + warnings.append("The stronger sampled-frame inspection provider was unavailable.") + continue + parsed = _parse_result(response.text, windows, sampled_by_video) + if parsed is not None: + answer, citations = parsed + return { + "status": "supported", + "answer": answer, + "citations": citations, + "windows": inspected, + "usage": usage, + "warnings": warnings, + } + warnings.append("The sampled-frame inspection did not return validated supporting evidence.") + + return { + "status": "insufficient", + "answer": None, + "citations": [], + "windows": inspected, + "usage": usage, + "warnings": warnings, + } diff --git a/src/av/search/rag.py b/src/av/search/rag.py index 9da726d..0da1a31 100644 --- a/src/av/search/rag.py +++ b/src/av/search/rag.py @@ -5,7 +5,93 @@ from av.core.config import AVConfig from av.db.repository import Repository, _fmt_timestamp from av.providers.openai import OpenAILLM +from av.search.inspection import inspect_with_stronger_vision +from av.search.refine import ( + RefinementError, + SystemOneClient, + judge_answer_support, + refine_search_results, +) from av.search.semantic import search +from av.search.usage import merge_usage, new_usage, record_usage + + +def _citations(results: list[dict]) -> list[dict]: + citations = [] + provenance_keys = ( + "artifact_id", + "chunk_start_sec", + "chunk_end_sec", + "scene_confidence", + "merged_artifact_ids", + "relevance_p", + "evidence_scope", + ) + for result in results: + citation = { + "video_id": result.get("video_id", ""), + "start_sec": result.get("timestamp_sec", 0), + "end_sec": result.get("end_sec"), + "source_type": result.get("source_type", ""), + "text": result.get("text", ""), + "score": result.get("score", 0), + } + for key in provenance_keys: + if key in result: + citation[key] = result.get(key) + citations.append(citation) + return citations + + +def _context(results: list[dict]) -> str: + parts: list[str] = [] + for result in results: + start = result.get("timestamp_formatted", "") + end_sec = result.get("end_sec") + end = _fmt_timestamp(float(end_sec)) if isinstance(end_sec, (int, float)) else start + parts.append( + f"[{result.get('filename', '')} @ {start}-{end} ({result.get('source_type', '')})] " + f"{result.get('text', '')}" + ) + return "\n\n".join(parts) + + +def _heuristic_confidence(results: list[dict]) -> float: + top_score = results[0].get("score", 0) if results else 0 + return min(round(float(top_score), 2), 1.0) if top_score else 0.5 + + +def _legacy_ask(question: str, results: list[dict], config: AVConfig) -> dict: + if not results: + return { + "answer": "No relevant content found in the indexed videos.", + "citations": [], + "confidence": 0.0, + } + answer = OpenAILLM(config).complete(question, _context(results)) + return { + "answer": answer, + "citations": _citations(results), + "confidence": _heuristic_confidence(results), + } + + +def _usage_from_completion(completion) -> dict: + if isinstance(getattr(completion, "usage", None), dict): + return dict(completion.usage) + usage = new_usage() + record_usage( + usage, + { + "input_tokens": completion.input_tokens, + "output_tokens": completion.output_tokens, + }, + ) + return usage + + +def _add_stage_usage(target: dict, source: dict) -> None: + merge_usage(target, source) def ask( @@ -15,6 +101,7 @@ def ask( *, video_id: str | None = None, top_k: int = 10, + refine: bool = True, ) -> dict: """Answer a question using RAG over video artifacts.""" # Step 1: Retrieve relevant context @@ -22,48 +109,158 @@ def ask( question, repo, config, limit=top_k, video_id=video_id ) - results = search_result.get("results", []) - if not results: + raw_results = search_result.get("results", []) + if not refine or not config.refine_enabled or not config.typesafe_api_key: + return _legacy_ask(question, raw_results, config) + + warnings: list[str] = [] + stage_usage: dict[str, dict | None] = { + "relevance": new_usage(), + "boundary": new_usage(), + "answer": new_usage(), + "support": new_usage(), + "vision": new_usage(), + "embedding": search_result.get("embedding_usage"), + } + if not raw_results: return { "answer": "No relevant content found in the indexed videos.", "citations": [], "confidence": 0.0, + "confidence_basis": "no_evidence", + "route": "refined_no_results", + "evidence_status": "no_retrieval_hits", + "refinement": {"status": "no_retrieval_hits", "raw_count": 0, "scene_count": 0}, + "warnings": warnings, + "inspected_windows": [], + "stage_usage": stage_usage, } - # Step 2: Build context string - context_parts: list[str] = [] - for r in results: - vid = r.get("video_id", "") - fn = r.get("filename", "") - ts = r.get("timestamp_formatted", "") - src = r.get("source_type", "") - text = r.get("text", "") - context_parts.append(f"[{fn} @ {ts} ({src})] {text}") + client = SystemOneClient(config) + try: + results, refinement, refinement_usage = refine_search_results( + question, raw_results, repo, config, client=client + ) + stage_usage.update(refinement_usage) + except RefinementError as exc: + for stage, usage in exc.stage_usage.items(): + stage_usage[stage] = usage + warnings.append("Jev refinement was unavailable; answering from raw retrieval without judged evidence confidence.") + completion = OpenAILLM(config).complete_with_usage(question, _context(raw_results)) + stage_usage["answer"] = _usage_from_completion(completion) + return { + "answer": completion.text, + "citations": _citations(raw_results), + "confidence": _heuristic_confidence(raw_results), + "confidence_basis": "retrieval_heuristic", + "route": "refinement_fallback", + "evidence_status": "raw_unjudged", + "refinement": {"status": "provider_fallback", "raw_count": len(raw_results)}, + "warnings": warnings, + "inspected_windows": [], + "stage_usage": stage_usage, + } - context = "\n\n".join(context_parts) + if not results: + return { + "answer": "No supported evidence was found for this question in the retrieved video moments.", + "citations": [], + "confidence": 0.0, + "confidence_basis": "jev_relevance", + "route": "refined_no_results", + "evidence_status": "all_sources_irrelevant", + "refinement": refinement, + "warnings": warnings, + "inspected_windows": [], + "stage_usage": stage_usage, + } - # Step 3: Generate answer - llm = OpenAILLM(config) - answer = llm.complete(question, context) + completion = OpenAILLM(config).complete_with_usage(question, _context(results)) + stage_usage["answer"] = _usage_from_completion(completion) + answer = completion.text + citations = _citations(results) + support_probability: float | None = None + support_failed = False + try: + support_probability, support_usage = judge_answer_support(client, question, answer, results) + stage_usage["support"] = support_usage + except RefinementError as exc: + if "support" in exc.stage_usage: + stage_usage["support"] = exc.stage_usage["support"] + support_failed = True + warnings.append("Answer support judgment was unavailable; the answer is not marked supported.") - # Step 4: Build citations - citations = [] - for r in results: - citations.append({ - "video_id": r.get("video_id", ""), - "start_sec": r.get("timestamp_sec", 0), - "end_sec": None, - "source_type": r.get("source_type", ""), - "text": r.get("text", ""), - "score": r.get("score", 0), - }) - - # Confidence is a rough heuristic based on top result score - top_score = results[0].get("score", 0) if results else 0 - confidence = min(round(float(top_score), 2), 1.0) if top_score else 0.5 + if support_probability is not None and support_probability >= config.refine_support_min: + return { + "answer": answer, + "citations": citations, + "confidence": support_probability, + "confidence_basis": "jev_answer_support", + "route": "refined", + "evidence_status": "supported", + "refinement": refinement, + "warnings": warnings, + "inspected_windows": [], + "stage_usage": stage_usage, + } + inspection = inspect_with_stronger_vision(question, answer, results, repo, config) + stage_usage["vision"] = inspection["usage"] + warnings.extend(inspection["warnings"]) + if inspection["status"] == "supported": + inspected_answer = inspection["answer"] + inspected_citations = inspection["citations"] + try: + inspected_support, inspected_usage = judge_answer_support( + client, + question, + inspected_answer, + [ + { + "video_id": citation["video_id"], + "timestamp_sec": citation["start_sec"], + "end_sec": citation["end_sec"], + "source_type": citation["source_type"], + "text": citation["text"], + } + for citation in inspected_citations + ], + ) + _add_stage_usage(stage_usage["support"], inspected_usage) + except RefinementError as exc: + if "support" in exc.stage_usage: + _add_stage_usage(stage_usage["support"], exc.stage_usage["support"]) + inspected_support = None + warnings.append("Inspected evidence could not be independently support-judged.") + if inspected_support is not None and inspected_support >= config.refine_support_min: + return { + "answer": inspected_answer, + "citations": inspected_citations, + "confidence": inspected_support, + "confidence_basis": "jev_answer_support_after_sampled_frames", + "route": "vision_inspected", + "evidence_status": "sampled_frames_supported", + "refinement": refinement, + "warnings": warnings, + "inspected_windows": inspection["windows"], + "stage_usage": stage_usage, + } + + if support_failed: + uncertain = f"Unverified: {answer}" if answer else "The answer could not be verified from the available evidence." + evidence_status = "support_unknown" + else: + uncertain = "The retrieved evidence was relevant, but it did not support a reliable answer." + evidence_status = "unsupported" return { - "answer": answer, + "answer": uncertain, "citations": citations, - "confidence": confidence, + "confidence": support_probability or 0.0, + "confidence_basis": "jev_answer_support" if support_probability is not None else "unknown", + "route": "refined_uncertain", + "evidence_status": evidence_status, + "refinement": refinement, + "warnings": warnings, + "inspected_windows": inspection["windows"], + "stage_usage": stage_usage, } diff --git a/src/av/search/refine.py b/src/av/search/refine.py new file mode 100644 index 0000000..10365ff --- /dev/null +++ b/src/av/search/refine.py @@ -0,0 +1,689 @@ +"""Jev/System One search refinement with bounded scene expansion. + +Source relevance, temporal grouping, and answer support are deliberately separate +decisions. Window sizes and batch sizes are public configuration, not hidden policy. +All provider calls use the documented ``/v1/systemone`` contract; no generative +completion is presented as Jev. +""" + +from __future__ import annotations + +import json +import math +import time +from dataclasses import dataclass, field +from typing import Any + +import requests + +from av.core.config import AVConfig +from av.db.models import ArtifactRecord +from av.db.repository import Repository, _fmt_timestamp +from av.search.usage import new_usage, record_usage + +CONTIGUOUS_GAP_SEC = 1.5 +MAX_SCENE_TEXT_CHARS = 6000 +_RETRYABLE_STATUS = {429, 500, 502, 503, 504} + + +class RefinementError(RuntimeError): + """A sanitized System One request or schema failure.""" + + def __init__( + self, + message: str, + *, + attempts: int = 0, + raw_usage: dict[str, Any] | None = None, + stage_usage: dict[str, dict] | None = None, + ) -> None: + super().__init__(message) + self.attempts = attempts + self.raw_usage = raw_usage + self.stage_usage = stage_usage or {} + + +def _record_client_error(usage: dict, error: RefinementError) -> None: + attempts = max(error.attempts, 1) + record_usage( + usage, + error.raw_usage, + requests=attempts, + ambiguous_attempts=attempts > 1 or error.raw_usage is None, + ) + + +def _record_call_usage(usage: dict, call_usage: dict[str, Any] | None) -> None: + attempts = 1 + cleaned = call_usage + if isinstance(call_usage, dict): + attempts_value = call_usage.get("_attempts", 1) + attempts = attempts_value if isinstance(attempts_value, int) and attempts_value > 0 else 1 + cleaned = {key: value for key, value in call_usage.items() if key != "_attempts"} + record_usage( + usage, + cleaned, + requests=attempts, + ambiguous_attempts=attempts > 1, + ) + + +def _probability(value: Any, name: str) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise RefinementError(f"System One returned an invalid probability for {name}") + value = float(value) + if not math.isfinite(value) or not 0 <= value <= 1: + raise RefinementError(f"System One returned an out-of-range probability for {name}") + return value + + +class SystemOneClient: + """Small synchronous client for the documented TypeSafe System One endpoint.""" + + def __init__(self, config: AVConfig, *, session: requests.Session | None = None) -> None: + if not config.typesafe_api_key: + raise RefinementError("TypeSafe API key is not configured") + self.endpoint = config.typesafe_endpoint + self.api_key = config.typesafe_api_key + self.model = config.typesafe_model + self.timeout = config.typesafe_timeout_sec + self.max_retries = config.typesafe_max_retries + self.session = session or requests.Session() + + def ask(self, state: Any, questions: dict[str, dict]) -> tuple[dict[str, dict], dict]: + payload = {"state": state, "model": self.model, "questions": questions} + headers = { + "Authorization": f"Bearer {self.api_key}", + "Content-Type": "application/json", + } + last_error = "request failed" + attempts = 0 + for attempt in range(self.max_retries + 1): + attempts += 1 + try: + response = self.session.post( + self.endpoint, + headers=headers, + json=payload, + timeout=self.timeout, + ) + except requests.RequestException as exc: + last_error = type(exc).__name__ + if attempt < self.max_retries: + time.sleep(min(0.25 * (2**attempt), 1.0)) + continue + raise RefinementError( + f"System One unavailable ({last_error})", + attempts=attempts, + ) from exc + if response.status_code in _RETRYABLE_STATUS and attempt < self.max_retries: + time.sleep(min(0.25 * (2**attempt), 1.0)) + continue + if not response.ok: + raise RefinementError( + f"System One request failed with HTTP {response.status_code}", + attempts=attempts, + ) + try: + data = response.json() + except ValueError as exc: + raise RefinementError( + "System One returned invalid JSON", + attempts=attempts, + ) from exc + if not isinstance(data, dict): + raise RefinementError( + "System One returned an invalid JSON document", + attempts=attempts, + ) + usage = data.get("usage") if isinstance(data.get("usage"), dict) else {} + answers = data.get("answers") + if not isinstance(answers, dict): + raise RefinementError( + "System One response is missing answers", + attempts=attempts, + raw_usage=usage, + ) + usage = {**usage, "_attempts": attempts} + return answers, usage + raise RefinementError(f"System One unavailable ({last_error})", attempts=attempts) + + +@dataclass +class Scene: + artifact_id: str + video_id: str + filename: str + source_type: str + start_sec: float + end_sec: float + text: str + retrieval_score: float + relevance_p: float + chunk_start_sec: float + chunk_end_sec: float + scene_confidence: float | None = None + merged_artifact_ids: list[str] = field(default_factory=list) + hit_texts: list[str] = field(default_factory=list) + is_broad: bool = False + + @property + def rank_score(self) -> float: + return self.relevance_p * self.retrieval_score + + def to_result(self, rank: int) -> dict: + return { + "rank": rank, + "score": self.retrieval_score, + "video_id": self.video_id, + "filename": self.filename, + "timestamp_sec": self.start_sec, + "end_sec": self.end_sec, + "timestamp_formatted": _fmt_timestamp(self.start_sec), + "source_type": self.source_type, + "text": self.text, + "artifact_id": self.artifact_id, + "relevance_p": self.relevance_p, + "scene_confidence": self.scene_confidence, + "merged_artifact_ids": self.merged_artifact_ids, + "chunk_start_sec": self.chunk_start_sec, + "chunk_end_sec": self.chunk_end_sec, + "rank_score": self.rank_score, + "evidence_scope": "broad" if self.is_broad else "scene", + } + + +def _artifact_end(artifact: ArtifactRecord | dict) -> float: + if isinstance(artifact, dict): + start = float(artifact.get("timestamp_sec", 0)) + end = artifact.get("end_sec") + else: + start = artifact.start_sec + end = artifact.end_sec + return max(start, float(end)) if end is not None else start + + +def _ordered_contiguous_events( + result: dict, + artifacts: list[ArtifactRecord], +) -> tuple[list[dict], int]: + hit_start = float(result.get("timestamp_sec", 0)) + hit_end = _artifact_end(result) + hit_id = str(result.get("artifact_id") or "") + rows = [ + { + "artifact_id": artifact.id, + "source_type": artifact.type, + "start": artifact.start_sec, + "end": _artifact_end(artifact), + "text": artifact.text, + } + for artifact in artifacts + ] + if hit_id and not any(row["artifact_id"] == hit_id for row in rows): + rows.append({ + "artifact_id": hit_id, + "source_type": str(result.get("source_type") or "artifact"), + "start": hit_start, + "end": hit_end, + "text": str(result.get("text") or ""), + }) + + # Build temporal events rather than treating every modality row as a chunk. + # Strictly overlapping rows attach to one event; touching fixed chunks remain + # separate events and are linked later by the contiguous-gap rule. + rows.sort(key=lambda row: (row["start"], -(row["end"] - row["start"]), row["artifact_id"])) + events: list[dict] = [] + for row in rows: + point = row["end"] <= row["start"] + target = None + for event in reversed(events): + overlaps = row["start"] < event["end"] and row["end"] > event["start"] + point_inside = point and event["start"] <= row["start"] <= event["end"] + if overlaps or point_inside: + target = event + break + if event["end"] < row["start"]: + break + label = ( + f"[{row['source_type']} {_fmt_timestamp(row['start'])}-{_fmt_timestamp(row['end'])}] " + f"{row['text']}" + ) + if target is None: + events.append({ + "artifact_ids": [row["artifact_id"]], + "start": row["start"], + "end": row["end"], + "texts": [label], + "text": label, + }) + else: + target["start"] = min(target["start"], row["start"]) + target["end"] = max(target["end"], row["end"]) + target["artifact_ids"].append(row["artifact_id"]) + if label not in target["texts"]: + target["texts"].append(label) + target["text"] = "\n".join(target["texts"]) + events.sort(key=lambda event: (event["start"], event["end"])) + hit_index = next( + (index for index, event in enumerate(events) if hit_id in event["artifact_ids"]), + -1, + ) + if hit_index < 0: + hit_index = next( + ( + index + for index, event in enumerate(events) + if event["start"] <= hit_start < max(event["end"], event["start"] + 0.001) + ), + -1, + ) + if hit_index < 0: + label = f"[{result.get('source_type', 'artifact')}] {result.get('text', '')}" + events.append({ + "artifact_ids": [hit_id], + "start": hit_start, + "end": hit_end, + "texts": [label], + "text": label, + }) + events.sort(key=lambda event: (event["start"], event["end"])) + hit_index = next(index for index, event in enumerate(events) if hit_id in event["artifact_ids"]) + + lo = hit_index + while lo > 0 and events[lo]["start"] - events[lo - 1]["end"] <= CONTIGUOUS_GAP_SEC: + lo -= 1 + hi = hit_index + while hi < len(events) - 1 and events[hi + 1]["start"] - events[hi]["end"] <= CONTIGUOUS_GAP_SEC: + hi += 1 + return events[lo : hi + 1], hit_index - lo + + +def _boundary_input(events: list[dict], hit_index: int, query: str, window: int) -> dict: + lo = max(0, hit_index - window) + hi = min(len(events) - 1, hit_index + window) + candidates: dict[str, dict] = {} + start_labels: list[str] = [] + end_labels: list[str] = [] + for index in range(lo, hi + 1): + offset = index - hit_index + label = f"e{offset}" + event = events[index] + candidates[label] = { + "start": event["start"], + "end": event["end"], + "text": event["text"], + } + if offset <= 0: + start_labels.append(label) + if offset >= 0: + end_labels.append(label) + return { + "query": query, + "hit_event": events[hit_index]["text"], + "surrounding_events": candidates, + "start_labels": start_labels, + "end_labels": end_labels, + "lo": lo, + "hi": hi, + } + + +def _choice_criteria(labels: list[str], candidates: dict[str, dict]) -> dict: + out = {} + for label in labels: + event = candidates[label] + out[label] = ( + {"what": "the hit event itself; the scene does not extend farther on this side", "text": event["text"]} + if label == "e0" + else {"seconds": f"{event['start']}-{event['end']}", "text": event["text"]} + ) + return out + + +def _read_choice(answer: Any, name: str, allowed: list[str]) -> tuple[str, float]: + if not isinstance(answer, dict) or answer.get("type") != "choice": + raise RefinementError(f"System One returned an invalid Choice answer for {name}") + choice = answer.get("choice") + if choice not in allowed: + raise RefinementError(f"System One returned an invalid boundary choice for {name}") + confidence = _probability(answer.get("confidence"), f"{name}.confidence") + return str(choice), confidence + + +def _judge_bounds( + client: SystemOneClient, + query: str, + events: list[dict], + hit_index: int, + window: int, + usage: dict, +) -> tuple[float, float, float, str, str, dict]: + data = _boundary_input(events, hit_index, query, window) + candidates = data["surrounding_events"] + questions = { + "start": { + "type": "choice", + "instructions": "Select the earliest candidate that belongs to the same continuous video moment as `hit_event` for the user's `query`. Use e0 when earlier candidates do not belong to that moment.", + "criteria": _choice_criteria(data["start_labels"], candidates), + }, + "end": { + "type": "choice", + "instructions": "Select the latest candidate that belongs to the same continuous video moment as `hit_event` for the user's `query`. Use e0 when later candidates do not belong to that moment.", + "criteria": _choice_criteria(data["end_labels"], candidates), + }, + } + state = { + "query": query, + "hit_event": data["hit_event"], + "surrounding_events": candidates, + "note": "Candidates are ordered temporal events from one video. e0 contains the retrieved hit; negative labels are earlier and positive labels are later.", + } + try: + answers, call_usage = client.ask(state, questions) + except RefinementError as exc: + _record_client_error(usage, exc) + exc.stage_usage["boundary"] = usage + raise + _record_call_usage(usage, call_usage) + try: + start_label, start_conf = _read_choice(answers.get("start"), "start", data["start_labels"]) + end_label, end_conf = _read_choice(answers.get("end"), "end", data["end_labels"]) + except RefinementError as exc: + exc.stage_usage["boundary"] = usage + raise + start_event = events[hit_index + int(start_label[1:])] + end_event = events[hit_index + int(end_label[1:])] + return start_event["start"], end_event["end"], min(start_conf, end_conf), start_label, end_label, data + + +def judge_relevance( + client: SystemOneClient, + query: str, + results: list[dict], + *, + batch_size: int = 10, +) -> tuple[dict[str, float], dict[str, int | None]]: + usage = new_usage() + probabilities: dict[str, float] = {} + for offset in range(0, len(results), batch_size): + batch = results[offset : offset + batch_size] + clips: dict[str, dict] = {} + questions: dict[str, dict] = {} + ids: dict[str, str] = {} + for index, result in enumerate(batch): + key = f"c{index}" + artifact_id = str(result.get("artifact_id") or f"result-{offset + index}") + ids[key] = artifact_id + clips[key] = { + "caption": result.get("text") if result.get("source_type") != "transcript" else "(none)", + "transcript": result.get("text") if result.get("source_type") == "transcript" else "(none)", + "source_type": result.get("source_type"), + "video_id": result.get("video_id"), + "start_sec": result.get("timestamp_sec"), + "end_sec": result.get("end_sec"), + } + questions[key] = { + "type": "noul", + "instructions": f"Does `clips.{key}` contain visual or spoken evidence about the subject or event requested by `query`?", + "criteria": { + "true": "The clip content materially concerns the requested subject or event.", + "false": "The clip is unrelated or mentions the subject only incidentally.", + }, + } + try: + answers, call_usage = client.ask({"query": query, "clips": clips}, questions) + except RefinementError as exc: + _record_client_error(usage, exc) + exc.stage_usage["relevance"] = usage + raise + _record_call_usage(usage, call_usage) + try: + for key, artifact_id in ids.items(): + answer = answers.get(key) + if not isinstance(answer, dict) or answer.get("type") != "noul": + raise RefinementError(f"System One response is missing Noul answer {key}") + probabilities[artifact_id] = _probability(answer.get("noul"), key) + except RefinementError as exc: + exc.stage_usage["relevance"] = usage + raise + return probabilities, usage + + +def _scene_from_result(result: dict, relevance_p: float) -> Scene: + start = float(result.get("timestamp_sec", 0)) + end = _artifact_end(result) + artifact_id = str(result.get("artifact_id") or "") + source_type = str(result.get("source_type") or "") + text = str(result.get("text") or "") + return Scene( + artifact_id=artifact_id, + video_id=str(result.get("video_id") or ""), + filename=str(result.get("filename") or ""), + source_type=source_type, + start_sec=start, + end_sec=end, + text=text, + retrieval_score=float(result.get("score") or 0), + relevance_p=relevance_p, + chunk_start_sec=start, + chunk_end_sec=end, + merged_artifact_ids=[artifact_id] if artifact_id else [], + hit_texts=[text] if text else [], + is_broad=source_type in {"summary", "report"}, + ) + + +def _expand_scene( + query: str, + result: dict, + relevance_p: float, + repo: Repository, + client: SystemOneClient, + usage: dict, + context_events: int, +) -> Scene: + scene = _scene_from_result(result, relevance_p) + if scene.is_broad: + return scene + artifacts = repo.get_refinement_window( + scene.video_id, + scene.chunk_start_sec, + before=context_events, + after=context_events, + ) + events, hit_index = _ordered_contiguous_events(result, artifacts) + if len(events) < 2: + return scene + start, end, confidence, _, _, _ = _judge_bounds( + client, query, events, hit_index, context_events, usage + ) + scene.start_sec = min(start, scene.chunk_start_sec) + scene.end_sec = max(end, scene.chunk_end_sec) + scene.scene_confidence = confidence + return scene + + +def merge_overlapping_scenes(scenes: list[Scene]) -> tuple[list[Scene], int]: + by_video: dict[str, list[Scene]] = {} + for scene in scenes: + if scene.is_broad: + continue + by_video.setdefault(scene.video_id, []).append(scene) + merged: list[Scene] = [scene for scene in scenes if scene.is_broad] + merged_count = 0 + for video_scenes in by_video.values(): + video_scenes.sort(key=lambda scene: (scene.start_sec, scene.end_sec)) + current = video_scenes[0] + for next_scene in video_scenes[1:]: + if next_scene.start_sec <= current.end_sec: + merged_count += 1 + # The higher retrieval score supplies representative fields; + # merged relevance uses the maximum accepted probability. + best = next_scene if next_scene.retrieval_score > current.retrieval_score else current + current = Scene( + artifact_id=best.artifact_id, + video_id=best.video_id, + filename=best.filename, + source_type=best.source_type, + start_sec=min(current.start_sec, next_scene.start_sec), + end_sec=max(current.end_sec, next_scene.end_sec), + text=best.text, + retrieval_score=best.retrieval_score, + relevance_p=max(current.relevance_p, next_scene.relevance_p), + chunk_start_sec=best.chunk_start_sec, + chunk_end_sec=best.chunk_end_sec, + scene_confidence=min( + value for value in (current.scene_confidence, next_scene.scene_confidence) if value is not None + ) if current.scene_confidence is not None or next_scene.scene_confidence is not None else None, + merged_artifact_ids=current.merged_artifact_ids + next_scene.merged_artifact_ids, + hit_texts=list(dict.fromkeys(current.hit_texts + next_scene.hit_texts)), + ) + else: + merged.append(current) + current = next_scene + merged.append(current) + return merged, merged_count + + +def _hydrate_scene_text(scene: Scene, repo: Repository) -> None: + if scene.is_broad: + return + artifacts = repo.get_artifacts_overlapping(scene.video_id, scene.start_sec, scene.end_sec) + parts = [f"[retrieved hit] {text}" for text in scene.hit_texts if text.strip()] + seen_text = {text.strip() for text in scene.hit_texts if text.strip()} + for artifact in artifacts: + text = artifact.text.strip() + if artifact.id in scene.merged_artifact_ids or not text or text in seen_text: + continue + part = f"[{_fmt_timestamp(artifact.start_sec)}-{_fmt_timestamp(_artifact_end(artifact))} {artifact.type}] {text}" + if part not in parts: + parts.append(part) + seen_text.add(text) + hydrated = "\n".join(parts) + if hydrated: + scene.text = hydrated[:MAX_SCENE_TEXT_CHARS] + + +def refine_search_results( + query: str, + results: list[dict], + repo: Repository, + config: AVConfig, + *, + client: SystemOneClient | None = None, +) -> tuple[list[dict], dict, dict[str, dict[str, int | None]]]: + client = client or SystemOneClient(config) + probabilities, relevance_usage = judge_relevance( + client, + query, + results, + batch_size=config.refine_batch_size, + ) + kept = [] + for result in results: + artifact_id = str(result.get("artifact_id") or "") + probability = probabilities.get(artifact_id) + if probability is None: + raise RefinementError( + "System One omitted a source relevance probability", + stage_usage={"relevance": relevance_usage}, + ) + if probability >= config.refine_relevance_min: + kept.append((result, probability)) + meta = { + "status": "success" if kept else "no_relevant_evidence", + "raw_count": len(results), + "dropped_count": len(results) - len(kept), + "merged_count": 0, + "capped_count": 0, + "scene_count": 0, + "video_count": 0, + "relevance_min": config.refine_relevance_min, + } + boundary_usage = new_usage() + if not kept: + return [], meta, {"relevance": relevance_usage, "boundary": boundary_usage} + + try: + expanded = [ + _expand_scene( + query, + result, + probability, + repo, + client, + boundary_usage, + config.refine_context_events, + ) + for result, probability in kept + ] + except RefinementError as exc: + exc.stage_usage.setdefault("relevance", relevance_usage) + exc.stage_usage["boundary"] = boundary_usage + raise + merged, merged_count = merge_overlapping_scenes(expanded) + for scene in merged: + _hydrate_scene_text(scene, repo) + merged.sort(key=lambda scene: scene.rank_score, reverse=True) + capped = merged[: config.refine_max_scenes] + meta.update({ + "merged_count": merged_count, + "capped_count": len(merged) - len(capped), + "scene_count": len(capped), + "video_count": len({scene.video_id for scene in capped}), + }) + return ( + [scene.to_result(index + 1) for index, scene in enumerate(capped)], + meta, + {"relevance": relevance_usage, "boundary": boundary_usage}, + ) + + +def judge_answer_support( + client: SystemOneClient, + question: str, + answer: str, + evidence: list[dict], +) -> tuple[float, dict[str, int | None]]: + state = { + "question": question, + "answer": answer, + "evidence": [ + { + "video_id": item.get("video_id"), + "start_sec": item.get("timestamp_sec"), + "end_sec": item.get("end_sec"), + "source_type": item.get("source_type"), + "text": item.get("text"), + } + for item in evidence + ], + } + questions = { + "is_supported": { + "type": "noul", + "instructions": "Is `answer` directly supported by `evidence` and sufficient to answer `question` without adding unsupported facts?", + "criteria": { + "true": "The evidence directly supports the answer's material claims and answers the question.", + "false": "The evidence is merely relevant, incomplete, contradictory, or does not support the answer's material claims.", + }, + } + } + usage = new_usage() + try: + answers, raw_usage = client.ask(state, questions) + except RefinementError as exc: + _record_client_error(usage, exc) + exc.stage_usage["support"] = usage + raise + _record_call_usage(usage, raw_usage) + try: + answer_data = answers.get("is_supported") + if not isinstance(answer_data, dict) or answer_data.get("type") != "noul": + raise RefinementError("System One response is missing the support Noul") + probability = _probability(answer_data.get("noul"), "is_supported") + except RefinementError as exc: + exc.stage_usage["support"] = usage + raise + return probability, usage diff --git a/src/av/search/semantic.py b/src/av/search/semantic.py index 2a2e5ad..e3151c7 100644 --- a/src/av/search/semantic.py +++ b/src/av/search/semantic.py @@ -30,6 +30,7 @@ def search( ) -> dict: """Search artifacts using FTS5, optionally reranked by cosine similarity.""" start_time = time.time() + embedding_usage = None # Step 1: FTS search (always available) fts_results = repo.search_fts(query, limit=limit * 3, video_id=video_id) @@ -49,6 +50,7 @@ def search( if embeddings: # Embed the query + embedder = None try: embedder = OpenAIEmbedder(get_openai_config(config) or config) query_vecs = embedder.embed([query]) @@ -73,6 +75,7 @@ def search( video_id=r.video_id, filename=r.filename, timestamp_sec=r.timestamp_sec, + end_sec=r.end_sec, timestamp_formatted=r.timestamp_formatted, source_type=r.source_type, text=r.text, @@ -82,6 +85,9 @@ def search( except Exception: # Fall back to FTS-only results if embedding fails fts_results = fts_results[:limit] + finally: + if embedder is not None: + embedding_usage = embedder.usage.snapshot() else: fts_results = fts_results[:limit] @@ -91,4 +97,5 @@ def search( "results": [r.model_dump() for r in fts_results], "total_results": len(fts_results), "search_time_ms": elapsed_ms, + "embedding_usage": embedding_usage, } diff --git a/src/av/search/usage.py b/src/av/search/usage.py new file mode 100644 index 0000000..c009ddb --- /dev/null +++ b/src/av/search/usage.py @@ -0,0 +1,61 @@ +"""Truthful per-stage token accounting helpers. + +Totals stay ``None`` once any contributing request omits that token dimension or +an attempted request fails before trustworthy usage is available. Request counts +still report the actual HTTP attempts. +""" + +from __future__ import annotations + +from typing import Any + + +def new_usage() -> dict[str, int | bool | None]: + return { + "requests": 0, + "input_tokens": None, + "output_tokens": None, + "input_tokens_complete": True, + "output_tokens_complete": True, + } + + +def record_usage( + total: dict[str, int | bool | None], + usage: dict[str, Any] | None, + *, + requests: int = 1, + ambiguous_attempts: bool = False, +) -> None: + total["requests"] = int(total.get("requests") or 0) + max(requests, 0) + for token_key, complete_key in ( + ("input_tokens", "input_tokens_complete"), + ("output_tokens", "output_tokens_complete"), + ): + if total.get(complete_key) is False: + total[token_key] = None + continue + value = usage.get(token_key) if isinstance(usage, dict) else None + if ambiguous_attempts or isinstance(value, bool) or not isinstance(value, int) or value < 0: + total[token_key] = None + total[complete_key] = False + continue + total[token_key] = int(total.get(token_key) or 0) + value + + +def merge_usage( + total: dict[str, int | bool | None], + addition: dict[str, int | bool | None], +) -> None: + total["requests"] = int(total.get("requests") or 0) + int(addition.get("requests") or 0) + for token_key, complete_key in ( + ("input_tokens", "input_tokens_complete"), + ("output_tokens", "output_tokens_complete"), + ): + if total.get(complete_key) is False or addition.get(complete_key) is False: + total[token_key] = None + total[complete_key] = False + continue + value = addition.get(token_key) + if isinstance(value, int): + total[token_key] = int(total.get(token_key) or 0) + value diff --git a/tests/test_ask_refinement.py b/tests/test_ask_refinement.py new file mode 100644 index 0000000..f3cd17f --- /dev/null +++ b/tests/test_ask_refinement.py @@ -0,0 +1,843 @@ +"""Offline tests for Jev search refinement and bounded sampled-frame inspection.""" + +from __future__ import annotations + +import json +from pathlib import Path +from unittest.mock import MagicMock, patch + +import pytest + +from av.core.config import AVConfig, get_config +from av.db.models import ArtifactRecord, VideoRecord +from av.db.repository import Repository +from av.providers.base import CompletionResult +from av.search.inspection import ( + ExplicitVisionClient, + InspectionWindow, + VisionResponse, + inspect_with_stronger_vision, + sample_window_timestamps, +) +from av.search.rag import _citations, ask +from av.search.refine import ( + RefinementError, + Scene, + SystemOneClient, + _hydrate_scene_text, + _ordered_contiguous_events, + judge_relevance, + merge_overlapping_scenes, + refine_search_results, +) +from av.search.semantic import search + + +def _video(video_id: str, path: Path, duration: float = 120.0) -> VideoRecord: + return VideoRecord( + id=video_id, + file_path=str(path), + file_hash=f"hash-{video_id}", + file_size_bytes=path.stat().st_size if path.exists() else 0, + filename=path.name, + duration_sec=duration, + status="complete", + ) + + +def _artifact( + artifact_id: str, + video_id: str, + start: float, + end: float, + text: str, + artifact_type: str = "caption", +) -> ArtifactRecord: + return ArtifactRecord( + id=artifact_id, + video_id=video_id, + type=artifact_type, + start_sec=start, + end_sec=end, + text=text, + ) + + +@pytest.fixture() +def repo(tmp_path: Path) -> Repository: + return Repository(tmp_path / "av.db") + + +def _seed_video(repo: Repository, tmp_path: Path, video_id: str, prefix: str = "event") -> Path: + path = tmp_path / f"{video_id}.mp4" + path.write_bytes(b"fake-video") + repo.insert_video(_video(video_id, path)) + repo.insert_artifacts_batch([ + _artifact(f"{video_id}-{i}", video_id, i * 10.0, (i + 1) * 10.0, f"{prefix} scene {i}") + for i in range(10) + ]) + return path + + +class FakeSystemOne: + def __init__( + self, + relevance: list[float] | None = None, + *, + support: float = 0.9, + edge_boundaries: bool = False, + usage: dict | None = None, + ) -> None: + self.relevance = list(relevance or []) + self.support = support + self.edge_boundaries = edge_boundaries + self.usage = usage or {} + self.calls: list[tuple[dict, dict]] = [] + self.relevance_offset = 0 + self.boundary_calls = 0 + + def ask(self, state, questions): + self.calls.append((state, questions)) + if "is_supported" in questions: + return {"is_supported": {"type": "noul", "noul": self.support}}, self.usage + if "start" in questions and "end" in questions: + self.boundary_calls += 1 + start_labels = list(questions["start"]["criteria"]) + end_labels = list(questions["end"]["criteria"]) + if self.edge_boundaries and self.boundary_calls == 1: + start = start_labels[0] + end = end_labels[-1] + elif self.edge_boundaries: + start = "e-5" if "e-5" in start_labels else start_labels[0] + end = "e4" if "e4" in end_labels else end_labels[-1] + else: + start = "e0" + end = "e0" + return { + "start": {"type": "choice", "choice": start, "confidence": 0.8}, + "end": {"type": "choice", "choice": end, "confidence": 0.7}, + }, self.usage + answers = {} + for index, key in enumerate(questions): + position = self.relevance_offset + index + probability = self.relevance[position] if position < len(self.relevance) else 1.0 + answers[key] = {"type": "noul", "noul": probability} + self.relevance_offset += len(questions) + return answers, self.usage + + +class FakeLLM: + def __init__(self, config: AVConfig) -> None: + self.config = config + + def complete(self, prompt: str, context: str) -> str: + return "legacy answer" + + def complete_with_usage(self, prompt: str, context: str) -> CompletionResult: + return CompletionResult("refined answer", input_tokens=None, output_tokens=None) + + +def test_temp_sqlite_search_preserves_end_times_and_video_isolation( + repo: Repository, tmp_path: Path +) -> None: + _seed_video(repo, tmp_path, "v1", prefix="door") + _seed_video(repo, tmp_path, "v2", prefix="door") + config = AVConfig(embed_model="") + result = search("door", repo, config, limit=4, video_id="v1") + assert result["results"] + assert {item["video_id"] for item in result["results"]} == {"v1"} + assert all(item["end_sec"] == item["timestamp_sec"] + 10 for item in result["results"]) + + +def test_relevance_is_batched_ten_and_rejects_bad_probabilities() -> None: + results = [ + {"artifact_id": f"a{i}", "video_id": "v", "timestamp_sec": i, "text": "x", "source_type": "caption"} + for i in range(11) + ] + fake = FakeSystemOne([0.5] * 11) + probabilities, usage = judge_relevance(fake, "query", results) + assert len(probabilities) == 11 + assert len(fake.calls) == 2 + assert usage["requests"] == 2 + + for bad in (-0.1, 1.1, "0.9", None): + with pytest.raises(RefinementError): + judge_relevance(FakeSystemOne([bad]), "query", results[:1]) + + +def test_boundary_window_is_single_pass_and_configurable( + repo: Repository, tmp_path: Path +) -> None: + _seed_video(repo, tmp_path, "v1") + raw = search("event", repo, AVConfig(embed_model=""), limit=1, video_id="v1")["results"] + # Put the hit in the middle so ±6 has real room on both sides. + raw[0] = repo.search_fts("scene 5", limit=1, video_id="v1")[0].model_dump() + fake = FakeSystemOne([0.9], edge_boundaries=True) + refined, meta, _ = refine_search_results( + "event", + raw, + repo, + AVConfig(typesafe_api_key="test", refine_context_events=4), + client=fake, + ) + assert fake.boundary_calls == 1 + boundary_questions = next(questions for _, questions in fake.calls if "start" in questions) + assert len(boundary_questions["start"]["criteria"]) <= 5 + assert len(boundary_questions["end"]["criteria"]) <= 5 + assert meta["scene_count"] == 1 + + +def test_overlap_merge_is_same_video_only_and_ranks_probability_times_score() -> None: + def scene(artifact_id: str, video: str, start: float, end: float, score: float, p: float) -> Scene: + return Scene( + artifact_id=artifact_id, + video_id=video, + filename=f"{video}.mp4", + source_type="caption", + start_sec=start, + end_sec=end, + text=artifact_id, + retrieval_score=score, + relevance_p=p, + chunk_start_sec=start, + chunk_end_sec=end, + merged_artifact_ids=[artifact_id], + ) + + merged, count = merge_overlapping_scenes([ + scene("a", "v1", 0, 20, 100, 0.5), + scene("b", "v1", 10, 30, 60, 1.0), + scene("c", "v2", 10, 30, 1.0, 1.0), + ]) + assert count == 1 + assert len(merged) == 2 + v1 = next(item for item in merged if item.video_id == "v1") + assert v1.artifact_id == "a" + assert v1.relevance_p == 1.0 + assert v1.rank_score == 100 + assert set(v1.merged_artifact_ids) == {"a", "b"} + + +def test_refinement_caps_to_top_eight_scenes(repo: Repository, tmp_path: Path) -> None: + raw: list[dict] = [] + for index in range(9): + video_id = f"v{index}" + path = tmp_path / f"{video_id}.mp4" + path.write_bytes(b"fake") + repo.insert_video(_video(video_id, path)) + artifact = _artifact(f"a{index}", video_id, 0, 10, f"target {index}") + repo.insert_artifact(artifact) + raw.append({ + "rank": index + 1, + "score": 0.1 + index / 10, + "video_id": video_id, + "filename": path.name, + "timestamp_sec": 0.0, + "end_sec": 10.0, + "timestamp_formatted": "00:00:00", + "source_type": "caption", + "text": artifact.text, + "artifact_id": artifact.id, + }) + refined, meta, _ = refine_search_results( + "target", + raw, + repo, + AVConfig(typesafe_api_key="test", refine_max_scenes=8), + client=FakeSystemOne([1.0] * 9), + ) + assert len(refined) == 8 + assert meta["capped_count"] == 1 + assert refined[0]["video_id"] == "v8" + + +def test_whole_video_summary_stays_broad_and_does_not_merge_with_local_scene( + repo: Repository, tmp_path: Path +) -> None: + path = tmp_path / "v1.mp4" + path.write_bytes(b"fake") + repo.insert_video(_video("v1", path, duration=100.0)) + repo.insert_artifacts_batch([ + _artifact("summary", "v1", 0, 100, "Whole-video overview", "summary"), + _artifact("local", "v1", 10, 20, "Person opens the door", "caption"), + ]) + raw = [item.model_dump() for item in repo.search_fts("overview OR door", limit=10, video_id="v1")] + refined, meta, _ = refine_search_results( + "door", + raw, + repo, + AVConfig(typesafe_api_key="test", refine_context_events=2), + client=FakeSystemOne([0.9, 0.9]), + ) + assert meta["merged_count"] == 0 + assert len(refined) == 2 + broad = next(item for item in refined if item["artifact_id"] == "summary") + local = next(item for item in refined if item["artifact_id"] == "local") + assert broad["evidence_scope"] == "broad" + assert broad["text"] == "Whole-video overview" + assert broad["scene_confidence"] is None + assert local["evidence_scope"] == "scene" + assert "Person opens the door" in local["text"] + + +def test_temporal_events_attach_interleaved_modalities_without_false_gap() -> None: + artifacts = [ + _artifact("cap0", "v", 0, 10, "caption zero", "caption"), + _artifact("t0", "v", 0, 1, "hello", "transcript"), + _artifact("t8", "v", 8, 9, "there", "transcript"), + _artifact("cap10", "v", 10, 20, "caption ten", "caption"), + ] + events, hit_index = _ordered_contiguous_events( + { + "artifact_id": "cap10", + "timestamp_sec": 10, + "end_sec": 20, + "source_type": "caption", + "text": "caption ten", + }, + artifacts, + ) + assert [(event["start"], event["end"]) for event in events] == [(0, 10), (10, 20)] + assert hit_index == 1 + assert "hello" in events[0]["text"] + assert "there" in events[0]["text"] + + +def test_strict_hydration_excludes_touching_chunks_and_preserves_hit_texts( + repo: Repository, tmp_path: Path +) -> None: + path = tmp_path / "v.mp4" + path.write_bytes(b"fake") + repo.insert_video(_video("v", path)) + repo.insert_artifacts_batch([ + _artifact("before", "v", 0, 10, "outside before"), + _artifact("hit", "v", 10, 20, "primary hit"), + _artifact("merged", "v", 12, 14, "merged hit", "transcript"), + _artifact("inside", "v", 15, 16, "surrounding inside", "transcript"), + _artifact("after", "v", 20, 30, "outside after"), + ]) + scene = Scene( + artifact_id="hit", + video_id="v", + filename="v.mp4", + source_type="caption", + start_sec=10, + end_sec=20, + text="primary hit", + retrieval_score=1, + relevance_p=1, + chunk_start_sec=10, + chunk_end_sec=20, + merged_artifact_ids=["hit", "merged"], + hit_texts=["primary hit", "merged hit"], + ) + _hydrate_scene_text(scene, repo) + assert scene.text.startswith("[retrieved hit] primary hit\n[retrieved hit] merged hit") + assert "surrounding inside" in scene.text + assert "outside before" not in scene.text + assert "outside after" not in scene.text + + +def test_citations_preserve_refinement_provenance() -> None: + citation = _citations([{ + "video_id": "v", + "timestamp_sec": 10, + "end_sec": 20, + "source_type": "caption", + "text": "hit", + "score": 0.7, + "artifact_id": "a", + "chunk_start_sec": 12, + "chunk_end_sec": 14, + "scene_confidence": 0.8, + "merged_artifact_ids": ["a", "b"], + "relevance_p": 0.9, + "evidence_scope": "scene", + }])[0] + assert citation["artifact_id"] == "a" + assert citation["chunk_start_sec"] == 12 + assert citation["chunk_end_sec"] == 14 + assert citation["scene_confidence"] == 0.8 + assert citation["merged_artifact_ids"] == ["a", "b"] + assert citation["relevance_p"] == 0.9 + + +def test_valid_all_irrelevant_is_no_results_not_raw_fallback(repo: Repository, tmp_path: Path) -> None: + _seed_video(repo, tmp_path, "v1", prefix="cake") + fake = FakeSystemOne([0.1] * 30) + config = AVConfig(typesafe_api_key="test", embed_model="", refine_relevance_min=0.5) + with patch("av.search.rag.SystemOneClient", return_value=fake), \ + patch("av.search.rag.OpenAILLM", side_effect=AssertionError("answer model must not run")): + result = ask("cake", repo, config, video_id="v1") + assert result["route"] == "refined_no_results" + assert result["evidence_status"] == "all_sources_irrelevant" + assert result["citations"] == [] + assert not result["warnings"] + + +def test_fast_path_support_skips_vision_and_usage_stays_unknown( + repo: Repository, tmp_path: Path +) -> None: + _seed_video(repo, tmp_path, "v1", prefix="cake") + fake = FakeSystemOne([0.9] * 30, support=0.88) + config = AVConfig(typesafe_api_key="test", embed_model="") + with patch("av.search.rag.SystemOneClient", return_value=fake), \ + patch("av.search.rag.OpenAILLM", FakeLLM), \ + patch("av.search.rag.inspect_with_stronger_vision") as inspect: + result = ask("cake", repo, config, video_id="v1") + inspect.assert_not_called() + assert result["route"] == "refined" + assert result["confidence"] == pytest.approx(0.88) + assert result["stage_usage"]["answer"]["input_tokens"] is None + assert result["stage_usage"]["relevance"]["input_tokens"] is None + + +def test_relevant_sources_can_still_fail_answer_support_and_escalate( + repo: Repository, tmp_path: Path +) -> None: + _seed_video(repo, tmp_path, "v1", prefix="cake") + fake = FakeSystemOne([0.9] * 30, support=0.1) + inspection = { + "status": "insufficient", + "answer": None, + "citations": [], + "windows": [{"video_id": "v1", "start_sec": 0, "end_sec": 10}], + "usage": {"requests": 1, "input_tokens": None, "output_tokens": None}, + "warnings": [], + } + with patch("av.search.rag.SystemOneClient", return_value=fake), \ + patch("av.search.rag.OpenAILLM", FakeLLM), \ + patch("av.search.rag.inspect_with_stronger_vision", return_value=inspection) as inspect: + result = ask("cake", repo, AVConfig(typesafe_api_key="test", embed_model=""), video_id="v1") + inspect.assert_called_once() + assert result["route"] == "refined_uncertain" + assert result["evidence_status"] == "unsupported" + assert "did not support" in result["answer"] + + +def test_refinement_outage_falls_back_raw_without_secret_leak( + repo: Repository, tmp_path: Path +) -> None: + _seed_video(repo, tmp_path, "v1", prefix="cake") + config = AVConfig(typesafe_api_key="top-secret", embed_model="") + with patch("av.search.rag.SystemOneClient"), \ + patch("av.search.rag.refine_search_results", side_effect=RefinementError("top-secret private input")), \ + patch("av.search.rag.OpenAILLM", FakeLLM): + result = ask("cake", repo, config, video_id="v1") + encoded = json.dumps(result) + assert result["route"] == "refinement_fallback" + assert result["evidence_status"] == "raw_unjudged" + assert "top-secret" not in encoded + assert "private input" not in encoded + + +def test_system_one_http_error_is_sanitized() -> None: + response = MagicMock(status_code=401, ok=False) + response.text = "secret-key and raw private state" + session = MagicMock() + session.post.return_value = response + client = SystemOneClient( + AVConfig(typesafe_api_key="secret-key", typesafe_max_retries=0), session=session + ) + with pytest.raises(RefinementError) as exc: + client.ask({"private": "state"}, {"q": {"type": "noul", "instructions": "x"}}) + assert "401" in str(exc.value) + assert "secret-key" not in str(exc.value) + assert "private state" not in str(exc.value) + + +@pytest.mark.parametrize("payload", [[], None, "invalid-root"]) +def test_system_one_non_object_json_is_sanitized(payload) -> None: + response = MagicMock(status_code=200, ok=True) + response.json.return_value = payload + session = MagicMock() + session.post.return_value = response + client = SystemOneClient( + AVConfig(typesafe_api_key="secret", typesafe_max_retries=0), + session=session, + ) + with pytest.raises(RefinementError, match="invalid JSON document") as exc: + client.ask({"private": "state"}, {"q": {"type": "noul", "instructions": "x"}}) + assert "secret" not in str(exc.value) + + +def test_partial_relevance_usage_is_preserved_when_later_retry_fails() -> None: + class PartialFailureClient: + calls = 0 + + def ask(self, state, questions): + self.calls += 1 + if self.calls == 1: + return ( + {key: {"type": "noul", "noul": 0.9} for key in questions}, + {"input_tokens": 10, "output_tokens": 2}, + ) + raise RefinementError("unavailable", attempts=2) + + results = [ + {"artifact_id": f"a{i}", "video_id": "v", "timestamp_sec": i, "text": "x", "source_type": "caption"} + for i in range(11) + ] + with pytest.raises(RefinementError) as exc: + judge_relevance(PartialFailureClient(), "query", results, batch_size=10) + usage = exc.value.stage_usage["relevance"] + assert usage["requests"] == 3 + assert usage["input_tokens"] is None + assert usage["output_tokens"] is None + assert usage["input_tokens_complete"] is False + + +def test_refinement_fallback_returns_partial_stage_usage(repo: Repository, tmp_path: Path) -> None: + _seed_video(repo, tmp_path, "v1", prefix="cake") + partial = { + "requests": 2, + "input_tokens": None, + "output_tokens": None, + "input_tokens_complete": False, + "output_tokens_complete": False, + } + error = RefinementError("failed", stage_usage={"relevance": partial}) + with patch("av.search.rag.SystemOneClient"), \ + patch("av.search.rag.refine_search_results", side_effect=error), \ + patch("av.search.rag.OpenAILLM", FakeLLM): + result = ask( + "cake", + repo, + AVConfig(typesafe_api_key="test", embed_model=""), + video_id="v1", + ) + assert result["stage_usage"]["relevance"]["requests"] == 2 + assert result["stage_usage"]["relevance"]["input_tokens_complete"] is False + + +def test_full_window_timestamp_plan_covers_start_and_end() -> None: + windows = [ + InspectionWindow("v1", "a.mp4", Path("a.mp4"), 10.0, 40.0, 10.0, 40.0), + InspectionWindow("v2", "b.mp4", Path("b.mp4"), 100.0, 120.0, 100.0, 120.0), + ] + plan = sample_window_timestamps(windows, 8) + first, second = plan.values() + assert first[0] == 10.0 + assert first[-1] == pytest.approx(39.999) + assert second[0] == 100.0 + assert second[-1] == pytest.approx(119.999) + assert sum(len(values) for values in plan.values()) == 8 + + +def test_inspection_handles_provider_media_frames_and_budget_failures( + repo: Repository, tmp_path: Path +) -> None: + missing = tmp_path / "missing.mp4" + repo.insert_video(_video("v1", missing)) + result = [{"video_id": "v1", "timestamp_sec": 0.0, "end_sec": 10.0}] + + not_configured = inspect_with_stronger_vision("q", "a", result, repo, AVConfig()) + assert not_configured["status"] == "not_configured" + + missing_media = inspect_with_stronger_vision( + "q", + "a", + result, + repo, + AVConfig(strong_vision_api_base_url="http://example/v1", strong_vision_model="model"), + ) + assert missing_media["status"] == "unavailable" + + path = tmp_path / "v2.mp4" + path.write_bytes(b"fake") + repo.insert_video(_video("v2", path)) + available = [{"video_id": "v2", "timestamp_sec": 0.0, "end_sec": 10.0}] + budget = inspect_with_stronger_vision( + "q", + "a", + available, + repo, + AVConfig( + strong_vision_api_base_url="http://example/v1", + strong_vision_model="model", + inspection_max_frames=0, + ), + ) + assert budget["status"] == "budget_exhausted" + + empty_frames = MagicMock(paths=[], timestamps=[]) + with patch("av.search.inspection.sample_at", return_value=empty_frames): + empty = inspect_with_stronger_vision( + "q", + "a", + available, + repo, + AVConfig( + strong_vision_api_base_url="http://example/v1", + strong_vision_model="model", + inspection_max_frames=4, + ), + ) + assert empty["status"] == "insufficient" + assert any("no usable" in warning for warning in empty["warnings"]) + + +def test_inspection_validates_absolute_timestamped_evidence( + repo: Repository, tmp_path: Path +) -> None: + path = _seed_video(repo, tmp_path, "v1") + requested: list[float] = [] + + def fake_sample(video_path, timestamps, **kwargs): + requested.extend(timestamps) + frames = [] + for index, timestamp in enumerate(timestamps): + frame = Path(kwargs["out_dir"]) / f"{index}.jpg" + frame.parent.mkdir(parents=True, exist_ok=True) + frame.write_bytes(b"jpg") + frames.append(frame) + return MagicMock(paths=frames, timestamps=timestamps) + + class FakeVLM: + def __init__(self, *args, **kwargs): + pass + + def ask(self, images, prompt): + assert "absolute_time" in prompt + return MagicMock( + ok=True, + text=json.dumps({ + "supported": True, + "answer": "A person enters.", + "evidence": [{"video_id": "v1", "timestamp_sec": requested[0], "description": "Person in doorway"}], + }), + input_tokens=None, + output_tokens=None, + ) + + config = AVConfig( + strong_vision_api_base_url="http://example/v1", + strong_vision_model="strong-model", + inspection_max_frames=6, + ) + with patch("av.search.inspection.sample_at", side_effect=fake_sample): + out = inspect_with_stronger_vision( + "Who enters?", + "Unknown", + [{"video_id": "v1", "timestamp_sec": 10.0, "end_sec": 40.0}], + repo, + config, + provider_factory=FakeVLM, + ) + assert path.exists() + assert out["status"] == "supported" + assert out["citations"][0]["source_type"] == "sampled_frame_inspection" + assert min(requested) == 10.0 + assert max(requested) == pytest.approx(39.999) + + +def test_explicit_vision_client_uses_only_supplied_credential_and_one_attempt( + tmp_path: Path, +) -> None: + frame = tmp_path / "frame.jpg" + frame.write_bytes(b"jpg") + response = MagicMock(ok=False, status_code=503) + session = MagicMock() + session.post.return_value = response + config = AVConfig( + strong_vision_api_base_url="https://public.example/v1", + strong_vision_api_key="", + strong_vision_model="cheap-vlm", + ) + client = ExplicitVisionClient(config, session=session) + result = client.ask([frame], "inspect") + assert result.ok is False + session.post.assert_called_once() + kwargs = session.post.call_args.kwargs + assert kwargs["headers"] == {"Content-Type": "application/json"} + assert session.post.call_args.args[0] == "https://public.example/v1/chat/completions" + + +def test_inspection_provider_initialization_error_is_sanitized( + repo: Repository, tmp_path: Path +) -> None: + _seed_video(repo, tmp_path, "v1") + + class BrokenProvider: + def __init__(self, config): + raise RuntimeError("https://private.example secret-token") + + out = inspect_with_stronger_vision( + "q", + "a", + [{"video_id": "v1", "timestamp_sec": 0.0, "end_sec": 10.0}], + repo, + AVConfig( + strong_vision_api_base_url="https://private.example/v1", + strong_vision_api_key="secret-token", + strong_vision_model="model", + ), + provider_factory=BrokenProvider, + ) + encoded = json.dumps(out) + assert out["status"] == "unavailable" + assert "private.example" not in encoded + assert "secret-token" not in encoded + assert out["usage"]["requests"] == 0 + + +def test_inspection_rejects_unsampled_timestamp_and_reports_truncation( + repo: Repository, tmp_path: Path +) -> None: + _seed_video(repo, tmp_path, "v1") + + def fake_sample(video_path, timestamps, **kwargs): + frame = Path(kwargs["out_dir"]) / "one.jpg" + frame.parent.mkdir(parents=True, exist_ok=True) + frame.write_bytes(b"jpg") + return MagicMock(paths=[frame], timestamps=[timestamps[0]]) + + class UnsampledProvider: + def __init__(self, config): + pass + + def ask(self, images, prompt): + assert "do not establish what happened between" in prompt + return VisionResponse( + ok=True, + text=json.dumps({ + "supported": True, + "answer": "unsupported timestamp", + "evidence": [{ + "video_id": "v1", + "timestamp_sec": 15.0, + "description": "not actually sampled", + }], + }), + input_tokens=5, + output_tokens=2, + ) + + config = AVConfig( + strong_vision_api_base_url="https://public.example/v1", + strong_vision_model="cheap-vlm", + inspection_max_seconds=5, + inspection_max_frames=4, + ) + with patch("av.search.inspection.sample_at", side_effect=fake_sample): + out = inspect_with_stronger_vision( + "q", + "a", + [{"video_id": "v1", "timestamp_sec": 10.0, "end_sec": 40.0}], + repo, + config, + provider_factory=UnsampledProvider, + ) + assert out["status"] == "insufficient" + window = out["windows"][0] + assert window["requested_start_sec"] == 10.0 + assert window["requested_end_sec"] == 40.0 + assert window["start_sec"] == 10.0 + assert window["end_sec"] == 15.0 + assert window["truncated"] is True + assert window["all_requested_frames_extracted"] is False + + +def test_inspection_attempt_cap_and_partial_usage_are_truthful( + repo: Repository, tmp_path: Path +) -> None: + _seed_video(repo, tmp_path, "v1") + calls = 0 + + def fake_sample(video_path, timestamps, **kwargs): + frames = [] + for index, _ in enumerate(timestamps): + frame = Path(kwargs["out_dir"]) / f"{index}.jpg" + frame.parent.mkdir(parents=True, exist_ok=True) + frame.write_bytes(b"jpg") + frames.append(frame) + return MagicMock(paths=frames, timestamps=timestamps) + + class TwoAttemptProvider: + def __init__(self, config): + pass + + def ask(self, images, prompt): + nonlocal calls + calls += 1 + return VisionResponse( + ok=True, + text='{"supported": false, "answer": "", "evidence": []}', + input_tokens=10 if calls == 1 else None, + output_tokens=1 if calls == 1 else None, + ) + + config = AVConfig( + strong_vision_api_base_url="https://public.example/v1", + strong_vision_model="cheap-vlm", + inspection_max_frames=8, + inspection_max_attempts=2, + inspection_dense_pass=True, + ) + with patch("av.search.inspection.sample_at", side_effect=fake_sample): + out = inspect_with_stronger_vision( + "q", + "a", + [{"video_id": "v1", "timestamp_sec": 10.0, "end_sec": 40.0}], + repo, + config, + provider_factory=TwoAttemptProvider, + ) + assert calls == 2 + assert out["usage"]["requests"] == 2 + assert out["usage"]["input_tokens"] is None + assert out["usage"]["input_tokens_complete"] is False + + +def test_broad_evidence_does_not_drive_inspection(repo: Repository, tmp_path: Path) -> None: + _seed_video(repo, tmp_path, "v1") + config = AVConfig( + strong_vision_api_base_url="https://public.example/v1", + strong_vision_model="cheap-vlm", + inspection_max_windows=1, + inspection_max_frames=0, + ) + out = inspect_with_stronger_vision( + "q", + "a", + [ + {"video_id": "v1", "timestamp_sec": 0.0, "end_sec": 100.0, "evidence_scope": "broad"}, + {"video_id": "v1", "timestamp_sec": 20.0, "end_sec": 30.0, "evidence_scope": "scene"}, + ], + repo, + config, + ) + assert out["status"] == "budget_exhausted" + assert any("Broad summary" in warning for warning in out["warnings"]) + + +def test_config_file_env_priority_and_secret_fields(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + config_file = tmp_path / "config.json" + config_file.write_text(json.dumps({ + "typesafe_api_key": "file-key", + "typesafe_model": "jev-file", + "refine_relevance_min": 0.7, + "strong_vision_api_key": "vision-file-key", + })) + monkeypatch.setattr("av.core.config.CONFIG_FILE_PATH", config_file) + monkeypatch.setenv("AV_TYPESAFE_MODEL", "jev-env") + monkeypatch.setenv("TYPESAFE_API_KEY", "env-key") + config = get_config() + assert config.typesafe_api_key == "env-key" + assert config.typesafe_model == "jev-env" + assert config.refine_relevance_min == pytest.approx(0.7) + assert config.strong_vision_api_key == "vision-file-key" + + +def test_no_refine_preserves_legacy_contract(repo: Repository, tmp_path: Path) -> None: + _seed_video(repo, tmp_path, "v1", prefix="cake") + with patch("av.search.rag.OpenAILLM", FakeLLM): + result = ask( + "cake", + repo, + AVConfig(typesafe_api_key="configured", embed_model=""), + video_id="v1", + refine=False, + ) + assert set(result) == {"answer", "citations", "confidence"} + assert result["answer"] == "legacy answer" diff --git a/tests/test_config.py b/tests/test_config.py index 47ed68f..c211d5f 100644 --- a/tests/test_config.py +++ b/tests/test_config.py @@ -9,7 +9,7 @@ import pytest -from av.core.config import AVConfig, _load_config_file, get_config, save_config +from av.core.config import AVConfig, _load_config_file, get_config, get_openai_config, save_config from av.core.constants import CONFIG_FILE_PATH, PROVIDER_PRESETS @@ -101,6 +101,25 @@ def test_env_var_overrides_config_file(tmp_path: Path, monkeypatch: pytest.Monke assert config.provider == "anthropic" +def test_openai_fallback_does_not_read_oauth_unless_enabled() -> None: + config = AVConfig(provider="anthropic", openai_api_key="", allow_oauth_fallback=False) + with patch("av.providers.openai._openclaw_oauth_token") as openclaw, \ + patch("av.providers.openai._codex_oauth_token") as codex: + assert get_openai_config(config) is None + openclaw.assert_not_called() + codex.assert_not_called() + + +def test_openai_fallback_can_use_oauth_only_when_explicitly_enabled() -> None: + config = AVConfig(provider="anthropic", openai_api_key="", allow_oauth_fallback=True) + with patch("av.providers.openai._openclaw_oauth_token", return_value="oauth-explicit"), \ + patch("av.providers.openai._codex_oauth_token") as codex: + fallback = get_openai_config(config) + assert fallback is not None + assert fallback.api_key == "oauth-explicit" + codex.assert_not_called() + + def test_db_path_override() -> None: custom = Path("/tmp/test.db") config = get_config(db_path=custom) diff --git a/tests/test_ingest_usage.py b/tests/test_ingest_usage.py new file mode 100644 index 0000000..5078e3f --- /dev/null +++ b/tests/test_ingest_usage.py @@ -0,0 +1,181 @@ +"""Offline receipts for API-only ingestion stages.""" + +from __future__ import annotations + +from dataclasses import dataclass +import json +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import patch + +import pytest + +from av.core.config import AVConfig +from av.db.repository import Repository +from av.pipeline.ingest import ingest_video +from av.pipeline.transcript_sidecar import TranscriptSidecarError +from av.db.models import VideoRecord +from av.providers.base import Caption +from av.providers.usage import ProviderUsage + + +@dataclass +class _Meta: + duration_sec: float = 20.0 + width: int = 640 + height: int = 360 + fps: float = 24.0 + codec: str = "h264" + bitrate: int = 1_000_000 + file_size_bytes: int = 4 + + +class _Captioner: + def __init__(self, config: AVConfig) -> None: + self.usage = ProviderUsage() + + def caption_frames(self, frame_paths, timestamps, prompt=None): + for _ in frame_paths: + usage = SimpleNamespace( + prompt_tokens=11, + completion_tokens=3, + prompt_tokens_details=SimpleNamespace(cached_tokens=2), + ) + self.usage.record_success(usage) + return [ + Caption(timestamp_sec=timestamp, text=f"frame at {timestamp}", frame_path=str(path)) + for path, timestamp in zip(frame_paths, timestamps) + ] + + +def test_dense_ingest_reports_actual_usage_and_effective_budgets(tmp_path: Path) -> None: + video = tmp_path / "video.mp4" + video.write_bytes(b"fake") + frames_dir = tmp_path / "frames" + frames_dir.mkdir() + frame_a = frames_dir / "frame_000001.jpg" + frame_b = frames_dir / "frame_000002.jpg" + frame_a.write_bytes(b"jpg") + frame_b.write_bytes(b"jpg") + repo = Repository(tmp_path / "av.db") + config = AVConfig( + provider="openai", + api_key="explicit", + transcribe_model="", + embed_model="", + api_timeout_sec=19, + api_max_retries=1, + allow_oauth_fallback=False, + allow_codex_fallback=False, + ) + with patch("av.pipeline.ingest.get_video_info", return_value=_Meta()), \ + patch("av.pipeline.ingest.extract_frames", return_value=[(frame_a, 0.0), (frame_b, 10.0)]), \ + patch("av.pipeline.ingest.OpenAICaptioner", _Captioner), \ + patch("av.pipeline.ingest.export_dense_outputs"): + result = ingest_video( + video, + repo, + config, + dense_vision=True, + no_embed=True, + max_frames=2, + dense_output_dir=tmp_path / "dense", + ) + usage = result["stage_usage"]["caption"] + assert usage["requests"] == 2 + assert usage["successful_requests"] == 2 + assert usage["failed_requests"] == 0 + assert usage["input_tokens"] == 22 + assert usage["output_tokens"] == 6 + assert usage["cached_input_tokens"] == 4 + assert usage["input_tokens_complete"] is True + settings = result["ingest_settings"] + assert settings["api_timeout_sec"] == 19 + assert settings["api_max_retries"] == 1 + assert settings["allow_oauth_fallback"] is False + assert settings["allow_codex_fallback"] is False + assert settings["caption_concurrency"] == 1 + assert settings["max_frames"] == 2 + assert settings["dense_caption_frames"] == 2 + + +def test_provider_usage_keeps_unknown_dimensions_null_after_missing_usage() -> None: + usage = ProviderUsage() + usage.record_success(SimpleNamespace(prompt_tokens=5, completion_tokens=2)) + usage.record_success(None) + receipt = usage.snapshot() + assert receipt["requests"] == 2 + assert receipt["input_tokens"] is None + assert receipt["output_tokens"] is None + assert receipt["cached_input_tokens"] is None + assert receipt["input_tokens_complete"] is False + assert receipt["cached_input_tokens_complete"] is False + + +def test_transcript_sidecar_bypasses_builtin_asr_and_uses_normal_artifacts(tmp_path: Path) -> None: + video = tmp_path / "video.mp4" + video.write_bytes(b"fake") + sidecar = tmp_path / "transcript.json" + sidecar.write_text(json.dumps({ + "model": "external-cheap-asr", + "provenance": {"method": "public-script"}, + "segments": [ + {"start_sec": 1.0, "end_sec": 2.5, "text": " Exact accepted text "}, + ], + })) + repo = Repository(tmp_path / "av.db") + config = AVConfig( + api_key="explicit", + transcribe_model="whisper-1", + embed_model="", + ) + with patch("av.pipeline.ingest.get_video_info", return_value=_Meta()), \ + patch("av.pipeline.ingest.extract_audio", side_effect=AssertionError("ASR audio must not run")), \ + patch("av.pipeline.ingest.OpenAITranscriber", side_effect=AssertionError("ASR provider must not run")): + result = ingest_video( + video, + repo, + config, + transcript_json=sidecar, + no_embed=True, + ) + artifacts = repo.get_artifacts(result["video_id"], "transcript") + assert len(artifacts) == 1 + assert artifacts[0].start_sec == 1.0 + assert artifacts[0].end_sec == 2.5 + assert artifacts[0].text == " Exact accepted text " + assert json.loads(artifacts[0].meta_json) == { + "model": "external-cheap-asr", + "provenance": {"method": "public-script"}, + } + assert result["stage_usage"]["transcription"]["requests"] == 0 + assert result["ingest_settings"]["transcription_source"] == "sidecar" + + +def test_invalid_sidecar_is_validated_before_force_deletes_existing_video(tmp_path: Path) -> None: + video = tmp_path / "video.mp4" + video.write_bytes(b"fake") + sidecar = tmp_path / "bad.json" + sidecar.write_text('{"segments":[{"start_sec":0,"end_sec":99,"text":"too long"}]}') + repo = Repository(tmp_path / "av.db") + repo.insert_video(VideoRecord( + id="existing", + file_path=str(video), + file_hash="same-hash", + file_size_bytes=4, + filename=video.name, + duration_sec=20, + status="complete", + )) + with patch("av.pipeline.ingest.file_hash", return_value="same-hash"), \ + patch("av.pipeline.ingest.get_video_info", return_value=_Meta()): + with pytest.raises(TranscriptSidecarError): + ingest_video( + video, + repo, + AVConfig(transcribe_model="", embed_model=""), + transcript_json=sidecar, + force=True, + no_embed=True, + ) + assert repo.get_video("existing").id == "existing" diff --git a/tests/test_provider.py b/tests/test_provider.py index d1ea820..9fb8cba 100644 --- a/tests/test_provider.py +++ b/tests/test_provider.py @@ -2,10 +2,14 @@ from __future__ import annotations -from unittest.mock import patch +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import MagicMock, patch + +import pytest from av.core.config import AVConfig -from av.providers.openai import _client, _resolve_api_key +from av.providers.openai import OpenAICaptioner, _client, _resolve_api_key def test_client_default_no_extra_headers() -> None: @@ -41,7 +45,76 @@ def test_resolve_api_key_no_key_placeholder() -> None: def test_resolve_api_key_oauth_fallback() -> None: - config = AVConfig(api_key="") + config = AVConfig(api_key="", allow_oauth_fallback=True) with patch("av.providers.openai._openclaw_oauth_token", return_value="oauth-token-123"), \ patch("av.providers.openai._codex_oauth_token", return_value=None): assert _resolve_api_key(config) == "oauth-token-123" + + +def test_resolve_api_key_default_does_not_read_oauth() -> None: + config = AVConfig(api_key="", allow_oauth_fallback=False) + with patch("av.providers.openai._openclaw_oauth_token") as openclaw, \ + patch("av.providers.openai._codex_oauth_token") as codex: + assert _resolve_api_key(config) == "no-key" + openclaw.assert_not_called() + codex.assert_not_called() + + +def test_client_disables_sdk_retries_and_uses_configured_timeout() -> None: + config = AVConfig( + api_key="explicit", + api_timeout_sec=17.0, + api_max_retries=2, + ) + client = _client(config) + assert client.max_retries == 0 + assert client.timeout == 17.0 + + +def test_caption_fallback_is_disabled_by_default(tmp_path: Path) -> None: + frame = tmp_path / "frame.jpg" + frame.write_bytes(b"jpg") + fake_client = MagicMock() + fake_client.chat.completions.create.side_effect = RuntimeError("model_not_found") + config = AVConfig( + api_key="explicit", + api_max_retries=0, + allow_codex_fallback=False, + ) + with patch("av.providers.openai._client", return_value=fake_client), \ + patch("av.providers.openai._codex_cli_caption") as codex: + captioner = OpenAICaptioner(config) + assert captioner.caption_frames([frame], [0.0]) == [] + with pytest.raises(RuntimeError, match="model_not_found"): + captioner.caption_chunk([frame], [0.0], "describe") + codex.assert_not_called() + assert captioner.usage.snapshot()["requests"] == 2 + + +def test_caption_retries_are_bounded_and_usage_marks_failed_attempt_unknown(tmp_path: Path) -> None: + frame = tmp_path / "frame.jpg" + frame.write_bytes(b"jpg") + usage = SimpleNamespace( + prompt_tokens=12, + completion_tokens=4, + prompt_tokens_details=SimpleNamespace(cached_tokens=3), + ) + response = SimpleNamespace( + choices=[SimpleNamespace(message=SimpleNamespace(content="caption"))], + usage=usage, + ) + fake_client = MagicMock() + fake_client.chat.completions.create.side_effect = [RuntimeError("temporary"), response] + config = AVConfig(api_key="explicit", api_max_retries=1) + with patch("av.providers.openai._client", return_value=fake_client), \ + patch("av.providers.openai.time.sleep"): + captioner = OpenAICaptioner(config) + assert captioner.caption_chunk([frame], [0.0], "describe") == "caption" + receipt = captioner.usage.snapshot() + assert fake_client.chat.completions.create.call_count == 2 + assert receipt["requests"] == 2 + assert receipt["failed_requests"] == 1 + assert receipt["successful_requests"] == 1 + assert receipt["input_tokens"] is None + assert receipt["input_tokens_complete"] is False + assert receipt["cached_input_tokens"] is None diff --git a/tests/test_transcript_sidecar.py b/tests/test_transcript_sidecar.py new file mode 100644 index 0000000..8a03725 --- /dev/null +++ b/tests/test_transcript_sidecar.py @@ -0,0 +1,223 @@ +"""Offline transcript-sidecar validation and CLI wiring tests.""" + +from dataclasses import FrozenInstanceError +import json +import os +from pathlib import Path +from unittest.mock import Mock + +import pytest +import typer +from typer.testing import CliRunner + +import av.cli.ingest as ingest_cli +from av.core.exceptions import IngestError +from av.pipeline.transcript_sidecar import load_transcript_sidecar +from av.pipeline.transcript_sidecar import TranscriptSidecarError + + +VALID_SEGMENT = {"start_sec": 0, "end_sec": 2.4, "text": "Hello"} + + +def write_sidecar(tmp_path: Path, data: object) -> Path: + path = tmp_path / "transcript.json" + path.write_text(json.dumps(data), encoding="utf-8") + return path + + +def test_object_preserves_segments_and_metadata(tmp_path): + segments = [ + {"start_sec": 5, "end_sec": 10, "text": " Second passage.\n"}, + {"start_sec": 0.125, "end_sec": 2.75, "text": "First passage — café"}, + ] + provenance = {"source": "public-example", "settings": {"language": "en"}, "tags": [1, True, None]} + path = write_sidecar(tmp_path, {"segments": segments, "model": "external-asr", "provenance": provenance}) + + result = load_transcript_sidecar(path, duration_sec=10) + + assert [(s.start_sec, s.end_sec, s.text) for s in result.segments] == [ + (s["start_sec"], s["end_sec"], s["text"]) for s in segments + ] + assert isinstance(result.segments, tuple) + assert result.model == "external-asr" + assert result.provenance == provenance + with pytest.raises(FrozenInstanceError): + result.segments[0].text = "changed" + + +def test_simple_segment_list(tmp_path): + result = load_transcript_sidecar(write_sidecar(tmp_path, [VALID_SEGMENT]), duration_sec=2.4) + assert result.segments[0].end_sec == 2.4 + assert result.model is None + assert result.provenance is None + + +@pytest.mark.parametrize("root", [[], {"segments": []}]) +def test_empty_segments_are_valid_for_silence(tmp_path, root): + assert load_transcript_sidecar(write_sidecar(tmp_path, root), duration_sec=10).segments == () + + +@pytest.mark.parametrize("duration", [0, -1, True, False, "10", None, float("nan"), float("inf"), 10 ** 500]) +def test_invalid_actual_duration(tmp_path, duration): + with pytest.raises(TranscriptSidecarError, match="duration"): + load_transcript_sidecar(write_sidecar(tmp_path, [VALID_SEGMENT]), duration_sec=duration) + + +@pytest.mark.parametrize( + ("start", "end"), + [(-0.1, 1), (1, 1), (2, 1), (0, 10.000001), (True, 1), (0, False), + ("0", 1), (0, "1"), (None, 1), ([], 1), ({}, 1), (0, float("nan")), + (float("inf"), 1), (0, float("-inf")), (0, 10 ** 500)], +) +def test_rejects_invalid_timestamp_bounds_and_types(tmp_path, start, end): + path = write_sidecar(tmp_path, [{"start_sec": start, "end_sec": end, "text": "Hello"}]) + with pytest.raises(TranscriptSidecarError): + load_transcript_sidecar(path, duration_sec=10) + + +@pytest.mark.parametrize("text", ["", " \n\t ", None, 1, True, [], {}]) +def test_rejects_invalid_text_without_echoing_it(tmp_path, text): + segment = {**VALID_SEGMENT, "text": text} + with pytest.raises(TranscriptSidecarError, match="text must be a nonempty string"): + load_transcript_sidecar(write_sidecar(tmp_path, [segment]), duration_sec=10) + + +@pytest.mark.parametrize("root", [None, True, 1, "text", {}, {"text": "Hello"}, {"segments": {}}, {"segments": None}]) +def test_rejects_invalid_root_and_segments_container(tmp_path, root): + with pytest.raises(TranscriptSidecarError): + load_transcript_sidecar(write_sidecar(tmp_path, root), duration_sec=10) + + +@pytest.mark.parametrize("segment", [None, True, "Hello", [], {}, {"start_sec": 0, "end_sec": 1}, {**VALID_SEGMENT, "video_id": "override"}]) +def test_rejects_invalid_or_extra_segment_fields(tmp_path, segment): + with pytest.raises(TranscriptSidecarError, match="requires only"): + load_transcript_sidecar(write_sidecar(tmp_path, [segment]), duration_sec=10) + + +def test_rejects_extra_root_fields(tmp_path): + with pytest.raises(TranscriptSidecarError, match="permits only"): + load_transcript_sidecar(write_sidecar(tmp_path, {"segments": [], "video_id": "override"}), duration_sec=10) + + +@pytest.mark.parametrize("model", [None, True, 12, [], {}, "", " "]) +def test_rejects_invalid_model(tmp_path, model): + with pytest.raises(TranscriptSidecarError, match="model"): + load_transcript_sidecar(write_sidecar(tmp_path, {"segments": [], "model": model}), duration_sec=10) + + +@pytest.mark.parametrize("provenance", [None, True, 1, "source", []]) +def test_rejects_nonobject_provenance(tmp_path, provenance): + with pytest.raises(TranscriptSidecarError, match="provenance"): + load_transcript_sidecar(write_sidecar(tmp_path, {"segments": [], "provenance": provenance}), duration_sec=10) + + +@pytest.mark.parametrize("raw", [ + "{", + '{"segments": [], "segments": []}', + '[{"start_sec": 0, "end_sec": 1, "text": "Hello", "text": "Other"}]', + '{"segments": [], "provenance": {"nested": [1e999]}}', + '{"segments": [], "provenance": {"nested": [NaN]}}', + '{"segments": [], "provenance": {"same": 1, "same": 2}}', +]) +def test_rejects_malformed_or_ambiguous_json(tmp_path, raw): + path = tmp_path / "transcript.json" + path.write_text(raw) + with pytest.raises(TranscriptSidecarError): + load_transcript_sidecar(path, duration_sec=10) + + +def test_file_read_errors_are_ingest_errors(tmp_path): + with pytest.raises(IngestError, match="Could not read"): + load_transcript_sidecar(tmp_path / "missing.json", duration_sec=10) + path = tmp_path / "invalid.json" + path.write_bytes(b"\xff") + with pytest.raises(IngestError, match="Could not read"): + load_transcript_sidecar(path, duration_sec=10) + + +@pytest.fixture +def cli_setup(tmp_path, monkeypatch): + # Use real isolated config and SQLite; only replace the pipeline boundary. + monkeypatch.chdir(tmp_path) + monkeypatch.setattr("av.core.config.CONFIG_FILE_PATH", tmp_path / "config.json") + for key in list(os.environ): + if key.startswith("AV_") or key in {"OPENAI_API_KEY", "TYPESAFE_API_KEY", "TYPESAFE_DEFAULT_MODEL"}: + monkeypatch.delenv(key) + app = typer.Typer() + + @app.callback() + def callback(): + pass + + ingest_cli.register(app) + pipeline = Mock(return_value={"status": "complete", "artifacts_count": 1}) + monkeypatch.setattr(ingest_cli, "ingest_video", pipeline) + video = tmp_path / "video.mp4" + video.write_bytes(b"offline-test-video") + sidecar = write_sidecar(tmp_path, [VALID_SEGMENT]) + return app, pipeline, video, sidecar, tmp_path / "av.db" + + +def test_cli_forwards_resolved_sidecar_path(cli_setup): + app, pipeline, video, sidecar, db = cli_setup + result = CliRunner().invoke(app, ["ingest", str(video), "--transcript-json", sidecar.name, "--db", str(db)]) + assert result.exit_code == 0, result.output + assert json.loads(result.stdout)["status"] == "complete" + assert pipeline.call_args.kwargs["transcript_json"] == sidecar.resolve() + assert pipeline.call_args.args[0] == video + + +def test_cli_without_sidecar_preserves_default(cli_setup): + app, pipeline, video, _, db = cli_setup + result = CliRunner().invoke(app, ["ingest", str(video), "--db", str(db)]) + assert result.exit_code == 0, result.output + assert pipeline.call_args.kwargs["transcript_json"] is None + + +def test_cli_rejects_directory_even_with_one_video(cli_setup): + app, pipeline, video, sidecar, db = cli_setup + result = CliRunner().invoke(app, ["ingest", str(video.parent), "--transcript-json", str(sidecar), "--db", str(db)]) + assert result.exit_code == 1 + assert "not a directory" in result.output + pipeline.assert_not_called() + assert not db.exists() + + +def test_cli_rejects_multiple_discovered_inputs(cli_setup, monkeypatch): + app, pipeline, video, sidecar, db = cli_setup + monkeypatch.setattr(ingest_cli, "discover_videos", lambda _: [video, video]) + result = CliRunner().invoke(app, ["ingest", str(video), "--transcript-json", str(sidecar), "--db", str(db)]) + assert result.exit_code == 1 + assert "exactly one video" in result.output + pipeline.assert_not_called() + assert not db.exists() + + +@pytest.mark.parametrize("use_directory", [False, True]) +def test_cli_rejects_invalid_sidecar_path_before_ingest(cli_setup, use_directory): + app, pipeline, video, _, db = cli_setup + path = video.parent if use_directory else video.parent / "missing.json" + result = CliRunner().invoke(app, ["ingest", str(video), "--transcript-json", str(path), "--db", str(db)]) + assert result.exit_code == 2 + pipeline.assert_not_called() + assert not db.exists() + + +def test_cli_accepts_one_downloaded_url(cli_setup, monkeypatch): + app, pipeline, video, sidecar, db = cli_setup + download = Mock(return_value=video) + monkeypatch.setattr(ingest_cli, "download_video", download) + url = "https://example.com/video.mp4" + result = CliRunner().invoke(app, ["ingest", url, "--transcript-json", str(sidecar), "--db", str(db)]) + assert result.exit_code == 0, result.output + download.assert_called_once_with(url) + assert pipeline.call_args.kwargs["transcript_json"] == sidecar + + +def test_cli_reports_explicit_sidecar_validation_failure(cli_setup): + app, pipeline, video, sidecar, db = cli_setup + pipeline.side_effect = TranscriptSidecarError("Transcript segment 1 is invalid.") + result = CliRunner().invoke(app, ["ingest", str(video), "--transcript-json", str(sidecar), "--db", str(db)]) + assert result.exit_code == 1 + assert json.loads(result.stdout)["status"] == "error" + assert "Transcript segment 1 is invalid." in result.output From 75125c3104278dd137e4960927e0afb047c81f90 Mon Sep 17 00:00:00 2001 From: Sean Phan Date: Fri, 18 Sep 2026 18:43:57 +0000 Subject: [PATCH 02/12] fix: bound ask output and preserve usage receipts --- CLAUDE.md | 1 + README.md | 2 + src/av/cli/config_cmd.py | 1 + src/av/core/config.py | 3 + src/av/providers/openai.py | 1 + src/av/search/rag.py | 122 ++++++++++++++++++++++++++++-- src/av/search/refine.py | 7 +- tests/test_ask_refinement.py | 143 ++++++++++++++++++++++++++++++++++- tests/test_config.py | 24 ++++++ tests/test_provider.py | 16 +++- 10 files changed, 306 insertions(+), 14 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 95e04ae..f0b33e8 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -196,6 +196,7 @@ When a capability is unavailable (e.g. Anthropic has no Whisper), the pipeline s | `AV_VISION_MODEL` | `gpt-4-1` | Vision/caption model | | `AV_EMBED_MODEL` | `text-embedding-3-small` | Embedding model | | `AV_CHAT_MODEL` | `gpt-4-1` | Chat/RAG model | +| `AV_CHAT_MAX_OUTPUT_TOKENS` | `1024` | Positive output-token cap for each answer response | | `AV_DB_PATH` | `~/.config/av/av.db` | Database location | | `AV_TYPESAFE_API_KEY` | (none) | Jev/System One key; enables ask refinement by default | | `AV_TYPESAFE_ENDPOINT` | `https://api.typesafe.ai/v1/systemone` | Explicit System One endpoint | diff --git a/README.md b/README.md index 586df67..7c82f5a 100644 --- a/README.md +++ b/README.md @@ -307,6 +307,7 @@ export AV_TRANSCRIBE_MODEL="whisper" export AV_VISION_MODEL="gpt-4-1" export AV_EMBED_MODEL="text-embedding-3-small" export AV_CHAT_MODEL="gpt-4-1" +export AV_CHAT_MAX_OUTPUT_TOKENS="1024" # positive cap for each answer response # Optional Jev/System One refinement (automatic when a key is present) export AV_TYPESAFE_API_KEY="..." # TYPESAFE_API_KEY also works @@ -338,6 +339,7 @@ API requests use the configured timeout and explicit retry limit. Ingestion JSON includes `stage_usage` for transcription, captioning, caption summarization, and embeddings, plus the effective frame/request settings. Request failures are counted; token totals become `null` with a completeness flag when any provider omits usage. +Ask JSON likewise reports the effective chat model/output cap and per-stage usage. No dollar total is inferred. ## Requirements diff --git a/src/av/cli/config_cmd.py b/src/av/cli/config_cmd.py index f64451d..4bdaa01 100644 --- a/src/av/cli/config_cmd.py +++ b/src/av/cli/config_cmd.py @@ -65,6 +65,7 @@ def config_show() -> None: "vision_model": config.vision_model, "embed_model": config.embed_model or "(disabled)", "chat_model": config.chat_model, + "chat_max_output_tokens": config.chat_max_output_tokens, "typesafe_api_key": "***" if config.typesafe_api_key else "(not set)", "typesafe_endpoint": config.typesafe_endpoint, "typesafe_model": config.typesafe_model, diff --git a/src/av/core/config.py b/src/av/core/config.py index b0b9407..062c6ab 100644 --- a/src/av/core/config.py +++ b/src/av/core/config.py @@ -63,6 +63,7 @@ class AVConfig(BaseSettings): vision_model: str = Field(default=DEFAULT_VISION_MODEL) embed_model: str = Field(default=DEFAULT_EMBED_MODEL) chat_model: str = Field(default=DEFAULT_CHAT_MODEL) + chat_max_output_tokens: int = Field(default=1024, gt=0) # Optional System One query refinement. A credential enables refinement by # default; callers can still opt out per request. @@ -116,6 +117,7 @@ def get_config(db_path: Path | None = None) -> AVConfig: "vision_model", "embed_model", "chat_model", + "chat_max_output_tokens", "typesafe_api_key", "typesafe_endpoint", "typesafe_model", @@ -191,4 +193,5 @@ def get_openai_config(config: AVConfig) -> AVConfig | None: embed_model="text-embedding-3-small", vision_model=config.vision_model, chat_model=config.chat_model, + chat_max_output_tokens=config.chat_max_output_tokens, ) diff --git a/src/av/providers/openai.py b/src/av/providers/openai.py index c31e9da..f358dbc 100644 --- a/src/av/providers/openai.py +++ b/src/av/providers/openai.py @@ -392,6 +392,7 @@ def complete_with_usage(self, prompt: str, context: str) -> CompletionResult: "content": f"Context from video analysis:\n\n{context}\n\nQuestion: {prompt}", }, ], + max_tokens=self.config.chat_max_output_tokens, ), ) usage = getattr(response, "usage", None) diff --git a/src/av/search/rag.py b/src/av/search/rag.py index 0da1a31..77d9054 100644 --- a/src/av/search/rag.py +++ b/src/av/search/rag.py @@ -61,18 +61,75 @@ def _heuristic_confidence(results: list[dict]) -> float: return min(round(float(top_score), 2), 1.0) if top_score else 0.5 -def _legacy_ask(question: str, results: list[dict], config: AVConfig) -> dict: +def _ask_settings(config: AVConfig) -> dict: + return { + "chat_model": config.chat_model, + "chat_max_output_tokens": config.chat_max_output_tokens, + } + + +def _llm_usage_snapshot(llm: OpenAILLM | None) -> dict: + usage = getattr(llm, "usage", None) + snapshot = getattr(usage, "snapshot", None) + if callable(snapshot): + receipt = snapshot() + if isinstance(receipt, dict): + return receipt + return new_usage() + + +def _legacy_ask( + question: str, + results: list[dict], + config: AVConfig, + embedding_usage: dict | None, +) -> dict: + warnings: list[str] = [] + stage_usage = { + "embedding": embedding_usage, + "answer": new_usage(), + } if not results: return { "answer": "No relevant content found in the indexed videos.", "citations": [], "confidence": 0.0, + "confidence_basis": "no_evidence", + "route": "legacy_no_results", + "evidence_status": "no_retrieval_hits", + "warnings": warnings, + "stage_usage": stage_usage, + "ask_settings": _ask_settings(config), + } + llm: OpenAILLM | None = None + try: + llm = OpenAILLM(config) + completion = llm.complete_with_usage(question, _context(results)) + stage_usage["answer"] = _usage_from_completion(completion) + except Exception: + stage_usage["answer"] = _llm_usage_snapshot(llm) + warnings.append("Answer generation was unavailable; no answer was produced.") + return { + "answer": "Answer generation was unavailable. Retrieved video moments are included as citations.", + "citations": _citations(results), + "confidence": 0.0, + "confidence_basis": "unknown", + "route": "legacy_answer_failed", + "evidence_status": "answer_unavailable", + "warnings": warnings, + "stage_usage": stage_usage, + "ask_settings": _ask_settings(config), } - answer = OpenAILLM(config).complete(question, _context(results)) return { - "answer": answer, + "answer": completion.text, "citations": _citations(results), "confidence": _heuristic_confidence(results), + "confidence_basis": "retrieval_heuristic", + "route": "legacy", + "evidence_status": "raw_unjudged", + "warnings": warnings, + "stage_usage": stage_usage, + "ask_settings": _ask_settings(config), } @@ -111,7 +168,12 @@ def ask( raw_results = search_result.get("results", []) if not refine or not config.refine_enabled or not config.typesafe_api_key: - return _legacy_ask(question, raw_results, config) + return _legacy_ask( + question, + raw_results, + config, + search_result.get("embedding_usage"), + ) warnings: list[str] = [] stage_usage: dict[str, dict | None] = { @@ -134,6 +196,7 @@ def ask( "warnings": warnings, "inspected_windows": [], "stage_usage": stage_usage, + "ask_settings": _ask_settings(config), } client = SystemOneClient(config) @@ -146,8 +209,27 @@ def ask( for stage, usage in exc.stage_usage.items(): stage_usage[stage] = usage warnings.append("Jev refinement was unavailable; answering from raw retrieval without judged evidence confidence.") - completion = OpenAILLM(config).complete_with_usage(question, _context(raw_results)) - stage_usage["answer"] = _usage_from_completion(completion) + llm: OpenAILLM | None = None + try: + llm = OpenAILLM(config) + completion = llm.complete_with_usage(question, _context(raw_results)) + stage_usage["answer"] = _usage_from_completion(completion) + except Exception: + stage_usage["answer"] = _llm_usage_snapshot(llm) + warnings.append("Answer generation was unavailable; no answer was produced.") + return { + "answer": "Answer generation was unavailable. Retrieved video moments are included as citations.", + "citations": _citations(raw_results), + "confidence": 0.0, + "confidence_basis": "unknown", + "route": "refinement_fallback_answer_failed", + "evidence_status": "answer_unavailable", + "refinement": {"status": "provider_fallback", "raw_count": len(raw_results)}, + "warnings": warnings, + "inspected_windows": [], + "stage_usage": stage_usage, + "ask_settings": _ask_settings(config), + } return { "answer": completion.text, "citations": _citations(raw_results), @@ -159,6 +241,7 @@ def ask( "warnings": warnings, "inspected_windows": [], "stage_usage": stage_usage, + "ask_settings": _ask_settings(config), } if not results: @@ -173,10 +256,30 @@ def ask( "warnings": warnings, "inspected_windows": [], "stage_usage": stage_usage, + "ask_settings": _ask_settings(config), } - completion = OpenAILLM(config).complete_with_usage(question, _context(results)) - stage_usage["answer"] = _usage_from_completion(completion) + llm = None + try: + llm = OpenAILLM(config) + completion = llm.complete_with_usage(question, _context(results)) + stage_usage["answer"] = _usage_from_completion(completion) + except Exception: + stage_usage["answer"] = _llm_usage_snapshot(llm) + warnings.append("Answer generation was unavailable; no answer was produced.") + return { + "answer": "Answer generation was unavailable. Relevant video moments are included as citations.", + "citations": _citations(results), + "confidence": 0.0, + "confidence_basis": "unknown", + "route": "refined_answer_failed", + "evidence_status": "answer_unavailable", + "refinement": refinement, + "warnings": warnings, + "inspected_windows": [], + "stage_usage": stage_usage, + "ask_settings": _ask_settings(config), + } answer = completion.text citations = _citations(results) support_probability: float | None = None @@ -202,6 +305,7 @@ def ask( "warnings": warnings, "inspected_windows": [], "stage_usage": stage_usage, + "ask_settings": _ask_settings(config), } inspection = inspect_with_stronger_vision(question, answer, results, repo, config) @@ -244,6 +348,7 @@ def ask( "warnings": warnings, "inspected_windows": inspection["windows"], "stage_usage": stage_usage, + "ask_settings": _ask_settings(config), } if support_failed: @@ -263,4 +368,5 @@ def ask( "warnings": warnings, "inspected_windows": inspection["windows"], "stage_usage": stage_usage, + "ask_settings": _ask_settings(config), } diff --git a/src/av/search/refine.py b/src/av/search/refine.py index 10365ff..7d062ff 100644 --- a/src/av/search/refine.py +++ b/src/av/search/refine.py @@ -493,13 +493,16 @@ def _expand_scene( after=context_events, ) events, hit_index = _ordered_contiguous_events(result, artifacts) + containing_event = events[hit_index] + scene.start_sec = min(scene.start_sec, containing_event["start"]) + scene.end_sec = max(scene.end_sec, containing_event["end"]) if len(events) < 2: return scene start, end, confidence, _, _, _ = _judge_bounds( client, query, events, hit_index, context_events, usage ) - scene.start_sec = min(start, scene.chunk_start_sec) - scene.end_sec = max(end, scene.chunk_end_sec) + scene.start_sec = min(start, scene.start_sec) + scene.end_sec = max(end, scene.end_sec) scene.scene_confidence = confidence return scene diff --git a/tests/test_ask_refinement.py b/tests/test_ask_refinement.py index f3cd17f..7c0a65b 100644 --- a/tests/test_ask_refinement.py +++ b/tests/test_ask_refinement.py @@ -12,6 +12,7 @@ from av.db.models import ArtifactRecord, VideoRecord from av.db.repository import Repository from av.providers.base import CompletionResult +from av.providers.usage import ProviderUsage from av.search.inspection import ( ExplicitVisionClient, InspectionWindow, @@ -303,6 +304,58 @@ def test_temporal_events_attach_interleaved_modalities_without_false_gap() -> No assert "there" in events[0]["text"] +@pytest.mark.parametrize( + ( + "caption_start", + "caption_end", + "transcript_start", + "transcript_end", + "expected_start", + "expected_end", + ), + [ + (10.0, 20.0, 5.0, 25.0, 5.0, 25.0), + (10.0, 10.0, 10.0, 15.0, 10.0, 15.0), + ], +) +def test_scene_uses_containing_temporal_event_bounds_without_boundary_call( + repo: Repository, + tmp_path: Path, + caption_start: float, + caption_end: float, + transcript_start: float, + transcript_end: float, + expected_start: float, + expected_end: float, +) -> None: + path = tmp_path / "v.mp4" + path.write_bytes(b"fake") + repo.insert_video(_video("v", path)) + repo.insert_artifacts_batch([ + _artifact("caption-hit", "v", caption_start, caption_end, "target caption", "caption"), + _artifact( + "transcript-overlap", + "v", + transcript_start, + transcript_end, + "overlapping speech", + "transcript", + ), + ]) + raw = [repo.search_fts("target", limit=1, video_id="v")[0].model_dump()] + client = FakeSystemOne([0.9]) + refined, _, _ = refine_search_results( + "target", + raw, + repo, + AVConfig(typesafe_api_key="test", refine_context_events=3), + client=client, + ) + assert client.boundary_calls == 0 + assert refined[0]["timestamp_sec"] == expected_start + assert refined[0]["end_sec"] == expected_end + + def test_strict_hydration_excludes_touching_chunks_and_preserves_hit_texts( repo: Repository, tmp_path: Path ) -> None: @@ -510,6 +563,74 @@ def test_refinement_fallback_returns_partial_stage_usage(repo: Repository, tmp_p assert result["stage_usage"]["relevance"]["input_tokens_complete"] is False +def test_refined_answer_failure_preserves_receipts_and_is_sanitized( + repo: Repository, tmp_path: Path +) -> None: + _seed_video(repo, tmp_path, "v1", prefix="cake") + + class FailingLLM: + def __init__(self, config: AVConfig) -> None: + self.usage = ProviderUsage() + + def complete_with_usage(self, prompt: str, context: str) -> CompletionResult: + self.usage.record_failure() + raise RuntimeError("https://private.example/v1 secret-token") + + fake = FakeSystemOne([0.9] * 30) + with patch("av.search.rag.SystemOneClient", return_value=fake), \ + patch("av.search.rag.OpenAILLM", FailingLLM), \ + patch("av.search.rag.judge_answer_support") as support, \ + patch("av.search.rag.inspect_with_stronger_vision") as inspect: + result = ask( + "cake", + repo, + AVConfig(typesafe_api_key="test", embed_model=""), + video_id="v1", + ) + support.assert_not_called() + inspect.assert_not_called() + assert result["route"] == "refined_answer_failed" + assert result["evidence_status"] == "answer_unavailable" + assert result["stage_usage"]["relevance"]["requests"] > 0 + assert result["stage_usage"]["answer"]["requests"] == 1 + assert result["stage_usage"]["answer"]["failed_requests"] == 1 + assert result["ask_settings"]["chat_max_output_tokens"] == 1024 + encoded = json.dumps(result) + assert "private.example" not in encoded + assert "secret-token" not in encoded + + +def test_legacy_answer_failure_preserves_attempted_usage_and_is_sanitized( + repo: Repository, tmp_path: Path +) -> None: + _seed_video(repo, tmp_path, "v1", prefix="cake") + + class FailingLLM: + def __init__(self, config: AVConfig) -> None: + self.usage = ProviderUsage() + + def complete_with_usage(self, prompt: str, context: str) -> CompletionResult: + self.usage.record_failure() + raise RuntimeError("https://private.example/v1 secret-token") + + with patch("av.search.rag.OpenAILLM", FailingLLM): + result = ask( + "cake", + repo, + AVConfig(embed_model=""), + video_id="v1", + refine=False, + ) + assert result["route"] == "legacy_answer_failed" + assert result["evidence_status"] == "answer_unavailable" + assert "embedding" in result["stage_usage"] + assert result["stage_usage"]["answer"]["requests"] == 1 + assert result["stage_usage"]["answer"]["failed_requests"] == 1 + encoded = json.dumps(result) + assert "private.example" not in encoded + assert "secret-token" not in encoded + + def test_full_window_timestamp_plan_covers_start_and_end() -> None: windows = [ InspectionWindow("v1", "a.mp4", Path("a.mp4"), 10.0, 40.0, 10.0, 40.0), @@ -831,7 +952,18 @@ def test_config_file_env_priority_and_secret_fields(tmp_path: Path, monkeypatch: def test_no_refine_preserves_legacy_contract(repo: Repository, tmp_path: Path) -> None: _seed_video(repo, tmp_path, "v1", prefix="cake") - with patch("av.search.rag.OpenAILLM", FakeLLM): + raw = [repo.search_fts("cake", limit=1, video_id="v1")[0].model_dump()] + embedding_usage = { + "requests": 1, + "input_tokens": 4, + "output_tokens": 0, + "input_tokens_complete": True, + "output_tokens_complete": True, + } + with patch( + "av.search.rag.search", + return_value={"results": raw, "embedding_usage": embedding_usage}, + ), patch("av.search.rag.OpenAILLM", FakeLLM): result = ask( "cake", repo, @@ -839,5 +971,10 @@ def test_no_refine_preserves_legacy_contract(repo: Repository, tmp_path: Path) - video_id="v1", refine=False, ) - assert set(result) == {"answer", "citations", "confidence"} - assert result["answer"] == "legacy answer" + assert result["answer"] == "refined answer" + assert result["route"] == "legacy" + assert result["evidence_status"] == "raw_unjudged" + assert result["confidence_basis"] == "retrieval_heuristic" + assert result["warnings"] == [] + assert result["stage_usage"]["embedding"] == embedding_usage + assert result["stage_usage"]["answer"]["requests"] == 1 diff --git a/tests/test_config.py b/tests/test_config.py index c211d5f..8ec2799 100644 --- a/tests/test_config.py +++ b/tests/test_config.py @@ -9,6 +9,7 @@ import pytest +from av.cli.config_cmd import config_show from av.core.config import AVConfig, _load_config_file, get_config, get_openai_config, save_config from av.core.constants import CONFIG_FILE_PATH, PROVIDER_PRESETS @@ -101,6 +102,29 @@ def test_env_var_overrides_config_file(tmp_path: Path, monkeypatch: pytest.Monke assert config.provider == "anthropic" +def test_chat_output_cap_loads_from_config_and_env_with_positive_validation( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + cfg_file = tmp_path / "config.json" + cfg_file.write_text(json.dumps({"chat_max_output_tokens": 700})) + monkeypatch.setattr("av.core.config.CONFIG_FILE_PATH", cfg_file) + monkeypatch.delenv("AV_CHAT_MAX_OUTPUT_TOKENS", raising=False) + assert get_config().chat_max_output_tokens == 700 + + monkeypatch.setenv("AV_CHAT_MAX_OUTPUT_TOKENS", "900") + assert get_config().chat_max_output_tokens == 900 + + with pytest.raises(ValueError): + AVConfig(chat_max_output_tokens=0) + + +def test_config_show_includes_chat_output_cap() -> None: + with patch("av.cli.config_cmd.get_config", return_value=AVConfig(chat_max_output_tokens=777)), \ + patch("av.cli.config_cmd.output_json") as output: + config_show() + assert output.call_args.args[0]["chat_max_output_tokens"] == 777 + + def test_openai_fallback_does_not_read_oauth_unless_enabled() -> None: config = AVConfig(provider="anthropic", openai_api_key="", allow_oauth_fallback=False) with patch("av.providers.openai._openclaw_oauth_token") as openclaw, \ diff --git a/tests/test_provider.py b/tests/test_provider.py index 9fb8cba..071931b 100644 --- a/tests/test_provider.py +++ b/tests/test_provider.py @@ -9,7 +9,7 @@ import pytest from av.core.config import AVConfig -from av.providers.openai import OpenAICaptioner, _client, _resolve_api_key +from av.providers.openai import OpenAICaptioner, OpenAILLM, _client, _resolve_api_key def test_client_default_no_extra_headers() -> None: @@ -71,6 +71,20 @@ def test_client_disables_sdk_retries_and_uses_configured_timeout() -> None: assert client.timeout == 17.0 +def test_llm_request_uses_configured_output_token_cap() -> None: + response = SimpleNamespace( + choices=[SimpleNamespace(message=SimpleNamespace(content="answer"))], + usage=SimpleNamespace(prompt_tokens=4, completion_tokens=2), + ) + fake_client = MagicMock() + fake_client.chat.completions.create.return_value = response + config = AVConfig(api_key="explicit", chat_max_output_tokens=321) + with patch("av.providers.openai._client", return_value=fake_client): + completion = OpenAILLM(config).complete_with_usage("question", "context") + assert completion.text == "answer" + assert fake_client.chat.completions.create.call_args.kwargs["max_tokens"] == 321 + + def test_caption_fallback_is_disabled_by_default(tmp_path: Path) -> None: frame = tmp_path / "frame.jpg" frame.write_bytes(b"jpg") From 86ebf7c81bc3391213162a66641db42c8fd8ec70 Mon Sep 17 00:00:00 2001 From: Sean Phan Date: Fri, 18 Sep 2026 19:00:57 +0000 Subject: [PATCH 03/12] fix(search): retrieve candidates from natural-language questions --- src/av/search/query.py | 43 ++++++++ src/av/search/rag.py | 2 +- src/av/search/semantic.py | 10 +- tests/test_natural_language_search.py | 137 ++++++++++++++++++++++++++ 4 files changed, 189 insertions(+), 3 deletions(-) create mode 100644 src/av/search/query.py create mode 100644 tests/test_natural_language_search.py diff --git a/src/av/search/query.py b/src/av/search/query.py new file mode 100644 index 0000000..7f2219c --- /dev/null +++ b/src/av/search/query.py @@ -0,0 +1,43 @@ +"""Bounded lexical candidate queries for natural-language questions. + +This is a recall-oriented lookup, not a query rewrite or answer model. It does +not change the original question passed to embeddings, Jev, or answer synthesis. +""" + +from __future__ import annotations + +import re +import unicodedata + +MAX_CANDIDATE_TERMS = 24 +MAX_TERM_CHARACTERS = 80 +_WORDS = re.compile(r"[^\W_]+", re.UNICODE) +# Generic English question/function words, not video-domain vocabulary. +_QUESTION_WORDS = frozenset(""" +a an the am is are was were be been being do does did can could would should +shall will may might must have has had having i me my we us our you your he +him his she her it its they them their this that these those there here then +to of for from in on at by with as about and or but if so what which who whom +whose when where why how please tell show s t +""".split()) + + +def natural_language_fts_query(question: str) -> str: + """Return a safe OR of distinct content terms, or empty when none remain. + + Unicode letters/digits are retained; punctuation and FTS operators are never + interpreted as syntax. Apply the term budget after removing question words + and duplicates so a long question preamble cannot consume the candidate cap. + """ + terms: list[str] = [] + seen: set[str] = set() + normalized = unicodedata.normalize("NFC", question).casefold() + for match in _WORDS.finditer(normalized): + term = match.group() + if term in _QUESTION_WORDS or term in seen or len(term) > MAX_TERM_CHARACTERS: + continue + seen.add(term) + terms.append('"' + term.replace('"', '""') + '"') + if len(terms) >= MAX_CANDIDATE_TERMS: + break + return " OR ".join(terms) diff --git a/src/av/search/rag.py b/src/av/search/rag.py index 77d9054..0f712c2 100644 --- a/src/av/search/rag.py +++ b/src/av/search/rag.py @@ -163,7 +163,7 @@ def ask( """Answer a question using RAG over video artifacts.""" # Step 1: Retrieve relevant context search_result = search( - question, repo, config, limit=top_k, video_id=video_id + question, repo, config, limit=top_k, video_id=video_id, natural_language=True ) raw_results = search_result.get("results", []) diff --git a/src/av/search/semantic.py b/src/av/search/semantic.py index e3151c7..e07e3d0 100644 --- a/src/av/search/semantic.py +++ b/src/av/search/semantic.py @@ -9,6 +9,7 @@ from av.db.models import SearchResult from av.db.repository import Repository from av.providers.openai import OpenAIEmbedder +from av.search.query import natural_language_fts_query def _cosine_similarity(a: list[float], b: list[float]) -> float: @@ -27,13 +28,18 @@ def search( *, limit: int = 10, video_id: str | None = None, + natural_language: bool = False, ) -> dict: """Search artifacts using FTS5, optionally reranked by cosine similarity.""" start_time = time.time() embedding_usage = None - # Step 1: FTS search (always available) - fts_results = repo.search_fts(query, limit=limit * 3, video_id=video_id) + # av ask uses lexical candidates; av search retains explicit FTS semantics. + candidate_query = natural_language_fts_query(query) if natural_language else query + fts_results = ( + repo.search_fts(candidate_query, limit=limit * 3, video_id=video_id) + if candidate_query else [] + ) if not fts_results: elapsed_ms = int((time.time() - start_time) * 1000) diff --git a/tests/test_natural_language_search.py b/tests/test_natural_language_search.py new file mode 100644 index 0000000..208fc0b --- /dev/null +++ b/tests/test_natural_language_search.py @@ -0,0 +1,137 @@ +"""Offline regressions for question candidate retrieval before Jev refinement.""" + +from pathlib import Path +from unittest.mock import MagicMock, patch + +import pytest + +from av.core.config import AVConfig +from av.db.models import ArtifactRecord, VideoRecord +from av.db.repository import Repository +from av.providers.base import CompletionResult +from av.providers.usage import ProviderUsage +from av.search.query import MAX_CANDIDATE_TERMS, natural_language_fts_query +from av.search.rag import ask +from av.search.semantic import search + + +@pytest.fixture() +def repo(tmp_path: Path): + repository = Repository(tmp_path / "questions.db") + for video_id in ("local", "other"): + repository.insert_video(VideoRecord( + id=video_id, file_path=f"{video_id}.mp4", file_hash=video_id, + file_size_bytes=0, filename=f"{video_id}.mp4", duration_sec=120, + status="complete", + )) + repository.insert_artifacts_batch([ + ArtifactRecord(id="blue", video_id="local", type="caption", start_sec=10, + end_sec=20, text="The truck was blue"), + ArtifactRecord(id="irrelevant", video_id="local", type="caption", start_sec=60, + end_sec=70, text="The truck advertisement listed rental prices"), + ArtifactRecord(id="red", video_id="other", type="caption", start_sec=10, + end_sec=20, text="The truck was red"), + ArtifactRecord(id="unicode", video_id="local", type="caption", start_sec=90, + end_sec=100, text="A café named Étoile appears beside a 東京 sign"), + ]) + try: + yield repository + finally: + repository.close() + + +@pytest.mark.parametrize("question", ["What color was the truck?", "What color was the truck"]) +def test_natural_question_retrieves_candidates_and_preserves_video_isolation(repo, question): + result = search(question, repo, AVConfig(embed_model=""), video_id="local", natural_language=True) + assert result["query"] == question + assert {row["artifact_id"] for row in result["results"]} == {"blue", "irrelevant"} + assert {row["video_id"] for row in result["results"]} == {"local"} + + +def test_unicode_punctuation_and_fts_operators_are_literal_candidates(repo): + result = search('What is "CAFÉ"? OR 東京: (Étoile*)', repo, AVConfig(embed_model=""), + video_id="local", natural_language=True) + assert [row["artifact_id"] for row in result["results"]] == ["unicode"] + assert natural_language_fts_query('truck NOT "red"') == '"truck" OR "not" OR "red"' + + +@pytest.mark.parametrize("question", ["What is it?", "?! () :", "Where did the submarine surface?"]) +def test_empty_or_unrelated_question_returns_no_hits_or_model_calls(repo, question): + with patch("av.search.rag.SystemOneClient") as judge, patch("av.search.rag.OpenAILLM") as llm: + result = ask(question, repo, AVConfig(typesafe_api_key="test", embed_model=""), video_id="local") + assert result["route"] == "refined_no_results" + assert result["citations"] == [] + judge.assert_not_called() + llm.assert_not_called() + + +def test_candidate_budget_applies_after_question_words_and_duplicates(): + question = "what " * 100 + "truck " * 100 + " ".join(f"object{i}" for i in range(40)) + terms = natural_language_fts_query(question).split(" OR ") + assert len(terms) == MAX_CANDIDATE_TERMS + assert terms[0] == '"truck"' + assert terms[1] == '"object0"' + assert len(set(terms)) == len(terms) + + +def test_advanced_fts_search_semantics_stay_unchanged(repo): + config = AVConfig(embed_model="") + result = search("truck NOT red", repo, config) + assert {row["artifact_id"] for row in result["results"]} == {"blue", "irrelevant"} + phrase = search('"truck was blue"', repo, config) + assert [row["artifact_id"] for row in phrase["results"]] == ["blue"] + assert search("What color was the truck", repo, config)["results"] == [] + + +def test_embedding_receives_original_question(repo): + question = "What color was the truck?" + embedder = MagicMock() + embedder.embed.return_value = [[1.0, 0.0]] + embedder.usage.snapshot.return_value = {"requests": 1} + with patch.object(repo, "get_embeddings_for_artifacts", return_value={"blue": [1.0, 0.0]}), \ + patch("av.search.semantic.OpenAIEmbedder", return_value=embedder): + result = search(question, repo, AVConfig(), video_id="local", natural_language=True) + embedder.embed.assert_called_once_with([question]) + assert result["query"] == question + assert result["results"][0]["artifact_id"] == "blue" + + +@pytest.mark.parametrize("reject_all", [False, True]) +def test_ask_passes_original_question_to_jev_and_filters_broad_candidates(repo, reject_all): + question = "What color was the truck?" + relevance_texts = [] + + class Judge: + def ask(self, state, questions): + assert state.get("query", state.get("question")) == question + if "is_supported" in questions: + return {"is_supported": {"type": "noul", "noul": 0.99}}, {} + assert "clips" in state + relevance_texts.extend(clip["caption"] for clip in state["clips"].values()) + return { + key: {"type": "noul", "noul": 0.99 if "blue" in clip["caption"] and not reject_all else 0.01} + for key, clip in state["clips"].items() + }, {} + + class Answer: + def __init__(self, config): + self.usage = ProviderUsage() + + def complete_with_usage(self, prompt, context): + assert prompt == question + assert "blue" in context + assert "rental" not in context + return CompletionResult("The truck was blue", input_tokens=20, output_tokens=5) + + with patch("av.search.rag.SystemOneClient", return_value=Judge()), \ + patch("av.search.rag.OpenAILLM", side_effect=Answer) as llm: + result = ask(question, repo, AVConfig(typesafe_api_key="test", embed_model=""), video_id="local") + assert len(relevance_texts) == 2 + if reject_all: + assert result["route"] == "refined_no_results" + assert result["citations"] == [] + llm.assert_not_called() + else: + assert result["route"] == "refined" + assert result["answer"] == "The truck was blue" + assert [citation["artifact_id"] for citation in result["citations"]] == ["blue"] From bb22c76fe8f2da0b92c5ca9f637af2059fd69551 Mon Sep 17 00:00:00 2001 From: Sean Phan Date: Fri, 18 Sep 2026 19:54:20 +0000 Subject: [PATCH 04/12] fix: support provider completion token caps --- src/av/cli/config_cmd.py | 8 +++-- src/av/core/config.py | 11 ++++++ src/av/providers/openai.py | 12 +++++-- tests/test_config.py | 72 ++++++++++++++++++++++++++++++++++++-- tests/test_provider.py | 32 +++++++++++++++++ 5 files changed, 127 insertions(+), 8 deletions(-) diff --git a/src/av/cli/config_cmd.py b/src/av/cli/config_cmd.py index 4bdaa01..a77ca48 100644 --- a/src/av/cli/config_cmd.py +++ b/src/av/cli/config_cmd.py @@ -27,20 +27,21 @@ def _validate_key(provider: str, config_data: dict) -> bool: """Make a lightweight API call to verify the key works. Returns True on success.""" - from av.providers.openai import _client + from av.providers.openai import _client, _completion_token_limit temp_config = AVConfig( provider=provider, api_base_url=config_data["api_base_url"], api_key=config_data.get("api_key", ""), chat_model=config_data["chat_model"], + api_token_limit_parameter=config_data.get("api_token_limit_parameter", "max_tokens"), ) client = _client(temp_config) try: client.chat.completions.create( model=config_data["chat_model"], messages=[{"role": "user", "content": "hi"}], - max_tokens=1, + **_completion_token_limit(temp_config, 1), ) return True except Exception as e: @@ -59,10 +60,13 @@ def config_show() -> None: "openai_api_key": "***" if config.openai_api_key else "(not set)", "api_timeout_sec": config.api_timeout_sec, "api_max_retries": config.api_max_retries, + "api_token_limit_parameter": config.api_token_limit_parameter, "allow_oauth_fallback": config.allow_oauth_fallback, "allow_codex_fallback": config.allow_codex_fallback, "transcribe_model": config.transcribe_model or "(disabled)", "vision_model": config.vision_model, + "vision_max_output_tokens": config.vision_max_output_tokens, + "vision_chunk_max_output_tokens": config.vision_chunk_max_output_tokens, "embed_model": config.embed_model or "(disabled)", "chat_model": config.chat_model, "chat_max_output_tokens": config.chat_max_output_tokens, diff --git a/src/av/core/config.py b/src/av/core/config.py index 062c6ab..11f4895 100644 --- a/src/av/core/config.py +++ b/src/av/core/config.py @@ -5,6 +5,7 @@ import json import os from pathlib import Path +from typing import Literal from pydantic import Field from pydantic_settings import BaseSettings, SettingsConfigDict @@ -55,12 +56,16 @@ class AVConfig(BaseSettings): openai_api_key: str = Field(default="") api_timeout_sec: float = Field(default=120.0, gt=0) api_max_retries: int = Field(default=1, ge=0, le=3) + # Select one request field; provider semantics and enforcement can differ. + api_token_limit_parameter: Literal["max_tokens", "max_completion_tokens"] = "max_tokens" allow_oauth_fallback: bool = Field(default=False) allow_codex_fallback: bool = Field(default=False) # Models transcribe_model: str = Field(default=DEFAULT_TRANSCRIBE_MODEL) vision_model: str = Field(default=DEFAULT_VISION_MODEL) + vision_max_output_tokens: int = Field(default=200, gt=0) + vision_chunk_max_output_tokens: int = Field(default=500, gt=0) embed_model: str = Field(default=DEFAULT_EMBED_MODEL) chat_model: str = Field(default=DEFAULT_CHAT_MODEL) chat_max_output_tokens: int = Field(default=1024, gt=0) @@ -111,10 +116,13 @@ def get_config(db_path: Path | None = None) -> AVConfig: "openai_api_key", "api_timeout_sec", "api_max_retries", + "api_token_limit_parameter", "allow_oauth_fallback", "allow_codex_fallback", "transcribe_model", "vision_model", + "vision_max_output_tokens", + "vision_chunk_max_output_tokens", "embed_model", "chat_model", "chat_max_output_tokens", @@ -187,11 +195,14 @@ def get_openai_config(config: AVConfig) -> AVConfig | None: api_key=key, api_timeout_sec=config.api_timeout_sec, api_max_retries=config.api_max_retries, + api_token_limit_parameter=config.api_token_limit_parameter, allow_oauth_fallback=False, allow_codex_fallback=config.allow_codex_fallback, transcribe_model="whisper-1", embed_model="text-embedding-3-small", vision_model=config.vision_model, + vision_max_output_tokens=config.vision_max_output_tokens, + vision_chunk_max_output_tokens=config.vision_chunk_max_output_tokens, chat_model=config.chat_model, chat_max_output_tokens=config.chat_max_output_tokens, ) diff --git a/src/av/providers/openai.py b/src/av/providers/openai.py index f358dbc..2e420b0 100644 --- a/src/av/providers/openai.py +++ b/src/av/providers/openai.py @@ -123,6 +123,11 @@ def _client(config: AVConfig) -> OpenAI: return OpenAI(**kwargs) +def _completion_token_limit(config: AVConfig, limit: int) -> dict[str, int]: + """Send exactly the configured token-limit field to compatible providers.""" + return {config.api_token_limit_parameter: limit} + + def _call_with_retries(config: AVConfig, usage: ProviderUsage, operation): last_error: Exception | None = None for attempt in range(config.api_max_retries + 1): @@ -267,7 +272,7 @@ def caption_frames( ], } ], - max_tokens=200, + **_completion_token_limit(self.config, self.config.vision_max_output_tokens), ), ) text = response.choices[0].message.content or "" @@ -315,7 +320,7 @@ def caption_chunk( lambda: self.client.chat.completions.create( model=self.config.vision_model, messages=[{"role": "user", "content": content}], - max_tokens=500, + **_completion_token_limit(self.config, self.config.vision_chunk_max_output_tokens), ), ) return (response.choices[0].message.content or "").strip() @@ -392,7 +397,7 @@ def complete_with_usage(self, prompt: str, context: str) -> CompletionResult: "content": f"Context from video analysis:\n\n{context}\n\nQuestion: {prompt}", }, ], - max_tokens=self.config.chat_max_output_tokens, + **_completion_token_limit(self.config, self.config.chat_max_output_tokens), ), ) usage = getattr(response, "usage", None) @@ -417,6 +422,7 @@ def summarize(self, system_prompt: str, user_content: str) -> str: {"role": "system", "content": system_prompt}, {"role": "user", "content": user_content}, ], + **_completion_token_limit(self.config, self.config.chat_max_output_tokens), ), ) return (response.choices[0].message.content or "").strip() diff --git a/tests/test_config.py b/tests/test_config.py index 8ec2799..8f5044c 100644 --- a/tests/test_config.py +++ b/tests/test_config.py @@ -118,11 +118,65 @@ def test_chat_output_cap_loads_from_config_and_env_with_positive_validation( AVConfig(chat_max_output_tokens=0) +def test_provider_token_limit_settings_load_from_config_and_env( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + cfg_file = tmp_path / "config.json" + cfg_file.write_text(json.dumps({ + "api_token_limit_parameter": "max_completion_tokens", + "vision_max_output_tokens": 32, + "vision_chunk_max_output_tokens": 64, + })) + monkeypatch.setattr("av.core.config.CONFIG_FILE_PATH", cfg_file) + for name in ( + "AV_API_TOKEN_LIMIT_PARAMETER", + "AV_VISION_MAX_OUTPUT_TOKENS", + "AV_VISION_CHUNK_MAX_OUTPUT_TOKENS", + ): + monkeypatch.delenv(name, raising=False) + + config = get_config() + assert config.api_token_limit_parameter == "max_completion_tokens" + assert config.vision_max_output_tokens == 32 + assert config.vision_chunk_max_output_tokens == 64 + + monkeypatch.setenv("AV_API_TOKEN_LIMIT_PARAMETER", "max_tokens") + monkeypatch.setenv("AV_VISION_MAX_OUTPUT_TOKENS", "48") + monkeypatch.setenv("AV_VISION_CHUNK_MAX_OUTPUT_TOKENS", "80") + config = get_config() + assert config.api_token_limit_parameter == "max_tokens" + assert config.vision_max_output_tokens == 48 + assert config.vision_chunk_max_output_tokens == 80 + + +@pytest.mark.parametrize( + ("setting", "value"), + [ + ("api_token_limit_parameter", "unsupported"), + ("vision_max_output_tokens", 0), + ("vision_chunk_max_output_tokens", 0), + ], +) +def test_provider_token_limit_settings_reject_invalid_values(setting: str, value: object) -> None: + with pytest.raises(ValueError): + AVConfig(**{setting: value}) + + def test_config_show_includes_chat_output_cap() -> None: - with patch("av.cli.config_cmd.get_config", return_value=AVConfig(chat_max_output_tokens=777)), \ + config = AVConfig( + api_token_limit_parameter="max_completion_tokens", + vision_max_output_tokens=32, + vision_chunk_max_output_tokens=64, + chat_max_output_tokens=777, + ) + with patch("av.cli.config_cmd.get_config", return_value=config), \ patch("av.cli.config_cmd.output_json") as output: config_show() - assert output.call_args.args[0]["chat_max_output_tokens"] == 777 + shown = output.call_args.args[0] + assert shown["api_token_limit_parameter"] == "max_completion_tokens" + assert shown["vision_max_output_tokens"] == 32 + assert shown["vision_chunk_max_output_tokens"] == 64 + assert shown["chat_max_output_tokens"] == 777 def test_openai_fallback_does_not_read_oauth_unless_enabled() -> None: @@ -135,12 +189,24 @@ def test_openai_fallback_does_not_read_oauth_unless_enabled() -> None: def test_openai_fallback_can_use_oauth_only_when_explicitly_enabled() -> None: - config = AVConfig(provider="anthropic", openai_api_key="", allow_oauth_fallback=True) + config = AVConfig( + provider="anthropic", + openai_api_key="", + allow_oauth_fallback=True, + api_token_limit_parameter="max_completion_tokens", + vision_max_output_tokens=32, + vision_chunk_max_output_tokens=64, + chat_max_output_tokens=96, + ) with patch("av.providers.openai._openclaw_oauth_token", return_value="oauth-explicit"), \ patch("av.providers.openai._codex_oauth_token") as codex: fallback = get_openai_config(config) assert fallback is not None assert fallback.api_key == "oauth-explicit" + assert fallback.api_token_limit_parameter == "max_completion_tokens" + assert fallback.vision_max_output_tokens == 32 + assert fallback.vision_chunk_max_output_tokens == 64 + assert fallback.chat_max_output_tokens == 96 codex.assert_not_called() diff --git a/tests/test_provider.py b/tests/test_provider.py index 071931b..f2dc87f 100644 --- a/tests/test_provider.py +++ b/tests/test_provider.py @@ -85,6 +85,38 @@ def test_llm_request_uses_configured_output_token_cap() -> None: assert fake_client.chat.completions.create.call_args.kwargs["max_tokens"] == 321 +def test_selected_completion_token_field_is_used_exclusively_for_all_chat_calls( + tmp_path: Path, +) -> None: + frame = tmp_path / "frame.jpg" + frame.write_bytes(b"jpg") + response = SimpleNamespace( + choices=[SimpleNamespace(message=SimpleNamespace(content="result"))], + usage=SimpleNamespace(prompt_tokens=4, completion_tokens=2), + ) + fake_client = MagicMock() + fake_client.chat.completions.create.return_value = response + config = AVConfig( + api_key="explicit", + api_token_limit_parameter="max_completion_tokens", + vision_max_output_tokens=32, + vision_chunk_max_output_tokens=64, + chat_max_output_tokens=96, + ) + + with patch("av.providers.openai._client", return_value=fake_client): + captioner = OpenAICaptioner(config) + assert captioner.caption_frames([frame], [0.0])[0].text == "result" + assert captioner.caption_chunk([frame], [0.0], "describe") == "result" + llm = OpenAILLM(config) + assert llm.complete("question", "context") == "result" + assert llm.summarize("system", "content") == "result" + + calls = fake_client.chat.completions.create.call_args_list + assert [call.kwargs["max_completion_tokens"] for call in calls] == [32, 64, 96, 96] + assert all("max_tokens" not in call.kwargs for call in calls) + + def test_caption_fallback_is_disabled_by_default(tmp_path: Path) -> None: frame = tmp_path / "frame.jpg" frame.write_bytes(b"jpg") From 86b849d90525896c21a99b626dfa80861ef47dcf Mon Sep 17 00:00:00 2001 From: Sean Phan Date: Fri, 18 Sep 2026 20:21:21 +0000 Subject: [PATCH 05/12] docs: add reproducible Jev query cookbook --- README.md | 5 +- cookbook/README.md | 18 + cookbook/cost-model/README.md | 108 +++++ cookbook/cost-model/build_nb.py | 134 ++++++ cookbook/cost-model/model.py | 273 ++++++++++++ cookbook/cost-model/notebook.ipynb | 393 ++++++++++++++++++ cookbook/cost-model/scenario.pending.json | 125 ++++++ cookbook/cost-model/scenario.receipts.json | 251 +++++++++++ cookbook/cost-model/test_model.py | 187 +++++++++ cookbook/jev-refined-ask/README.md | 188 +++++++++ .../jev-refined-ask/test_transcribe_gemini.py | 74 ++++ cookbook/jev-refined-ask/transcribe_gemini.py | 352 ++++++++++++++++ cookbook/receipts/README.md | 17 + cookbook/receipts/asr.json | 77 ++++ .../receipts/cap-probe-32-incompatible.json | 32 ++ cookbook/receipts/caption-aborted.json | 106 +++++ cookbook/receipts/caption-smoke.json | 39 ++ cookbook/receipts/gemini38-baseline.json | 58 +++ 18 files changed, 2435 insertions(+), 2 deletions(-) create mode 100644 cookbook/README.md create mode 100644 cookbook/cost-model/README.md create mode 100644 cookbook/cost-model/build_nb.py create mode 100644 cookbook/cost-model/model.py create mode 100644 cookbook/cost-model/notebook.ipynb create mode 100644 cookbook/cost-model/scenario.pending.json create mode 100644 cookbook/cost-model/scenario.receipts.json create mode 100644 cookbook/cost-model/test_model.py create mode 100644 cookbook/jev-refined-ask/README.md create mode 100644 cookbook/jev-refined-ask/test_transcribe_gemini.py create mode 100644 cookbook/jev-refined-ask/transcribe_gemini.py create mode 100644 cookbook/receipts/README.md create mode 100644 cookbook/receipts/asr.json create mode 100644 cookbook/receipts/cap-probe-32-incompatible.json create mode 100644 cookbook/receipts/caption-aborted.json create mode 100644 cookbook/receipts/caption-smoke.json create mode 100644 cookbook/receipts/gemini38-baseline.json diff --git a/README.md b/README.md index 7c82f5a..90e979a 100644 --- a/README.md +++ b/README.md @@ -55,8 +55,9 @@ usage remains `null`. Missing, malformed, or out-of-range System One probabilities are treated as a refinement failure: `av` reports the fallback and does not invent a confidence. -See the [AV ask refinement cookbook](cookbook/README.md) for a reproducible recipe, -offline cost arithmetic, and receipt provenance. +See the [AV ask refinement cookbook](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook) +for the unmerged reproducible recipe, offline cost arithmetic, and sanitized +receipt provenance. FTS5 remains the first retrieval stage. An unscoped query with zero FTS matches does not scan the video archive or invoke sampled-frame inspection. diff --git a/cookbook/README.md b/cookbook/README.md new file mode 100644 index 0000000..ca01ca5 --- /dev/null +++ b/cookbook/README.md @@ -0,0 +1,18 @@ +# AV cookbook + +Runnable recipes for the open-source **av** CLI live here alongside the code. + +| Recipe | What it demonstrates | Evidence status | +|---|---|---| +| [Cost model](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook/cost-model) | Separate tokens, estimates, unknown costs, ingestion, queries, failures, reservations, and cap accounting | Completed component receipts; paired AV comparison incomplete | +| [Jev-refined ask](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook/jev-refined-ask) | Build a source-bound transcript sidecar, retrieve indexed moments, refine evidence, answer, and check support | Runnable recipe; no speed, cost, or quality parity claim | +| [Sanitized receipts](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook/receipts) | Completed ASR/baseline, caption smoke/abort, and blocked cap probe | No media, transcript/caption corpus, credentials, or private routes | + +The [original public cost notebook](https://github.com/PixelML/cookbook/tree/main/agentic-video/cost-model) +remains available at its existing URL. Its Composer demonstrations are historical +context, not measurements of this CLI. New AV reproduction results belong in +this cookbook with their own media, model, configuration, and cost provenance. + +The current evidence includes a completed Gemini 3.8 direct-video baseline, but +there is **no completed AV Grok+Jev comparison yet**. Do not infer speed, cost, or +quality parity from the component receipts. diff --git a/cookbook/cost-model/README.md b/cookbook/cost-model/README.md new file mode 100644 index 0000000..e77e3b6 --- /dev/null +++ b/cookbook/cost-model/README.md @@ -0,0 +1,108 @@ +# AV cost model: account for every stage + +Run the deterministic arithmetic offline from the repository root: + + python3 cookbook/cost-model/model.py --queries 1 100 1000 + python3 -m unittest discover -s cookbook/cost-model -p 'test_*.py' -v + python3 cookbook/cost-model/build_nb.py --check + +Python 3.11+ is sufficient. These commands need no API keys, network, AV +installation, notebook server, or downloaded media. The [editable notebook](https://github.com/PixelML/av/blob/codex/jev-query-cascade/cookbook/cost-model/notebook.ipynb) +uses the same [model.py](https://github.com/PixelML/av/blob/codex/jev-query-cascade/cookbook/cost-model/model.py). +Its checked-in source and outputs are generated by `build_nb.py`. + +## Evidence currently included + +[`scenario.receipts.json`](https://github.com/PixelML/av/blob/codex/jev-query-cascade/cookbook/cost-model/scenario.receipts.json) +integrates the sanitized receipts in [`cookbook/receipts`](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook/receipts). +The evidence is partial: + +| Attempt | Outcome | Recorded list-rate estimate | +|---|---|---:| +| External ASR ingestion | completed | $0.0685709 | +| ASR alignment experiment | failed and retained | $0.0069552 | +| Grok caption smoke | completed | $0.0006359 | +| Grok caption ingestion | aborted after 4 responses; 0 captions persisted | $0.0070886 | +| Gemini 3.8 native-video baseline | completed | $0.308076 | +| 32-token caption cap probe | HTTP 502/no route; usage unknown | unknown | + +Known token-derived list-rate estimates total **$0.3913266**. They are not billed +dollars. Provider/account or proxy billing remains unknown. Compute, storage, and +network allocation also remains unknown. + +The cumulative experiment cap is **$5**. Conservative reservations remain visible +and separate from spend: **$0.90** caption ingestion, **$0.20** query/judge/answer, +and **$0.01** cap probe. Known estimates plus reservations are **$1.5013266**, +leaving **$3.4986734** of estimate headroom. + +**No completed AV Grok+Jev comparison exists yet.** The receipts do not establish +speed, cost, or quality parity with the Gemini 3.8 baseline. The baseline used +native video/audio at its recorded sampling configuration; the incomplete AV arm +is not an equal-input comparison. + +Use [`scenario.pending.json`](https://github.com/PixelML/av/blob/codex/jev-query-cascade/cookbook/cost-model/scenario.pending.json) +as a blank template for another run. + +## What the model keeps separate + +| Category | Treatment | +|---|---| +| Measured tokens | Request/token counters are preserved independently of dollars | +| List-rate estimates | Measured tokens × recorded public rates; exact decimal arithmetic | +| Billed/provider/proxy cost | Unknown until invoice or account evidence exists | +| Compute/storage/network | Unknown until an allocation method is recorded | +| One-time ingestion | Charged once, never multiplied by query count | +| Per-query stages | Retrieval, judge, answer, and optional fallback scale with query count | +| Failures | Failed and aborted attempts remain in the experiment ledger | +| Reservations | Count against the cumulative cap but are not called spend | + +Every cost item has `usd`, `basis`, and an evidence `note`. Bases are `measured`, +`estimated`, `assumed`, `unknown`, or `not_used`. Dollar values use decimal +strings in generated output. Unknown values stay null; they never become zero. + +The experiment ledger is independent of projected repeatable costs. It records +successful requests, unsuccessful attempts, and smoke/probe calls exactly once. +A failed response can incur cost even when its output is rejected. Reservations +are a separate guardrail and are never added to the spend subtotal. + +## Fill a scenario from another AV run + +1. Record public media provenance, content hash, AV revision, exact model IDs, + sampling settings, commands, and observation period. +2. Capture one-time ingestion and every query stage separately. Preserve token + usage, elapsed time, retries, failures, and missing meters. +3. Copy `scenario.pending.json`, enter evidence and reservations, then run + `python3 cookbook/cost-model/model.py YOUR_SCENARIO.json --queries 100 1000`. +4. Inspect `unknown_terms`, `unknown_costs`, and cap accounting before reporting + any total or comparison. + +The ASR sidecar is imported with `av ingest VIDEO --transcript-json FILE`; its +external usage is not AV provider usage. A direct-video baseline is another run, +not an AV answer setting. Record its model, audio inclusion, video sampling, +resolution, cache settings, answer budget, and pricing basis independently. + +## The accounting + +For Q queries, the indexed model is: + + one_time_ingestion + + Q × (retrieval + judge + answer + fallback_frequency × fallback_cost) + + period_hosting + period_storage + +The uncached baseline is `Q × uncached_request_cost`. A cache-policy baseline is: + + Q × ((1 - hit_rate) × uncached_request_cost + hit_rate × hit_request_cost) + + cache_write_and_refresh_cost + cache_storage_cost + +`complete_total_usd` is null whenever a modeled term is missing. Ratios are +suppressed unless both the indexed path and baseline have complete totals. Even a +complete modeled total can contain estimates or assumptions and cannot establish +quality parity. + +## Historical context + +The [original public cookbook](https://github.com/PixelML/cookbook/tree/main/agentic-video/cost-model) +contains earlier Composer demonstrations, not AV measurements. Its assumptions +and pooled-repeat accuracy figures do not establish quality parity or gains for +this AV path. New AV claims require a completed paired workload with provenance, +sampling, cache policy, missing costs, and answer-quality review attached. diff --git a/cookbook/cost-model/build_nb.py b/cookbook/cost-model/build_nb.py new file mode 100644 index 0000000..afaa5d2 --- /dev/null +++ b/cookbook/cost-model/build_nb.py @@ -0,0 +1,134 @@ +#!/usr/bin/env python3 +"""Build or check the cost notebook using only the standard library.""" +from __future__ import annotations + +import argparse +import contextlib +import io +import json +import os +from pathlib import Path + +HERE = Path(__file__).resolve().parent + +# Edit these cells, then regenerate. The notebook does not duplicate model.py. +CELLS = [ + ("markdown", """# AV stage-cost model (offline) + +This notebook calculates with the same model.py used by the command line. +The checked-in receipt scenario includes completed ASR and direct-video baseline +attempts, but no completed AV Grok+Jev comparison. Measured tokens, list-rate +estimates, unknown costs, failures, and reservations remain separate. See +README.md for provenance and limitations. These cells never call a provider or +fetch media. +"""), + ("code", """import json +import sys +from pathlib import Path + +root = Path.cwd() +recipe = root if (root / "model.py").is_file() else root / "cookbook" / "cost-model" +sys.path.insert(0, str(recipe.resolve())) +from model import experiment_report, jsonable, scenario, token_cost + +input_path = recipe / "scenario.receipts.json" +inputs = json.loads(input_path.read_text()) +print(json.dumps(inputs["experiment"], indent=2)) +accounting = experiment_report(inputs) +print(json.dumps(jsonable({ + "measured_usage": accounting["measured_usage"], + "known_list_estimates_usd": accounting["list_rate_estimates"]["known_subtotal_usd"], + "unknown_costs": accounting["unknown_costs"], + "failures": accounting["failures"], + "reservations": accounting["reservations"], + "cumulative_cap_usd": accounting["cumulative_cap_usd"], + "known_plus_reserved_usd": accounting["known_plus_reserved_usd"], + "remaining_cap_usd": accounting["remaining_cap_usd"], +}), indent=2)) +"""), + ("markdown", """## Compare query volumes without hiding unknowns + +Indexing is charged once. Retrieval, judge, answer, and optional stronger +inspection are multiplied by query volume. Hosting and storage refer to the +same observation period. A complete modeled total can contain assumptions; it +is not necessarily a measured bill. This receipt scenario remains incomplete +because no completed AV Grok+Jev query exists. Repeated-query projections are +not new benchmark measurements. +"""), + ("code", """for queries in (1, 100, 1000): + result = scenario(inputs, queries) + print(json.dumps(jsonable({ + "queries": queries, + "known_subtotal_usd": result["indexed"]["known_subtotal_usd"], + "complete_total_usd": result["indexed"]["complete_total_usd"], + "unknown_terms": result["indexed"]["unknown_terms"], + "uncached_to_indexed_before_unknown_costs_ratio": result["uncached_to_indexed_before_unknown_costs_ratio"], + }), indent=2)) +"""), + ("markdown", """## Edit cache and fallback assumptions explicitly + +In the input JSON, record cache policy (TTL, reuse scope, refresh count and +storage duration), hit rate, all-in cache-hit request cost, writes and storage. +Reuse can span requests and users of one application. Do not infer a universal +cache policy from one request with zero cached tokens. The stronger fallback +frequency and per-invocation price are separate inputs; a missing price stays +unknown when the fallback runs. Historical Composer inputs are not AV evidence. +"""), + ("code", """result = scenario(inputs, 100) +print(json.dumps(jsonable({ + "cache_policy": result["cache_policy"], + "cache_hit_rate": result["cache_hit_rate"], + "cache_baseline": result["cache_policy_baseline"], + "fallback": inputs["query"]["stronger_fallback"], +}), indent=2)) +"""), +] + + +def build_notebook(): + cells = [] + namespace = {"__name__": "cost_notebook"} + execution = 0 + old_cwd = Path.cwd() + try: + os.chdir(HERE) + for index, (kind, source) in enumerate(CELLS): + cell = {"cell_type": kind, "id": f"av-cost-{index:02d}", "metadata": {}, "source": source.splitlines(keepends=True)} + if kind == "code": + execution += 1 + output = io.StringIO() + with contextlib.redirect_stdout(output): + exec(compile(source, f"cost-model-cell-{index}", "exec"), namespace) + cell["execution_count"] = execution + cell["outputs"] = [{"name": "stdout", "output_type": "stream", "text": output.getvalue().splitlines(keepends=True)}] + cells.append(cell) + finally: + os.chdir(old_cwd) + return { + "cells": cells, + "metadata": { + "kernelspec": {"display_name": "Python 3", "language": "python", "name": "python3"}, + "language_info": {"name": "python", "version": "3.11"}, + }, + "nbformat": 4, + "nbformat_minor": 5, + } + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--check", action="store_true", help="fail if source or outputs are stale") + args = parser.parse_args() + content = json.dumps(build_notebook(), indent=2, ensure_ascii=False) + "\n" + path = HERE / "notebook.ipynb" + if args.check: + if not path.exists() or path.read_text() != content: + raise SystemExit("Notebook is stale; run python3 cookbook/cost-model/build_nb.py") + print("Notebook source and outputs are current.") + else: + path.write_text(content) + print("Wrote notebook.ipynb (offline, deterministic source and outputs).") + + +if __name__ == "__main__": + main() diff --git a/cookbook/cost-model/model.py b/cookbook/cost-model/model.py new file mode 100644 index 0000000..8dd092e --- /dev/null +++ b/cookbook/cost-model/model.py @@ -0,0 +1,273 @@ +#!/usr/bin/env python3 +"""Offline stage-cost accounting. Python stdlib only; no credentials or network.""" +from __future__ import annotations + +import argparse +import json +from decimal import Decimal +from pathlib import Path + +BASES = {"measured", "estimated", "assumed", "unknown", "not_used"} +ZERO = Decimal("0") +ONE = Decimal("1") + + +def number(value, label: str) -> Decimal: + if isinstance(value, bool) or not isinstance(value, (int, float, str, Decimal)): + raise ValueError(f"{label} must be a non-negative finite number") + try: + result = Decimal(str(value)) + except Exception as exc: + raise ValueError(f"{label} must be numeric") from exc + if not result.is_finite() or result < ZERO: + raise ValueError(f"{label} must be a non-negative finite number") + return result + + +def fraction(value, label: str) -> Decimal: + result = number(value, label) + if result > ONE: + raise ValueError(f"{label} must be between 0 and 1") + return result + + +def token_cost(input_tokens, output_tokens, input_usd_per_million, output_usd_per_million): + """Return an estimated dollar amount; unreported usage remains unknown.""" + if any(v is None for v in (input_tokens, output_tokens, input_usd_per_million, output_usd_per_million)): + return None + for label, value in (("input_tokens", input_tokens), ("output_tokens", output_tokens)): + parsed = number(value, label) + if parsed != parsed.to_integral_value(): + raise ValueError(f"{label} must be an integer") + return ( + number(input_tokens, "input_tokens") * number(input_usd_per_million, "input rate") + + number(output_tokens, "output_tokens") * number(output_usd_per_million, "output rate") + ) / Decimal("1000000") + + +def expense(item: dict, label: str): + basis = item.get("basis") + if basis not in BASES: + raise ValueError(f"{label}: invalid basis") + if not isinstance(item.get("note"), str) or not item["note"].strip(): + raise ValueError(f"{label}: provide an evidence or assumption note") + value = item.get("usd") + if basis == "unknown": + if value is not None: + raise ValueError(f"{label}: unknown expense must have usd=null") + return None + if value is None: + raise ValueError(f"{label}: known expense needs usd") + value = number(value, label) + if basis == "not_used" and value != ZERO: + raise ValueError(f"{label}: not_used must have usd=0") + return value + + +def term(label: str, item: dict, multiplier=ONE): + value = expense(item, label) + if multiplier is None: + value = None + elif multiplier == ZERO: + value = ZERO + elif value is not None: + value *= multiplier + result = {"name": label, "usd": value, "basis": item["basis"], "note": item["note"]} + for name in ("category", "outcome", "receipt", "usage_ref"): + if name in item: + result[name] = item[name] + return result + + +def account(terms: list[dict]) -> dict: + missing = [entry["name"] for entry in terms if entry["usd"] is None] + known = sum((entry["usd"] for entry in terms if entry["usd"] is not None), ZERO) + return { + "terms": terms, + "known_subtotal_usd": known, + "unknown_terms": missing, + "complete_total_usd": None if missing else known, + "complete_means": "all modeled costs supplied; estimated and assumed inputs retain their provenance", + } + + +def experiment_spend(data: dict) -> dict: + """Actual experiment ledger, independent of projected query volume. + + Include failed/preflight/trial calls as separate items. The ledger must also + include successful calls and unknown incurred costs before it is complete. + """ + entries = data.get("experiment_spend") + if not isinstance(entries, list) or not entries: + return account([{ + "name": "experiment.not_recorded", "usd": None, "basis": "unknown", + "note": "No experiment spend ledger was supplied.", + }]) + return account([term(f"experiment.{i}.{entry['name']}", entry) for i, entry in enumerate(entries)]) + + +def measured_usage(data: dict) -> list[dict]: + """Validate token meters separately from any dollar estimate.""" + entries = data.get("measured_usage", []) + if not isinstance(entries, list): + raise ValueError("measured_usage must be a list") + result = [] + token_fields = ("input_tokens", "output_tokens", "thinking_tokens", "cached_tokens") + for index, entry in enumerate(entries): + if not isinstance(entry.get("name"), str) or not entry["name"].strip(): + raise ValueError(f"measured_usage.{index}: name is required") + row = {"name": entry["name"]} + for field in token_fields: + value = entry.get(field) + if value is None: + row[field] = None + continue + parsed = number(value, f"measured_usage.{index}.{field}") + if parsed != parsed.to_integral_value(): + raise ValueError(f"measured_usage.{index}.{field} must be an integer") + row[field] = int(parsed) + for field in ("requests", "outcome", "receipt", "note"): + if field in entry: + row[field] = entry[field] + result.append(row) + return result + + +def reservation_account(data: dict) -> dict: + """Track conservative cap reservations without treating them as spend.""" + entries = data.get("reservations", []) + if not isinstance(entries, list): + raise ValueError("reservations must be a list") + terms = [] + for index, entry in enumerate(entries): + if not isinstance(entry.get("name"), str) or not entry["name"].strip(): + raise ValueError(f"reservations.{index}: name is required") + if not isinstance(entry.get("note"), str) or not entry["note"].strip(): + raise ValueError(f"reservations.{index}: note is required") + terms.append({ + "name": f"reservation.{index}.{entry['name']}", + "usd": number(entry.get("usd"), f"reservations.{index}.usd"), + "status": entry.get("status", "retained"), + "note": entry["note"], + }) + total = sum((entry["usd"] for entry in terms), ZERO) + return {"terms": terms, "total_reserved_usd": total} + + +def experiment_report(data: dict) -> dict: + """Separate measurements, estimates, unknowns, failures, and reservations.""" + spend = experiment_spend(data) + unknown = account([ + term(f"unknown.{i}.{entry['name']}", entry) + for i, entry in enumerate(data.get("unknown_costs", [])) + ]) + reservations = reservation_account(data) + cap = number(data.get("cumulative_cap_usd"), "cumulative_cap_usd") + committed = spend["known_subtotal_usd"] + reservations["total_reserved_usd"] + if committed > cap: + raise ValueError("known list estimates plus reservations exceed cumulative cap") + failures = [ + entry for entry in spend["terms"] + unknown["terms"] + if entry.get("outcome") in {"failed", "aborted", "incompatible"} + ] + return { + "measured_usage": measured_usage(data), + "list_rate_estimates": spend, + "unknown_costs": unknown, + "failures": failures, + "reservations": reservations, + "cumulative_cap_usd": cap, + "known_plus_reserved_usd": committed, + "remaining_cap_usd": cap - committed, + "cap_warning": "Reservations are conservative guardrails, not billed or estimated spend.", + } + + +def scenario(data: dict, queries: int) -> dict: + if data.get("schema_version") != 1: + raise ValueError("schema_version must be 1") + if isinstance(queries, bool) or not isinstance(queries, int) or queries <= 0: + raise ValueError("queries must be a positive integer") + q = Decimal(queries) + indexing = data["indexing"] + if not isinstance(indexing, list) or not indexing: + raise ValueError("indexing must list at least one cost or explicit unknown") + indexed_terms = [term(f"indexing.{i}.{entry['name']}", entry) for i, entry in enumerate(indexing)] + for name in ("retrieval", "judge", "answer"): + indexed_terms.append(term(f"query.{name}", data["query"][name], q)) + fallback = data["query"]["stronger_fallback"] + frequency = fallback.get("frequency") + if frequency is not None: + frequency = fraction(frequency, "stronger_fallback.frequency") + indexed_terms.append(term("query.stronger_fallback", fallback["per_invocation"], None if frequency is None else q * frequency)) + for name in ("hosting", "storage"): + indexed_terms.append(term(f"period.{name}", data["period"][name])) + + baseline = data["baseline"] + cache = baseline["cache"] + if not isinstance(cache.get("policy"), str) or not cache["policy"].strip(): + raise ValueError("cache.policy must explain disabled or configured retention/reuse policy") + hit_rate = cache.get("hit_rate") + if hit_rate is not None: + hit_rate = fraction(hit_rate, "cache.hit_rate") + if cache["policy"] == "disabled" and hit_rate != ZERO: + raise ValueError("disabled cache requires hit_rate=0") + uncached = account([term("baseline.uncached_queries", baseline["uncached_per_query"], q)]) + cached_terms = [ + term("baseline.cache_misses", baseline["uncached_per_query"], None if hit_rate is None else q * (ONE - hit_rate)), + term("baseline.cache_hits", cache["hit_per_query"], None if hit_rate is None else q * hit_rate), + term("baseline.cache_writes", cache["write_period"]), + term("baseline.cache_storage", cache["storage_period"]), + ] + cached = account(cached_terms) + indexed = account(indexed_terms) + + def ratio(candidate): + # Suppress comparisons unless both sides have complete modeled totals. + total = candidate["complete_total_usd"] + denominator = indexed["complete_total_usd"] + if total is None or denominator is None or denominator <= ZERO: + return None + return total / denominator + + return { + "queries": queries, + "indexing_charged_once": True, + "indexed": indexed, + "uncached_baseline": uncached, + "cache_policy_baseline": cached, + "cache_policy": cache["policy"], + "cache_hit_rate": hit_rate, + "uncached_to_indexed_before_unknown_costs_ratio": ratio(uncached), + "cache_to_indexed_before_unknown_costs_ratio": ratio(cached), + "warning": "Ratios are emitted only for complete modeled totals and still inherit estimates or assumptions. They do not establish quality parity or library-scale behavior.", + } + + +def jsonable(value): + """Use decimal strings to preserve exact cost arithmetic in JSON output.""" + if isinstance(value, Decimal): + return format(value, "f") + if isinstance(value, dict): + return {key: jsonable(item) for key, item in value.items()} + if isinstance(value, list): + return [jsonable(item) for item in value] + return value + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("input", type=Path, nargs="?", default=Path(__file__).with_name("scenario.receipts.json")) + parser.add_argument("--queries", type=int, nargs="+", default=[1, 100, 1000]) + args = parser.parse_args() + data = json.loads(args.input.read_text()) + report = { + "experiment": data["experiment"], + "experiment_accounting": experiment_report(data), + "scenarios": [scenario(data, q) for q in args.queries], + } + print(json.dumps(jsonable(report), indent=2)) + + +if __name__ == "__main__": + main() diff --git a/cookbook/cost-model/notebook.ipynb b/cookbook/cost-model/notebook.ipynb new file mode 100644 index 0000000..74769d5 --- /dev/null +++ b/cookbook/cost-model/notebook.ipynb @@ -0,0 +1,393 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "id": "av-cost-00", + "metadata": {}, + "source": [ + "# AV stage-cost model (offline)\n", + "\n", + "This notebook calculates with the same model.py used by the command line.\n", + "The checked-in receipt scenario includes completed ASR and direct-video baseline\n", + "attempts, but no completed AV Grok+Jev comparison. Measured tokens, list-rate\n", + "estimates, unknown costs, failures, and reservations remain separate. See\n", + "README.md for provenance and limitations. These cells never call a provider or\n", + "fetch media.\n" + ] + }, + { + "cell_type": "code", + "id": "av-cost-01", + "metadata": {}, + "source": [ + "import json\n", + "import sys\n", + "from pathlib import Path\n", + "\n", + "root = Path.cwd()\n", + "recipe = root if (root / \"model.py\").is_file() else root / \"cookbook\" / \"cost-model\"\n", + "sys.path.insert(0, str(recipe.resolve()))\n", + "from model import experiment_report, jsonable, scenario, token_cost\n", + "\n", + "input_path = recipe / \"scenario.receipts.json\"\n", + "inputs = json.loads(input_path.read_text())\n", + "print(json.dumps(inputs[\"experiment\"], indent=2))\n", + "accounting = experiment_report(inputs)\n", + "print(json.dumps(jsonable({\n", + " \"measured_usage\": accounting[\"measured_usage\"],\n", + " \"known_list_estimates_usd\": accounting[\"list_rate_estimates\"][\"known_subtotal_usd\"],\n", + " \"unknown_costs\": accounting[\"unknown_costs\"],\n", + " \"failures\": accounting[\"failures\"],\n", + " \"reservations\": accounting[\"reservations\"],\n", + " \"cumulative_cap_usd\": accounting[\"cumulative_cap_usd\"],\n", + " \"known_plus_reserved_usd\": accounting[\"known_plus_reserved_usd\"],\n", + " \"remaining_cap_usd\": accounting[\"remaining_cap_usd\"],\n", + "}), indent=2))\n" + ], + "execution_count": 1, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "{\n", + " \"status\": \"partial_reproduction\",\n", + " \"implementation\": \"av\",\n", + " \"receipt_directory\": \"../receipts\",\n", + " \"note\": \"Completed ASR and direct-video baseline receipts exist, but no completed AV Grok+Jev comparison exists yet.\",\n", + " \"projection\": \"Repeated-query totals are projections from entered stage costs, not additional measured runs.\"\n", + "}\n", + "{\n", + " \"measured_usage\": [\n", + " {\n", + " \"name\": \"completed_asr\",\n", + " \"input_tokens\": 118228,\n", + " \"output_tokens\": 13241,\n", + " \"thinking_tokens\": null,\n", + " \"cached_tokens\": null,\n", + " \"requests\": 75,\n", + " \"outcome\": \"completed\",\n", + " \"receipt\": \"../receipts/asr.json\",\n", + " \"note\": \"Complete source audio submitted in fixed windows; transcript accuracy and word completeness were not verified.\"\n", + " },\n", + " {\n", + " \"name\": \"failed_asr_alignment\",\n", + " \"input_tokens\": 7634,\n", + " \"output_tokens\": 1866,\n", + " \"thinking_tokens\": null,\n", + " \"cached_tokens\": null,\n", + " \"requests\": 1,\n", + " \"outcome\": \"failed\",\n", + " \"receipt\": \"../receipts/asr.json\"\n", + " },\n", + " {\n", + " \"name\": \"caption_smoke\",\n", + " \"input_tokens\": 484,\n", + " \"output_tokens\": 93,\n", + " \"thinking_tokens\": 0,\n", + " \"cached_tokens\": 192,\n", + " \"requests\": 1,\n", + " \"outcome\": \"completed\",\n", + " \"receipt\": \"../receipts/caption-smoke.json\"\n", + " },\n", + " {\n", + " \"name\": \"aborted_caption_attempt\",\n", + " \"input_tokens\": 5004,\n", + " \"output_tokens\": 656,\n", + " \"thinking_tokens\": 0,\n", + " \"cached_tokens\": 768,\n", + " \"requests\": 4,\n", + " \"outcome\": \"aborted\",\n", + " \"receipt\": \"../receipts/caption-aborted.json\"\n", + " },\n", + " {\n", + " \"name\": \"gemini38_native_video_baseline\",\n", + " \"input_tokens\": 409548,\n", + " \"output_tokens\": 39,\n", + " \"thinking_tokens\": 205,\n", + " \"cached_tokens\": null,\n", + " \"requests\": 1,\n", + " \"outcome\": \"completed\",\n", + " \"receipt\": \"../receipts/gemini38-baseline.json\"\n", + " }\n", + " ],\n", + " \"known_list_estimates_usd\": \"0.3913266\",\n", + " \"unknown_costs\": {\n", + " \"terms\": [\n", + " {\n", + " \"name\": \"unknown.0.provider_or_proxy_billed_total\",\n", + " \"usd\": null,\n", + " \"basis\": \"unknown\",\n", + " \"note\": \"Account and proxy billed dollars were not available; list-rate estimates are not invoices.\",\n", + " \"category\": \"billing\"\n", + " },\n", + " {\n", + " \"name\": \"unknown.1.cap_probe_dispatch\",\n", + " \"usd\": null,\n", + " \"basis\": \"unknown\",\n", + " \"note\": \"The 32-token probe returned HTTP 502/no route with no usage; paid upstream dispatch is unverified.\",\n", + " \"category\": \"failed_attempt\",\n", + " \"outcome\": \"incompatible\",\n", + " \"receipt\": \"../receipts/cap-probe-32-incompatible.json\"\n", + " },\n", + " {\n", + " \"name\": \"unknown.2.compute_storage_network\",\n", + " \"usd\": null,\n", + " \"basis\": \"unknown\",\n", + " \"note\": \"Local compute, media/index storage, upload, and network costs were not allocated.\",\n", + " \"category\": \"infrastructure\"\n", + " }\n", + " ],\n", + " \"known_subtotal_usd\": \"0\",\n", + " \"unknown_terms\": [\n", + " \"unknown.0.provider_or_proxy_billed_total\",\n", + " \"unknown.1.cap_probe_dispatch\",\n", + " \"unknown.2.compute_storage_network\"\n", + " ],\n", + " \"complete_total_usd\": null,\n", + " \"complete_means\": \"all modeled costs supplied; estimated and assumed inputs retain their provenance\"\n", + " },\n", + " \"failures\": [\n", + " {\n", + " \"name\": \"experiment.1.failed_asr_alignment\",\n", + " \"usd\": \"0.0069552\",\n", + " \"basis\": \"estimated\",\n", + " \"note\": \"Rejected alignment attempt retained as experiment spend.\",\n", + " \"category\": \"failed_attempt\",\n", + " \"outcome\": \"failed\",\n", + " \"receipt\": \"../receipts/asr.json\",\n", + " \"usage_ref\": \"failed_asr_alignment\"\n", + " },\n", + " {\n", + " \"name\": \"experiment.3.aborted_caption_attempt\",\n", + " \"usd\": \"0.0070886\",\n", + " \"basis\": \"estimated\",\n", + " \"note\": \"Four measured responses before the guard stopped ingestion; no captions were persisted.\",\n", + " \"category\": \"failed_attempt\",\n", + " \"outcome\": \"aborted\",\n", + " \"receipt\": \"../receipts/caption-aborted.json\",\n", + " \"usage_ref\": \"aborted_caption_attempt\"\n", + " },\n", + " {\n", + " \"name\": \"unknown.1.cap_probe_dispatch\",\n", + " \"usd\": null,\n", + " \"basis\": \"unknown\",\n", + " \"note\": \"The 32-token probe returned HTTP 502/no route with no usage; paid upstream dispatch is unverified.\",\n", + " \"category\": \"failed_attempt\",\n", + " \"outcome\": \"incompatible\",\n", + " \"receipt\": \"../receipts/cap-probe-32-incompatible.json\"\n", + " }\n", + " ],\n", + " \"reservations\": {\n", + " \"terms\": [\n", + " {\n", + " \"name\": \"reservation.0.caption_ingest\",\n", + " \"usd\": \"0.90\",\n", + " \"status\": \"retained_in_cumulative_plan\",\n", + " \"note\": \"Conservative reservation, not spend.\"\n", + " },\n", + " {\n", + " \"name\": \"reservation.1.query_judge_and_answer\",\n", + " \"usd\": \"0.20\",\n", + " \"status\": \"retained_in_cumulative_plan\",\n", + " \"note\": \"Conservative reservation, not spend.\"\n", + " },\n", + " {\n", + " \"name\": \"reservation.2.max_completion_tokens_cap_probe\",\n", + " \"usd\": \"0.01\",\n", + " \"status\": \"retained_after_no_route\",\n", + " \"note\": \"Retained because usage and upstream dispatch were unknown.\"\n", + " }\n", + " ],\n", + " \"total_reserved_usd\": \"1.11\"\n", + " },\n", + " \"cumulative_cap_usd\": \"5\",\n", + " \"known_plus_reserved_usd\": \"1.5013266\",\n", + " \"remaining_cap_usd\": \"3.4986734\"\n", + "}\n" + ] + } + ] + }, + { + "cell_type": "markdown", + "id": "av-cost-02", + "metadata": {}, + "source": [ + "## Compare query volumes without hiding unknowns\n", + "\n", + "Indexing is charged once. Retrieval, judge, answer, and optional stronger\n", + "inspection are multiplied by query volume. Hosting and storage refer to the\n", + "same observation period. A complete modeled total can contain assumptions; it\n", + "is not necessarily a measured bill. This receipt scenario remains incomplete\n", + "because no completed AV Grok+Jev query exists. Repeated-query projections are\n", + "not new benchmark measurements.\n" + ] + }, + { + "cell_type": "code", + "id": "av-cost-03", + "metadata": {}, + "source": [ + "for queries in (1, 100, 1000):\n", + " result = scenario(inputs, queries)\n", + " print(json.dumps(jsonable({\n", + " \"queries\": queries,\n", + " \"known_subtotal_usd\": result[\"indexed\"][\"known_subtotal_usd\"],\n", + " \"complete_total_usd\": result[\"indexed\"][\"complete_total_usd\"],\n", + " \"unknown_terms\": result[\"indexed\"][\"unknown_terms\"],\n", + " \"uncached_to_indexed_before_unknown_costs_ratio\": result[\"uncached_to_indexed_before_unknown_costs_ratio\"],\n", + " }), indent=2))\n" + ], + "execution_count": 2, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "{\n", + " \"queries\": 1,\n", + " \"known_subtotal_usd\": \"0.0685709\",\n", + " \"complete_total_usd\": null,\n", + " \"unknown_terms\": [\n", + " \"indexing.0.caption\",\n", + " \"indexing.3.preprocessing\",\n", + " \"query.retrieval\",\n", + " \"query.judge\",\n", + " \"query.answer\",\n", + " \"period.hosting\",\n", + " \"period.storage\"\n", + " ],\n", + " \"uncached_to_indexed_before_unknown_costs_ratio\": null\n", + "}\n", + "{\n", + " \"queries\": 100,\n", + " \"known_subtotal_usd\": \"0.0685709\",\n", + " \"complete_total_usd\": null,\n", + " \"unknown_terms\": [\n", + " \"indexing.0.caption\",\n", + " \"indexing.3.preprocessing\",\n", + " \"query.retrieval\",\n", + " \"query.judge\",\n", + " \"query.answer\",\n", + " \"period.hosting\",\n", + " \"period.storage\"\n", + " ],\n", + " \"uncached_to_indexed_before_unknown_costs_ratio\": null\n", + "}\n", + "{\n", + " \"queries\": 1000,\n", + " \"known_subtotal_usd\": \"0.0685709\",\n", + " \"complete_total_usd\": null,\n", + " \"unknown_terms\": [\n", + " \"indexing.0.caption\",\n", + " \"indexing.3.preprocessing\",\n", + " \"query.retrieval\",\n", + " \"query.judge\",\n", + " \"query.answer\",\n", + " \"period.hosting\",\n", + " \"period.storage\"\n", + " ],\n", + " \"uncached_to_indexed_before_unknown_costs_ratio\": null\n", + "}\n" + ] + } + ] + }, + { + "cell_type": "markdown", + "id": "av-cost-04", + "metadata": {}, + "source": [ + "## Edit cache and fallback assumptions explicitly\n", + "\n", + "In the input JSON, record cache policy (TTL, reuse scope, refresh count and\n", + "storage duration), hit rate, all-in cache-hit request cost, writes and storage.\n", + "Reuse can span requests and users of one application. Do not infer a universal\n", + "cache policy from one request with zero cached tokens. The stronger fallback\n", + "frequency and per-invocation price are separate inputs; a missing price stays\n", + "unknown when the fallback runs. Historical Composer inputs are not AV evidence.\n" + ] + }, + { + "cell_type": "code", + "id": "av-cost-05", + "metadata": {}, + "source": [ + "result = scenario(inputs, 100)\n", + "print(json.dumps(jsonable({\n", + " \"cache_policy\": result[\"cache_policy\"],\n", + " \"cache_hit_rate\": result[\"cache_hit_rate\"],\n", + " \"cache_baseline\": result[\"cache_policy_baseline\"],\n", + " \"fallback\": inputs[\"query\"][\"stronger_fallback\"],\n", + "}), indent=2))\n" + ], + "execution_count": 3, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "{\n", + " \"cache_policy\": \"disabled for the single recorded baseline request; no reusable cache was explicitly created\",\n", + " \"cache_hit_rate\": \"0\",\n", + " \"cache_baseline\": {\n", + " \"terms\": [\n", + " {\n", + " \"name\": \"baseline.cache_misses\",\n", + " \"usd\": \"30.807600\",\n", + " \"basis\": \"estimated\",\n", + " \"note\": \"One completed Gemini 3.8 native-video query: measured provider tokens multiplied by recorded upstream list rates; actual billing is unknown.\"\n", + " },\n", + " {\n", + " \"name\": \"baseline.cache_hits\",\n", + " \"usd\": \"0\",\n", + " \"basis\": \"unknown\",\n", + " \"note\": \"No cache-hit request was measured.\"\n", + " },\n", + " {\n", + " \"name\": \"baseline.cache_writes\",\n", + " \"usd\": \"0\",\n", + " \"basis\": \"not_used\",\n", + " \"note\": \"No explicit cache write was requested.\"\n", + " },\n", + " {\n", + " \"name\": \"baseline.cache_storage\",\n", + " \"usd\": \"0\",\n", + " \"basis\": \"not_used\",\n", + " \"note\": \"No explicit cache storage was requested.\"\n", + " }\n", + " ],\n", + " \"known_subtotal_usd\": \"30.807600\",\n", + " \"unknown_terms\": [],\n", + " \"complete_total_usd\": \"30.807600\",\n", + " \"complete_means\": \"all modeled costs supplied; estimated and assumed inputs retain their provenance\"\n", + " },\n", + " \"fallback\": {\n", + " \"frequency\": 0,\n", + " \"per_invocation\": {\n", + " \"usd\": 0,\n", + " \"basis\": \"not_used\",\n", + " \"note\": \"No completed stronger-inspection query exists.\"\n", + " }\n", + " }\n", + "}\n" + ] + } + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "name": "python", + "version": "3.11" + } + }, + "nbformat": 4, + "nbformat_minor": 5 +} diff --git a/cookbook/cost-model/scenario.pending.json b/cookbook/cost-model/scenario.pending.json new file mode 100644 index 0000000..47ce336 --- /dev/null +++ b/cookbook/cost-model/scenario.pending.json @@ -0,0 +1,125 @@ +{ + "schema_version": 1, + "experiment": { + "status": "pending", + "implementation": "av", + "receipt": null, + "note": "Input template, not a benchmark result. Fill with a sanitized live AV receipt before making a claim.", + "projection": "Each query uses the entered average stage costs. Repeated-query totals are projections, not measured runs." + }, + "indexing": [ + { + "name": "caption", + "usd": null, + "basis": "unknown", + "note": "AV caption model usage and applicable provider price pending." + }, + { + "name": "external_asr", + "usd": null, + "basis": "unknown", + "note": "External ASR usage is separate from AV sidecar import; pending." + }, + { + "name": "embedding", + "usd": null, + "basis": "unknown", + "note": "Record model usage or explicit not_used if embeddings are disabled." + }, + { + "name": "preprocessing", + "usd": null, + "basis": "unknown", + "note": "One-time media preparation compute and network allocation is unmetered." + } + ], + "query": { + "retrieval": { + "usd": null, + "basis": "unknown", + "note": "Query embedding, retrieval compute, and network cost not metered." + }, + "judge": { + "usd": null, + "basis": "unknown", + "note": "Sum source relevance, scene refinement, answer support, and retries from the AV run." + }, + "answer": { + "usd": null, + "basis": "unknown", + "note": "Use AV answer usage and documented rate; token-derived dollars are estimates." + }, + "stronger_fallback": { + "frequency": 0, + "per_invocation": { + "usd": 0, + "basis": "not_used", + "note": "Disabled for the pending basic reproduction; set frequency and cost if enabled." + } + } + }, + "period": { + "label": "same period as entered query volume", + "hosting": { + "usd": null, + "basis": "unknown", + "note": "Host cost allocation is unmetered." + }, + "storage": { + "usd": null, + "basis": "unknown", + "note": "Index and media storage allocation is unmetered." + } + }, + "baseline": { + "uncached_per_query": { + "usd": null, + "basis": "unknown", + "note": "No paired direct-video baseline run is included yet." + }, + "cache": { + "policy": "unknown; record TTL, reuse scope, cache write/refresh count, and observation period", + "hit_rate": null, + "hit_per_query": { + "usd": null, + "basis": "unknown", + "note": "All-in cost of a cache-hit request, including uncached prompt and output." + }, + "write_period": { + "usd": null, + "basis": "unknown", + "note": "Cache creation and refresh costs for the same period." + }, + "storage_period": { + "usd": null, + "basis": "unknown", + "note": "Cache token-hours or corresponding storage cost for the same period." + } + } + }, + "experiment_spend": [ + { + "name": "all_incurred_costs", + "usd": null, + "basis": "unknown", + "note": "Pending actual run ledger: include successful requests, failed ASR attempts, preflight/trial calls, and unmetered incurred infrastructure. Keep trial costs separate from repeatable indexing inputs." + } + ], + "measured_usage": [], + "unknown_costs": [ + { + "name": "provider_or_proxy_billed_total", + "usd": null, + "basis": "unknown", + "note": "Replace with billing evidence if available; list-rate estimates are not billed dollars." + }, + { + "name": "compute_storage_network", + "usd": null, + "basis": "unknown", + "note": "Record compute, storage, and network allocation separately." + } + ], + "reservations": [], + "cumulative_cap_usd": "5" +} diff --git a/cookbook/cost-model/scenario.receipts.json b/cookbook/cost-model/scenario.receipts.json new file mode 100644 index 0000000..55935f5 --- /dev/null +++ b/cookbook/cost-model/scenario.receipts.json @@ -0,0 +1,251 @@ +{ + "schema_version": 1, + "experiment": { + "status": "partial_reproduction", + "implementation": "av", + "receipt_directory": "../receipts", + "note": "Completed ASR and direct-video baseline receipts exist, but no completed AV Grok+Jev comparison exists yet.", + "projection": "Repeated-query totals are projections from entered stage costs, not additional measured runs." + }, + "indexing": [ + { + "name": "caption", + "usd": null, + "basis": "unknown", + "note": "The caption ingestion was aborted after four responses and persisted no frame captions; no repeatable completed caption-ingestion cost exists." + }, + { + "name": "external_asr", + "usd": "0.0685709", + "basis": "estimated", + "note": "Measured tokens for the completed 75-window ASR run multiplied by recorded upstream list rates; billed dollars are unknown." + }, + { + "name": "embedding", + "usd": 0, + "basis": "not_used", + "note": "The documented partial reproduction used transcript-only FTS retrieval with embeddings disabled." + }, + { + "name": "preprocessing", + "usd": null, + "basis": "unknown", + "note": "Local media extraction, compute, storage, and network allocation were not priced." + } + ], + "query": { + "retrieval": { + "usd": null, + "basis": "unknown", + "note": "One offline zero-hit diagnostic exists, but no completed representative AV query workload was priced." + }, + "judge": { + "usd": null, + "basis": "unknown", + "note": "No completed Jev relevance/support sequence exists for the comparison." + }, + "answer": { + "usd": null, + "basis": "unknown", + "note": "No completed AV answer exists for the comparison." + }, + "stronger_fallback": { + "frequency": 0, + "per_invocation": { + "usd": 0, + "basis": "not_used", + "note": "No completed stronger-inspection query exists." + } + } + }, + "period": { + "label": "same period as entered query volume", + "hosting": { + "usd": null, + "basis": "unknown", + "note": "Compute and hosting allocation were not metered." + }, + "storage": { + "usd": null, + "basis": "unknown", + "note": "Media, index, transcript, and receipt storage allocation were not priced." + } + }, + "baseline": { + "uncached_per_query": { + "usd": "0.308076", + "basis": "estimated", + "note": "One completed Gemini 3.8 native-video query: measured provider tokens multiplied by recorded upstream list rates; actual billing is unknown." + }, + "cache": { + "policy": "disabled for the single recorded baseline request; no reusable cache was explicitly created", + "hit_rate": 0, + "hit_per_query": { + "usd": null, + "basis": "unknown", + "note": "No cache-hit request was measured." + }, + "write_period": { + "usd": 0, + "basis": "not_used", + "note": "No explicit cache write was requested." + }, + "storage_period": { + "usd": 0, + "basis": "not_used", + "note": "No explicit cache storage was requested." + } + } + }, + "measured_usage": [ + { + "name": "completed_asr", + "requests": 75, + "input_tokens": 118228, + "output_tokens": 13241, + "thinking_tokens": null, + "cached_tokens": null, + "outcome": "completed", + "receipt": "../receipts/asr.json", + "note": "Complete source audio submitted in fixed windows; transcript accuracy and word completeness were not verified." + }, + { + "name": "failed_asr_alignment", + "requests": 1, + "input_tokens": 7634, + "output_tokens": 1866, + "thinking_tokens": null, + "cached_tokens": null, + "outcome": "failed", + "receipt": "../receipts/asr.json" + }, + { + "name": "caption_smoke", + "requests": 1, + "input_tokens": 484, + "output_tokens": 93, + "thinking_tokens": 0, + "cached_tokens": 192, + "outcome": "completed", + "receipt": "../receipts/caption-smoke.json" + }, + { + "name": "aborted_caption_attempt", + "requests": 4, + "input_tokens": 5004, + "output_tokens": 656, + "thinking_tokens": 0, + "cached_tokens": 768, + "outcome": "aborted", + "receipt": "../receipts/caption-aborted.json" + }, + { + "name": "gemini38_native_video_baseline", + "requests": 1, + "input_tokens": 409548, + "output_tokens": 39, + "thinking_tokens": 205, + "cached_tokens": null, + "outcome": "completed", + "receipt": "../receipts/gemini38-baseline.json" + } + ], + "experiment_spend": [ + { + "name": "completed_asr", + "usd": "0.0685709", + "basis": "estimated", + "category": "one_time_ingestion", + "outcome": "completed", + "receipt": "../receipts/asr.json", + "usage_ref": "completed_asr", + "note": "Upstream list-rate estimate from measured tokens; not billed dollars." + }, + { + "name": "failed_asr_alignment", + "usd": "0.0069552", + "basis": "estimated", + "category": "failed_attempt", + "outcome": "failed", + "receipt": "../receipts/asr.json", + "usage_ref": "failed_asr_alignment", + "note": "Rejected alignment attempt retained as experiment spend." + }, + { + "name": "caption_smoke", + "usd": "0.0006359", + "basis": "estimated", + "category": "smoke_test", + "outcome": "completed", + "receipt": "../receipts/caption-smoke.json", + "usage_ref": "caption_smoke", + "note": "Cache-aware upstream list-rate estimate; proxy billing is unknown." + }, + { + "name": "aborted_caption_attempt", + "usd": "0.0070886", + "basis": "estimated", + "category": "failed_attempt", + "outcome": "aborted", + "receipt": "../receipts/caption-aborted.json", + "usage_ref": "aborted_caption_attempt", + "note": "Four measured responses before the guard stopped ingestion; no captions were persisted." + }, + { + "name": "gemini38_native_video_baseline", + "usd": "0.308076", + "basis": "estimated", + "category": "direct_video_baseline_query", + "outcome": "completed", + "receipt": "../receipts/gemini38-baseline.json", + "usage_ref": "gemini38_native_video_baseline", + "note": "Full-rate upstream list estimate from measured input, output, and thinking tokens; not billed dollars." + } + ], + "unknown_costs": [ + { + "name": "provider_or_proxy_billed_total", + "usd": null, + "basis": "unknown", + "category": "billing", + "note": "Account and proxy billed dollars were not available; list-rate estimates are not invoices." + }, + { + "name": "cap_probe_dispatch", + "usd": null, + "basis": "unknown", + "category": "failed_attempt", + "outcome": "incompatible", + "receipt": "../receipts/cap-probe-32-incompatible.json", + "note": "The 32-token probe returned HTTP 502/no route with no usage; paid upstream dispatch is unverified." + }, + { + "name": "compute_storage_network", + "usd": null, + "basis": "unknown", + "category": "infrastructure", + "note": "Local compute, media/index storage, upload, and network costs were not allocated." + } + ], + "reservations": [ + { + "name": "caption_ingest", + "usd": "0.90", + "status": "retained_in_cumulative_plan", + "note": "Conservative reservation, not spend." + }, + { + "name": "query_judge_and_answer", + "usd": "0.20", + "status": "retained_in_cumulative_plan", + "note": "Conservative reservation, not spend." + }, + { + "name": "max_completion_tokens_cap_probe", + "usd": "0.01", + "status": "retained_after_no_route", + "note": "Retained because usage and upstream dispatch were unknown." + } + ], + "cumulative_cap_usd": "5" +} diff --git a/cookbook/cost-model/test_model.py b/cookbook/cost-model/test_model.py new file mode 100644 index 0000000..f243c76 --- /dev/null +++ b/cookbook/cost-model/test_model.py @@ -0,0 +1,187 @@ +"""Arithmetic and incomplete-accounting invariants; no provider calls.""" +from __future__ import annotations + +import json +import unittest +from decimal import Decimal +from pathlib import Path + +from build_nb import build_notebook +from model import experiment_report, experiment_spend, scenario, token_cost + +HERE = Path(__file__).resolve().parent +RECEIPTS = HERE.parent / "receipts" + + +def cost(value, basis="assumed"): + return {"usd": value, "basis": basis, "note": "Synthetic arithmetic test input, not a benchmark receipt."} + + +def fixture(): + return { + "schema_version": 1, + "indexing": [{"name": "synthetic", **cost("2")}], + "query": { + "retrieval": cost("0.01"), + "judge": cost("0.02"), + "answer": cost("0.03"), + "stronger_fallback": {"frequency": 0, "per_invocation": cost(0, "not_used")}, + }, + "period": {"hosting": cost("3"), "storage": cost("4")}, + "baseline": { + "uncached_per_query": cost("1"), + "cache": { + "policy": "disabled", "hit_rate": 0, + "hit_per_query": cost(None, "unknown"), + "write_period": cost(0, "not_used"), + "storage_period": cost(0, "not_used"), + }, + }, + } + + +class CostAccountingTests(unittest.TestCase): + def test_one_time_index_is_not_multiplied_by_query_count(self): + # 2 indexing + 10*(.01+.02+.03) queries + 3 hosting + 4 storage + self.assertEqual(scenario(fixture(), 10)["indexed"]["complete_total_usd"], Decimal("9.60")) + + def test_token_cost_is_exact_and_missing_usage_stays_unknown(self): + self.assertEqual(token_cost(1000, 100, "2", "5"), Decimal("0.0025")) + self.assertIsNone(token_cost(None, 100, "2", "5")) + with self.assertRaises(ValueError): + token_cost(1.5, 0, 1, 1) + + def test_unknown_infrastructure_keeps_total_and_ratio_unknown(self): + data = fixture() + data["query"]["retrieval"] = cost(None, "unknown") + data["period"] = {"hosting": cost(None, "unknown"), "storage": cost(None, "unknown")} + result = scenario(data, 10) + # Known costs: 2 indexing + 10*(.02 judge + .03 answer) = 2.50 + self.assertEqual(result["indexed"]["known_subtotal_usd"], Decimal("2.50")) + self.assertIsNone(result["indexed"]["complete_total_usd"]) + self.assertIsNone(result["uncached_to_indexed_before_unknown_costs_ratio"]) + self.assertEqual(result["indexed"]["unknown_terms"], ["query.retrieval", "period.hosting", "period.storage"]) + + def test_fallback_unknown_price_is_unknown_only_when_it_can_run(self): + data = fixture() + data["query"]["stronger_fallback"]["per_invocation"] = cost(None, "unknown") + self.assertEqual(scenario(data, 10)["indexed"]["complete_total_usd"], Decimal("9.60")) + data["query"]["stronger_fallback"]["frequency"] = "0.2" + self.assertIsNone(scenario(data, 10)["indexed"]["complete_total_usd"]) + data["query"]["stronger_fallback"]["per_invocation"] = cost("0.5") + self.assertEqual(scenario(data, 10)["indexed"]["complete_total_usd"], Decimal("10.60")) + + def test_cache_hit_rate_combines_hits_misses_writes_and_storage(self): + data = fixture() + data["baseline"]["cache"] = { + "policy": "Synthetic TTL=1h, one reusable prefix, one write in period", + "hit_rate": "0.75", "hit_per_query": cost("0.10"), + "write_period": cost("2"), "storage_period": cost("3"), + } + # 10*(.25*1 + .75*.1) + 2 write + 3 storage + self.assertEqual(scenario(data, 10)["cache_policy_baseline"]["complete_total_usd"], Decimal("8.250")) + data["baseline"]["cache"]["hit_rate"] = None + result = scenario(data, 10) + self.assertIsNone(result["cache_policy_baseline"]["complete_total_usd"]) + self.assertIsNone(result["cache_to_indexed_before_unknown_costs_ratio"]) + + def test_disabled_cache_matches_uncached_requests(self): + result = scenario(fixture(), 10) + self.assertEqual(result["cache_policy_baseline"]["complete_total_usd"], result["uncached_baseline"]["complete_total_usd"]) + + def test_pending_receipt_cannot_emit_total_or_ratio(self): + data = json.loads((HERE / "scenario.pending.json").read_text()) + result = scenario(data, 100) + self.assertIsNone(result["indexed"]["complete_total_usd"]) + self.assertIsNone(result["uncached_to_indexed_before_unknown_costs_ratio"]) + self.assertIn("query.judge", result["indexed"]["unknown_terms"]) + + def test_receipt_ledger_preserves_cap_and_separates_unknowns(self): + data = json.loads((HERE / "scenario.receipts.json").read_text()) + report = experiment_report(data) + self.assertEqual(report["list_rate_estimates"]["known_subtotal_usd"], + Decimal("0.3913266")) + self.assertEqual(report["reservations"]["total_reserved_usd"], Decimal("1.11")) + self.assertEqual(report["known_plus_reserved_usd"], Decimal("1.5013266")) + self.assertEqual(report["remaining_cap_usd"], Decimal("3.4986734")) + self.assertIsNone(report["unknown_costs"]["complete_total_usd"]) + self.assertEqual( + [item["outcome"] for item in report["failures"]], + ["failed", "aborted", "incompatible"], + ) + self.assertEqual(report["measured_usage"][0]["input_tokens"], 118228) + + def test_sanitized_receipts_are_present_and_baseline_estimate_recomputes(self): + required = { + "asr.json", + "gemini38-baseline.json", + "caption-aborted.json", + "caption-smoke.json", + "cap-probe-32-incompatible.json", + } + self.assertTrue(required.issubset({path.name for path in RECEIPTS.glob("*.json")})) + receipt = json.loads((RECEIPTS / "gemini38-baseline.json").read_text()) + usage = receipt["usage"] + recomputed = token_cost( + usage["input_tokens"], + usage["output_tokens"] + usage["thinking_tokens"], + "0.75", + "3.75", + ) + self.assertEqual(recomputed, Decimal("0.308076")) + self.assertEqual( + Decimal(str(receipt["cost"]["full_rate_list_estimate_usd"])), + recomputed, + ) + self.assertFalse(receipt["limitations"]["paired_av_grok_jev_run_completed"]) + + def test_incomplete_av_side_suppresses_baseline_ratio(self): + data = json.loads((HERE / "scenario.receipts.json").read_text()) + result = scenario(data, 1) + self.assertEqual(result["uncached_baseline"]["complete_total_usd"], + Decimal("0.308076")) + self.assertIsNone(result["indexed"]["complete_total_usd"]) + self.assertIsNone(result["uncached_to_indexed_before_unknown_costs_ratio"]) + + def test_cap_rejects_estimates_plus_reservations_above_limit(self): + data = json.loads((HERE / "scenario.receipts.json").read_text()) + data["cumulative_cap_usd"] = "1.50" + with self.assertRaisesRegex(ValueError, "exceed cumulative cap"): + experiment_report(data) + + def test_invalid_costs_and_frequencies_fail(self): + for value in (-1, "NaN", "Infinity", True): + with self.subTest(value=value), self.assertRaises(ValueError): + data = fixture() + data["query"]["answer"] = cost(value) + scenario(data, 1) + for rate in (-0.1, 1.1, True): + with self.subTest(rate=rate), self.assertRaises(ValueError): + data = fixture() + data["query"]["stronger_fallback"]["frequency"] = rate + scenario(data, 1) + data = fixture() + data["query"]["answer"] = cost(0, "unknown") + with self.assertRaises(ValueError): + scenario(data, 1) + + def test_trial_spend_is_visible_without_changing_repeatable_projection(self): + data = fixture() + data["experiment_spend"] = [ + {"name": "successful_calls", **cost("2")}, + {"name": "failed_trial_calls", **cost("0.5")}, + ] + self.assertEqual(experiment_spend(data)["complete_total_usd"], Decimal("2.5")) + self.assertEqual(scenario(data, 10)["indexed"]["complete_total_usd"], Decimal("9.60")) + data["experiment_spend"].append({"name": "unmetered_attempt", **cost(None, "unknown")}) + self.assertIsNone(experiment_spend(data)["complete_total_usd"]) + self.assertEqual(experiment_spend(data)["known_subtotal_usd"], Decimal("2.5")) + self.assertIsNone(experiment_spend(fixture())["complete_total_usd"]) + + def test_notebook_source_and_executed_outputs_match_generator(self): + checked_in = json.loads((HERE / "notebook.ipynb").read_text()) + self.assertEqual(checked_in, build_notebook()) + + +if __name__ == "__main__": + unittest.main() diff --git a/cookbook/jev-refined-ask/README.md b/cookbook/jev-refined-ask/README.md new file mode 100644 index 0000000..893963c --- /dev/null +++ b/cookbook/jev-refined-ask/README.md @@ -0,0 +1,188 @@ +# Ask with Jev evidence refinement + +Use AV's indexed moments as candidate evidence, then refine relevance and scene +context before answering. This recipe describes the open-source CLI behavior; +it does not claim that an AV run reproduces historical Composer cost or quality. +See the [cost model](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook/cost-model) +and [sanitized receipts](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook/receipts) +for separate stage accounting and evidence provenance. + +## Run it + +Configure the ordinary AV answer provider first, then set your TypeSafe key in +the environment. Do not save real credentials in notebooks or receipts. + + av config setup + export AV_TYPESAFE_API_KEY='your-typesafe-key' + av ingest video.mp4 --captions + av ask "What happens when the speaker approaches the microphone?" + av ask "What happens when the speaker approaches the microphone?" --no-refine + +These commands make provider calls. Ingestion, answer generation, and configured +judgments can incur charges; the cost calculator itself stays offline. The +**TYPESAFE_API_KEY** alias also enables refinement. If **AV_TYPESAFE_API_KEY** is +set, it takes precedence. Without either key, AV uses the existing retrieval and +answer path. **--no-refine** opts out for a request; **AV_REFINE_ENABLED=false** +disables refinement through configuration. + +## Import an externally generated transcript + +If your caption provider does not transcribe audio, generate a transcript with a +separate ASR tool and use the public import seam: + + av ingest video.mp4 --transcript-json transcript.json --no-embed + +The input can be a root segment array or an object with a required **segments** +array and optional **model** string and **provenance** object: + + {"segments": [{"start_sec": 0, "end_sec": 2.4, "text": "Hello"}], + "model": "your-asr-model", "provenance": {"method": "external ASR"}} + +Timestamps must be numeric, finite, ordered within each segment, non-negative, +and within the actual video duration. The import preserves segment order, text, +and timestamps. It accepts one video at a time; an empty array is valid for +silence. Import does not meter the external ASR run or verify transcript accuracy. +Keep that run's token usage, approximate-timestamp caveat, and price separate from +AV's ingestion receipt. Use only public metadata in a published sidecar. + +The included [`transcribe_gemini.py`](https://github.com/PixelML/av/blob/codex/jev-query-cascade/cookbook/jev-refined-ask/transcribe_gemini.py) +is a standard-library helper for Gemini audio transcription. It does not bundle +media or transcripts. It accepts any readable local source supported by ffmpeg: + + export GEMINI_API_KEY='your-key' + python3 cookbook/jev-refined-ask/transcribe_gemini.py \ + ./video.mp4 ./asr-run --model gemini-3.5-flash-lite \ + --budget-usd 0.50 --prior-reserved-usd 0 + av ingest ./video.mp4 --transcript-json ./asr-run/transcript.json --no-embed + +The helper supports positive sub-second inputs, folds only a remainder shorter +than one second into the preceding window, and uses no automatic retries. Before +either provider action it writes a conservative reservation to an fsync'd ledger. +Reservations survive failures and restarts; use `--prior-reserved-usd` for attempts +made outside the output directory. + +`run-manifest.json` binds the output directory to the source hash, source size, +duration, model, prompt, generation configuration, pricing inputs, window plan, +and recipe revision. Completed chunk and raw-response caches also carry that +binding. JSON artifacts are atomically replaced. A valid saved provider response +can be recovered after interruption without issuing a duplicate generation call; +a mismatch stops and requires a new directory. + +For a non-default model, pass both `--input-usd-per-million` and +`--output-usd-per-million`. These rates enforce a local reservation cap and +produce a list-rate estimate only. Actual billed/account cost, compute, storage, +and network remain unknown unless separately evidenced. + +## Dense captions with a separate transcript + +For a bounded image-caption recipe, select the caption provider and model with +**AV_API_BASE_URL**, **AV_API_KEY**, and **AV_VISION_MODEL**, then import the +external transcript in the same ingestion: + + av ingest video.mp4 --transcript-json transcript.json --dense-vision \ + --fps-sample 0.0666667 --max-frames 300 --no-embed --db ./recipe.db + av ask "Your question" --top-k 5 --db ./recipe.db + +The example samples about one still every 15 seconds, with a maximum of 300 +frames. That budget is appropriate for about 75 minutes; the frame limit caps +extraction instead of redistributing samples across a longer video. Record the +actual **dense_caption_frames** and sampling settings from ingestion output. +Use a fresh database for this FTS-only example: **--no-embed** prevents new +embeddings, but does not remove embeddings from an existing database. + +**--dense-vision** runs this dense-frame path. **--captions** additionally selects +the cascade caption/summary path; combining them changes the workload and cost. +Set **AV_CHAT_MODEL** and, if needed, the provider environment for the answer +command separately. The caption VLM, external ASR, answer model, and direct-video +baseline are independent choices. Neither this command example nor an output +receipt establishes answer quality without checking the answer against the media. + +## What happens + +1. Retrieve timestamped index artifacts for the question. The index may contain + transcripts, captions, or both; material not represented in it can be missed. +2. Ask Jev for source relevance. Relevant text is not yet proof of an answer. +3. Refine bounded scene context, combine overlaps, rank candidates, and cap the + context supplied to the configured answer model. +4. Generate an answer with citations, then ask a separate support question about + whether that evidence supports the answer's material claims. +5. If support is insufficient or unavailable, optionally inspect bounded sampled + frames using an explicitly configured stronger model; otherwise return an + uncertain result. A successful inspection also receives a support check. + +A successful judgment that rejects every source returns no supported evidence; +it does not silently revert to rejected raw hits. If refinement itself fails, +AV warns and can answer from raw retrieval with heuristic confidence. A failed +support judgment is labeled unknown, not supported. + +## Configuration + +| Environment variable | Default | Purpose | +|---|---|---| +| AV_CHAT_MAX_OUTPUT_TOKENS | 1024 | Output-token cap passed to the configured answer provider | +| AV_TYPESAFE_MODEL | jev-latest | Jev model used for judgments | +| AV_TYPESAFE_TIMEOUT_SEC | 30 | Timeout for a judgment request | +| AV_TYPESAFE_MAX_RETRIES | 1 | Retry bound | +| AV_REFINE_RELEVANCE_MIN | 0.5 | Source relevance threshold | +| AV_REFINE_SUPPORT_MIN | 0.5 | Separate answer-support threshold | +| AV_REFINE_MAX_SCENES | 8 | Maximum merged scenes sent to synthesis | +| AV_REFINE_BATCH_SIZE | 10 | Maximum hits in a relevance request | +| AV_REFINE_CONTEXT_EVENTS | 3 | Bounded neighboring-event context on each side | + +The explicit TypeSafe endpoint is configurable with **AV_TYPESAFE_ENDPOINT**. +The thresholds are routing settings, not calibrated probabilities of correctness. +FTS retrieval remains the first stage: an unscoped query with no matching hits +does not inspect the whole archive with the stronger model. + +## Optional stronger inspection + +Configure **AV_STRONG_VISION_API_BASE_URL**, **AV_STRONG_VISION_MODEL**, and +**AV_STRONG_VISION_API_KEY** when the endpoint requires authentication. AV does +not choose a stronger paid provider automatically. Source media must still be +available at the indexed location, and ffmpeg must be installed. + +| Budget setting | Default | Meaning | +|---|---|---| +| AV_INSPECTION_MAX_WINDOWS | 2 | Maximum selected scene windows | +| AV_INSPECTION_MAX_SECONDS | 120 | Total duration represented by selected windows | +| AV_INSPECTION_MAX_FRAMES | 12 | Total sampled-frame budget across attempts | +| AV_INSPECTION_MAX_ATTEMPTS | 1 | At most 2 attempts may be configured | +| AV_INSPECTION_DENSE_PASS | false | Allow a second sampling pass within the total budget | + +This path sends sampled still images. It does not send native video or audio and +cannot establish continuity, timing between unsampled frames, or inaudible spoken +content. More frames are not a guarantee of better evidence. Record stronger +inspection costs and its invocation rate separately when comparing workloads. + +## Read the result + +The answer, citations, and confidence remain available. Refined results also expose +**route**, **evidence_status**, **confidence_basis**, **refinement**, **warnings**, +**inspected_windows**, **ask_settings**, and **stage_usage**. Treat confidence +according to its basis. The ordinary path also reports answer/query-embedding +usage and its answer settings. A route ending in **answer_failed** preserves +available stage usage and sets **evidence_status=answer_unavailable**; inspect +these fields instead of treating any JSON response as a successful answer. + +| Result | Interpretation | +|---|---| +| refined / supported | Jev support check met the configured threshold | +| refined_no_results | No retrieval hits or all retrieved sources rejected | +| refinement_fallback / raw_unjudged | Refinement failed; raw retrieval answer has not been judged | +| refined_uncertain | Evidence did not support a reliable answer or support remains unknown | +| vision_inspected / sampled_frames_supported | Bounded frames yielded an answer that passed support checking | +| legacy, refined, or refinement_fallback route ending in answer_failed | Answer provider failed; usage can be incomplete and incurred costs remain possible | + +Stage usage covers relevance, boundary, answer, support, vision, and query +embedding when reported. +Request counts can include retry attempts; inspect metering completeness before +using aggregate usage as a full cost receipt. +Missing token counts and unpriced retrieval infrastructure are unknown, not zero. The +metadata describes execution and evidence routing; it is not a complete billing +receipt. No benchmark in this recipe establishes quality improvement, parity +with direct-video models, or a universal cost ratio. + +The current receipts include completed ASR and Gemini 3.8 baseline calls, an +aborted caption attempt, a successful image smoke call, and a 32-token cap probe +blocked by HTTP 502/no route. **No completed AV Grok+Jev comparison exists yet.** +Do not claim speed, cost, or quality parity from these component attempts. diff --git a/cookbook/jev-refined-ask/test_transcribe_gemini.py b/cookbook/jev-refined-ask/test_transcribe_gemini.py new file mode 100644 index 0000000..08a2f42 --- /dev/null +++ b/cookbook/jev-refined-ask/test_transcribe_gemini.py @@ -0,0 +1,74 @@ +"""Offline invariants for the public Gemini transcript helper.""" +from __future__ import annotations + +import importlib.util +import json +from pathlib import Path +import tempfile +import unittest + + +HERE = Path(__file__).resolve().parent +SPEC = importlib.util.spec_from_file_location( + "transcribe_gemini", HERE / "transcribe_gemini.py" +) +MODULE = importlib.util.module_from_spec(SPEC) +assert SPEC.loader is not None +SPEC.loader.exec_module(MODULE) + + +class TranscriptRecipeTests(unittest.TestCase): + def test_sub_second_source_is_one_fractional_window(self): + self.assertEqual(MODULE.windows(0.25), [(0, 0.25)]) + + def test_short_tail_is_folded_into_previous_window(self): + self.assertEqual(MODULE.windows(60.5), [(0, 60.5)]) + self.assertEqual(MODULE.windows(61), [(0, 60), (60, 61)]) + + def test_manifest_binds_nonempty_directory_to_exact_source_and_config(self): + with tempfile.TemporaryDirectory() as temporary: + directory = Path(temporary) / "run" + manifest = {"source_sha256": "abc", "model": "example", "prompt": "v1"} + MODULE.bind_directory(directory, manifest) + MODULE.bind_directory(directory, manifest) + with self.assertRaisesRegex(ValueError, "manifest mismatch"): + MODULE.bind_directory(directory, {**manifest, "prompt": "v2"}) + + def test_prior_and_persisted_reservations_survive_resume(self): + with tempfile.TemporaryDirectory() as temporary: + path = Path(temporary) / "usage.jsonl" + first = MODULE.Ledger(path, "5", "0.90") + attempt = first.reserve({"stage": "asr"}, "0.20") + first.record({**attempt, "status": "failed"}) + # External reservations are supplied cumulatively on every resume; + # this directory contributes its persisted 0.20 reservation. + resumed = MODULE.Ledger(path, "5", "0.91") + self.assertEqual(str(resumed.reserved), "1.11") + self.assertEqual(str(resumed.remaining), "3.89") + + def test_response_cache_is_bound_and_replayable_without_network(self): + binding = { + "manifest_sha256": "manifest", + "source_sha256": "source", + "model": "model", + "chunk": 0, + "start_sec": 0, + "end_sec": 0.5, + } + response = { + "usageMetadata": {"promptTokenCount": 10, "candidatesTokenCount": 2}, + "candidates": [{ + "finishReason": "STOP", + "content": {"parts": [{"text": json.dumps({ + "text": "hello", "silence": False + })}]}, + }], + } + artifact = MODULE.response_artifact(response, binding, 0.1) + self.assertEqual(MODULE.parse_response_artifact(artifact, binding), ("hello", False)) + with self.assertRaisesRegex(ValueError, "cached response mismatch"): + MODULE.parse_response_artifact(artifact, {**binding, "model": "other"}) + + +if __name__ == "__main__": + unittest.main() diff --git a/cookbook/jev-refined-ask/transcribe_gemini.py b/cookbook/jev-refined-ask/transcribe_gemini.py new file mode 100644 index 0000000..d1fe32d --- /dev/null +++ b/cookbook/jev-refined-ask/transcribe_gemini.py @@ -0,0 +1,352 @@ +#!/usr/bin/env python3 +"""Create an AV transcript sidecar using metered Gemini audio windows. + +Requires Python 3.11+, ffmpeg, ffprobe, and GEMINI_API_KEY. No automatic retries. +Run --help for arguments. Timestamps are audio-window bounds, not word alignment. +Every cache artifact is bound to the source hash and complete run configuration. +""" +import argparse +import base64 +import concurrent.futures +from decimal import Decimal, InvalidOperation +import hashlib +import json +import math +import os +from pathlib import Path +import subprocess +import threading +import time +import urllib.parse +import urllib.request +import uuid + +DEFAULT_MODEL = "gemini-3.5-flash-lite" +API = "https://generativelanguage.googleapis.com/v1beta/models/" +PROMPT = ('Transcribe every audible spoken word faithfully in its original language. ' + 'Return JSON only: {"text":"verbatim speech","silence":false}. ' + 'Do not summarize or infer missing words. Preserve numbers and percentages as heard. ' + 'Use [inaudible] for unintelligible speech. If there is no speech, return ' + '{"text":"","silence":true}. Do not provide timestamps.') +GENERATION = { + "temperature": 0, "maxOutputTokens": 2048, "responseMimeType": "application/json", + "responseSchema": {"type": "OBJECT", "properties": { + "text": {"type": "STRING"}, "silence": {"type": "BOOLEAN"}}, + "required": ["text", "silence"]}, + "thinkingConfig": {"thinkingLevel": "minimal"}, +} +MAX_INPUT_TOKENS = 3500 +MAX_AUDIO_BYTES = 3000000 +RECIPE_REVISION = 2 + + +def require(condition, message): + if not condition: + raise ValueError(message) + + +def amount(value): + try: + result = Decimal(str(value)) + except (InvalidOperation, ValueError): + raise ValueError("amount must be finite and non-negative") from None + require(result.is_finite() and result >= 0, "amount must be finite and non-negative") + return result + + +def windows(duration): + require(isinstance(duration, (int, float)) and not isinstance(duration, bool) + and math.isfinite(duration) and 0 < duration <= 4501, + "source duration must be positive and at most 4501 seconds") + starts = list(range(0, math.ceil(duration), 60)) + if len(starts) > 1 and duration - starts[-1] < 1: + starts.pop() # Fold a short remainder into the preceding audio window. + require(len(starts) <= 75, "maximum 75 audio windows") + return [(start, starts[i + 1] if i + 1 < len(starts) else duration) + for i, start in enumerate(starts)] + + +def parse_transcription(value): + # The measured run included one singleton-array response. Preserve its raw + # response while accepting exactly this normalization, not arbitrary arrays. + if isinstance(value, list) and len(value) == 1: + value = value[0] + require(isinstance(value, dict) and isinstance(value.get("text"), str) + and isinstance(value.get("silence"), bool), "invalid response schema") + text = value["text"].strip() + require(value["silence"] == (not bool(text)), "inconsistent silence/text") + return text, value["silence"] + + +def fingerprint(value): + return hashlib.sha256(json.dumps(value, sort_keys=True, separators=(",", ":")) + .encode()).hexdigest() + + +def write_json(path, value): + """Atomically replace a JSON artifact so interrupted writes are never cache hits.""" + temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp") + try: + with temporary.open("x") as stream: + stream.write(json.dumps(value, ensure_ascii=False, indent=2) + "\n") + stream.flush() + os.fsync(stream.fileno()) + os.replace(temporary, path) + directory_fd = os.open(path.parent, os.O_RDONLY) + try: + os.fsync(directory_fd) + finally: + os.close(directory_fd) + finally: + temporary.unlink(missing_ok=True) + + +def bind_directory(directory, manifest): + directory.mkdir(parents=True, exist_ok=True) + path = directory / "run-manifest.json" + if path.exists(): + require(json.loads(path.read_text()) == manifest, "cached manifest mismatch; use a new directory") + else: + require(not any(directory.iterdir()), "use an empty output directory") + write_json(path, manifest) + + +class Ledger: + """Persist reservations before sending requests, including abandoned attempts. + + A reservation is a conservative list-price guard, not a measured bill. It is + never released on failure, so a manual resume cannot forget prior attempts. + """ + def __init__(self, path, budget, prior): + self.path = path + self.lock = threading.Lock() + self.budget = amount(budget) + self.reserved = amount(prior) + if path.exists(): + for line in path.read_text().splitlines(): + row = json.loads(line) + if row["status"] == "reserved": + self.reserved += amount(row["reserved_max_usd"]) + require(self.reserved <= self.budget, "prior reservations exceed budget") + + def _write(self, row): + with self.path.open("a") as stream: + stream.write(json.dumps(row) + "\n") + stream.flush() + os.fsync(stream.fileno()) + + def reserve(self, base, ceiling): + with self.lock: + ceiling = amount(ceiling) + require(self.reserved + ceiling <= self.budget, "ASR reservation budget exceeded") + attempt = dict(base, attempt_id=uuid.uuid4().hex, + reserved_max_usd=str(ceiling)) + self._write(dict(attempt, status="reserved")) + self.reserved += ceiling + return attempt + + def record(self, row): + with self.lock: + self._write(row) + + @property + def remaining(self): + return self.budget - self.reserved + + +def post(key, model, action, payload): + request = urllib.request.Request( + API + urllib.parse.quote(model, safe="") + ":" + action, + data=json.dumps(payload).encode(), + headers={"x-goog-api-key": key, "Content-Type": "application/json"}) + with urllib.request.urlopen(request, timeout=90) as response: + return json.load(response) + + +def response_artifact(response, binding, elapsed): + usage = response.get("usageMetadata") or {} + candidate = (response.get("candidates") or [{}])[0] + finish = candidate.get("finishReason") + text = "".join(part.get("text", "") for part in candidate.get("content", {}).get("parts", [])) + return dict(binding, usage=usage, finish_reason=finish, text=text, + wall_seconds=elapsed) + + +def parse_response_artifact(artifact, binding): + for name, expected in binding.items(): + require(artifact.get(name) == expected, "cached response mismatch; use a new directory") + require(artifact.get("finish_reason") == "STOP", + "cached generation is incomplete; use a new directory") + return parse_transcription(json.loads(artifact["text"])) + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("source", type=Path) + parser.add_argument("output_dir", type=Path) + parser.add_argument("--model", default=DEFAULT_MODEL) + parser.add_argument("--budget-usd", default="0.50", + help="Conservative upstream list-price reservation limit, not a billing limit") + parser.add_argument("--prior-reserved-usd", default="0", + help="Reservations from earlier attempts outside this output directory") + parser.add_argument("--input-usd-per-million") + parser.add_argument("--output-usd-per-million") + parser.add_argument("--first-only", action="store_true", + help="Process only the first window; do not produce a full transcript") + args = parser.parse_args(argv) + require(bool(args.model.strip()), "model must not be empty") + budget, prior = amount(args.budget_usd), amount(args.prior_reserved_usd) + require(budget > 0 and prior <= budget, "invalid budget or prior reservation") + require(bool(args.input_usd_per_million) == bool(args.output_usd_per_million), + "supply both custom rate arguments") + custom_rates = args.input_usd_per_million is not None + require(args.model == DEFAULT_MODEL or custom_rates, + "a different model requires explicit input and output rates and compatible API semantics") + input_rate = amount(args.input_usd_per_million if custom_rates else "0.30") + output_rate = amount(args.output_usd_per_million if custom_rates else "2.50") + rates = {"input_per_million": str(input_rate), "output_per_million": str(output_rate), + "basis": "upstream list-price estimate; account billing is unverified", + "source": "user supplied" if custom_rates else "https://ai.google.dev/gemini-api/docs/pricing", + "as_of": None if custom_rates else "2026-09-18", + "input_policy": "all input conservatively priced at the audio rate; no cache discount"} + key = os.environ.get("GEMINI_API_KEY", "") + require(bool(key), "GEMINI_API_KEY is required") + require(args.source.is_file(), "source must be a readable file") + duration = float(json.loads(subprocess.check_output([ + "ffprobe", "-v", "quiet", "-show_format", "-of", "json", str(args.source)]))["format"]["duration"]) + intervals = windows(duration) + digest = hashlib.sha256() + with args.source.open("rb") as stream: + for chunk in iter(lambda: stream.read(1048576), b""): + digest.update(chunk) + source_hash = digest.hexdigest() + manifest = {"schema_version": 1, "recipe_revision": RECIPE_REVISION, + "source_sha256": source_hash, "source_bytes": args.source.stat().st_size, + "model": args.model, + "source_duration_sec": duration, "windows": intervals, + "prompt": PROMPT, "generation_config": GENERATION, + "max_input_tokens": MAX_INPUT_TOKENS, "max_audio_bytes": MAX_AUDIO_BYTES, + "audio": {"format": "flac", "channels": 1, "sample_rate_hz": 16000}, + "max_workers": 2, "retry_count": 0, "rates": rates, + "prior_reserved_usd": str(prior), "budget_usd": str(budget)} + # Normalize tuples to JSON arrays before equality checking on resume. + manifest = json.loads(json.dumps(manifest)) + manifest_id = fingerprint(manifest) + bind_directory(args.output_dir, manifest) + ledger = Ledger(args.output_dir / "asr-usage.jsonl", budget, prior) + stopped = threading.Event() + + def one(index): + require(not stopped.is_set(), "another window failed; no new request started") + start, end = intervals[index] + output = args.output_dir / f"chunk-{index:03d}.json" + binding = {"manifest_sha256": manifest_id, "source_sha256": source_hash, + "model": args.model, "chunk": index, "start_sec": start, + "end_sec": end} + if output.exists(): + cached = json.loads(output.read_text()) + require(all(cached.get(name) == expected for name, expected in binding.items()), + "cached chunk mismatch; use a new directory") + parse_transcription(cached) + return cached + response_path = args.output_dir / f"chunk-{index:03d}.response.json" + if response_path.exists(): + artifact = json.loads(response_path.read_text()) + text, silent = parse_response_artifact(artifact, binding) + row = dict(binding, text=text, silence=silent) + write_json(output, row) + print(json.dumps({"chunk": index, "status": "recovered_from_response"}), flush=True) + return row + audio = args.output_dir / f"chunk-{index:03d}.flac" + # Rebuild unfinished windows; an existing audio file is not a valid cache. + subprocess.run(["ffmpeg", "-v", "error", "-threads", "2", "-ss", str(start), + "-i", str(args.source), "-t", str(end - start), "-vn", "-ac", "1", + "-ar", "16000", "-c:a", "flac", "-y", str(audio)], check=True, timeout=90) + require(0 < audio.stat().st_size < MAX_AUDIO_BYTES, "audio bytes ceiling") + contents = [{"role": "user", "parts": [{"text": PROMPT}, {"inlineData": { + "mimeType": "audio/flac", "data": base64.b64encode(audio.read_bytes()).decode()}}]}] + # Reserve a conservative maximum before either provider request. Count-token + # calls are not assumed free, and reservations remain charged to the local + # cap after failures or interrupted runs. + ceiling = (MAX_INPUT_TOKENS * input_rate + + GENERATION["maxOutputTokens"] * output_rate) / 1000000 + base = ledger.reserve(dict(binding, stage="asr", provider_requests_max=2, + automatic_retries=0, rates=rates), ceiling) + probe_began = time.monotonic() + try: + count_response = post(key, args.model, "countTokens", {"contents": contents}) + except Exception as exc: + ledger.record(dict(base, status="failed", failed_action="countTokens", + error_type=type(exc).__name__, http_status=getattr(exc, "code", None), + input_tokens=None, output_tokens=None, completeness="unknown", + wall_seconds=time.monotonic() - probe_began)) + raise RuntimeError("ASR token-count request failed; reservation retained") from None + count = count_response.get("totalTokens") + require(isinstance(count, int) and not isinstance(count, bool) + and 0 < count <= MAX_INPUT_TOKENS, "input token ceiling") + require(not stopped.is_set(), "another window failed; no new generation started") + request_base = dict(base, input_count_probe=count, + count_tokens_wall_seconds=time.monotonic() - probe_began) + began = time.monotonic() + try: + response = post(key, args.model, "generateContent", { + "contents": contents, "generationConfig": GENERATION}) + except Exception as exc: + ledger.record(dict(request_base, status="failed", failed_action="generateContent", + error_type=type(exc).__name__, + http_status=getattr(exc, "code", None), input_tokens=None, + output_tokens=None, completeness="unknown", + wall_seconds=time.monotonic() - began)) + raise RuntimeError("ASR request failed; receipt saved") from None + elapsed = time.monotonic() - began + artifact = response_artifact(response, binding, elapsed) + # Persist the provider response before validating it. A safe resume can + # recover a valid response without issuing a duplicate paid request. + write_json(response_path, artifact) + usage = artifact["usage"] + ledger.record(dict(request_base, status="response", + finish_reason=artifact["finish_reason"], + input_tokens=usage.get("promptTokenCount"), + output_tokens=usage.get("candidatesTokenCount"), + thinking_tokens=usage.get("thoughtsTokenCount"), + cached_tokens=usage.get("cachedContentTokenCount"), + raw_usage=usage, wall_seconds=elapsed, + cache_policy="no explicit cache requested; absent cache meter is unknown")) + text, silent = parse_response_artifact(artifact, binding) + row = dict(binding, text=text, silence=silent) + write_json(output, row) + print(json.dumps({"chunk": index, "status": "valid_silence" if silent else "transcribed"}), flush=True) + return row + + def guarded(index): + try: + return one(index) + except Exception: + stopped.set() + raise + + indices = [0] if args.first_only else range(len(intervals)) + with concurrent.futures.ThreadPoolExecutor(max_workers=2) as pool: + rows = list(pool.map(guarded, indices)) + if not args.first_only: + rows.sort(key=lambda row: row["chunk"]) + segments = [{key: row[key] for key in ("start_sec", "end_sec", "text")} + for row in rows if not row["silence"]] + write_json(args.output_dir / "transcript.json", { + "model": args.model, "segments": segments, + "provenance": {"source_sha256": source_hash, "source_duration_sec": duration, + "stage": "external Gemini ASR sidecar; separate AV import", + "window_seconds": 60, "chunk_count": len(rows), + "silent_windows": sum(row["silence"] for row in rows), + "entire_audio_submitted": True, + "timestamp_quality": "coarse fixed audio windows, not speech alignment", + "human_transcript_review": False, + "omissions": "model transcription may contain errors or omissions; coverage means input submitted, not verified word completeness", + "rates": rates}}) + print(json.dumps({"status": "completed", "windows": len(rows), "artifacts": len(segments), + "source_duration_sec": duration, "reserved_usd": str(ledger.reserved), + "reservation_remaining_usd": str(ledger.remaining)})) + + +if __name__ == "__main__": + main() diff --git a/cookbook/receipts/README.md b/cookbook/receipts/README.md new file mode 100644 index 0000000..05679d6 --- /dev/null +++ b/cookbook/receipts/README.md @@ -0,0 +1,17 @@ +# Sanitized reproduction receipts + +These JSON files contain usage and execution metadata only. They do not include +media, transcript or caption corpora, credentials, upload URIs, or private routes. +The source media is not redistributed. + +| Receipt | Result | +|---|---| +| [asr.json](asr.json) | Completed 75-window external ASR plus one retained failed alignment attempt | +| [gemini38-baseline.json](gemini38-baseline.json) | Completed native-video Gemini 3.8 baseline query | +| [caption-aborted.json](caption-aborted.json) | Four metered caption responses; ingestion aborted and no captions persisted | +| [caption-smoke.json](caption-smoke.json) | Successful one-frame caption smoke request | +| [cap-probe-32-incompatible.json](cap-probe-32-incompatible.json) | 32-token cap probe blocked by HTTP 502/no available route; usage unknown | + +The receipts are evidence for those individual attempts only. No completed AV +Grok+Jev comparison exists yet. They do not establish speed, cost, or quality +parity between the direct-video baseline and an AV pipeline. diff --git a/cookbook/receipts/asr.json b/cookbook/receipts/asr.json new file mode 100644 index 0000000..a88b1b3 --- /dev/null +++ b/cookbook/receipts/asr.json @@ -0,0 +1,77 @@ +{ + "schema_version": 1, + "source": { + "sha256": "863559e8630cae8019139edc03e0a9ea0437aac2c2f48a811f2d31f9aa14a575", + "duration_seconds": 4500.044, + "provenance_url": "https://www.youtube.com/watch?v=Bpw9HyfUpCw", + "original_clip_start_sec": 9300, + "original_clip_end_sec": 13800, + "public_availability_currently_verified": false, + "license": "unknown; media not redistributed" + }, + "model_requested": "gemini-3.5-flash-lite", + "model_returned": null, + "rates": { + "input_per_million": 0.3, + "output_per_million": 2.5, + "source": "https://ai.google.dev/gemini-api/docs/pricing", + "as_of": "2026-09-18", + "tier": "standard paid list rates; account billing/free tier unverified" + }, + "successful_indexing_asr": { + "generation_requests": 75, + "input_tokens": 118228, + "output_tokens": 13241, + "cached_tokens": null, + "thinking_tokens": null, + "token_metering": "measured API response usage", + "summed_request_wall_sec": 258.87398497699905, + "end_to_end_wall_sec": null, + "usd": 0.06857089999999999, + "basis": "estimated", + "note": "Measured tokens multiplied by official list rates, not billed dollars; cache and thinking absent meter remain unknown." + }, + "failed_alignment_experiment": { + "generation_requests": 1, + "input_tokens": 7634, + "output_tokens": 1866, + "cached_tokens": null, + "thinking_tokens": null, + "token_metering": "measured API response usage", + "summed_request_wall_sec": 8.694261409000319, + "end_to_end_wall_sec": null, + "usd": 0.0069552, + "basis": "estimated", + "note": "Measured tokens multiplied by official list rates, not billed dollars; cache and thinking absent meter remain unknown." + }, + "sampling": { + "audio": "complete source audio, mono16k FLAC", + "windows": 75, + "window_seconds": 60, + "last_window_seconds": 60.044, + "transcript_artifacts": 75, + "silence_windows": 0, + "timestamps": "coarse fixed source windows; not speech alignment", + "word_completeness": "unverified", + "all_source_audio_submitted": true + }, + "request_policy": { + "automatic_retries": 0, + "concurrency": 2, + "timeout_seconds": 90, + "max_input_count_tokens": 3500, + "max_output_tokens": 2048, + "explicit_cache_requested": false, + "cached_token_hit_rate": null + }, + "validation": { + "all75accepted": true, + "singleton_array_chunk12_normalized_without_new_call": true, + "prior300sec_attempt_rejected": "invalid generated timestamps; charged attempt retained" + }, + "infra_cost": { + "usd": null, + "basis": "unknown", + "note": "Media extraction/hosting/storage/network not billed in receipt" + } +} diff --git a/cookbook/receipts/cap-probe-32-incompatible.json b/cookbook/receipts/cap-probe-32-incompatible.json new file mode 100644 index 0000000..bdae6e7 --- /dev/null +++ b/cookbook/receipts/cap-probe-32-incompatible.json @@ -0,0 +1,32 @@ +{ + "schema_version": 1, + "stage": "max_completion_tokens_frame_caption_probe", + "status": "incompatible", + "recorded_at": "2026-09-18T20:04:07Z", + "runtime_commit": "bb22c76fe8f2da0b92c5ca9f637af2059fd69551", + "model_requested": "grok-4.20-0309-non-reasoning", + "token_limit_parameter": "max_completion_tokens", + "requested_output_cap": 32, + "automatic_retries": 0, + "hidden_fallbacks": false, + "av_request_attempts": 1, + "successful_provider_responses": 0, + "http_status": 502, + "failure_class": "configured_proxy_model_route_unavailable", + "failure_message": "The configured proxy reported that no provider route was available for the exact requested model.", + "usage": null, + "usage_complete": false, + "cap_verified": false, + "paid_upstream_dispatch": "unverified", + "list_estimate_usd": null, + "proxy_billed_usd": null, + "stop_decision": "No retry, full ingestion, query, or additional paid request was started because the gate returned no usage.", + "source_sha256": "863559e8630cae8019139edc03e0a9ea0437aac2c2f48a811f2d31f9aa14a575", + "frame_timestamp_sec": 0.0, + "frame_sha256": "0907fcb6670ebb7de71954b822b7d06306337eec88d43ac8ce2ace6f3d74acd0", + "frame_dimensions": [1280, 720], + "jpeg_quality": 2, + "wall_seconds": 0.4747030300022743, + "private_proxy_details_included": false, + "harness_sha256": "8183c674109d3bc719635e09f8a7b61f71d9a53080a128a0f6feb7fa4ac8d51f" +} diff --git a/cookbook/receipts/caption-aborted.json b/cookbook/receipts/caption-aborted.json new file mode 100644 index 0000000..1b12408 --- /dev/null +++ b/cookbook/receipts/caption-aborted.json @@ -0,0 +1,106 @@ +{ + "stage": "aborted_caption_ingest", + "model_requested": "grok-4.20-0309-non-reasoning", + "model_returned": ["grok-4.20-0309-non-reasoning"], + "runtime_head": "75125c3104278dd137e4960927e0afb047c81f90", + "requests": 4, + "input_tokens": 5004, + "output_tokens": 656, + "cached_tokens": 768, + "wall_seconds": 14.9669291709979, + "usage_complete": true, + "cache_aware_usd_estimate": 0.0070886000000000005, + "miss_only_usd_estimate": 0.007895, + "basis": "estimated", + "proxy_billed_usd": null, + "stop_reason": "Observed completion_tokens 230 exceeds requested deprecated max_tokens=200; meter guard stopped before request5", + "settings": { + "max_frames": 300, + "frame_resolution": [1280, 720], + "requested_output_cap": 200, + "retries": 0, + "fps": 0.06666666666666667 + }, + "frames_completed": 4, + "frame_captions_persisted": 0, + "per_request_receipts": [ + { + "request": 1, + "status": "response", + "model_returned": "grok-4.20-0309-non-reasoning", + "wall_seconds": 3.9929721240005165, + "usage": { + "completion_tokens": 150, + "prompt_tokens": 1251, + "total_tokens": 1401, + "completion_tokens_details": { + "accepted_prediction_tokens": null, + "audio_tokens": null, + "reasoning_tokens": 0, + "rejected_prediction_tokens": null + }, + "prompt_tokens_details": {"audio_tokens": null, "cached_tokens": 192} + }, + "choices": [{"finish_reason": "stop", "content_nonempty": true, "content_chars": 754}] + }, + { + "request": 2, + "status": "response", + "model_returned": "grok-4.20-0309-non-reasoning", + "wall_seconds": 3.2215615309978602, + "usage": { + "completion_tokens": 142, + "prompt_tokens": 1251, + "total_tokens": 1393, + "completion_tokens_details": { + "accepted_prediction_tokens": null, + "audio_tokens": null, + "reasoning_tokens": 0, + "rejected_prediction_tokens": null + }, + "prompt_tokens_details": {"audio_tokens": null, "cached_tokens": 192} + }, + "choices": [{"finish_reason": "stop", "content_nonempty": true, "content_chars": 682}] + }, + { + "request": 3, + "status": "response", + "model_returned": "grok-4.20-0309-non-reasoning", + "wall_seconds": 3.1171810109990474, + "usage": { + "completion_tokens": 134, + "prompt_tokens": 1251, + "total_tokens": 1385, + "completion_tokens_details": { + "accepted_prediction_tokens": null, + "audio_tokens": null, + "reasoning_tokens": 0, + "rejected_prediction_tokens": null + }, + "prompt_tokens_details": {"audio_tokens": null, "cached_tokens": 192} + }, + "choices": [{"finish_reason": "stop", "content_nonempty": true, "content_chars": 649}] + }, + { + "request": 4, + "status": "response", + "model_returned": "grok-4.20-0309-non-reasoning", + "wall_seconds": 4.6352145050004765, + "usage": { + "completion_tokens": 230, + "prompt_tokens": 1251, + "total_tokens": 1481, + "completion_tokens_details": { + "accepted_prediction_tokens": null, + "audio_tokens": null, + "reasoning_tokens": 0, + "rejected_prediction_tokens": null + }, + "prompt_tokens_details": {"audio_tokens": null, "cached_tokens": 192} + }, + "choices": [{"finish_reason": "stop", "content_nonempty": true, "content_chars": 1129}] + } + ], + "rates_source": "https://docs.x.ai/developers/models/grok-4.20-0309-non-reasoning.md", + "rates_as_of": "2026-09-18" +} diff --git a/cookbook/receipts/caption-smoke.json b/cookbook/receipts/caption-smoke.json new file mode 100644 index 0000000..a5b520e --- /dev/null +++ b/cookbook/receipts/caption-smoke.json @@ -0,0 +1,39 @@ +{ + "stage": "caption_smoke", + "model_requested": "grok-4.20-0309-non-reasoning", + "request_cap": 1, + "max_output_tokens": 200, + "timeout_seconds": 90, + "retries": 0, + "reserved_upstream_list_usd": 2.501, + "proxy_billed_usd": null, + "rates": { + "input_per_million_under_200k": 1.25, + "output_per_million_under_200k": 2.5, + "input_per_million_200k_plus": 2.5, + "output_per_million_200k_plus": 5.0, + "source": "https://docs.x.ai/developers/models/grok-4.20-0309-non-reasoning.md", + "as_of": "2026-09-18" + }, + "frame_timestamp_requested_sec": 7.5, + "frame_sha256": "f611aefc8c2d4c63c1cb5ae61b6e1878777d2be6ae8fc54ce9a27844c342c42e", + "frame_width": 640, + "status": "response", + "model_returned": "grok-4.20-0309-non-reasoning", + "wall_seconds": 2.135960043000523, + "usage": { + "completion_tokens": 93, + "total_tokens": 577, + "prompt_tokens": 484, + "prompt_tokens_details": { + "cached_tokens": 192 + }, + "completion_tokens_details": { + "reasoning_tokens": 0 + } + }, + "cache_aware_usd_estimate": 0.0006359, + "miss_only_usd_estimate": 0.0008375, + "basis": "estimated", + "note": "Official upstream list rates; proxy billing and mapping beyond returned model ID unverified." +} diff --git a/cookbook/receipts/gemini38-baseline.json b/cookbook/receipts/gemini38-baseline.json new file mode 100644 index 0000000..bd3931d --- /dev/null +++ b/cookbook/receipts/gemini38-baseline.json @@ -0,0 +1,58 @@ +{ + "schema_version": 1, + "stage": "native_video_baseline_query", + "status": "completed", + "recorded_at": "2026-09-18", + "source": { + "sha256": "863559e8630cae8019139edc03e0a9ea0437aac2c2f48a811f2d31f9aa14a575", + "duration_seconds": 4500.044, + "native_video_and_audio": true, + "media_redistributed": false + }, + "model_requested": "gemini-3.8-flash", + "model_returned": "gemini-3.8-flash", + "question": "What percentage did manufacturing expand by in the second quarter and what drove it?", + "answer": "Manufacturing expanded by 12.2% in the second quarter, driven largely by AI-related demand for electronics and precision engineering (19:51–20:00).", + "sampling": { + "fps": 1.0, + "media_resolution": "low", + "media_processing": "static", + "input": "native video and audio", + "comparison_note": "This is not sampling-equivalent to the incomplete AV arm." + }, + "request_policy": { + "generation_attempts": 1, + "automatic_retries": 0, + "max_output_tokens_including_thoughts": 8192, + "thinking_level": "high", + "explicit_cache": false + }, + "usage": { + "input_tokens": 409548, + "output_tokens": 39, + "thinking_tokens": 205, + "cached_tokens": null, + "token_metering": "provider usage metadata" + }, + "timing": { + "upload_seconds": 12.834643709000375, + "file_processing_seconds": 169.24631727199812, + "precount_seconds": 0.3055749349987309, + "query_seconds": 50.40440310399936 + }, + "cost": { + "full_rate_list_estimate_usd": 0.308076, + "worst_case_reserved_list_estimate_usd": 0.37845825, + "basis": "Measured tokens multiplied by published paid-tier list rates; not a billing invoice.", + "provider_or_proxy_billed_usd": null, + "compute_storage_network_usd": null + }, + "limitations": { + "answer_quality_reviewed": false, + "paired_av_grok_jev_run_completed": false, + "speed_cost_or_quality_parity_claimed": false, + "missing_usage_policy": "Absent fields remain unknown; no absent field is treated as measured zero." + }, + "private_upload_uri_included": false, + "credentials_included": false +} From 7e1a0ce1271e4d7b88d2b6cb5b0bbb9103a01ee6 Mon Sep 17 00:00:00 2001 From: Sean Phan Date: Fri, 18 Sep 2026 20:50:06 +0000 Subject: [PATCH 06/12] docs: address Jev cookbook review findings --- README.md | 16 +++++-- cookbook/README.md | 6 +-- cookbook/cost-model/README.md | 29 +++++++----- cookbook/cost-model/build_nb.py | 9 ++-- cookbook/cost-model/model.py | 44 +++++++++++++---- cookbook/cost-model/notebook.ipynb | 55 +++++++++++++++++----- cookbook/cost-model/scenario.receipts.json | 23 +++++++-- cookbook/cost-model/test_model.py | 55 +++++++++++++++++++++- cookbook/jev-refined-ask/README.md | 16 +++++-- cookbook/receipts/README.md | 9 ++-- 10 files changed, 206 insertions(+), 56 deletions(-) diff --git a/README.md b/README.md index 90e979a..be1c207 100644 --- a/README.md +++ b/README.md @@ -55,9 +55,8 @@ usage remains `null`. Missing, malformed, or out-of-range System One probabilities are treated as a refinement failure: `av` reports the fallback and does not invent a confidence. -See the [AV ask refinement cookbook](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook) -for the unmerged reproducible recipe, offline cost arithmetic, and sanitized -receipt provenance. +See the [AV ask refinement cookbook](cookbook/README.md) for the reproducible +recipe, offline cost arithmetic, and sanitized receipt provenance. FTS5 remains the first retrieval stage. An unscoped query with zero FTS matches does not scan the video archive or invoke sampled-frame inspection. @@ -302,10 +301,13 @@ export AV_API_KEY="sk-..." export AV_API_BASE_URL="https://api.openai.com/v1" # or any OpenAI-compatible endpoint export AV_API_TIMEOUT_SEC="120" export AV_API_MAX_RETRIES="1" +export AV_API_TOKEN_LIMIT_PARAMETER="max_tokens" # or max_completion_tokens when required export AV_ALLOW_OAUTH_FALLBACK="false" # never read local auth caches unless explicitly enabled export AV_ALLOW_CODEX_FALLBACK="false" # never spawn Codex unless explicitly enabled export AV_TRANSCRIBE_MODEL="whisper" export AV_VISION_MODEL="gpt-4-1" +export AV_VISION_MAX_OUTPUT_TOKENS="200" # single-frame caption response +export AV_VISION_CHUNK_MAX_OUTPUT_TOKENS="500" # multi-frame chunk caption response export AV_EMBED_MODEL="text-embedding-3-small" export AV_CHAT_MODEL="gpt-4-1" export AV_CHAT_MAX_OUTPUT_TOKENS="1024" # positive cap for each answer response @@ -336,6 +338,14 @@ export AV_API_BASE_URL="http://your-sglang-host:30000/v1" export DEEPSEEK_API_KEY="..." # only if your server requires one ``` +`AV_API_TOKEN_LIMIT_PARAMETER` selects the request field sent by AV's primary +OpenAI-compatible chat provider. The two vision limits apply to its single-frame +and multi-frame caption calls; `AV_CHAT_MAX_OUTPUT_TOKENS` applies to its ordinary +answer and summarization calls. These settings do not configure Jev/System One, +stronger sampled-frame inspection, `av bench`, or `av sentinel`. They also do not +prove that an upstream provider accepts or enforces the requested cap; verify the +returned usage and finish reason for the exact endpoint and model. + API requests use the configured timeout and explicit retry limit. Ingestion JSON includes `stage_usage` for transcription, captioning, caption summarization, and embeddings, plus the effective frame/request settings. Request failures are counted; diff --git a/cookbook/README.md b/cookbook/README.md index ca01ca5..df2fe84 100644 --- a/cookbook/README.md +++ b/cookbook/README.md @@ -4,9 +4,9 @@ Runnable recipes for the open-source **av** CLI live here alongside the code. | Recipe | What it demonstrates | Evidence status | |---|---|---| -| [Cost model](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook/cost-model) | Separate tokens, estimates, unknown costs, ingestion, queries, failures, reservations, and cap accounting | Completed component receipts; paired AV comparison incomplete | -| [Jev-refined ask](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook/jev-refined-ask) | Build a source-bound transcript sidecar, retrieve indexed moments, refine evidence, answer, and check support | Runnable recipe; no speed, cost, or quality parity claim | -| [Sanitized receipts](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook/receipts) | Completed ASR/baseline, caption smoke/abort, and blocked cap probe | No media, transcript/caption corpus, credentials, or private routes | +| [Cost model](cost-model/README.md) | Separate tokens, estimates, unknown costs, ingestion, queries, failures, reservations, and cap accounting | Completed component receipts; paired AV comparison incomplete | +| [Jev-refined ask](jev-refined-ask/README.md) | Build a source-bound transcript sidecar, retrieve indexed moments, refine evidence, answer, and check support | Runnable recipe; no speed, cost, or quality parity claim | +| [Sanitized receipts](receipts/README.md) | Completed ASR/baseline, caption smoke/abort, and blocked cap probe | No media, transcript/caption corpus, credentials, upload URIs, or private routes; baseline question and returned answer/rationale retained | The [original public cost notebook](https://github.com/PixelML/cookbook/tree/main/agentic-video/cost-model) remains available at its existing URL. Its Composer demonstrations are historical diff --git a/cookbook/cost-model/README.md b/cookbook/cost-model/README.md index e77e3b6..7304230 100644 --- a/cookbook/cost-model/README.md +++ b/cookbook/cost-model/README.md @@ -7,14 +7,14 @@ Run the deterministic arithmetic offline from the repository root: python3 cookbook/cost-model/build_nb.py --check Python 3.11+ is sufficient. These commands need no API keys, network, AV -installation, notebook server, or downloaded media. The [editable notebook](https://github.com/PixelML/av/blob/codex/jev-query-cascade/cookbook/cost-model/notebook.ipynb) -uses the same [model.py](https://github.com/PixelML/av/blob/codex/jev-query-cascade/cookbook/cost-model/model.py). +installation, notebook server, or downloaded media. The [editable notebook](notebook.ipynb) +uses the same [model.py](model.py). Its checked-in source and outputs are generated by `build_nb.py`. ## Evidence currently included -[`scenario.receipts.json`](https://github.com/PixelML/av/blob/codex/jev-query-cascade/cookbook/cost-model/scenario.receipts.json) -integrates the sanitized receipts in [`cookbook/receipts`](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook/receipts). +[`scenario.receipts.json`](scenario.receipts.json) integrates the sanitized +receipts in [`cookbook/receipts`](../receipts/README.md). The evidence is partial: | Attempt | Outcome | Recorded list-rate estimate | @@ -30,18 +30,21 @@ Known token-derived list-rate estimates total **$0.3913266**. They are not bille dollars. Provider/account or proxy billing remains unknown. Compute, storage, and network allocation also remains unknown. -The cumulative experiment cap is **$5**. Conservative reservations remain visible -and separate from spend: **$0.90** caption ingestion, **$0.20** query/judge/answer, -and **$0.01** cap probe. Known estimates plus reservations are **$1.5013266**, -leaving **$3.4986734** of estimate headroom. +The cumulative experiment cap is **$5**. The reservation ledger records +**$3.98945825** of request ceilings: **$2.501** for the caption smoke and +**$0.37845825** for the native-video baseline were released after metering, while +**$0.90** caption ingestion, **$0.20** query/judge/answer, and **$0.01** cap probe +remain retained. Only the retained **$1.11** counts against current headroom. +Known estimates plus retained reservations are **$1.5013266**, leaving +**$3.4986734** of estimate headroom. **No completed AV Grok+Jev comparison exists yet.** The receipts do not establish speed, cost, or quality parity with the Gemini 3.8 baseline. The baseline used native video/audio at its recorded sampling configuration; the incomplete AV arm is not an equal-input comparison. -Use [`scenario.pending.json`](https://github.com/PixelML/av/blob/codex/jev-query-cascade/cookbook/cost-model/scenario.pending.json) -as a blank template for another run. +Use [`scenario.pending.json`](scenario.pending.json) as a blank template for +another run. ## What the model keeps separate @@ -54,7 +57,7 @@ as a blank template for another run. | One-time ingestion | Charged once, never multiplied by query count | | Per-query stages | Retrieval, judge, answer, and optional fallback scale with query count | | Failures | Failed and aborted attempts remain in the experiment ledger | -| Reservations | Count against the cumulative cap but are not called spend | +| Reservations | Every ceiling is marked `retained`, `released_after_metering`, or `superseded`; only retained amounts count against current cap headroom | Every cost item has `usd`, `basis`, and an evidence `note`. Bases are `measured`, `estimated`, `assumed`, `unknown`, or `not_used`. Dollar values use decimal @@ -63,7 +66,9 @@ strings in generated output. Unknown values stay null; they never become zero. The experiment ledger is independent of projected repeatable costs. It records successful requests, unsuccessful attempts, and smoke/probe calls exactly once. A failed response can incur cost even when its output is rejected. Reservations -are a separate guardrail and are never added to the spend subtotal. +are a separate guardrail and are never added to the spend subtotal. Released and +superseded reservations stay visible as history without being double-counted +after a measured estimate or replacement reservation is recorded. ## Fill a scenario from another AV run diff --git a/cookbook/cost-model/build_nb.py b/cookbook/cost-model/build_nb.py index afaa5d2..d1b41d2 100644 --- a/cookbook/cost-model/build_nb.py +++ b/cookbook/cost-model/build_nb.py @@ -18,9 +18,10 @@ This notebook calculates with the same model.py used by the command line. The checked-in receipt scenario includes completed ASR and direct-video baseline attempts, but no completed AV Grok+Jev comparison. Measured tokens, list-rate -estimates, unknown costs, failures, and reservations remain separate. See -README.md for provenance and limitations. These cells never call a provider or -fetch media. +estimates, unknown costs, failures, and reservation states remain separate. Only +retained reservations count against current cap headroom; released and superseded +ceilings remain visible as history. See README.md for provenance and limitations. +These cells never call a provider or fetch media. """), ("code", """import json import sys @@ -42,7 +43,7 @@ "failures": accounting["failures"], "reservations": accounting["reservations"], "cumulative_cap_usd": accounting["cumulative_cap_usd"], - "known_plus_reserved_usd": accounting["known_plus_reserved_usd"], + "known_estimates_plus_retained_usd": accounting["known_estimates_plus_retained_usd"], "remaining_cap_usd": accounting["remaining_cap_usd"], }), indent=2)) """), diff --git a/cookbook/cost-model/model.py b/cookbook/cost-model/model.py index 8dd092e..f27be56 100644 --- a/cookbook/cost-model/model.py +++ b/cookbook/cost-model/model.py @@ -8,6 +8,7 @@ from pathlib import Path BASES = {"measured", "estimated", "assumed", "unknown", "not_used"} +RESERVATION_STATES = {"retained", "released_after_metering", "superseded"} ZERO = Decimal("0") ONE = Decimal("1") @@ -134,24 +135,44 @@ def measured_usage(data: dict) -> list[dict]: def reservation_account(data: dict) -> dict: - """Track conservative cap reservations without treating them as spend.""" + """Track reservation history while charging only retained ceilings to the cap.""" entries = data.get("reservations", []) if not isinstance(entries, list): raise ValueError("reservations must be a list") terms = [] + status_totals = {status: ZERO for status in sorted(RESERVATION_STATES)} for index, entry in enumerate(entries): if not isinstance(entry.get("name"), str) or not entry["name"].strip(): raise ValueError(f"reservations.{index}: name is required") if not isinstance(entry.get("note"), str) or not entry["note"].strip(): raise ValueError(f"reservations.{index}: note is required") - terms.append({ + status = entry.get("status") + if status not in RESERVATION_STATES: + raise ValueError( + f"reservations.{index}: status must be retained, " + "released_after_metering, or superseded" + ) + value = number(entry.get("usd"), f"reservations.{index}.usd") + row = { "name": f"reservation.{index}.{entry['name']}", - "usd": number(entry.get("usd"), f"reservations.{index}.usd"), - "status": entry.get("status", "retained"), + "usd": value, + "status": status, + "counts_against_cap": status == "retained", "note": entry["note"], - }) - total = sum((entry["usd"] for entry in terms), ZERO) - return {"terms": terms, "total_reserved_usd": total} + } + for field in ("receipt", "receipt_field", "superseded_by"): + if field in entry: + row[field] = entry[field] + terms.append(row) + status_totals[status] += value + total_recorded = sum(status_totals.values(), ZERO) + total_retained = status_totals["retained"] + return { + "terms": terms, + "status_totals_usd": status_totals, + "total_recorded_usd": total_recorded, + "total_retained_usd": total_retained, + } def experiment_report(data: dict) -> dict: @@ -163,7 +184,7 @@ def experiment_report(data: dict) -> dict: ]) reservations = reservation_account(data) cap = number(data.get("cumulative_cap_usd"), "cumulative_cap_usd") - committed = spend["known_subtotal_usd"] + reservations["total_reserved_usd"] + committed = spend["known_subtotal_usd"] + reservations["total_retained_usd"] if committed > cap: raise ValueError("known list estimates plus reservations exceed cumulative cap") failures = [ @@ -177,9 +198,12 @@ def experiment_report(data: dict) -> dict: "failures": failures, "reservations": reservations, "cumulative_cap_usd": cap, - "known_plus_reserved_usd": committed, + "known_estimates_plus_retained_usd": committed, "remaining_cap_usd": cap - committed, - "cap_warning": "Reservations are conservative guardrails, not billed or estimated spend.", + "cap_warning": ( + "Reservations are conservative guardrails, not billed or estimated spend; " + "only retained reservations count against current headroom." + ), } diff --git a/cookbook/cost-model/notebook.ipynb b/cookbook/cost-model/notebook.ipynb index 74769d5..7a7af0d 100644 --- a/cookbook/cost-model/notebook.ipynb +++ b/cookbook/cost-model/notebook.ipynb @@ -10,9 +10,10 @@ "This notebook calculates with the same model.py used by the command line.\n", "The checked-in receipt scenario includes completed ASR and direct-video baseline\n", "attempts, but no completed AV Grok+Jev comparison. Measured tokens, list-rate\n", - "estimates, unknown costs, failures, and reservations remain separate. See\n", - "README.md for provenance and limitations. These cells never call a provider or\n", - "fetch media.\n" + "estimates, unknown costs, failures, and reservation states remain separate. Only\n", + "retained reservations count against current cap headroom; released and superseded\n", + "ceilings remain visible as history. See README.md for provenance and limitations.\n", + "These cells never call a provider or fetch media.\n" ] }, { @@ -40,7 +41,7 @@ " \"failures\": accounting[\"failures\"],\n", " \"reservations\": accounting[\"reservations\"],\n", " \"cumulative_cap_usd\": accounting[\"cumulative_cap_usd\"],\n", - " \"known_plus_reserved_usd\": accounting[\"known_plus_reserved_usd\"],\n", + " \"known_estimates_plus_retained_usd\": accounting[\"known_estimates_plus_retained_usd\"],\n", " \"remaining_cap_usd\": accounting[\"remaining_cap_usd\"],\n", "}), indent=2))\n" ], @@ -181,28 +182,56 @@ " \"reservations\": {\n", " \"terms\": [\n", " {\n", - " \"name\": \"reservation.0.caption_ingest\",\n", + " \"name\": \"reservation.0.caption_smoke_request_ceiling\",\n", + " \"usd\": \"2.501\",\n", + " \"status\": \"released_after_metering\",\n", + " \"counts_against_cap\": false,\n", + " \"note\": \"The receipt-level ceiling was released after the successful response reported usage; the measured cache-aware estimate remains in experiment_spend.\",\n", + " \"receipt\": \"../receipts/caption-smoke.json\",\n", + " \"receipt_field\": \"reserved_upstream_list_usd\"\n", + " },\n", + " {\n", + " \"name\": \"reservation.1.gemini38_native_video_baseline_ceiling\",\n", + " \"usd\": \"0.37845825\",\n", + " \"status\": \"released_after_metering\",\n", + " \"counts_against_cap\": false,\n", + " \"note\": \"The worst-case request ceiling was released after provider usage was metered; the full-rate estimate remains in experiment_spend.\",\n", + " \"receipt\": \"../receipts/gemini38-baseline.json\",\n", + " \"receipt_field\": \"cost.worst_case_reserved_list_estimate_usd\"\n", + " },\n", + " {\n", + " \"name\": \"reservation.2.caption_ingest\",\n", " \"usd\": \"0.90\",\n", - " \"status\": \"retained_in_cumulative_plan\",\n", + " \"status\": \"retained\",\n", + " \"counts_against_cap\": true,\n", " \"note\": \"Conservative reservation, not spend.\"\n", " },\n", " {\n", - " \"name\": \"reservation.1.query_judge_and_answer\",\n", + " \"name\": \"reservation.3.query_judge_and_answer\",\n", " \"usd\": \"0.20\",\n", - " \"status\": \"retained_in_cumulative_plan\",\n", + " \"status\": \"retained\",\n", + " \"counts_against_cap\": true,\n", " \"note\": \"Conservative reservation, not spend.\"\n", " },\n", " {\n", - " \"name\": \"reservation.2.max_completion_tokens_cap_probe\",\n", + " \"name\": \"reservation.4.max_completion_tokens_cap_probe\",\n", " \"usd\": \"0.01\",\n", - " \"status\": \"retained_after_no_route\",\n", - " \"note\": \"Retained because usage and upstream dispatch were unknown.\"\n", + " \"status\": \"retained\",\n", + " \"counts_against_cap\": true,\n", + " \"note\": \"Retained because usage and upstream dispatch were unknown.\",\n", + " \"receipt\": \"../receipts/cap-probe-32-incompatible.json\"\n", " }\n", " ],\n", - " \"total_reserved_usd\": \"1.11\"\n", + " \"status_totals_usd\": {\n", + " \"released_after_metering\": \"2.87945825\",\n", + " \"retained\": \"1.11\",\n", + " \"superseded\": \"0\"\n", + " },\n", + " \"total_recorded_usd\": \"3.98945825\",\n", + " \"total_retained_usd\": \"1.11\"\n", " },\n", " \"cumulative_cap_usd\": \"5\",\n", - " \"known_plus_reserved_usd\": \"1.5013266\",\n", + " \"known_estimates_plus_retained_usd\": \"1.5013266\",\n", " \"remaining_cap_usd\": \"3.4986734\"\n", "}\n" ] diff --git a/cookbook/cost-model/scenario.receipts.json b/cookbook/cost-model/scenario.receipts.json index 55935f5..66df191 100644 --- a/cookbook/cost-model/scenario.receipts.json +++ b/cookbook/cost-model/scenario.receipts.json @@ -228,22 +228,39 @@ } ], "reservations": [ + { + "name": "caption_smoke_request_ceiling", + "usd": "2.501", + "status": "released_after_metering", + "receipt": "../receipts/caption-smoke.json", + "receipt_field": "reserved_upstream_list_usd", + "note": "The receipt-level ceiling was released after the successful response reported usage; the measured cache-aware estimate remains in experiment_spend." + }, + { + "name": "gemini38_native_video_baseline_ceiling", + "usd": "0.37845825", + "status": "released_after_metering", + "receipt": "../receipts/gemini38-baseline.json", + "receipt_field": "cost.worst_case_reserved_list_estimate_usd", + "note": "The worst-case request ceiling was released after provider usage was metered; the full-rate estimate remains in experiment_spend." + }, { "name": "caption_ingest", "usd": "0.90", - "status": "retained_in_cumulative_plan", + "status": "retained", "note": "Conservative reservation, not spend." }, { "name": "query_judge_and_answer", "usd": "0.20", - "status": "retained_in_cumulative_plan", + "status": "retained", "note": "Conservative reservation, not spend." }, { "name": "max_completion_tokens_cap_probe", "usd": "0.01", - "status": "retained_after_no_route", + "status": "retained", + "receipt": "../receipts/cap-probe-32-incompatible.json", "note": "Retained because usage and upstream dispatch were unknown." } ], diff --git a/cookbook/cost-model/test_model.py b/cookbook/cost-model/test_model.py index f243c76..da8e68c 100644 --- a/cookbook/cost-model/test_model.py +++ b/cookbook/cost-model/test_model.py @@ -101,8 +101,13 @@ def test_receipt_ledger_preserves_cap_and_separates_unknowns(self): report = experiment_report(data) self.assertEqual(report["list_rate_estimates"]["known_subtotal_usd"], Decimal("0.3913266")) - self.assertEqual(report["reservations"]["total_reserved_usd"], Decimal("1.11")) - self.assertEqual(report["known_plus_reserved_usd"], Decimal("1.5013266")) + self.assertEqual(report["reservations"]["total_recorded_usd"], Decimal("3.98945825")) + self.assertEqual(report["reservations"]["total_retained_usd"], Decimal("1.11")) + self.assertEqual( + report["reservations"]["status_totals_usd"]["released_after_metering"], + Decimal("2.87945825"), + ) + self.assertEqual(report["known_estimates_plus_retained_usd"], Decimal("1.5013266")) self.assertEqual(report["remaining_cap_usd"], Decimal("3.4986734")) self.assertIsNone(report["unknown_costs"]["complete_total_usd"]) self.assertEqual( @@ -135,6 +140,52 @@ def test_sanitized_receipts_are_present_and_baseline_estimate_recomputes(self): ) self.assertFalse(receipt["limitations"]["paired_av_grok_jev_run_completed"]) + def test_every_receipt_reservation_is_reconciled_with_an_explicit_state(self): + data = json.loads((HERE / "scenario.receipts.json").read_text()) + reconciled = { + (entry.get("receipt"), entry.get("receipt_field")): entry + for entry in data["reservations"] + if entry.get("receipt_field") + } + expected = { + ("../receipts/caption-smoke.json", "reserved_upstream_list_usd"), + ("../receipts/gemini38-baseline.json", "cost.worst_case_reserved_list_estimate_usd"), + } + for key in expected: + with self.subTest(receipt=key[0], field=key[1]): + self.assertIn(key, reconciled) + receipt = json.loads((HERE / key[0]).resolve().read_text()) + receipt_value = receipt + for field in key[1].split("."): + receipt_value = receipt_value[field] + self.assertEqual(Decimal(reconciled[key]["usd"]), Decimal(str(receipt_value))) + self.assertIn( + reconciled[key]["status"], + {"retained", "released_after_metering", "superseded"}, + ) + + def test_only_retained_reservations_count_against_headroom(self): + data = fixture() + data["reservations"] = [ + {"name": "active", "usd": "1.00", "status": "retained", "note": "active"}, + { + "name": "metered", + "usd": "2.00", + "status": "released_after_metering", + "note": "metered", + }, + {"name": "old", "usd": "3.00", "status": "superseded", "note": "replaced"}, + ] + data["cumulative_cap_usd"] = "10" + report = experiment_report(data) + self.assertEqual(report["reservations"]["total_recorded_usd"], Decimal("6.00")) + self.assertEqual(report["reservations"]["total_retained_usd"], Decimal("1.00")) + self.assertEqual(report["known_estimates_plus_retained_usd"], Decimal("1.00")) + + data["reservations"][0]["status"] = "retained_in_cumulative_plan" + with self.assertRaisesRegex(ValueError, "status must be"): + experiment_report(data) + def test_incomplete_av_side_suppresses_baseline_ratio(self): data = json.loads((HERE / "scenario.receipts.json").read_text()) result = scenario(data, 1) diff --git a/cookbook/jev-refined-ask/README.md b/cookbook/jev-refined-ask/README.md index 893963c..33b682f 100644 --- a/cookbook/jev-refined-ask/README.md +++ b/cookbook/jev-refined-ask/README.md @@ -3,8 +3,8 @@ Use AV's indexed moments as candidate evidence, then refine relevance and scene context before answering. This recipe describes the open-source CLI behavior; it does not claim that an AV run reproduces historical Composer cost or quality. -See the [cost model](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook/cost-model) -and [sanitized receipts](https://github.com/PixelML/av/tree/codex/jev-query-cascade/cookbook/receipts) +See the [cost model](../cost-model/README.md) and +[sanitized receipts](../receipts/README.md) for separate stage accounting and evidence provenance. ## Run it @@ -45,7 +45,7 @@ silence. Import does not meter the external ASR run or verify transcript accurac Keep that run's token usage, approximate-timestamp caveat, and price separate from AV's ingestion receipt. Use only public metadata in a published sidecar. -The included [`transcribe_gemini.py`](https://github.com/PixelML/av/blob/codex/jev-query-cascade/cookbook/jev-refined-ask/transcribe_gemini.py) +The included [`transcribe_gemini.py`](transcribe_gemini.py) is a standard-library helper for Gemini audio transcription. It does not bundle media or transcripts. It accepts any readable local source supported by ffmpeg: @@ -119,6 +119,9 @@ support judgment is labeled unknown, not supported. | Environment variable | Default | Purpose | |---|---|---| +| AV_API_TOKEN_LIMIT_PARAMETER | max_tokens | Request field used by the primary OpenAI-compatible caption/answer provider; choose max_completion_tokens only when the endpoint requires it | +| AV_VISION_MAX_OUTPUT_TOKENS | 200 | Requested output cap for each primary-provider single-frame caption call | +| AV_VISION_CHUNK_MAX_OUTPUT_TOKENS | 500 | Requested output cap for each primary-provider multi-frame caption call | | AV_CHAT_MAX_OUTPUT_TOKENS | 1024 | Output-token cap passed to the configured answer provider | | AV_TYPESAFE_MODEL | jev-latest | Jev model used for judgments | | AV_TYPESAFE_TIMEOUT_SEC | 30 | Timeout for a judgment request | @@ -134,6 +137,13 @@ The thresholds are routing settings, not calibrated probabilities of correctness FTS retrieval remains the first stage: an unscoped query with no matching hits does not inspect the whole archive with the stronger model. +The token-limit parameter and the vision/chat values above cover only AV's +primary OpenAI-compatible frame-caption, chunk-caption, answer, and summarization +calls. They do not configure Jev/System One, stronger sampled-frame inspection, +`av bench`, or `av sentinel`, and they are not a provider-side enforcement or +billing guarantee. Confirm support from the exact endpoint/model and inspect +returned usage and finish reasons. + ## Optional stronger inspection Configure **AV_STRONG_VISION_API_BASE_URL**, **AV_STRONG_VISION_MODEL**, and diff --git a/cookbook/receipts/README.md b/cookbook/receipts/README.md index 05679d6..a16feaf 100644 --- a/cookbook/receipts/README.md +++ b/cookbook/receipts/README.md @@ -1,8 +1,11 @@ # Sanitized reproduction receipts -These JSON files contain usage and execution metadata only. They do not include -media, transcript or caption corpora, credentials, upload URIs, or private routes. -The source media is not redistributed. +These JSON files contain selected sanitized result evidence plus usage and +execution metadata. They do not include media, transcript or caption corpora, +credentials, upload URIs, or private routes. The source media is not redistributed. +The native-video baseline receipt intentionally retains its benchmark question +and returned answer, including the short explanatory rationale and timestamp, +because those fields are needed to interpret the recorded result. | Receipt | Result | |---|---| From 6a1cde2fb0009759e45983ff1f9b9d5afc50f9fa Mon Sep 17 00:00:00 2001 From: Sean Phan Date: Sat, 19 Sep 2026 02:04:32 +0000 Subject: [PATCH 07/12] docs: record direct cap-probe incompatibility --- cookbook/README.md | 2 +- cookbook/cost-model/README.md | 18 +++--- cookbook/cost-model/notebook.ipynb | 60 +++++++++---------- cookbook/cost-model/scenario.receipts.json | 39 ++++++------ cookbook/cost-model/test_model.py | 13 ++-- cookbook/receipts/README.md | 1 + .../cap-probe-32-direct-incompatible.json | 48 +++++++++++++++ 7 files changed, 116 insertions(+), 65 deletions(-) create mode 100644 cookbook/receipts/cap-probe-32-direct-incompatible.json diff --git a/cookbook/README.md b/cookbook/README.md index df2fe84..dfd379d 100644 --- a/cookbook/README.md +++ b/cookbook/README.md @@ -6,7 +6,7 @@ Runnable recipes for the open-source **av** CLI live here alongside the code. |---|---|---| | [Cost model](cost-model/README.md) | Separate tokens, estimates, unknown costs, ingestion, queries, failures, reservations, and cap accounting | Completed component receipts; paired AV comparison incomplete | | [Jev-refined ask](jev-refined-ask/README.md) | Build a source-bound transcript sidecar, retrieve indexed moments, refine evidence, answer, and check support | Runnable recipe; no speed, cost, or quality parity claim | -| [Sanitized receipts](receipts/README.md) | Completed ASR/baseline, caption smoke/abort, and blocked cap probe | No media, transcript/caption corpus, credentials, upload URIs, or private routes; baseline question and returned answer/rationale retained | +| [Sanitized receipts](receipts/README.md) | Completed ASR/baseline, caption smoke/abort, and both incompatible cap probes | No media, transcript/caption corpus, credentials, upload URIs, or private routes; baseline question and returned answer/rationale retained | The [original public cost notebook](https://github.com/PixelML/cookbook/tree/main/agentic-video/cost-model) remains available at its existing URL. Its Composer demonstrations are historical diff --git a/cookbook/cost-model/README.md b/cookbook/cost-model/README.md index 7304230..495218d 100644 --- a/cookbook/cost-model/README.md +++ b/cookbook/cost-model/README.md @@ -24,19 +24,21 @@ The evidence is partial: | Grok caption smoke | completed | $0.0006359 | | Grok caption ingestion | aborted after 4 responses; 0 captions persisted | $0.0070886 | | Gemini 3.8 native-video baseline | completed | $0.308076 | -| 32-token caption cap probe | HTTP 502/no route; usage unknown | unknown | +| 32-token caption cap probe through the first configured route | HTTP 502/no route; usage unknown | +| Direct 32-token caption cap probe | exact model returned 309 completion tokens against requested cap 32 | $0.0020309 | -Known token-derived list-rate estimates total **$0.3913266**. They are not billed +Known token-derived list-rate estimates total **$0.3933575**. They are not billed dollars. Provider/account or proxy billing remains unknown. Compute, storage, and network allocation also remains unknown. The cumulative experiment cap is **$5**. The reservation ledger records -**$3.98945825** of request ceilings: **$2.501** for the caption smoke and -**$0.37845825** for the native-video baseline were released after metering, while -**$0.90** caption ingestion, **$0.20** query/judge/answer, and **$0.01** cap probe -remain retained. Only the retained **$1.11** counts against current headroom. -Known estimates plus retained reservations are **$1.5013266**, leaving -**$3.4986734** of estimate headroom. +**$3.97945825** of request ceilings: **$2.501** for the caption smoke and +**$0.37845825** for the native-video baseline were released after metering. +The **$0.01** first-route cap-probe ceiling was closed after a second direct +probe measured **$0.0020309**, leaving **$0.90** caption ingestion and +**$0.20** query/judge/answer retained. Only the retained **$1.10** counts against +current headroom. Known estimates plus retained reservations are **$1.4933575**, +leaving **$3.5066425** of estimate headroom. **No completed AV Grok+Jev comparison exists yet.** The receipts do not establish speed, cost, or quality parity with the Gemini 3.8 baseline. The baseline used diff --git a/cookbook/cost-model/notebook.ipynb b/cookbook/cost-model/notebook.ipynb index 7a7af0d..57c3bd4 100644 --- a/cookbook/cost-model/notebook.ipynb +++ b/cookbook/cost-model/notebook.ipynb @@ -55,7 +55,7 @@ " \"status\": \"partial_reproduction\",\n", " \"implementation\": \"av\",\n", " \"receipt_directory\": \"../receipts\",\n", - " \"note\": \"Completed ASR and direct-video baseline receipts exist, but no completed AV Grok+Jev comparison exists yet.\",\n", + " \"note\": \"Completed ASR and direct-video baseline receipts exist, and the direct 32-token compatibility probe completed with usage, but the exact model ignored the requested output cap. No completed AV Grok+Jev comparison exists yet.\",\n", " \"projection\": \"Repeated-query totals are projections from entered stage costs, not additional measured runs.\"\n", "}\n", "{\n", @@ -110,9 +110,20 @@ " \"requests\": 1,\n", " \"outcome\": \"completed\",\n", " \"receipt\": \"../receipts/gemini38-baseline.json\"\n", + " },\n", + " {\n", + " \"name\": \"direct_max_completion_tokens_cap_probe\",\n", + " \"input_tokens\": 1168,\n", + " \"output_tokens\": 309,\n", + " \"thinking_tokens\": 0,\n", + " \"cached_tokens\": 192,\n", + " \"requests\": 1,\n", + " \"outcome\": \"incompatible\",\n", + " \"receipt\": \"../receipts/cap-probe-32-direct-incompatible.json\",\n", + " \"note\": \"Exact model returned usage but ignored requested max_completion_tokens=32.\"\n", " }\n", " ],\n", - " \"known_list_estimates_usd\": \"0.3913266\",\n", + " \"known_list_estimates_usd\": \"0.3933575\",\n", " \"unknown_costs\": {\n", " \"terms\": [\n", " {\n", @@ -123,16 +134,7 @@ " \"category\": \"billing\"\n", " },\n", " {\n", - " \"name\": \"unknown.1.cap_probe_dispatch\",\n", - " \"usd\": null,\n", - " \"basis\": \"unknown\",\n", - " \"note\": \"The 32-token probe returned HTTP 502/no route with no usage; paid upstream dispatch is unverified.\",\n", - " \"category\": \"failed_attempt\",\n", - " \"outcome\": \"incompatible\",\n", - " \"receipt\": \"../receipts/cap-probe-32-incompatible.json\"\n", - " },\n", - " {\n", - " \"name\": \"unknown.2.compute_storage_network\",\n", + " \"name\": \"unknown.1.compute_storage_network\",\n", " \"usd\": null,\n", " \"basis\": \"unknown\",\n", " \"note\": \"Local compute, media/index storage, upload, and network costs were not allocated.\",\n", @@ -142,8 +144,7 @@ " \"known_subtotal_usd\": \"0\",\n", " \"unknown_terms\": [\n", " \"unknown.0.provider_or_proxy_billed_total\",\n", - " \"unknown.1.cap_probe_dispatch\",\n", - " \"unknown.2.compute_storage_network\"\n", + " \"unknown.1.compute_storage_network\"\n", " ],\n", " \"complete_total_usd\": null,\n", " \"complete_means\": \"all modeled costs supplied; estimated and assumed inputs retain their provenance\"\n", @@ -170,13 +171,14 @@ " \"usage_ref\": \"aborted_caption_attempt\"\n", " },\n", " {\n", - " \"name\": \"unknown.1.cap_probe_dispatch\",\n", - " \"usd\": null,\n", - " \"basis\": \"unknown\",\n", - " \"note\": \"The 32-token probe returned HTTP 502/no route with no usage; paid upstream dispatch is unverified.\",\n", - " \"category\": \"failed_attempt\",\n", + " \"name\": \"experiment.5.direct_max_completion_tokens_cap_probe\",\n", + " \"usd\": \"0.0020309\",\n", + " \"basis\": \"estimated\",\n", + " \"note\": \"Cache-aware upstream list-rate estimate; the model returned 309 completion tokens despite requested cap 32.\",\n", + " \"category\": \"compatibility_probe\",\n", " \"outcome\": \"incompatible\",\n", - " \"receipt\": \"../receipts/cap-probe-32-incompatible.json\"\n", + " \"receipt\": \"../receipts/cap-probe-32-direct-incompatible.json\",\n", + " \"usage_ref\": \"direct_max_completion_tokens_cap_probe\"\n", " }\n", " ],\n", " \"reservations\": {\n", @@ -212,27 +214,19 @@ " \"status\": \"retained\",\n", " \"counts_against_cap\": true,\n", " \"note\": \"Conservative reservation, not spend.\"\n", - " },\n", - " {\n", - " \"name\": \"reservation.4.max_completion_tokens_cap_probe\",\n", - " \"usd\": \"0.01\",\n", - " \"status\": \"retained\",\n", - " \"counts_against_cap\": true,\n", - " \"note\": \"Retained because usage and upstream dispatch were unknown.\",\n", - " \"receipt\": \"../receipts/cap-probe-32-incompatible.json\"\n", " }\n", " ],\n", " \"status_totals_usd\": {\n", " \"released_after_metering\": \"2.87945825\",\n", - " \"retained\": \"1.11\",\n", + " \"retained\": \"1.10\",\n", " \"superseded\": \"0\"\n", " },\n", - " \"total_recorded_usd\": \"3.98945825\",\n", - " \"total_retained_usd\": \"1.11\"\n", + " \"total_recorded_usd\": \"3.97945825\",\n", + " \"total_retained_usd\": \"1.10\"\n", " },\n", " \"cumulative_cap_usd\": \"5\",\n", - " \"known_estimates_plus_retained_usd\": \"1.5013266\",\n", - " \"remaining_cap_usd\": \"3.4986734\"\n", + " \"known_estimates_plus_retained_usd\": \"1.4933575\",\n", + " \"remaining_cap_usd\": \"3.5066425\"\n", "}\n" ] } diff --git a/cookbook/cost-model/scenario.receipts.json b/cookbook/cost-model/scenario.receipts.json index 66df191..ab05fc4 100644 --- a/cookbook/cost-model/scenario.receipts.json +++ b/cookbook/cost-model/scenario.receipts.json @@ -4,7 +4,7 @@ "status": "partial_reproduction", "implementation": "av", "receipt_directory": "../receipts", - "note": "Completed ASR and direct-video baseline receipts exist, but no completed AV Grok+Jev comparison exists yet.", + "note": "Completed ASR and direct-video baseline receipts exist, and the direct 32-token compatibility probe completed with usage, but the exact model ignored the requested output cap. No completed AV Grok+Jev comparison exists yet.", "projection": "Repeated-query totals are projections from entered stage costs, not additional measured runs." }, "indexing": [ @@ -148,6 +148,17 @@ "cached_tokens": null, "outcome": "completed", "receipt": "../receipts/gemini38-baseline.json" + }, + { + "name": "direct_max_completion_tokens_cap_probe", + "requests": 1, + "input_tokens": 1168, + "output_tokens": 309, + "thinking_tokens": 0, + "cached_tokens": 192, + "outcome": "incompatible", + "receipt": "../receipts/cap-probe-32-direct-incompatible.json", + "note": "Exact model returned usage but ignored requested max_completion_tokens=32." } ], "experiment_spend": [ @@ -200,6 +211,16 @@ "receipt": "../receipts/gemini38-baseline.json", "usage_ref": "gemini38_native_video_baseline", "note": "Full-rate upstream list estimate from measured input, output, and thinking tokens; not billed dollars." + }, + { + "name": "direct_max_completion_tokens_cap_probe", + "usd": "0.0020309", + "basis": "estimated", + "category": "compatibility_probe", + "outcome": "incompatible", + "receipt": "../receipts/cap-probe-32-direct-incompatible.json", + "usage_ref": "direct_max_completion_tokens_cap_probe", + "note": "Cache-aware upstream list-rate estimate; the model returned 309 completion tokens despite requested cap 32." } ], "unknown_costs": [ @@ -210,15 +231,6 @@ "category": "billing", "note": "Account and proxy billed dollars were not available; list-rate estimates are not invoices." }, - { - "name": "cap_probe_dispatch", - "usd": null, - "basis": "unknown", - "category": "failed_attempt", - "outcome": "incompatible", - "receipt": "../receipts/cap-probe-32-incompatible.json", - "note": "The 32-token probe returned HTTP 502/no route with no usage; paid upstream dispatch is unverified." - }, { "name": "compute_storage_network", "usd": null, @@ -255,13 +267,6 @@ "usd": "0.20", "status": "retained", "note": "Conservative reservation, not spend." - }, - { - "name": "max_completion_tokens_cap_probe", - "usd": "0.01", - "status": "retained", - "receipt": "../receipts/cap-probe-32-incompatible.json", - "note": "Retained because usage and upstream dispatch were unknown." } ], "cumulative_cap_usd": "5" diff --git a/cookbook/cost-model/test_model.py b/cookbook/cost-model/test_model.py index da8e68c..46af9c9 100644 --- a/cookbook/cost-model/test_model.py +++ b/cookbook/cost-model/test_model.py @@ -100,15 +100,15 @@ def test_receipt_ledger_preserves_cap_and_separates_unknowns(self): data = json.loads((HERE / "scenario.receipts.json").read_text()) report = experiment_report(data) self.assertEqual(report["list_rate_estimates"]["known_subtotal_usd"], - Decimal("0.3913266")) - self.assertEqual(report["reservations"]["total_recorded_usd"], Decimal("3.98945825")) - self.assertEqual(report["reservations"]["total_retained_usd"], Decimal("1.11")) + Decimal("0.3933575")) + self.assertEqual(report["reservations"]["total_recorded_usd"], Decimal("3.97945825")) + self.assertEqual(report["reservations"]["total_retained_usd"], Decimal("1.10")) self.assertEqual( report["reservations"]["status_totals_usd"]["released_after_metering"], Decimal("2.87945825"), ) - self.assertEqual(report["known_estimates_plus_retained_usd"], Decimal("1.5013266")) - self.assertEqual(report["remaining_cap_usd"], Decimal("3.4986734")) + self.assertEqual(report["known_estimates_plus_retained_usd"], Decimal("1.4933575")) + self.assertEqual(report["remaining_cap_usd"], Decimal("3.5066425")) self.assertIsNone(report["unknown_costs"]["complete_total_usd"]) self.assertEqual( [item["outcome"] for item in report["failures"]], @@ -123,6 +123,7 @@ def test_sanitized_receipts_are_present_and_baseline_estimate_recomputes(self): "caption-aborted.json", "caption-smoke.json", "cap-probe-32-incompatible.json", + "cap-probe-32-direct-incompatible.json", } self.assertTrue(required.issubset({path.name for path in RECEIPTS.glob("*.json")})) receipt = json.loads((RECEIPTS / "gemini38-baseline.json").read_text()) @@ -196,7 +197,7 @@ def test_incomplete_av_side_suppresses_baseline_ratio(self): def test_cap_rejects_estimates_plus_reservations_above_limit(self): data = json.loads((HERE / "scenario.receipts.json").read_text()) - data["cumulative_cap_usd"] = "1.50" + data["cumulative_cap_usd"] = "1.4933574" with self.assertRaisesRegex(ValueError, "exceed cumulative cap"): experiment_report(data) diff --git a/cookbook/receipts/README.md b/cookbook/receipts/README.md index a16feaf..350cb1d 100644 --- a/cookbook/receipts/README.md +++ b/cookbook/receipts/README.md @@ -14,6 +14,7 @@ because those fields are needed to interpret the recorded result. | [caption-aborted.json](caption-aborted.json) | Four metered caption responses; ingestion aborted and no captions persisted | | [caption-smoke.json](caption-smoke.json) | Successful one-frame caption smoke request | | [cap-probe-32-incompatible.json](cap-probe-32-incompatible.json) | 32-token cap probe blocked by HTTP 502/no available route; usage unknown | +| [cap-probe-32-direct-incompatible.json](cap-probe-32-direct-incompatible.json) | Direct exact-model 32-token cap probe returned usage but produced 309 completion tokens; output cap ignored | The receipts are evidence for those individual attempts only. No completed AV Grok+Jev comparison exists yet. They do not establish speed, cost, or quality diff --git a/cookbook/receipts/cap-probe-32-direct-incompatible.json b/cookbook/receipts/cap-probe-32-direct-incompatible.json new file mode 100644 index 0000000..1d5a9f5 --- /dev/null +++ b/cookbook/receipts/cap-probe-32-direct-incompatible.json @@ -0,0 +1,48 @@ +{ + "schema_version": 1, + "stage": "max_completion_tokens_frame_caption_probe", + "status": "incompatible", + "runtime_commit": "7e1a0ce1271e4d7b88d2b6cb5b0bbb9103a01ee6", + "model_requested": "grok-4.20-0309-non-reasoning", + "model_returned": "grok-4.20-0309-non-reasoning", + "token_limit_parameter": "max_completion_tokens", + "requested_output_cap": 32, + "automatic_retries": 0, + "hidden_fallbacks": false, + "source_sha256": "863559e8630cae8019139edc03e0a9ea0437aac2c2f48a811f2d31f9aa14a575", + "frame_timestamp_sec": 0.0, + "frame_sha256": "0907fcb6670ebb7de71954b822b7d06306337eec88d43ac8ce2ace6f3d74acd0", + "frame_dimensions": [ + 1280, + 720 + ], + "jpeg_quality": 2, + "wall_seconds": 5.13789195100253, + "usage": { + "completion_tokens": 309, + "prompt_tokens": 1168, + "total_tokens": 1477, + "completion_tokens_details": { + "accepted_prediction_tokens": null, + "audio_tokens": null, + "reasoning_tokens": 0, + "rejected_prediction_tokens": null + }, + "prompt_tokens_details": { + "audio_tokens": null, + "cached_tokens": 192 + } + }, + "usage_complete": true, + "cap_enforced": false, + "finish_reason": "stop", + "caption_nonempty": true, + "caption_chars": 1410, + "error_type": null, + "cache_aware_list_estimate_usd": 0.0020309, + "proxy_billed_usd": null, + "basis": "estimated from measured usage and published upstream list rates", + "endpoint_class": "direct_agenticflow", + "private_endpoint_or_credential_included": false, + "stop_decision": "No full ingestion or query followed because the provider returned 309 completion tokens despite requested max_completion_tokens=32." +} From e24845de7b58eb55062bbea4d5efe72e8c37d30b Mon Sep 17 00:00:00 2001 From: Sean Phan Date: Sat, 19 Sep 2026 02:23:39 +0000 Subject: [PATCH 08/12] docs: record restored cap violation --- cookbook/README.md | 11 +-- cookbook/cost-model/README.md | 20 +++--- cookbook/cost-model/notebook.ipynb | 67 ++++++++++++++----- cookbook/cost-model/scenario.receipts.json | 45 +++++++++++-- cookbook/cost-model/test_model.py | 18 ++--- cookbook/jev-refined-ask/README.md | 9 ++- cookbook/receipts/README.md | 1 + .../cap-probe-32-direct-incompatible.json | 1 - .../cap-probe-32-restored-incompatible.json | 55 +++++++++++++++ 9 files changed, 182 insertions(+), 45 deletions(-) create mode 100644 cookbook/receipts/cap-probe-32-restored-incompatible.json diff --git a/cookbook/README.md b/cookbook/README.md index dfd379d..2e65775 100644 --- a/cookbook/README.md +++ b/cookbook/README.md @@ -6,13 +6,16 @@ Runnable recipes for the open-source **av** CLI live here alongside the code. |---|---|---| | [Cost model](cost-model/README.md) | Separate tokens, estimates, unknown costs, ingestion, queries, failures, reservations, and cap accounting | Completed component receipts; paired AV comparison incomplete | | [Jev-refined ask](jev-refined-ask/README.md) | Build a source-bound transcript sidecar, retrieve indexed moments, refine evidence, answer, and check support | Runnable recipe; no speed, cost, or quality parity claim | -| [Sanitized receipts](receipts/README.md) | Completed ASR/baseline, caption smoke/abort, and both incompatible cap probes | No media, transcript/caption corpus, credentials, upload URIs, or private routes; baseline question and returned answer/rationale retained | +| [Sanitized receipts](receipts/README.md) | Completed ASR/baseline, caption smoke/abort, and incompatible cap probes including the restored-route retry | No media, transcript/caption corpus, credentials, upload URIs, or private routes; baseline question and returned answer/rationale retained | The [original public cost notebook](https://github.com/PixelML/cookbook/tree/main/agentic-video/cost-model) remains available at its existing URL. Its Composer demonstrations are historical context, not measurements of this CLI. New AV reproduction results belong in this cookbook with their own media, model, configuration, and cost provenance. -The current evidence includes a completed Gemini 3.8 direct-video baseline, but -there is **no completed AV Grok+Jev comparison yet**. Do not infer speed, cost, or -quality parity from the component receipts. +The current evidence includes a completed Gemini 3.8 direct-video baseline. The +restored Grok route returned the exact requested model and usage, but returned +260 completion tokens against `max_completion_tokens=32`. The guard therefore +stopped before ingestion or query. There is **no completed AV Grok+Jev +comparison yet**; do not infer speed, cost, or quality parity from component +receipts. diff --git a/cookbook/cost-model/README.md b/cookbook/cost-model/README.md index 495218d..b8ab722 100644 --- a/cookbook/cost-model/README.md +++ b/cookbook/cost-model/README.md @@ -26,24 +26,26 @@ The evidence is partial: | Gemini 3.8 native-video baseline | completed | $0.308076 | | 32-token caption cap probe through the first configured route | HTTP 502/no route; usage unknown | | Direct 32-token caption cap probe | exact model returned 309 completion tokens against requested cap 32 | $0.0020309 | +| Restored-route 32-token caption cap probe at `6a1cde2` | exact model returned 260 completion tokens against requested cap 32 | $0.0009004 | -Known token-derived list-rate estimates total **$0.3933575**. They are not billed +Known token-derived list-rate estimates total **$0.3942579**. They are not billed dollars. Provider/account or proxy billing remains unknown. Compute, storage, and network allocation also remains unknown. The cumulative experiment cap is **$5**. The reservation ledger records -**$3.97945825** of request ceilings: **$2.501** for the caption smoke and -**$0.37845825** for the native-video baseline were released after metering. -The **$0.01** first-route cap-probe ceiling was closed after a second direct -probe measured **$0.0020309**, leaving **$0.90** caption ingestion and -**$0.20** query/judge/answer retained. Only the retained **$1.10** counts against -current headroom. Known estimates plus retained reservations are **$1.4933575**, -leaving **$3.5066425** of estimate headroom. +**$3.99945825** of historical request ceilings. The two **$0.01** probe +reservations were released after their later metered responses. The **$0.90** +caption-ingestion and **$0.20** query/judge/answer reservations were released +unused when the restored-route cap gate failed. No reservation remains retained. +Known estimates plus retained reservations are therefore **$0.3942579**, leaving +**$4.6057421** of estimate headroom. **No completed AV Grok+Jev comparison exists yet.** The receipts do not establish speed, cost, or quality parity with the Gemini 3.8 baseline. The baseline used native video/audio at its recorded sampling configuration; the incomplete AV arm -is not an equal-input comparison. +is not an equal-input comparison. The restored route did not make the selected +output-cap field enforceable, so the guard stopped before the 300-frame ingestion +and before the exact benchmark question. Use [`scenario.pending.json`](scenario.pending.json) as a blank template for another run. diff --git a/cookbook/cost-model/notebook.ipynb b/cookbook/cost-model/notebook.ipynb index 57c3bd4..790ab61 100644 --- a/cookbook/cost-model/notebook.ipynb +++ b/cookbook/cost-model/notebook.ipynb @@ -55,7 +55,7 @@ " \"status\": \"partial_reproduction\",\n", " \"implementation\": \"av\",\n", " \"receipt_directory\": \"../receipts\",\n", - " \"note\": \"Completed ASR and direct-video baseline receipts exist, and the direct 32-token compatibility probe completed with usage, but the exact model ignored the requested output cap. No completed AV Grok+Jev comparison exists yet.\",\n", + " \"note\": \"Completed ASR and direct-video baseline receipts exist. Two exact-model 32-token compatibility probes completed with usage, including a restored-route retry at AV commit 6a1cde2, but both ignored the requested output cap. No paid ingestion/query followed the latest failure and no completed AV Grok+Jev comparison exists yet.\",\n", " \"projection\": \"Repeated-query totals are projections from entered stage costs, not additional measured runs.\"\n", "}\n", "{\n", @@ -121,9 +121,20 @@ " \"outcome\": \"incompatible\",\n", " \"receipt\": \"../receipts/cap-probe-32-direct-incompatible.json\",\n", " \"note\": \"Exact model returned usage but ignored requested max_completion_tokens=32.\"\n", + " },\n", + " {\n", + " \"name\": \"restored_route_max_completion_tokens_cap_probe\",\n", + " \"input_tokens\": 1168,\n", + " \"output_tokens\": 260,\n", + " \"thinking_tokens\": 0,\n", + " \"cached_tokens\": 1152,\n", + " \"requests\": 1,\n", + " \"outcome\": \"incompatible\",\n", + " \"receipt\": \"../receipts/cap-probe-32-restored-incompatible.json\",\n", + " \"note\": \"Restored route returned the exact model and complete usage but ignored requested max_completion_tokens=32.\"\n", " }\n", " ],\n", - " \"known_list_estimates_usd\": \"0.3933575\",\n", + " \"known_list_estimates_usd\": \"0.3942579\",\n", " \"unknown_costs\": {\n", " \"terms\": [\n", " {\n", @@ -179,6 +190,16 @@ " \"outcome\": \"incompatible\",\n", " \"receipt\": \"../receipts/cap-probe-32-direct-incompatible.json\",\n", " \"usage_ref\": \"direct_max_completion_tokens_cap_probe\"\n", + " },\n", + " {\n", + " \"name\": \"experiment.6.restored_route_max_completion_tokens_cap_probe\",\n", + " \"usd\": \"0.0009004\",\n", + " \"basis\": \"estimated\",\n", + " \"note\": \"Cache-aware upstream list-rate estimate; the restored route returned 260 completion tokens despite requested cap 32.\",\n", + " \"category\": \"compatibility_probe\",\n", + " \"outcome\": \"incompatible\",\n", + " \"receipt\": \"../receipts/cap-probe-32-restored-incompatible.json\",\n", + " \"usage_ref\": \"restored_route_max_completion_tokens_cap_probe\"\n", " }\n", " ],\n", " \"reservations\": {\n", @@ -204,29 +225,45 @@ " {\n", " \"name\": \"reservation.2.caption_ingest\",\n", " \"usd\": \"0.90\",\n", - " \"status\": \"retained\",\n", - " \"counts_against_cap\": true,\n", - " \"note\": \"Conservative reservation, not spend.\"\n", + " \"status\": \"superseded\",\n", + " \"counts_against_cap\": false,\n", + " \"note\": \"Released unused because the restored-route cap gate failed before ingestion.\"\n", " },\n", " {\n", " \"name\": \"reservation.3.query_judge_and_answer\",\n", " \"usd\": \"0.20\",\n", - " \"status\": \"retained\",\n", - " \"counts_against_cap\": true,\n", - " \"note\": \"Conservative reservation, not spend.\"\n", + " \"status\": \"superseded\",\n", + " \"counts_against_cap\": false,\n", + " \"note\": \"Released unused because ingestion did not complete and no query was allowed.\"\n", + " },\n", + " {\n", + " \"name\": \"reservation.4.earlier_cap_probe_ceiling\",\n", + " \"usd\": \"0.01\",\n", + " \"status\": \"released_after_metering\",\n", + " \"counts_against_cap\": false,\n", + " \"note\": \"The ceiling covered the earlier no-route attempt and later direct measured probe; it was released after measured usage was recorded.\",\n", + " \"receipt\": \"../receipts/cap-probe-32-direct-incompatible.json\"\n", + " },\n", + " {\n", + " \"name\": \"reservation.5.restored_route_cap_probe_ceiling\",\n", + " \"usd\": \"0.01\",\n", + " \"status\": \"released_after_metering\",\n", + " \"counts_against_cap\": false,\n", + " \"note\": \"Fresh restored-route ceiling released after the one authorized request returned complete usage.\",\n", + " \"receipt\": \"../receipts/cap-probe-32-restored-incompatible.json\"\n", " }\n", " ],\n", " \"status_totals_usd\": {\n", - " \"released_after_metering\": \"2.87945825\",\n", - " \"retained\": \"1.10\",\n", - " \"superseded\": \"0\"\n", + " \"released_after_metering\": \"2.89945825\",\n", + " \"retained\": \"0\",\n", + " \"superseded\": \"1.10\"\n", " },\n", - " \"total_recorded_usd\": \"3.97945825\",\n", - " \"total_retained_usd\": \"1.10\"\n", + " \"total_recorded_usd\": \"3.99945825\",\n", + " \"total_retained_usd\": \"0\"\n", " },\n", " \"cumulative_cap_usd\": \"5\",\n", - " \"known_estimates_plus_retained_usd\": \"1.4933575\",\n", - " \"remaining_cap_usd\": \"3.5066425\"\n", + " \"known_estimates_plus_retained_usd\": \"0.3942579\",\n", + " \"remaining_cap_usd\": \"4.6057421\"\n", "}\n" ] } diff --git a/cookbook/cost-model/scenario.receipts.json b/cookbook/cost-model/scenario.receipts.json index ab05fc4..6973c23 100644 --- a/cookbook/cost-model/scenario.receipts.json +++ b/cookbook/cost-model/scenario.receipts.json @@ -4,7 +4,7 @@ "status": "partial_reproduction", "implementation": "av", "receipt_directory": "../receipts", - "note": "Completed ASR and direct-video baseline receipts exist, and the direct 32-token compatibility probe completed with usage, but the exact model ignored the requested output cap. No completed AV Grok+Jev comparison exists yet.", + "note": "Completed ASR and direct-video baseline receipts exist. Two exact-model 32-token compatibility probes completed with usage, including a restored-route retry at AV commit 6a1cde2, but both ignored the requested output cap. No paid ingestion/query followed the latest failure and no completed AV Grok+Jev comparison exists yet.", "projection": "Repeated-query totals are projections from entered stage costs, not additional measured runs." }, "indexing": [ @@ -159,6 +159,17 @@ "outcome": "incompatible", "receipt": "../receipts/cap-probe-32-direct-incompatible.json", "note": "Exact model returned usage but ignored requested max_completion_tokens=32." + }, + { + "name": "restored_route_max_completion_tokens_cap_probe", + "requests": 1, + "input_tokens": 1168, + "output_tokens": 260, + "thinking_tokens": 0, + "cached_tokens": 1152, + "outcome": "incompatible", + "receipt": "../receipts/cap-probe-32-restored-incompatible.json", + "note": "Restored route returned the exact model and complete usage but ignored requested max_completion_tokens=32." } ], "experiment_spend": [ @@ -221,6 +232,16 @@ "receipt": "../receipts/cap-probe-32-direct-incompatible.json", "usage_ref": "direct_max_completion_tokens_cap_probe", "note": "Cache-aware upstream list-rate estimate; the model returned 309 completion tokens despite requested cap 32." + }, + { + "name": "restored_route_max_completion_tokens_cap_probe", + "usd": "0.0009004", + "basis": "estimated", + "category": "compatibility_probe", + "outcome": "incompatible", + "receipt": "../receipts/cap-probe-32-restored-incompatible.json", + "usage_ref": "restored_route_max_completion_tokens_cap_probe", + "note": "Cache-aware upstream list-rate estimate; the restored route returned 260 completion tokens despite requested cap 32." } ], "unknown_costs": [ @@ -259,14 +280,28 @@ { "name": "caption_ingest", "usd": "0.90", - "status": "retained", - "note": "Conservative reservation, not spend." + "status": "superseded", + "note": "Released unused because the restored-route cap gate failed before ingestion." }, { "name": "query_judge_and_answer", "usd": "0.20", - "status": "retained", - "note": "Conservative reservation, not spend." + "status": "superseded", + "note": "Released unused because ingestion did not complete and no query was allowed." + }, + { + "name": "earlier_cap_probe_ceiling", + "usd": "0.01", + "status": "released_after_metering", + "receipt": "../receipts/cap-probe-32-direct-incompatible.json", + "note": "The ceiling covered the earlier no-route attempt and later direct measured probe; it was released after measured usage was recorded." + }, + { + "name": "restored_route_cap_probe_ceiling", + "usd": "0.01", + "status": "released_after_metering", + "receipt": "../receipts/cap-probe-32-restored-incompatible.json", + "note": "Fresh restored-route ceiling released after the one authorized request returned complete usage." } ], "cumulative_cap_usd": "5" diff --git a/cookbook/cost-model/test_model.py b/cookbook/cost-model/test_model.py index 46af9c9..c937293 100644 --- a/cookbook/cost-model/test_model.py +++ b/cookbook/cost-model/test_model.py @@ -100,19 +100,20 @@ def test_receipt_ledger_preserves_cap_and_separates_unknowns(self): data = json.loads((HERE / "scenario.receipts.json").read_text()) report = experiment_report(data) self.assertEqual(report["list_rate_estimates"]["known_subtotal_usd"], - Decimal("0.3933575")) - self.assertEqual(report["reservations"]["total_recorded_usd"], Decimal("3.97945825")) - self.assertEqual(report["reservations"]["total_retained_usd"], Decimal("1.10")) + Decimal("0.3942579")) + self.assertEqual(report["reservations"]["total_recorded_usd"], Decimal("3.99945825")) + self.assertEqual(report["reservations"]["total_retained_usd"], Decimal("0")) self.assertEqual( report["reservations"]["status_totals_usd"]["released_after_metering"], - Decimal("2.87945825"), + Decimal("2.89945825"), ) - self.assertEqual(report["known_estimates_plus_retained_usd"], Decimal("1.4933575")) - self.assertEqual(report["remaining_cap_usd"], Decimal("3.5066425")) + self.assertEqual(report["reservations"]["status_totals_usd"]["superseded"], Decimal("1.10")) + self.assertEqual(report["known_estimates_plus_retained_usd"], Decimal("0.3942579")) + self.assertEqual(report["remaining_cap_usd"], Decimal("4.6057421")) self.assertIsNone(report["unknown_costs"]["complete_total_usd"]) self.assertEqual( [item["outcome"] for item in report["failures"]], - ["failed", "aborted", "incompatible"], + ["failed", "aborted", "incompatible", "incompatible"], ) self.assertEqual(report["measured_usage"][0]["input_tokens"], 118228) @@ -124,6 +125,7 @@ def test_sanitized_receipts_are_present_and_baseline_estimate_recomputes(self): "caption-smoke.json", "cap-probe-32-incompatible.json", "cap-probe-32-direct-incompatible.json", + "cap-probe-32-restored-incompatible.json", } self.assertTrue(required.issubset({path.name for path in RECEIPTS.glob("*.json")})) receipt = json.loads((RECEIPTS / "gemini38-baseline.json").read_text()) @@ -197,7 +199,7 @@ def test_incomplete_av_side_suppresses_baseline_ratio(self): def test_cap_rejects_estimates_plus_reservations_above_limit(self): data = json.loads((HERE / "scenario.receipts.json").read_text()) - data["cumulative_cap_usd"] = "1.4933574" + data["cumulative_cap_usd"] = "0.3942578" with self.assertRaisesRegex(ValueError, "exceed cumulative cap"): experiment_report(data) diff --git a/cookbook/jev-refined-ask/README.md b/cookbook/jev-refined-ask/README.md index 33b682f..ae23c34 100644 --- a/cookbook/jev-refined-ask/README.md +++ b/cookbook/jev-refined-ask/README.md @@ -193,6 +193,9 @@ receipt. No benchmark in this recipe establishes quality improvement, parity with direct-video models, or a universal cost ratio. The current receipts include completed ASR and Gemini 3.8 baseline calls, an -aborted caption attempt, a successful image smoke call, and a 32-token cap probe -blocked by HTTP 502/no route. **No completed AV Grok+Jev comparison exists yet.** -Do not claim speed, cost, or quality parity from these component attempts. +aborted caption attempt, a successful image smoke call, one no-route probe, and +two exact-model probes that returned 309 and 260 completion tokens against a +requested cap of 32. The latest probe ran at AV commit `6a1cde2`, with automatic +retries and hidden fallbacks disabled. **No paid ingestion or query followed, +and no completed AV Grok+Jev comparison exists yet.** Do not claim speed, cost, +or quality parity from these component attempts. diff --git a/cookbook/receipts/README.md b/cookbook/receipts/README.md index 350cb1d..577bcc3 100644 --- a/cookbook/receipts/README.md +++ b/cookbook/receipts/README.md @@ -15,6 +15,7 @@ because those fields are needed to interpret the recorded result. | [caption-smoke.json](caption-smoke.json) | Successful one-frame caption smoke request | | [cap-probe-32-incompatible.json](cap-probe-32-incompatible.json) | 32-token cap probe blocked by HTTP 502/no available route; usage unknown | | [cap-probe-32-direct-incompatible.json](cap-probe-32-direct-incompatible.json) | Direct exact-model 32-token cap probe returned usage but produced 309 completion tokens; output cap ignored | +| [cap-probe-32-restored-incompatible.json](cap-probe-32-restored-incompatible.json) | Restored-route exact-model probe at commit `6a1cde2` produced 260 completion tokens against cap 32; ingestion/query stopped | The receipts are evidence for those individual attempts only. No completed AV Grok+Jev comparison exists yet. They do not establish speed, cost, or quality diff --git a/cookbook/receipts/cap-probe-32-direct-incompatible.json b/cookbook/receipts/cap-probe-32-direct-incompatible.json index 1d5a9f5..65fb0ec 100644 --- a/cookbook/receipts/cap-probe-32-direct-incompatible.json +++ b/cookbook/receipts/cap-probe-32-direct-incompatible.json @@ -42,7 +42,6 @@ "cache_aware_list_estimate_usd": 0.0020309, "proxy_billed_usd": null, "basis": "estimated from measured usage and published upstream list rates", - "endpoint_class": "direct_agenticflow", "private_endpoint_or_credential_included": false, "stop_decision": "No full ingestion or query followed because the provider returned 309 completion tokens despite requested max_completion_tokens=32." } diff --git a/cookbook/receipts/cap-probe-32-restored-incompatible.json b/cookbook/receipts/cap-probe-32-restored-incompatible.json new file mode 100644 index 0000000..14d1e83 --- /dev/null +++ b/cookbook/receipts/cap-probe-32-restored-incompatible.json @@ -0,0 +1,55 @@ +{ + "schema_version": 1, + "stage": "restored_route_max_completion_tokens_frame_caption_probe", + "status": "incompatible", + "runtime_commit": "6a1cde2fb0009759e45983ff1f9b9d5afc50f9fa", + "model_requested": "grok-4.20-0309-non-reasoning", + "model_returned": "grok-4.20-0309-non-reasoning", + "token_limit_parameter": "max_completion_tokens", + "requested_output_cap": 32, + "automatic_retries": 0, + "hidden_fallbacks": false, + "request_count": 1, + "successful_provider_responses": 1, + "source_sha256": "863559e8630cae8019139edc03e0a9ea0437aac2c2f48a811f2d31f9aa14a575", + "frame_timestamp_sec": 0.0, + "frame_sha256": "0907fcb6670ebb7de71954b822b7d06306337eec88d43ac8ce2ace6f3d74acd0", + "frame_dimensions": [ + 1280, + 720 + ], + "jpeg_quality": 2, + "wall_seconds": 4.679620450006041, + "usage": { + "completion_tokens": 260, + "prompt_tokens": 1168, + "total_tokens": 1428, + "completion_tokens_details": { + "accepted_prediction_tokens": null, + "audio_tokens": null, + "reasoning_tokens": 0, + "rejected_prediction_tokens": null + }, + "prompt_tokens_details": { + "audio_tokens": null, + "cached_tokens": 1152 + } + }, + "usage_complete": true, + "cap_enforced": false, + "finish_reason": "stop", + "caption_nonempty": true, + "caption_chars": 1229, + "provider_response_metadata": { + "response_id_present": true, + "created": 0, + "system_fingerprint": null, + "service_tier": null + }, + "error_type": null, + "cache_aware_list_estimate_usd": 0.0009004, + "proxy_billed_usd": null, + "basis": "estimated from measured usage and published upstream list rates", + "private_endpoint_or_credential_included": false, + "stop_decision": "No full ingestion or query followed because the provider returned 260 completion tokens despite requested max_completion_tokens=32." +} From dd9dfa2f3dd1d801589569b2872f21c2fa268d8c Mon Sep 17 00:00:00 2001 From: Sean Phan Date: Sat, 19 Sep 2026 03:41:58 +0000 Subject: [PATCH 09/12] docs: record zero-request live ingestion failure --- cookbook/README.md | 8 +++-- cookbook/cost-model/README.md | 14 +++++--- cookbook/cost-model/notebook.ipynb | 25 ++++++++++++-- cookbook/cost-model/scenario.receipts.json | 29 +++++++++++++++- cookbook/cost-model/test_model.py | 7 ++-- cookbook/receipts/README.md | 1 + .../live-ingestion-local-probe-timeout.json | 33 +++++++++++++++++++ 7 files changed, 103 insertions(+), 14 deletions(-) create mode 100644 cookbook/receipts/live-ingestion-local-probe-timeout.json diff --git a/cookbook/README.md b/cookbook/README.md index 2e65775..3775717 100644 --- a/cookbook/README.md +++ b/cookbook/README.md @@ -6,7 +6,7 @@ Runnable recipes for the open-source **av** CLI live here alongside the code. |---|---|---| | [Cost model](cost-model/README.md) | Separate tokens, estimates, unknown costs, ingestion, queries, failures, reservations, and cap accounting | Completed component receipts; paired AV comparison incomplete | | [Jev-refined ask](jev-refined-ask/README.md) | Build a source-bound transcript sidecar, retrieve indexed moments, refine evidence, answer, and check support | Runnable recipe; no speed, cost, or quality parity claim | -| [Sanitized receipts](receipts/README.md) | Completed ASR/baseline, caption smoke/abort, and incompatible cap probes including the restored-route retry | No media, transcript/caption corpus, credentials, upload URIs, or private routes; baseline question and returned answer/rationale retained | +| [Sanitized receipts](receipts/README.md) | Completed ASR/baseline, caption smoke/abort, incompatible cap probes, and the fresh zero-request local media-probe failure | No media, transcript/caption corpus, credentials, upload URIs, or private routes; baseline question and returned answer/rationale retained | The [original public cost notebook](https://github.com/PixelML/cookbook/tree/main/agentic-video/cost-model) remains available at its existing URL. Its Composer demonstrations are historical @@ -15,7 +15,9 @@ this cookbook with their own media, model, configuration, and cost provenance. The current evidence includes a completed Gemini 3.8 direct-video baseline. The restored Grok route returned the exact requested model and usage, but returned -260 completion tokens against `max_completion_tokens=32`. The guard therefore -stopped before ingestion or query. There is **no completed AV Grok+Jev +260 completion tokens against `max_completion_tokens=32`. A later explicitly +authorized fresh attempt imported the completed transcript, then stopped on a +local media-probe timeout before frame extraction: 0 captions and 0 provider +requests. There is **no completed AV Grok+Jev comparison yet**; do not infer speed, cost, or quality parity from component receipts. diff --git a/cookbook/cost-model/README.md b/cookbook/cost-model/README.md index b8ab722..3b473b7 100644 --- a/cookbook/cost-model/README.md +++ b/cookbook/cost-model/README.md @@ -27,25 +27,29 @@ The evidence is partial: | 32-token caption cap probe through the first configured route | HTTP 502/no route; usage unknown | | Direct 32-token caption cap probe | exact model returned 309 completion tokens against requested cap 32 | $0.0020309 | | Restored-route 32-token caption cap probe at `6a1cde2` | exact model returned 260 completion tokens against requested cap 32 | $0.0009004 | +| Fresh 300-frame ingestion attempt at `e24845d` | local media probe timed out before frame extraction; 75 transcript artifacts, 0 captions, 0 provider requests | $0 | Known token-derived list-rate estimates total **$0.3942579**. They are not billed dollars. Provider/account or proxy billing remains unknown. Compute, storage, and network allocation also remains unknown. The cumulative experiment cap is **$5**. The reservation ledger records -**$3.99945825** of historical request ceilings. The two **$0.01** probe +**$4.89945825** of historical request ceilings. The two **$0.01** probe reservations were released after their later metered responses. The **$0.90** caption-ingestion and **$0.20** query/judge/answer reservations were released -unused when the restored-route cap gate failed. No reservation remains retained. +unused when the restored-route cap gate failed. A fresh **$0.90** ingestion +reservation was later retained for the authorized 300-frame attempt and released +unused after the local media probe timed out before provider request 1. No +reservation remains retained. Known estimates plus retained reservations are therefore **$0.3942579**, leaving **$4.6057421** of estimate headroom. **No completed AV Grok+Jev comparison exists yet.** The receipts do not establish speed, cost, or quality parity with the Gemini 3.8 baseline. The baseline used native video/audio at its recorded sampling configuration; the incomplete AV arm -is not an equal-input comparison. The restored route did not make the selected -output-cap field enforceable, so the guard stopped before the 300-frame ingestion -and before the exact benchmark question. +is not an equal-input comparison. The later fresh attempt also did not reach the +provider: local frame-extraction preflight stopped before request 1, and the exact +benchmark question was not run against incomplete evidence. Use [`scenario.pending.json`](scenario.pending.json) as a blank template for another run. diff --git a/cookbook/cost-model/notebook.ipynb b/cookbook/cost-model/notebook.ipynb index 790ab61..53fe5db 100644 --- a/cookbook/cost-model/notebook.ipynb +++ b/cookbook/cost-model/notebook.ipynb @@ -55,7 +55,7 @@ " \"status\": \"partial_reproduction\",\n", " \"implementation\": \"av\",\n", " \"receipt_directory\": \"../receipts\",\n", - " \"note\": \"Completed ASR and direct-video baseline receipts exist. Two exact-model 32-token compatibility probes completed with usage, including a restored-route retry at AV commit 6a1cde2, but both ignored the requested output cap. No paid ingestion/query followed the latest failure and no completed AV Grok+Jev comparison exists yet.\",\n", + " \"note\": \"Completed ASR and direct-video baseline receipts exist. Two exact-model 32-token compatibility probes completed with usage but ignored the requested output cap. A later fresh attempt at AV commit e24845d imported the transcript, then stopped on a local media-probe timeout before frame extraction with zero provider requests. No completed AV Grok+Jev comparison exists yet.\",\n", " \"projection\": \"Repeated-query totals are projections from entered stage costs, not additional measured runs.\"\n", "}\n", "{\n", @@ -132,6 +132,17 @@ " \"outcome\": \"incompatible\",\n", " \"receipt\": \"../receipts/cap-probe-32-restored-incompatible.json\",\n", " \"note\": \"Restored route returned the exact model and complete usage but ignored requested max_completion_tokens=32.\"\n", + " },\n", + " {\n", + " \"name\": \"live_ingestion_local_probe_timeout\",\n", + " \"input_tokens\": null,\n", + " \"output_tokens\": null,\n", + " \"thinking_tokens\": null,\n", + " \"cached_tokens\": null,\n", + " \"requests\": 0,\n", + " \"outcome\": \"blocked_before_provider_request\",\n", + " \"receipt\": \"../receipts/live-ingestion-local-probe-timeout.json\",\n", + " \"note\": \"The completed transcript imported, but a local media probe timed out before 300-frame extraction; no provider request was attempted.\"\n", " }\n", " ],\n", " \"known_list_estimates_usd\": \"0.3942579\",\n", @@ -251,14 +262,22 @@ " \"counts_against_cap\": false,\n", " \"note\": \"Fresh restored-route ceiling released after the one authorized request returned complete usage.\",\n", " \"receipt\": \"../receipts/cap-probe-32-restored-incompatible.json\"\n", + " },\n", + " {\n", + " \"name\": \"reservation.6.fresh_live_caption_ingest\",\n", + " \"usd\": \"0.90\",\n", + " \"status\": \"superseded\",\n", + " \"counts_against_cap\": false,\n", + " \"note\": \"Released unused after local frame-extraction preflight timed out before provider request 1.\",\n", + " \"receipt\": \"../receipts/live-ingestion-local-probe-timeout.json\"\n", " }\n", " ],\n", " \"status_totals_usd\": {\n", " \"released_after_metering\": \"2.89945825\",\n", " \"retained\": \"0\",\n", - " \"superseded\": \"1.10\"\n", + " \"superseded\": \"2.00\"\n", " },\n", - " \"total_recorded_usd\": \"3.99945825\",\n", + " \"total_recorded_usd\": \"4.89945825\",\n", " \"total_retained_usd\": \"0\"\n", " },\n", " \"cumulative_cap_usd\": \"5\",\n", diff --git a/cookbook/cost-model/scenario.receipts.json b/cookbook/cost-model/scenario.receipts.json index 6973c23..3e9cb31 100644 --- a/cookbook/cost-model/scenario.receipts.json +++ b/cookbook/cost-model/scenario.receipts.json @@ -4,7 +4,7 @@ "status": "partial_reproduction", "implementation": "av", "receipt_directory": "../receipts", - "note": "Completed ASR and direct-video baseline receipts exist. Two exact-model 32-token compatibility probes completed with usage, including a restored-route retry at AV commit 6a1cde2, but both ignored the requested output cap. No paid ingestion/query followed the latest failure and no completed AV Grok+Jev comparison exists yet.", + "note": "Completed ASR and direct-video baseline receipts exist. Two exact-model 32-token compatibility probes completed with usage but ignored the requested output cap. A later fresh attempt at AV commit e24845d imported the transcript, then stopped on a local media-probe timeout before frame extraction with zero provider requests. No completed AV Grok+Jev comparison exists yet.", "projection": "Repeated-query totals are projections from entered stage costs, not additional measured runs." }, "indexing": [ @@ -170,6 +170,17 @@ "outcome": "incompatible", "receipt": "../receipts/cap-probe-32-restored-incompatible.json", "note": "Restored route returned the exact model and complete usage but ignored requested max_completion_tokens=32." + }, + { + "name": "live_ingestion_local_probe_timeout", + "requests": 0, + "input_tokens": null, + "output_tokens": null, + "thinking_tokens": null, + "cached_tokens": null, + "outcome": "blocked_before_provider_request", + "receipt": "../receipts/live-ingestion-local-probe-timeout.json", + "note": "The completed transcript imported, but a local media probe timed out before 300-frame extraction; no provider request was attempted." } ], "experiment_spend": [ @@ -242,6 +253,15 @@ "receipt": "../receipts/cap-probe-32-restored-incompatible.json", "usage_ref": "restored_route_max_completion_tokens_cap_probe", "note": "Cache-aware upstream list-rate estimate; the restored route returned 260 completion tokens despite requested cap 32." + }, + { + "name": "live_ingestion_local_probe_timeout", + "usd": 0, + "basis": "not_used", + "category": "failed_attempt", + "outcome": "blocked_before_provider_request", + "receipt": "../receipts/live-ingestion-local-probe-timeout.json", + "note": "No provider request was attempted; local compute, storage, and network allocation remain unknown." } ], "unknown_costs": [ @@ -302,6 +322,13 @@ "status": "released_after_metering", "receipt": "../receipts/cap-probe-32-restored-incompatible.json", "note": "Fresh restored-route ceiling released after the one authorized request returned complete usage." + }, + { + "name": "fresh_live_caption_ingest", + "usd": "0.90", + "status": "superseded", + "receipt": "../receipts/live-ingestion-local-probe-timeout.json", + "note": "Released unused after local frame-extraction preflight timed out before provider request 1." } ], "cumulative_cap_usd": "5" diff --git a/cookbook/cost-model/test_model.py b/cookbook/cost-model/test_model.py index c937293..0a0dcbf 100644 --- a/cookbook/cost-model/test_model.py +++ b/cookbook/cost-model/test_model.py @@ -101,13 +101,13 @@ def test_receipt_ledger_preserves_cap_and_separates_unknowns(self): report = experiment_report(data) self.assertEqual(report["list_rate_estimates"]["known_subtotal_usd"], Decimal("0.3942579")) - self.assertEqual(report["reservations"]["total_recorded_usd"], Decimal("3.99945825")) + self.assertEqual(report["reservations"]["total_recorded_usd"], Decimal("4.89945825")) self.assertEqual(report["reservations"]["total_retained_usd"], Decimal("0")) self.assertEqual( report["reservations"]["status_totals_usd"]["released_after_metering"], Decimal("2.89945825"), ) - self.assertEqual(report["reservations"]["status_totals_usd"]["superseded"], Decimal("1.10")) + self.assertEqual(report["reservations"]["status_totals_usd"]["superseded"], Decimal("2.00")) self.assertEqual(report["known_estimates_plus_retained_usd"], Decimal("0.3942579")) self.assertEqual(report["remaining_cap_usd"], Decimal("4.6057421")) self.assertIsNone(report["unknown_costs"]["complete_total_usd"]) @@ -116,6 +116,8 @@ def test_receipt_ledger_preserves_cap_and_separates_unknowns(self): ["failed", "aborted", "incompatible", "incompatible"], ) self.assertEqual(report["measured_usage"][0]["input_tokens"], 118228) + self.assertEqual(report["measured_usage"][-1]["outcome"], "blocked_before_provider_request") + self.assertEqual(report["measured_usage"][-1]["requests"], 0) def test_sanitized_receipts_are_present_and_baseline_estimate_recomputes(self): required = { @@ -126,6 +128,7 @@ def test_sanitized_receipts_are_present_and_baseline_estimate_recomputes(self): "cap-probe-32-incompatible.json", "cap-probe-32-direct-incompatible.json", "cap-probe-32-restored-incompatible.json", + "live-ingestion-local-probe-timeout.json", } self.assertTrue(required.issubset({path.name for path in RECEIPTS.glob("*.json")})) receipt = json.loads((RECEIPTS / "gemini38-baseline.json").read_text()) diff --git a/cookbook/receipts/README.md b/cookbook/receipts/README.md index 577bcc3..74248a8 100644 --- a/cookbook/receipts/README.md +++ b/cookbook/receipts/README.md @@ -16,6 +16,7 @@ because those fields are needed to interpret the recorded result. | [cap-probe-32-incompatible.json](cap-probe-32-incompatible.json) | 32-token cap probe blocked by HTTP 502/no available route; usage unknown | | [cap-probe-32-direct-incompatible.json](cap-probe-32-direct-incompatible.json) | Direct exact-model 32-token cap probe returned usage but produced 309 completion tokens; output cap ignored | | [cap-probe-32-restored-incompatible.json](cap-probe-32-restored-incompatible.json) | Restored-route exact-model probe at commit `6a1cde2` produced 260 completion tokens against cap 32; ingestion/query stopped | +| [live-ingestion-local-probe-timeout.json](live-ingestion-local-probe-timeout.json) | Fresh 300-frame attempt at commit `e24845d` stopped locally before provider request 1; transcript imported, 0 captions persisted, $0 provider estimate | The receipts are evidence for those individual attempts only. No completed AV Grok+Jev comparison exists yet. They do not establish speed, cost, or quality diff --git a/cookbook/receipts/live-ingestion-local-probe-timeout.json b/cookbook/receipts/live-ingestion-local-probe-timeout.json new file mode 100644 index 0000000..4556dd7 --- /dev/null +++ b/cookbook/receipts/live-ingestion-local-probe-timeout.json @@ -0,0 +1,33 @@ +{ + "schema_version": 1, + "stage": "grok_dense_ingestion", + "status": "blocked_before_provider_request", + "runtime_commit": "e24845de7b58eb55062bbea4d5efe72e8c37d30b", + "model_requested": "grok-4.20-0309-non-reasoning", + "settings": { + "sample_frames": 300, + "fps_sample": 0.06666666666666667, + "frame_dimensions": [1280, 720], + "jpeg_quality": 2, + "embeddings": false, + "transcript_windows": 75, + "automatic_retries": 0, + "hidden_fallbacks": false, + "completion_cap_advisory": true + }, + "wall_seconds": 42.45906633300183, + "request_attempts": 0, + "successful_provider_responses": 0, + "usage": null, + "cache_aware_list_estimate_usd": 0, + "artifacts_persisted": { + "transcript": 75, + "dense_caption": 0 + }, + "sanitized_failure": { + "stage": "local_frame_extraction_preflight", + "error_type": "FFprobeTimeout", + "message": "Dense vision was skipped because a local media probe exceeded its timeout." + }, + "stop_decision": "No retry and no query against incomplete evidence." +} From 59a7f7317329e4bba6e1b43cdf289a233e865a75 Mon Sep 17 00:00:00 2001 From: Sean Phan Date: Sat, 19 Sep 2026 06:05:40 +0000 Subject: [PATCH 10/12] docs: account for completed Grok live run --- cookbook/README.md | 15 ++- cookbook/cost-model/README.md | 39 +++--- cookbook/cost-model/notebook.ipynb | 87 ++++++++++---- cookbook/cost-model/scenario.receipts.json | 112 +++++++++++++++--- cookbook/cost-model/test_model.py | 19 +-- cookbook/receipts/README.md | 8 +- cookbook/receipts/grok-legacy-query.json | 53 +++++++++ cookbook/receipts/grok-live-ingestion.json | 59 +++++++++ cookbook/receipts/jev-credential-blocked.json | 17 +++ 9 files changed, 335 insertions(+), 74 deletions(-) create mode 100644 cookbook/receipts/grok-legacy-query.json create mode 100644 cookbook/receipts/grok-live-ingestion.json create mode 100644 cookbook/receipts/jev-credential-blocked.json diff --git a/cookbook/README.md b/cookbook/README.md index 3775717..15b66b6 100644 --- a/cookbook/README.md +++ b/cookbook/README.md @@ -13,11 +13,10 @@ remains available at its existing URL. Its Composer demonstrations are historica context, not measurements of this CLI. New AV reproduction results belong in this cookbook with their own media, model, configuration, and cost provenance. -The current evidence includes a completed Gemini 3.8 direct-video baseline. The -restored Grok route returned the exact requested model and usage, but returned -260 completion tokens against `max_completion_tokens=32`. A later explicitly -authorized fresh attempt imported the completed transcript, then stopped on a -local media-probe timeout before frame extraction: 0 captions and 0 provider -requests. There is **no completed AV Grok+Jev -comparison yet**; do not infer speed, cost, or quality parity from component -receipts. +The current evidence includes a completed Gemini 3.8 direct-video baseline and a +completed AV Grok-only path: 300/300 caption requests, 75 transcript windows, and +one correct answer with transcript citation. The selected route ignored output +caps in both 32-token probes, and six of the 300 caption responses exceeded the +requested 200-token advisory cap. No Jev request ran because the required +credential was absent, so there is **no completed AV Grok+Jev comparison**. Do +not infer all-in speed, cost, or quality parity from component receipts. diff --git a/cookbook/cost-model/README.md b/cookbook/cost-model/README.md index 3b473b7..464bbf6 100644 --- a/cookbook/cost-model/README.md +++ b/cookbook/cost-model/README.md @@ -15,7 +15,7 @@ Its checked-in source and outputs are generated by `build_nb.py`. [`scenario.receipts.json`](scenario.receipts.json) integrates the sanitized receipts in [`cookbook/receipts`](../receipts/README.md). -The evidence is partial: +The evidence is a partial reproduction: ingestion and one Grok-only answer completed, but the Jev arm made zero requests. | Attempt | Outcome | Recorded list-rate estimate | |---|---|---:| @@ -28,28 +28,35 @@ The evidence is partial: | Direct 32-token caption cap probe | exact model returned 309 completion tokens against requested cap 32 | $0.0020309 | | Restored-route 32-token caption cap probe at `6a1cde2` | exact model returned 260 completion tokens against requested cap 32 | $0.0009004 | | Fresh 300-frame ingestion attempt at `e24845d` | local media probe timed out before frame extraction; 75 transcript artifacts, 0 captions, 0 provider requests | $0 | +| Completed 300-frame Grok ingestion at `dd9dfa2` | 300/300 requests succeeded; 300 captions and 75 transcript windows persisted; six responses exceeded the requested 200-token advisory cap (maximum 235) | $0.5048525 | +| Completed Grok-only legacy query at `dd9dfa2` | correct answer with transcript citation `1140–1200s`; evidence remained raw/unjudged | $0.0015481 | +| Jev comparison arm | blocked before request because no configured `AV_TYPESAFE_API_KEY` credential was available | $0 | -Known token-derived list-rate estimates total **$0.3942579**. They are not billed +Known token-derived list-rate estimates total **$0.9006585**. They are not billed dollars. Provider/account or proxy billing remains unknown. Compute, storage, and network allocation also remains unknown. The cumulative experiment cap is **$5**. The reservation ledger records -**$4.89945825** of historical request ceilings. The two **$0.01** probe +**$5.99945825** of historical request ceilings. The two **$0.01** probe reservations were released after their later metered responses. The **$0.90** caption-ingestion and **$0.20** query/judge/answer reservations were released -unused when the restored-route cap gate failed. A fresh **$0.90** ingestion -reservation was later retained for the authorized 300-frame attempt and released -unused after the local media probe timed out before provider request 1. No -reservation remains retained. -Known estimates plus retained reservations are therefore **$0.3942579**, leaving -**$4.6057421** of estimate headroom. - -**No completed AV Grok+Jev comparison exists yet.** The receipts do not establish -speed, cost, or quality parity with the Gemini 3.8 baseline. The baseline used -native video/audio at its recorded sampling configuration; the incomplete AV arm -is not an equal-input comparison. The later fresh attempt also did not reach the -provider: local frame-extraction preflight stopped before request 1, and the exact -benchmark question was not run against incomplete evidence. +unused when the restored-route cap gate failed. A later **$0.90** ingestion +reservation was released after all 300 caption responses reported usage. A +**$0.20** metered-query reservation was released after the Grok answer reported +usage; the Jev arm made zero requests. No reservation remains retained. +Known estimates plus retained reservations are therefore **$0.9006585**, leaving +**$4.0993415** of estimate headroom. + +**No completed AV Grok+Jev comparison exists.** The completed run used 300 sampled +frames plus a coarse transcript sidecar and answered through the legacy route; no +Jev relevance or support request ran because the required credential was absent. +The recorded 2.0741-second answer latency is only that single Grok answer request, +not an all-in AV-versus-Gemini latency comparison, and no aggregate quality-parity +or savings claim is valid from one question. The Gemini baseline used native +video/audio at its recorded sampling configuration. Six caption responses also +exceeded the requested 200-token advisory cap; output caps are therefore not an +enforced guarantee on this route. Provider billing and compute, storage, and +network costs remain unknown. Use [`scenario.pending.json`](scenario.pending.json) as a blank template for another run. diff --git a/cookbook/cost-model/notebook.ipynb b/cookbook/cost-model/notebook.ipynb index 53fe5db..213cf84 100644 --- a/cookbook/cost-model/notebook.ipynb +++ b/cookbook/cost-model/notebook.ipynb @@ -52,10 +52,10 @@ "output_type": "stream", "text": [ "{\n", - " \"status\": \"partial_reproduction\",\n", + " \"status\": \"partial_reproduction_grok_only\",\n", " \"implementation\": \"av\",\n", " \"receipt_directory\": \"../receipts\",\n", - " \"note\": \"Completed ASR and direct-video baseline receipts exist. Two exact-model 32-token compatibility probes completed with usage but ignored the requested output cap. A later fresh attempt at AV commit e24845d imported the transcript, then stopped on a local media-probe timeout before frame extraction with zero provider requests. No completed AV Grok+Jev comparison exists yet.\",\n", + " \"note\": \"Completed external ASR, 300-frame Grok ingestion, and one Grok-only legacy query receipts exist. No Jev request ran because no configured AV_TYPESAFE_API_KEY credential was available, so no completed AV Grok+Jev comparison or quality-parity claim exists.\",\n", " \"projection\": \"Repeated-query totals are projections from entered stage costs, not additional measured runs.\"\n", "}\n", "{\n", @@ -143,9 +143,42 @@ " \"outcome\": \"blocked_before_provider_request\",\n", " \"receipt\": \"../receipts/live-ingestion-local-probe-timeout.json\",\n", " \"note\": \"The completed transcript imported, but a local media probe timed out before 300-frame extraction; no provider request was attempted.\"\n", + " },\n", + " {\n", + " \"name\": \"grok_dense_ingestion\",\n", + " \"input_tokens\": 375300,\n", + " \"output_tokens\": 38483,\n", + " \"thinking_tokens\": 0,\n", + " \"cached_tokens\": 57600,\n", + " \"requests\": 300,\n", + " \"outcome\": \"completed\",\n", + " \"receipt\": \"../receipts/grok-live-ingestion.json\",\n", + " \"note\": \"300/300 frame-caption requests succeeded and 75 transcript windows persisted; six completion counts exceeded the requested 200-token advisory cap.\"\n", + " },\n", + " {\n", + " \"name\": \"grok_legacy_query\",\n", + " \"input_tokens\": 1256,\n", + " \"output_tokens\": 45,\n", + " \"thinking_tokens\": 0,\n", + " \"cached_tokens\": 128,\n", + " \"requests\": 1,\n", + " \"outcome\": \"completed_grok_only\",\n", + " \"receipt\": \"../receipts/grok-legacy-query.json\",\n", + " \"note\": \"Correct answer with transcript citation; evidence remained raw and unjudged because no Jev request ran.\"\n", + " },\n", + " {\n", + " \"name\": \"jev_credential_blocker\",\n", + " \"input_tokens\": null,\n", + " \"output_tokens\": null,\n", + " \"thinking_tokens\": null,\n", + " \"cached_tokens\": null,\n", + " \"requests\": 0,\n", + " \"outcome\": \"blocked_before_request\",\n", + " \"receipt\": \"../receipts/jev-credential-blocked.json\",\n", + " \"note\": \"No configured AV_TYPESAFE_API_KEY credential was available.\"\n", " }\n", " ],\n", - " \"known_list_estimates_usd\": \"0.3942579\",\n", + " \"known_list_estimates_usd\": \"0.9006585\",\n", " \"unknown_costs\": {\n", " \"terms\": [\n", " {\n", @@ -245,7 +278,7 @@ " \"usd\": \"0.20\",\n", " \"status\": \"superseded\",\n", " \"counts_against_cap\": false,\n", - " \"note\": \"Released unused because ingestion did not complete and no query was allowed.\"\n", + " \"note\": \"Superseded by the metered-query reservation after the cap gate reopened and the Grok-only query ran.\"\n", " },\n", " {\n", " \"name\": \"reservation.4.earlier_cap_probe_ceiling\",\n", @@ -268,21 +301,39 @@ " \"usd\": \"0.90\",\n", " \"status\": \"superseded\",\n", " \"counts_against_cap\": false,\n", - " \"note\": \"Released unused after local frame-extraction preflight timed out before provider request 1.\",\n", + " \"note\": \"Released unused after local frame-extraction preflight timed out before provider request 1; a later fresh reservation covered the completed metered ingestion.\",\n", " \"receipt\": \"../receipts/live-ingestion-local-probe-timeout.json\"\n", + " },\n", + " {\n", + " \"name\": \"reservation.7.grok_dense_ingestion\",\n", + " \"usd\": \"0.90\",\n", + " \"status\": \"released_after_metering\",\n", + " \"counts_against_cap\": false,\n", + " \"note\": \"Released after 300 responses returned complete usage; the measured estimate remains in experiment_spend.\",\n", + " \"receipt\": \"../receipts/grok-live-ingestion.json\",\n", + " \"receipt_field\": \"reservation.reserved_usd\"\n", + " },\n", + " {\n", + " \"name\": \"reservation.8.query_judge_and_answer_metered\",\n", + " \"usd\": \"0.20\",\n", + " \"status\": \"released_after_metering\",\n", + " \"counts_against_cap\": false,\n", + " \"note\": \"Released after the Grok answer returned complete usage; the Jev arm remained blocked with zero requests and no balance retained.\",\n", + " \"receipt\": \"../receipts/grok-legacy-query.json\",\n", + " \"receipt_field\": \"reservation.reserved_usd\"\n", " }\n", " ],\n", " \"status_totals_usd\": {\n", - " \"released_after_metering\": \"2.89945825\",\n", + " \"released_after_metering\": \"3.99945825\",\n", " \"retained\": \"0\",\n", " \"superseded\": \"2.00\"\n", " },\n", - " \"total_recorded_usd\": \"4.89945825\",\n", + " \"total_recorded_usd\": \"5.99945825\",\n", " \"total_retained_usd\": \"0\"\n", " },\n", " \"cumulative_cap_usd\": \"5\",\n", - " \"known_estimates_plus_retained_usd\": \"0.3942579\",\n", - " \"remaining_cap_usd\": \"4.6057421\"\n", + " \"known_estimates_plus_retained_usd\": \"0.9006585\",\n", + " \"remaining_cap_usd\": \"4.0993415\"\n", "}\n" ] } @@ -326,14 +377,10 @@ "text": [ "{\n", " \"queries\": 1,\n", - " \"known_subtotal_usd\": \"0.0685709\",\n", + " \"known_subtotal_usd\": \"0.5749715\",\n", " \"complete_total_usd\": null,\n", " \"unknown_terms\": [\n", - " \"indexing.0.caption\",\n", " \"indexing.3.preprocessing\",\n", - " \"query.retrieval\",\n", - " \"query.judge\",\n", - " \"query.answer\",\n", " \"period.hosting\",\n", " \"period.storage\"\n", " ],\n", @@ -341,14 +388,10 @@ "}\n", "{\n", " \"queries\": 100,\n", - " \"known_subtotal_usd\": \"0.0685709\",\n", + " \"known_subtotal_usd\": \"0.7282334\",\n", " \"complete_total_usd\": null,\n", " \"unknown_terms\": [\n", - " \"indexing.0.caption\",\n", " \"indexing.3.preprocessing\",\n", - " \"query.retrieval\",\n", - " \"query.judge\",\n", - " \"query.answer\",\n", " \"period.hosting\",\n", " \"period.storage\"\n", " ],\n", @@ -356,14 +399,10 @@ "}\n", "{\n", " \"queries\": 1000,\n", - " \"known_subtotal_usd\": \"0.0685709\",\n", + " \"known_subtotal_usd\": \"2.1215234\",\n", " \"complete_total_usd\": null,\n", " \"unknown_terms\": [\n", - " \"indexing.0.caption\",\n", " \"indexing.3.preprocessing\",\n", - " \"query.retrieval\",\n", - " \"query.judge\",\n", - " \"query.answer\",\n", " \"period.hosting\",\n", " \"period.storage\"\n", " ],\n", diff --git a/cookbook/cost-model/scenario.receipts.json b/cookbook/cost-model/scenario.receipts.json index 3e9cb31..d5db6f5 100644 --- a/cookbook/cost-model/scenario.receipts.json +++ b/cookbook/cost-model/scenario.receipts.json @@ -1,18 +1,18 @@ { "schema_version": 1, "experiment": { - "status": "partial_reproduction", + "status": "partial_reproduction_grok_only", "implementation": "av", "receipt_directory": "../receipts", - "note": "Completed ASR and direct-video baseline receipts exist. Two exact-model 32-token compatibility probes completed with usage but ignored the requested output cap. A later fresh attempt at AV commit e24845d imported the transcript, then stopped on a local media-probe timeout before frame extraction with zero provider requests. No completed AV Grok+Jev comparison exists yet.", + "note": "Completed external ASR, 300-frame Grok ingestion, and one Grok-only legacy query receipts exist. No Jev request ran because no configured AV_TYPESAFE_API_KEY credential was available, so no completed AV Grok+Jev comparison or quality-parity claim exists.", "projection": "Repeated-query totals are projections from entered stage costs, not additional measured runs." }, "indexing": [ { - "name": "caption", - "usd": null, - "basis": "unknown", - "note": "The caption ingestion was aborted after four responses and persisted no frame captions; no repeatable completed caption-ingestion cost exists." + "name": "grok_caption_ingestion", + "usd": "0.5048525", + "basis": "estimated", + "note": "Measured usage for 300 successful frame-caption requests multiplied by recorded upstream list rates; six responses exceeded the requested 200-token advisory cap, and billed dollars are unknown." }, { "name": "external_asr", @@ -35,19 +35,19 @@ ], "query": { "retrieval": { - "usd": null, - "basis": "unknown", - "note": "One offline zero-hit diagnostic exists, but no completed representative AV query workload was priced." + "usd": 0, + "basis": "not_used", + "note": "The recorded Grok-only query used local FTS retrieval with embeddings disabled; provider model cost was zero and local compute remains unpriced." }, "judge": { - "usd": null, - "basis": "unknown", - "note": "No completed Jev relevance/support sequence exists for the comparison." + "usd": 0, + "basis": "not_used", + "note": "No Jev request ran because the required credential was absent; zero is not a measured Jev cost." }, "answer": { - "usd": null, - "basis": "unknown", - "note": "No completed AV answer exists for the comparison." + "usd": "0.0015481", + "basis": "estimated", + "note": "One Grok-only answer request with complete measured usage; billed dollars are unknown." }, "stronger_fallback": { "frequency": 0, @@ -181,6 +181,39 @@ "outcome": "blocked_before_provider_request", "receipt": "../receipts/live-ingestion-local-probe-timeout.json", "note": "The completed transcript imported, but a local media probe timed out before 300-frame extraction; no provider request was attempted." + }, + { + "name": "grok_dense_ingestion", + "requests": 300, + "input_tokens": 375300, + "output_tokens": 38483, + "thinking_tokens": 0, + "cached_tokens": 57600, + "outcome": "completed", + "receipt": "../receipts/grok-live-ingestion.json", + "note": "300/300 frame-caption requests succeeded and 75 transcript windows persisted; six completion counts exceeded the requested 200-token advisory cap." + }, + { + "name": "grok_legacy_query", + "requests": 1, + "input_tokens": 1256, + "output_tokens": 45, + "thinking_tokens": 0, + "cached_tokens": 128, + "outcome": "completed_grok_only", + "receipt": "../receipts/grok-legacy-query.json", + "note": "Correct answer with transcript citation; evidence remained raw and unjudged because no Jev request ran." + }, + { + "name": "jev_credential_blocker", + "requests": 0, + "input_tokens": null, + "output_tokens": null, + "thinking_tokens": null, + "cached_tokens": null, + "outcome": "blocked_before_request", + "receipt": "../receipts/jev-credential-blocked.json", + "note": "No configured AV_TYPESAFE_API_KEY credential was available." } ], "experiment_spend": [ @@ -262,6 +295,35 @@ "outcome": "blocked_before_provider_request", "receipt": "../receipts/live-ingestion-local-probe-timeout.json", "note": "No provider request was attempted; local compute, storage, and network allocation remain unknown." + }, + { + "name": "grok_dense_ingestion", + "usd": "0.5048525", + "basis": "estimated", + "category": "one_time_ingestion", + "outcome": "completed", + "receipt": "../receipts/grok-live-ingestion.json", + "usage_ref": "grok_dense_ingestion", + "note": "Cache-aware upstream list-rate estimate; the proxy bill and infrastructure costs remain unknown." + }, + { + "name": "grok_legacy_query", + "usd": "0.0015481", + "basis": "estimated", + "category": "av_grok_only_query", + "outcome": "completed_grok_only", + "receipt": "../receipts/grok-legacy-query.json", + "usage_ref": "grok_legacy_query", + "note": "Cache-aware upstream list-rate estimate for the answer request; no Jev request ran." + }, + { + "name": "jev_credential_blocker", + "usd": 0, + "basis": "not_used", + "category": "blocked_comparison_arm", + "outcome": "blocked_before_request", + "receipt": "../receipts/jev-credential-blocked.json", + "note": "No provider request was attempted because the required credential was absent." } ], "unknown_costs": [ @@ -307,7 +369,7 @@ "name": "query_judge_and_answer", "usd": "0.20", "status": "superseded", - "note": "Released unused because ingestion did not complete and no query was allowed." + "note": "Superseded by the metered-query reservation after the cap gate reopened and the Grok-only query ran." }, { "name": "earlier_cap_probe_ceiling", @@ -328,7 +390,23 @@ "usd": "0.90", "status": "superseded", "receipt": "../receipts/live-ingestion-local-probe-timeout.json", - "note": "Released unused after local frame-extraction preflight timed out before provider request 1." + "note": "Released unused after local frame-extraction preflight timed out before provider request 1; a later fresh reservation covered the completed metered ingestion." + }, + { + "name": "grok_dense_ingestion", + "usd": "0.90", + "status": "released_after_metering", + "receipt": "../receipts/grok-live-ingestion.json", + "receipt_field": "reservation.reserved_usd", + "note": "Released after 300 responses returned complete usage; the measured estimate remains in experiment_spend." + }, + { + "name": "query_judge_and_answer_metered", + "usd": "0.20", + "status": "released_after_metering", + "receipt": "../receipts/grok-legacy-query.json", + "receipt_field": "reservation.reserved_usd", + "note": "Released after the Grok answer returned complete usage; the Jev arm remained blocked with zero requests and no balance retained." } ], "cumulative_cap_usd": "5" diff --git a/cookbook/cost-model/test_model.py b/cookbook/cost-model/test_model.py index 0a0dcbf..54e1831 100644 --- a/cookbook/cost-model/test_model.py +++ b/cookbook/cost-model/test_model.py @@ -100,23 +100,23 @@ def test_receipt_ledger_preserves_cap_and_separates_unknowns(self): data = json.loads((HERE / "scenario.receipts.json").read_text()) report = experiment_report(data) self.assertEqual(report["list_rate_estimates"]["known_subtotal_usd"], - Decimal("0.3942579")) - self.assertEqual(report["reservations"]["total_recorded_usd"], Decimal("4.89945825")) + Decimal("0.9006585")) + self.assertEqual(report["reservations"]["total_recorded_usd"], Decimal("5.99945825")) self.assertEqual(report["reservations"]["total_retained_usd"], Decimal("0")) self.assertEqual( report["reservations"]["status_totals_usd"]["released_after_metering"], - Decimal("2.89945825"), + Decimal("3.99945825"), ) self.assertEqual(report["reservations"]["status_totals_usd"]["superseded"], Decimal("2.00")) - self.assertEqual(report["known_estimates_plus_retained_usd"], Decimal("0.3942579")) - self.assertEqual(report["remaining_cap_usd"], Decimal("4.6057421")) + self.assertEqual(report["known_estimates_plus_retained_usd"], Decimal("0.9006585")) + self.assertEqual(report["remaining_cap_usd"], Decimal("4.0993415")) self.assertIsNone(report["unknown_costs"]["complete_total_usd"]) self.assertEqual( [item["outcome"] for item in report["failures"]], ["failed", "aborted", "incompatible", "incompatible"], ) self.assertEqual(report["measured_usage"][0]["input_tokens"], 118228) - self.assertEqual(report["measured_usage"][-1]["outcome"], "blocked_before_provider_request") + self.assertEqual(report["measured_usage"][-1]["outcome"], "blocked_before_request") self.assertEqual(report["measured_usage"][-1]["requests"], 0) def test_sanitized_receipts_are_present_and_baseline_estimate_recomputes(self): @@ -129,6 +129,9 @@ def test_sanitized_receipts_are_present_and_baseline_estimate_recomputes(self): "cap-probe-32-direct-incompatible.json", "cap-probe-32-restored-incompatible.json", "live-ingestion-local-probe-timeout.json", + "grok-live-ingestion.json", + "grok-legacy-query.json", + "jev-credential-blocked.json", } self.assertTrue(required.issubset({path.name for path in RECEIPTS.glob("*.json")})) receipt = json.loads((RECEIPTS / "gemini38-baseline.json").read_text()) @@ -156,6 +159,8 @@ def test_every_receipt_reservation_is_reconciled_with_an_explicit_state(self): expected = { ("../receipts/caption-smoke.json", "reserved_upstream_list_usd"), ("../receipts/gemini38-baseline.json", "cost.worst_case_reserved_list_estimate_usd"), + ("../receipts/grok-live-ingestion.json", "reservation.reserved_usd"), + ("../receipts/grok-legacy-query.json", "reservation.reserved_usd"), } for key in expected: with self.subTest(receipt=key[0], field=key[1]): @@ -202,7 +207,7 @@ def test_incomplete_av_side_suppresses_baseline_ratio(self): def test_cap_rejects_estimates_plus_reservations_above_limit(self): data = json.loads((HERE / "scenario.receipts.json").read_text()) - data["cumulative_cap_usd"] = "0.3942578" + data["cumulative_cap_usd"] = "0.9006584" with self.assertRaisesRegex(ValueError, "exceed cumulative cap"): experiment_report(data) diff --git a/cookbook/receipts/README.md b/cookbook/receipts/README.md index 74248a8..18bc8a2 100644 --- a/cookbook/receipts/README.md +++ b/cookbook/receipts/README.md @@ -17,7 +17,11 @@ because those fields are needed to interpret the recorded result. | [cap-probe-32-direct-incompatible.json](cap-probe-32-direct-incompatible.json) | Direct exact-model 32-token cap probe returned usage but produced 309 completion tokens; output cap ignored | | [cap-probe-32-restored-incompatible.json](cap-probe-32-restored-incompatible.json) | Restored-route exact-model probe at commit `6a1cde2` produced 260 completion tokens against cap 32; ingestion/query stopped | | [live-ingestion-local-probe-timeout.json](live-ingestion-local-probe-timeout.json) | Fresh 300-frame attempt at commit `e24845d` stopped locally before provider request 1; transcript imported, 0 captions persisted, $0 provider estimate | +| [grok-live-ingestion.json](grok-live-ingestion.json) | Completed 300-frame Grok ingestion at `dd9dfa2`; six responses exceeded the requested 200-token advisory cap | +| [grok-legacy-query.json](grok-legacy-query.json) | Completed single Grok-only answer with transcript citation; no Jev relevance/support request | +| [jev-credential-blocked.json](jev-credential-blocked.json) | Jev arm stopped before request because `AV_TYPESAFE_API_KEY` was absent | -The receipts are evidence for those individual attempts only. No completed AV -Grok+Jev comparison exists yet. They do not establish speed, cost, or quality +The receipts are evidence for those individual attempts only. The completed query +was Grok-only through AV’s legacy route. No completed AV Grok+Jev comparison +exists. They do not establish speed, cost, or quality parity between the direct-video baseline and an AV pipeline. diff --git a/cookbook/receipts/grok-legacy-query.json b/cookbook/receipts/grok-legacy-query.json new file mode 100644 index 0000000..ae8c5d6 --- /dev/null +++ b/cookbook/receipts/grok-legacy-query.json @@ -0,0 +1,53 @@ +{ + "schema_version": 1, + "stage": "grok_legacy_query", + "status": "complete_grok_only", + "runtime_commit": "dd9dfa2f3dd1d801589569b2872f21c2fa268d8c", + "question": "What percentage did manufacturing expand by in the second quarter and what drove it?", + "answer": "Manufacturing expanded by 12.2% in the second quarter, driven largely by AI-related demand for electronics and precision engineering.", + "citations": [ + { + "start_sec": 1140.0, + "end_sec": 1200.0, + "source_type": "transcript" + } + ], + "route": "legacy", + "evidence_status": "raw_unjudged", + "jev_requests": 0, + "wall_seconds": 2.0740865, + "request_attempts": 1, + "successful_provider_responses": 1, + "model_requested": "grok-4.20-0309-non-reasoning", + "model_returned": "grok-4.20-0309-non-reasoning", + "settings": { + "chat_model": "grok-4.20-0309-non-reasoning", + "chat_max_output_tokens": 512, + "embeddings": false, + "stronger_inspection": false, + "automatic_retries": 0 + }, + "usage": { + "prompt_tokens": 1256, + "cached_prompt_tokens": 128, + "completion_tokens": 45, + "reasoning_tokens": 0 + }, + "cache_aware_list_estimate_usd": "0.0015481", + "usage_complete": true, + "proxy_billed_usd": null, + "basis": "estimated from measured usage and published upstream list rates", + "reservation": { + "name": "query_judge_and_answer", + "reserved_usd": "0.20", + "status": "released_after_metering" + }, + "limitations": { + "jev_relevance_or_support_check_completed": false, + "jev_absence_reason": "No configured AV_TYPESAFE_API_KEY credential was available; Grok was not substituted for Jev.", + "single_question_no_aggregate_quality_claim": true, + "query_latency_is_not_a_complete_av_vs_gemini_comparison": true, + "proxy_billed_usd_unknown": true, + "infrastructure_storage_network_unpriced": true + } +} diff --git a/cookbook/receipts/grok-live-ingestion.json b/cookbook/receipts/grok-live-ingestion.json new file mode 100644 index 0000000..6bfa347 --- /dev/null +++ b/cookbook/receipts/grok-live-ingestion.json @@ -0,0 +1,59 @@ +{ + "schema_version": 1, + "stage": "grok_dense_ingestion", + "status": "complete", + "runtime_commit": "dd9dfa2f3dd1d801589569b2872f21c2fa268d8c", + "model_requested": "grok-4.20-0309-non-reasoning", + "model_returned": [ + "grok-4.20-0309-non-reasoning" + ], + "settings": { + "sample_frames": 300, + "fps_sample": 0.06666666666666667, + "frame_dimensions": [ + 1280, + 720 + ], + "jpeg_quality": 2, + "embeddings": false, + "transcript_windows": 75, + "automatic_retries": 0, + "hidden_fallbacks": false, + "requested_caption_cap": 200, + "completion_cap_advisory": true, + "preextracted_exact_frames": true + }, + "wall_seconds": 902.9050934410043, + "request_attempts": 300, + "successful_provider_responses": 300, + "usage": { + "prompt_tokens": 375300, + "completion_tokens": 38483, + "cached_prompt_tokens": 57600, + "reasoning_tokens": 0 + }, + "completion_tokens": { + "minimum": 45, + "maximum": 235, + "over_requested_cap": 6 + }, + "cache_aware_list_estimate_usd": "0.5048525", + "artifacts_persisted": { + "transcript": 75, + "dense_caption": 300 + }, + "usage_complete": true, + "proxy_billed_usd": null, + "basis": "estimated from measured usage and published upstream list rates", + "reservation": { + "name": "grok_dense_ingestion", + "reserved_usd": "0.90", + "status": "released_after_metering" + }, + "limitations": { + "six_responses_exceeded_requested_200_token_advisory_cap": true, + "maximum_completion_tokens": 235, + "proxy_billed_usd_unknown": true, + "infrastructure_storage_network_unpriced": true + } +} diff --git a/cookbook/receipts/jev-credential-blocked.json b/cookbook/receipts/jev-credential-blocked.json new file mode 100644 index 0000000..2151942 --- /dev/null +++ b/cookbook/receipts/jev-credential-blocked.json @@ -0,0 +1,17 @@ +{ + "schema_version": 1, + "stage": "jev_comparison_arm", + "status": "blocked_before_request", + "runtime_commit": "dd9dfa2f3dd1d801589569b2872f21c2fa268d8c", + "request_attempts": 0, + "successful_provider_responses": 0, + "usage": null, + "list_estimate_usd": 0, + "blocker": "missing AV_TYPESAFE_API_KEY credential", + "resolution": "Stopped with no Jev request; do not substitute Grok for Jev or claim a completed comparison.", + "preserved_completed_work": [ + "grok_dense_ingestion", + "grok_legacy_query" + ], + "future_requirement": "Meter ordinary and list-token usage fields if exposed, retain raw usage, use the recorded $0.042/M input list rate, and fail closed on missing usage." +} From dad9dcf5b53baf080f1242b4159dc405ab06fbcd Mon Sep 17 00:00:00 2001 From: Sean Phan Date: Sat, 19 Sep 2026 12:36:05 +0000 Subject: [PATCH 11/12] docs: record completed Jev-refined query --- cookbook/README.md | 17 ++-- cookbook/cost-model/README.md | 32 ++++---- cookbook/cost-model/build_nb.py | 6 +- cookbook/cost-model/notebook.ipynb | 50 ++++++++---- cookbook/cost-model/scenario.receipts.json | 51 +++++++++--- cookbook/cost-model/test_model.py | 25 +++--- cookbook/jev-refined-ask/README.md | 13 +-- cookbook/receipts/README.md | 6 +- cookbook/receipts/gemini38-baseline.json | 8 +- cookbook/receipts/jev-refined-query.json | 92 ++++++++++++++++++++++ 10 files changed, 222 insertions(+), 78 deletions(-) create mode 100644 cookbook/receipts/jev-refined-query.json diff --git a/cookbook/README.md b/cookbook/README.md index 15b66b6..b48de07 100644 --- a/cookbook/README.md +++ b/cookbook/README.md @@ -4,19 +4,18 @@ Runnable recipes for the open-source **av** CLI live here alongside the code. | Recipe | What it demonstrates | Evidence status | |---|---|---| -| [Cost model](cost-model/README.md) | Separate tokens, estimates, unknown costs, ingestion, queries, failures, reservations, and cap accounting | Completed component receipts; paired AV comparison incomplete | +| [Cost model](cost-model/README.md) | Separate tokens, estimates, unknown costs, ingestion, queries, failures, reservations, and cap accounting | Completed one-question comparison; no aggregate parity claim | | [Jev-refined ask](jev-refined-ask/README.md) | Build a source-bound transcript sidecar, retrieve indexed moments, refine evidence, answer, and check support | Runnable recipe; no speed, cost, or quality parity claim | -| [Sanitized receipts](receipts/README.md) | Completed ASR/baseline, caption smoke/abort, incompatible cap probes, and the fresh zero-request local media-probe failure | No media, transcript/caption corpus, credentials, upload URIs, or private routes; baseline question and returned answer/rationale retained | +| [Sanitized receipts](receipts/README.md) | Completed ASR/baseline, caption smoke/abort, incompatible cap probes, Grok ingestion, and Jev-refined query | No media, transcript/caption corpus, credentials, upload URIs, or private routes; baseline question and returned answer/rationale retained | The [original public cost notebook](https://github.com/PixelML/cookbook/tree/main/agentic-video/cost-model) remains available at its existing URL. Its Composer demonstrations are historical context, not measurements of this CLI. New AV reproduction results belong in this cookbook with their own media, model, configuration, and cost provenance. -The current evidence includes a completed Gemini 3.8 direct-video baseline and a -completed AV Grok-only path: 300/300 caption requests, 75 transcript windows, and -one correct answer with transcript citation. The selected route ignored output -caps in both 32-token probes, and six of the 300 caption responses exceeded the -requested 200-token advisory cap. No Jev request ran because the required -credential was absent, so there is **no completed AV Grok+Jev comparison**. Do -not infer all-in speed, cost, or quality parity from component receipts. +The current evidence includes a completed Gemini 3.8 direct-video baseline, a +completed 300/300 Grok caption ingestion with 75 transcript windows, one Grok-only +answer, and one later Jev-refined answer for the same question. The selected route +ignored output caps in both 32-token probes, and six of the 300 caption responses +exceeded the requested 200-token advisory cap. The paired result is one question; +do not infer aggregate speed, cost, or quality parity from it. diff --git a/cookbook/cost-model/README.md b/cookbook/cost-model/README.md index 464bbf6..d3d1f65 100644 --- a/cookbook/cost-model/README.md +++ b/cookbook/cost-model/README.md @@ -15,7 +15,7 @@ Its checked-in source and outputs are generated by `build_nb.py`. [`scenario.receipts.json`](scenario.receipts.json) integrates the sanitized receipts in [`cookbook/receipts`](../receipts/README.md). -The evidence is a partial reproduction: ingestion and one Grok-only answer completed, but the Jev arm made zero requests. +The evidence now includes one completed Jev-refined query in addition to the earlier Grok-only attempt. | Attempt | Outcome | Recorded list-rate estimate | |---|---|---:| @@ -32,31 +32,27 @@ The evidence is a partial reproduction: ingestion and one Grok-only answer compl | Completed Grok-only legacy query at `dd9dfa2` | correct answer with transcript citation `1140–1200s`; evidence remained raw/unjudged | $0.0015481 | | Jev comparison arm | blocked before request because no configured `AV_TYPESAFE_API_KEY` credential was available | $0 | -Known token-derived list-rate estimates total **$0.9006585**. They are not billed +Known token-derived list-rate estimates total **$1.04309935**. They are not billed dollars. Provider/account or proxy billing remains unknown. Compute, storage, and network allocation also remains unknown. The cumulative experiment cap is **$5**. The reservation ledger records -**$5.99945825** of historical request ceilings. The two **$0.01** probe +**$6.19945825** of historical request ceilings. The two **$0.01** probe reservations were released after their later metered responses. The **$0.90** caption-ingestion and **$0.20** query/judge/answer reservations were released unused when the restored-route cap gate failed. A later **$0.90** ingestion reservation was released after all 300 caption responses reported usage. A -**$0.20** metered-query reservation was released after the Grok answer reported -usage; the Jev arm made zero requests. No reservation remains retained. -Known estimates plus retained reservations are therefore **$0.9006585**, leaving -**$4.0993415** of estimate headroom. - -**No completed AV Grok+Jev comparison exists.** The completed run used 300 sampled -frames plus a coarse transcript sidecar and answered through the legacy route; no -Jev relevance or support request ran because the required credential was absent. -The recorded 2.0741-second answer latency is only that single Grok answer request, -not an all-in AV-versus-Gemini latency comparison, and no aggregate quality-parity -or savings claim is valid from one question. The Gemini baseline used native -video/audio at its recorded sampling configuration. Six caption responses also -exceeded the requested 200-token advisory cap; output caps are therefore not an -enforced guarantee on this route. Provider billing and compute, storage, and -network costs remain unknown. +later **$0.20** reservation was released after the Jev-refined query reported complete usage. No reservation remains retained. Known estimates plus retained reservations are therefore **$1.04309935**, leaving **$3.95690065** of estimate headroom. + +One completed Jev-refined query exists. It used 300 sampled frames plus a coarse +transcript sidecar, made two Jev requests (relevance and support) and one Grok +answer request, and returned supported `1140–1200s` evidence. Its 3.7385-second +query wall time is a single question, not an all-in AV-versus-Gemini latency +comparison, and no aggregate quality-parity or savings claim is valid from it. +The Gemini baseline used native video/audio at its recorded sampling configuration. +Six caption responses also exceeded the requested 200-token advisory cap; output +caps are therefore not an enforced guarantee on this route. Provider billing and +compute, storage, and network costs remain unknown. Use [`scenario.pending.json`](scenario.pending.json) as a blank template for another run. diff --git a/cookbook/cost-model/build_nb.py b/cookbook/cost-model/build_nb.py index d1b41d2..c5ec610 100644 --- a/cookbook/cost-model/build_nb.py +++ b/cookbook/cost-model/build_nb.py @@ -16,8 +16,8 @@ ("markdown", """# AV stage-cost model (offline) This notebook calculates with the same model.py used by the command line. -The checked-in receipt scenario includes completed ASR and direct-video baseline -attempts, but no completed AV Grok+Jev comparison. Measured tokens, list-rate +The checked-in receipt scenario includes completed ASR, direct-video baseline, +ingestion, and one Jev-refined query attempt. Measured tokens, list-rate estimates, unknown costs, failures, and reservation states remain separate. Only retained reservations count against current cap headroom; released and superseded ceilings remain visible as history. See README.md for provenance and limitations. @@ -53,7 +53,7 @@ inspection are multiplied by query volume. Hosting and storage refer to the same observation period. A complete modeled total can contain assumptions; it is not necessarily a measured bill. This receipt scenario remains incomplete -because no completed AV Grok+Jev query exists. Repeated-query projections are +because it remains a single-question experiment. Repeated-query projections are not new benchmark measurements. """), ("code", """for queries in (1, 100, 1000): diff --git a/cookbook/cost-model/notebook.ipynb b/cookbook/cost-model/notebook.ipynb index 213cf84..44f97a8 100644 --- a/cookbook/cost-model/notebook.ipynb +++ b/cookbook/cost-model/notebook.ipynb @@ -8,8 +8,8 @@ "# AV stage-cost model (offline)\n", "\n", "This notebook calculates with the same model.py used by the command line.\n", - "The checked-in receipt scenario includes completed ASR and direct-video baseline\n", - "attempts, but no completed AV Grok+Jev comparison. Measured tokens, list-rate\n", + "The checked-in receipt scenario includes completed ASR, direct-video baseline,\n", + "ingestion, and one Jev-refined query attempt. Measured tokens, list-rate\n", "estimates, unknown costs, failures, and reservation states remain separate. Only\n", "retained reservations count against current cap headroom; released and superseded\n", "ceilings remain visible as history. See README.md for provenance and limitations.\n", @@ -52,10 +52,10 @@ "output_type": "stream", "text": [ "{\n", - " \"status\": \"partial_reproduction_grok_only\",\n", + " \"status\": \"completed_grok_jev_single_question\",\n", " \"implementation\": \"av\",\n", " \"receipt_directory\": \"../receipts\",\n", - " \"note\": \"Completed external ASR, 300-frame Grok ingestion, and one Grok-only legacy query receipts exist. No Jev request ran because no configured AV_TYPESAFE_API_KEY credential was available, so no completed AV Grok+Jev comparison or quality-parity claim exists.\",\n", + " \"note\": \"Completed external ASR, 300-frame Grok ingestion, one Grok-only legacy query, and one Jev-refined Grok answer receipt exist. This is a single-question comparison, not an aggregate quality claim.\",\n", " \"projection\": \"Repeated-query totals are projections from entered stage costs, not additional measured runs.\"\n", "}\n", "{\n", @@ -164,7 +164,7 @@ " \"requests\": 1,\n", " \"outcome\": \"completed_grok_only\",\n", " \"receipt\": \"../receipts/grok-legacy-query.json\",\n", - " \"note\": \"Correct answer with transcript citation; evidence remained raw and unjudged because no Jev request ran.\"\n", + " \"note\": \"Correct answer with transcript citation through the legacy route; retained separately from the later Jev-refined run.\"\n", " },\n", " {\n", " \"name\": \"jev_credential_blocker\",\n", @@ -176,9 +176,20 @@ " \"outcome\": \"blocked_before_request\",\n", " \"receipt\": \"../receipts/jev-credential-blocked.json\",\n", " \"note\": \"No configured AV_TYPESAFE_API_KEY credential was available.\"\n", + " },\n", + " {\n", + " \"name\": \"jev_refined_query\",\n", + " \"input_tokens\": 4533,\n", + " \"output_tokens\": 236,\n", + " \"thinking_tokens\": 0,\n", + " \"cached_tokens\": 128,\n", + " \"requests\": 3,\n", + " \"outcome\": \"completed\",\n", + " \"receipt\": \"../receipts/jev-refined-query.json\",\n", + " \"note\": \"Two Jev requests (relevance and support) plus one Grok answer request; complete usage was returned for each stage.\"\n", " }\n", " ],\n", - " \"known_list_estimates_usd\": \"0.9006585\",\n", + " \"known_list_estimates_usd\": \"1.04309935\",\n", " \"unknown_costs\": {\n", " \"terms\": [\n", " {\n", @@ -321,19 +332,28 @@ " \"note\": \"Released after the Grok answer returned complete usage; the Jev arm remained blocked with zero requests and no balance retained.\",\n", " \"receipt\": \"../receipts/grok-legacy-query.json\",\n", " \"receipt_field\": \"reservation.reserved_usd\"\n", + " },\n", + " {\n", + " \"name\": \"reservation.9.jev_refined_query_judge_and_answer\",\n", + " \"usd\": \"0.20\",\n", + " \"status\": \"released_after_metering\",\n", + " \"counts_against_cap\": false,\n", + " \"note\": \"Released after two Jev requests and one Grok answer returned complete usage.\",\n", + " \"receipt\": \"../receipts/jev-refined-query.json\",\n", + " \"receipt_field\": \"reservation.reserved_usd\"\n", " }\n", " ],\n", " \"status_totals_usd\": {\n", - " \"released_after_metering\": \"3.99945825\",\n", + " \"released_after_metering\": \"4.19945825\",\n", " \"retained\": \"0\",\n", " \"superseded\": \"2.00\"\n", " },\n", - " \"total_recorded_usd\": \"5.99945825\",\n", + " \"total_recorded_usd\": \"6.19945825\",\n", " \"total_retained_usd\": \"0\"\n", " },\n", " \"cumulative_cap_usd\": \"5\",\n", - " \"known_estimates_plus_retained_usd\": \"0.9006585\",\n", - " \"remaining_cap_usd\": \"4.0993415\"\n", + " \"known_estimates_plus_retained_usd\": \"1.04309935\",\n", + " \"remaining_cap_usd\": \"3.95690065\"\n", "}\n" ] } @@ -350,7 +370,7 @@ "inspection are multiplied by query volume. Hosting and storage refer to the\n", "same observation period. A complete modeled total can contain assumptions; it\n", "is not necessarily a measured bill. This receipt scenario remains incomplete\n", - "because no completed AV Grok+Jev query exists. Repeated-query projections are\n", + "because it remains a single-question experiment. Repeated-query projections are\n", "not new benchmark measurements.\n" ] }, @@ -377,7 +397,7 @@ "text": [ "{\n", " \"queries\": 1,\n", - " \"known_subtotal_usd\": \"0.5749715\",\n", + " \"known_subtotal_usd\": \"0.71586425\",\n", " \"complete_total_usd\": null,\n", " \"unknown_terms\": [\n", " \"indexing.3.preprocessing\",\n", @@ -388,7 +408,7 @@ "}\n", "{\n", " \"queries\": 100,\n", - " \"known_subtotal_usd\": \"0.7282334\",\n", + " \"known_subtotal_usd\": \"14.81750840\",\n", " \"complete_total_usd\": null,\n", " \"unknown_terms\": [\n", " \"indexing.3.preprocessing\",\n", @@ -399,7 +419,7 @@ "}\n", "{\n", " \"queries\": 1000,\n", - " \"known_subtotal_usd\": \"2.1215234\",\n", + " \"known_subtotal_usd\": \"143.01427340\",\n", " \"complete_total_usd\": null,\n", " \"unknown_terms\": [\n", " \"indexing.3.preprocessing\",\n", @@ -486,7 +506,7 @@ " \"per_invocation\": {\n", " \"usd\": 0,\n", " \"basis\": \"not_used\",\n", - " \"note\": \"No completed stronger-inspection query exists.\"\n", + " \"note\": \"No stronger-inspection request ran.\"\n", " }\n", " }\n", "}\n" diff --git a/cookbook/cost-model/scenario.receipts.json b/cookbook/cost-model/scenario.receipts.json index d5db6f5..c42dffe 100644 --- a/cookbook/cost-model/scenario.receipts.json +++ b/cookbook/cost-model/scenario.receipts.json @@ -1,10 +1,10 @@ { "schema_version": 1, "experiment": { - "status": "partial_reproduction_grok_only", + "status": "completed_grok_jev_single_question", "implementation": "av", "receipt_directory": "../receipts", - "note": "Completed external ASR, 300-frame Grok ingestion, and one Grok-only legacy query receipts exist. No Jev request ran because no configured AV_TYPESAFE_API_KEY credential was available, so no completed AV Grok+Jev comparison or quality-parity claim exists.", + "note": "Completed external ASR, 300-frame Grok ingestion, one Grok-only legacy query, and one Jev-refined Grok answer receipt exist. This is a single-question comparison, not an aggregate quality claim.", "projection": "Repeated-query totals are projections from entered stage costs, not additional measured runs." }, "indexing": [ @@ -37,24 +37,24 @@ "retrieval": { "usd": 0, "basis": "not_used", - "note": "The recorded Grok-only query used local FTS retrieval with embeddings disabled; provider model cost was zero and local compute remains unpriced." + "note": "The completed Jev-refined query used local FTS retrieval with embeddings disabled; local compute remains unpriced." }, "judge": { - "usd": 0, - "basis": "not_used", - "note": "No Jev request ran because the required credential was absent; zero is not a measured Jev cost." + "usd": "0.140784", + "basis": "estimated", + "note": "Two completed Jev requests: relevance and support. Measured input/output token usage multiplied by recorded list rates; billed dollars are unknown." }, "answer": { - "usd": "0.0015481", + "usd": "0.00165685", "basis": "estimated", - "note": "One Grok-only answer request with complete measured usage; billed dollars are unknown." + "note": "One Grok answer request with complete measured usage, including 128 cached prompt tokens; billed dollars are unknown." }, "stronger_fallback": { "frequency": 0, "per_invocation": { "usd": 0, "basis": "not_used", - "note": "No completed stronger-inspection query exists." + "note": "No stronger-inspection request ran." } } }, @@ -202,7 +202,7 @@ "cached_tokens": 128, "outcome": "completed_grok_only", "receipt": "../receipts/grok-legacy-query.json", - "note": "Correct answer with transcript citation; evidence remained raw and unjudged because no Jev request ran." + "note": "Correct answer with transcript citation through the legacy route; retained separately from the later Jev-refined run." }, { "name": "jev_credential_blocker", @@ -214,6 +214,17 @@ "outcome": "blocked_before_request", "receipt": "../receipts/jev-credential-blocked.json", "note": "No configured AV_TYPESAFE_API_KEY credential was available." + }, + { + "name": "jev_refined_query", + "requests": 3, + "input_tokens": 4533, + "output_tokens": 236, + "thinking_tokens": 0, + "cached_tokens": 128, + "outcome": "completed", + "receipt": "../receipts/jev-refined-query.json", + "note": "Two Jev requests (relevance and support) plus one Grok answer request; complete usage was returned for each stage." } ], "experiment_spend": [ @@ -314,7 +325,7 @@ "outcome": "completed_grok_only", "receipt": "../receipts/grok-legacy-query.json", "usage_ref": "grok_legacy_query", - "note": "Cache-aware upstream list-rate estimate for the answer request; no Jev request ran." + "note": "Cache-aware upstream list-rate estimate for the earlier Grok-only answer request; retained as a separate experiment attempt." }, { "name": "jev_credential_blocker", @@ -324,6 +335,16 @@ "outcome": "blocked_before_request", "receipt": "../receipts/jev-credential-blocked.json", "note": "No provider request was attempted because the required credential was absent." + }, + { + "name": "jev_refined_query", + "usd": "0.14244085", + "basis": "estimated", + "category": "av_grok_jev_query", + "outcome": "completed", + "receipt": "../receipts/jev-refined-query.json", + "usage_ref": "jev_refined_query", + "note": "Cache-aware upstream list-rate estimate for two Jev requests and one Grok answer request; no retry or fallback occurred." } ], "unknown_costs": [ @@ -407,6 +428,14 @@ "receipt": "../receipts/grok-legacy-query.json", "receipt_field": "reservation.reserved_usd", "note": "Released after the Grok answer returned complete usage; the Jev arm remained blocked with zero requests and no balance retained." + }, + { + "name": "jev_refined_query_judge_and_answer", + "usd": "0.20", + "status": "released_after_metering", + "receipt": "../receipts/jev-refined-query.json", + "receipt_field": "reservation.reserved_usd", + "note": "Released after two Jev requests and one Grok answer returned complete usage." } ], "cumulative_cap_usd": "5" diff --git a/cookbook/cost-model/test_model.py b/cookbook/cost-model/test_model.py index 54e1831..a7ced5a 100644 --- a/cookbook/cost-model/test_model.py +++ b/cookbook/cost-model/test_model.py @@ -100,24 +100,25 @@ def test_receipt_ledger_preserves_cap_and_separates_unknowns(self): data = json.loads((HERE / "scenario.receipts.json").read_text()) report = experiment_report(data) self.assertEqual(report["list_rate_estimates"]["known_subtotal_usd"], - Decimal("0.9006585")) - self.assertEqual(report["reservations"]["total_recorded_usd"], Decimal("5.99945825")) + Decimal("1.04309935")) + self.assertEqual(report["reservations"]["total_recorded_usd"], Decimal("6.19945825")) self.assertEqual(report["reservations"]["total_retained_usd"], Decimal("0")) self.assertEqual( report["reservations"]["status_totals_usd"]["released_after_metering"], - Decimal("3.99945825"), + Decimal("4.19945825"), ) self.assertEqual(report["reservations"]["status_totals_usd"]["superseded"], Decimal("2.00")) - self.assertEqual(report["known_estimates_plus_retained_usd"], Decimal("0.9006585")) - self.assertEqual(report["remaining_cap_usd"], Decimal("4.0993415")) + self.assertEqual(report["known_estimates_plus_retained_usd"], Decimal("1.04309935")) + self.assertEqual(report["remaining_cap_usd"], Decimal("3.95690065")) self.assertIsNone(report["unknown_costs"]["complete_total_usd"]) self.assertEqual( [item["outcome"] for item in report["failures"]], ["failed", "aborted", "incompatible", "incompatible"], ) self.assertEqual(report["measured_usage"][0]["input_tokens"], 118228) - self.assertEqual(report["measured_usage"][-1]["outcome"], "blocked_before_request") - self.assertEqual(report["measured_usage"][-1]["requests"], 0) + self.assertEqual(report["measured_usage"][-1]["name"], "jev_refined_query") + self.assertEqual(report["measured_usage"][-1]["outcome"], "completed") + self.assertEqual(report["measured_usage"][-1]["requests"], 3) def test_sanitized_receipts_are_present_and_baseline_estimate_recomputes(self): required = { @@ -132,6 +133,7 @@ def test_sanitized_receipts_are_present_and_baseline_estimate_recomputes(self): "grok-live-ingestion.json", "grok-legacy-query.json", "jev-credential-blocked.json", + "jev-refined-query.json", } self.assertTrue(required.issubset({path.name for path in RECEIPTS.glob("*.json")})) receipt = json.loads((RECEIPTS / "gemini38-baseline.json").read_text()) @@ -147,7 +149,11 @@ def test_sanitized_receipts_are_present_and_baseline_estimate_recomputes(self): Decimal(str(receipt["cost"]["full_rate_list_estimate_usd"])), recomputed, ) - self.assertFalse(receipt["limitations"]["paired_av_grok_jev_run_completed"]) + self.assertTrue(receipt["limitations"]["paired_av_grok_jev_run_completed"]) + self.assertEqual( + receipt["limitations"]["paired_av_grok_jev_run_receipt"], + "../receipts/jev-refined-query.json", + ) def test_every_receipt_reservation_is_reconciled_with_an_explicit_state(self): data = json.loads((HERE / "scenario.receipts.json").read_text()) @@ -161,6 +167,7 @@ def test_every_receipt_reservation_is_reconciled_with_an_explicit_state(self): ("../receipts/gemini38-baseline.json", "cost.worst_case_reserved_list_estimate_usd"), ("../receipts/grok-live-ingestion.json", "reservation.reserved_usd"), ("../receipts/grok-legacy-query.json", "reservation.reserved_usd"), + ("../receipts/jev-refined-query.json", "reservation.reserved_usd"), } for key in expected: with self.subTest(receipt=key[0], field=key[1]): @@ -207,7 +214,7 @@ def test_incomplete_av_side_suppresses_baseline_ratio(self): def test_cap_rejects_estimates_plus_reservations_above_limit(self): data = json.loads((HERE / "scenario.receipts.json").read_text()) - data["cumulative_cap_usd"] = "0.9006584" + data["cumulative_cap_usd"] = "1.04309934" with self.assertRaisesRegex(ValueError, "exceed cumulative cap"): experiment_report(data) diff --git a/cookbook/jev-refined-ask/README.md b/cookbook/jev-refined-ask/README.md index ae23c34..9177406 100644 --- a/cookbook/jev-refined-ask/README.md +++ b/cookbook/jev-refined-ask/README.md @@ -193,9 +193,10 @@ receipt. No benchmark in this recipe establishes quality improvement, parity with direct-video models, or a universal cost ratio. The current receipts include completed ASR and Gemini 3.8 baseline calls, an -aborted caption attempt, a successful image smoke call, one no-route probe, and -two exact-model probes that returned 309 and 260 completion tokens against a -requested cap of 32. The latest probe ran at AV commit `6a1cde2`, with automatic -retries and hidden fallbacks disabled. **No paid ingestion or query followed, -and no completed AV Grok+Jev comparison exists yet.** Do not claim speed, cost, -or quality parity from these component attempts. +aborted caption attempt, a successful image smoke call, two exact-model probes +that ignored a requested 32-token cap, a completed 300-frame Grok ingestion, and +a completed Jev-refined query. The refined query made two Jev requests +(relevance and support) plus one Grok answer request, used the same transcript +support window, and reported complete usage for each stage. It is still one +question: do not claim aggregate quality, total savings, or an all-in speed or +cost advantage from it. diff --git a/cookbook/receipts/README.md b/cookbook/receipts/README.md index 18bc8a2..1fdfbf6 100644 --- a/cookbook/receipts/README.md +++ b/cookbook/receipts/README.md @@ -20,8 +20,6 @@ because those fields are needed to interpret the recorded result. | [grok-live-ingestion.json](grok-live-ingestion.json) | Completed 300-frame Grok ingestion at `dd9dfa2`; six responses exceeded the requested 200-token advisory cap | | [grok-legacy-query.json](grok-legacy-query.json) | Completed single Grok-only answer with transcript citation; no Jev relevance/support request | | [jev-credential-blocked.json](jev-credential-blocked.json) | Jev arm stopped before request because `AV_TYPESAFE_API_KEY` was absent | +| [jev-refined-query.json](jev-refined-query.json) | Completed Jev-refined Grok answer at `59a7f73`: 2 Jev requests + 1 Grok answer, no retries/fallbacks, supported evidence | -The receipts are evidence for those individual attempts only. The completed query -was Grok-only through AV’s legacy route. No completed AV Grok+Jev comparison -exists. They do not establish speed, cost, or quality -parity between the direct-video baseline and an AV pipeline. +The receipts are evidence for those individual attempts only. The later Jev-refined query completed the intended AV Grok+Jev path for one question. These receipts do not establish aggregate speed, cost, or quality parity between the direct-video baseline and an AV pipeline. diff --git a/cookbook/receipts/gemini38-baseline.json b/cookbook/receipts/gemini38-baseline.json index bd3931d..5522923 100644 --- a/cookbook/receipts/gemini38-baseline.json +++ b/cookbook/receipts/gemini38-baseline.json @@ -12,7 +12,7 @@ "model_requested": "gemini-3.8-flash", "model_returned": "gemini-3.8-flash", "question": "What percentage did manufacturing expand by in the second quarter and what drove it?", - "answer": "Manufacturing expanded by 12.2% in the second quarter, driven largely by AI-related demand for electronics and precision engineering (19:51–20:00).", + "answer": "Manufacturing expanded by 12.2% in the second quarter, driven largely by AI-related demand for electronics and precision engineering (19:51\u201320:00).", "sampling": { "fps": 1.0, "media_resolution": "low", @@ -49,9 +49,11 @@ }, "limitations": { "answer_quality_reviewed": false, - "paired_av_grok_jev_run_completed": false, + "paired_av_grok_jev_run_completed": true, "speed_cost_or_quality_parity_claimed": false, - "missing_usage_policy": "Absent fields remain unknown; no absent field is treated as measured zero." + "missing_usage_policy": "Absent fields remain unknown; no absent field is treated as measured zero.", + "paired_av_grok_jev_run_receipt": "../receipts/jev-refined-query.json", + "paired_av_grok_jev_limit": "One unchanged question only; no aggregate quality, speed, or all-in cost claim." }, "private_upload_uri_included": false, "credentials_included": false diff --git a/cookbook/receipts/jev-refined-query.json b/cookbook/receipts/jev-refined-query.json new file mode 100644 index 0000000..b67a263 --- /dev/null +++ b/cookbook/receipts/jev-refined-query.json @@ -0,0 +1,92 @@ +{ + "schema_version": 1, + "stage": "jev_refined_query", + "status": "complete", + "runtime_commit": "59a7f7317329e4bba6e1b43cdf289a233e865a75", + "question": "What percentage did manufacturing expand by in the second quarter and what drove it?", + "answer": "Manufacturing expanded by 12.2% in the second quarter. It was driven largely by AI-related demand for electronics and precision engineering.", + "citations": [ + { + "start_sec": 1140.0, + "end_sec": 1200.0, + "source_type": "transcript" + } + ], + "route": "refined", + "evidence_status": "supported", + "confidence": 0.95, + "confidence_basis": "jev_answer_support", + "refinement": { + "status": "success", + "raw_count": 5, + "dropped_count": 4, + "merged_count": 0, + "capped_count": 0, + "scene_count": 1, + "video_count": 1, + "relevance_min": 0.5 + }, + "wall_seconds": 3.7385285390446192, + "request_counts": { + "provider_requests": 1, + "jev_requests": 2, + "automatic_retries": 0, + "fallbacks": 0 + }, + "models": { + "relevance_and_support": "jev-latest", + "answer": "grok-4.20-0309-non-reasoning" + }, + "model_returned": { + "answer": "grok-4.20-0309-non-reasoning" + }, + "stage_usage": { + "relevance": { + "requests": 1, + "input_tokens": 1888, + "output_tokens": 89, + "estimated_usd": "0.079296" + }, + "boundary": { + "requests": 0, + "input_tokens": null, + "output_tokens": null + }, + "answer": { + "requests": 1, + "input_tokens": 1181, + "cached_input_tokens": 128, + "output_tokens": 126, + "estimated_usd": "0.00165685" + }, + "support": { + "requests": 1, + "input_tokens": 1464, + "output_tokens": 21, + "estimated_usd": "0.061488" + }, + "vision": { + "requests": 0, + "input_tokens": null, + "output_tokens": null + } + }, + "cache_aware_list_estimate_usd": "0.14244085", + "usage_complete": true, + "known_cumulative_list_estimate_usd": "1.04309935", + "cumulative_cap_usd": "5", + "proxy_billed_usd": null, + "basis": "estimated from measured usage and recorded upstream list rates", + "reservation": { + "name": "query_judge_and_answer_jev", + "reserved_usd": "0.20", + "status": "released_after_metering" + }, + "limitations": { + "single_question_no_aggregate_quality_claim": true, + "query_latency_is_not_a_complete_av_vs_gemini_comparison": true, + "jev_returned_model_identifier_unavailable": true, + "proxy_billed_usd_unknown": true, + "infrastructure_storage_network_unpriced": true + } +} From bcf6c12439bea65f2eaea10f5449c644e8f4e765 Mon Sep 17 00:00:00 2001 From: Sean Phan Date: Sat, 19 Sep 2026 13:49:03 +0000 Subject: [PATCH 12/12] fix: correct Jev price units and cost receipt provenance --- cookbook/cost-model/README.md | 29 ++++++++++-- cookbook/cost-model/model.py | 8 ++++ cookbook/cost-model/notebook.ipynb | 31 +++++++------ cookbook/cost-model/scenario.receipts.json | 14 +++--- cookbook/cost-model/test_model.py | 53 ++++++++++++++++++++-- cookbook/receipts/README.md | 5 ++ cookbook/receipts/gemini38-baseline.json | 8 ++-- cookbook/receipts/grok-live-ingestion.json | 3 +- cookbook/receipts/jev-refined-query.json | 39 ++++++++++++++-- 9 files changed, 153 insertions(+), 37 deletions(-) diff --git a/cookbook/cost-model/README.md b/cookbook/cost-model/README.md index d3d1f65..2f80d03 100644 --- a/cookbook/cost-model/README.md +++ b/cookbook/cost-model/README.md @@ -30,9 +30,10 @@ The evidence now includes one completed Jev-refined query in addition to the ear | Fresh 300-frame ingestion attempt at `e24845d` | local media probe timed out before frame extraction; 75 transcript artifacts, 0 captions, 0 provider requests | $0 | | Completed 300-frame Grok ingestion at `dd9dfa2` | 300/300 requests succeeded; 300 captions and 75 transcript windows persisted; six responses exceeded the requested 200-token advisory cap (maximum 235) | $0.5048525 | | Completed Grok-only legacy query at `dd9dfa2` | correct answer with transcript citation `1140–1200s`; evidence remained raw/unjudged | $0.0015481 | -| Jev comparison arm | blocked before request because no configured `AV_TYPESAFE_API_KEY` credential was available | $0 | +| Earlier Jev comparison attempt | blocked before request because no configured `AV_TYPESAFE_API_KEY` credential was available | $0 | +| Completed Jev-refined Grok query | two Jev requests plus one Grok answer; corrected Jev units | $0.001797634 | -Known token-derived list-rate estimates total **$1.04309935**. They are not billed +Known token-derived list-rate estimates total **$0.902456134**. They are not billed dollars. Provider/account or proxy billing remains unknown. Compute, storage, and network allocation also remains unknown. @@ -42,7 +43,25 @@ reservations were released after their later metered responses. The **$0.90** caption-ingestion and **$0.20** query/judge/answer reservations were released unused when the restored-route cap gate failed. A later **$0.90** ingestion reservation was released after all 300 caption responses reported usage. A -later **$0.20** reservation was released after the Jev-refined query reported complete usage. No reservation remains retained. Known estimates plus retained reservations are therefore **$1.04309935**, leaving **$3.95690065** of estimate headroom. +later **$0.20** reservation was released after the Jev-refined query reported complete usage. No reservation remains retained. Known estimates plus retained reservations are therefore **$0.902456134**, leaving **$4.097543866** of estimate headroom. + +### Pricing correction — 2026-09-19 + +The original observer treated TypeSafe's **$42 per billion input tokens** as +$42 per million. The [official model documentation](https://docs.typesafe.ai/models.md) +lists **$0.042 per million input tokens and free output**. The corrected Jev +estimate is `(1,888 + 1,464) × $42 / 1,000,000,000 = $0.000140784`. +Adding the unchanged Grok answer estimate of $0.00165685 yields **$0.001797634** +for this query's metered model calls. The earlier query estimate was $0.14244085; +the $0.140643216 correction reduces the cumulative experiment estimate from +$1.04309935 to $0.902456134. + +The [query receipt](../receipts/jev-refined-query.json) retains the original +derived values, published rate and unit, source URL, and dated correction. Raw +provider responses, measured token counts, returned answer, and timings were not +changed. These remain list-price estimates, not verified billed spend. The +offline regression derives the rate from the published billion-token unit and +recomputes each stage and its ledger entry, including free output. One completed Jev-refined query exists. It used 300 sampled frames plus a coarse transcript sidecar, made two Jev requests (relevance and support) and one Grok @@ -50,6 +69,10 @@ answer request, and returned supported `1140–1200s` evidence. Its 3.7385-secon query wall time is a single question, not an all-in AV-versus-Gemini latency comparison, and no aggregate quality-parity or savings claim is valid from it. The Gemini baseline used native video/audio at its recorded sampling configuration. +No explicit Gemini cache was created, but implicit cached usage was not reported; +the full-rate estimate does not establish a measured cache miss. The 902.905-second +ingestion timer excludes prior frame extraction and external ASR, so it is not +end-to-end ingestion latency. Six caption responses also exceeded the requested 200-token advisory cap; output caps are therefore not an enforced guarantee on this route. Provider billing and compute, storage, and network costs remain unknown. diff --git a/cookbook/cost-model/model.py b/cookbook/cost-model/model.py index f27be56..076cdf6 100644 --- a/cookbook/cost-model/model.py +++ b/cookbook/cost-model/model.py @@ -32,6 +32,14 @@ def fraction(value, label: str) -> Decimal: return result +def rate_per_million(published_usd, published_token_unit: str) -> Decimal: + """Normalize a published token-price unit without silently assuming millions.""" + denominators = {"million": Decimal("1000000"), "billion": Decimal("1000000000")} + if published_token_unit not in denominators: + raise ValueError("published_token_unit must be million or billion") + return number(published_usd, "published rate") * Decimal("1000000") / denominators[published_token_unit] + + def token_cost(input_tokens, output_tokens, input_usd_per_million, output_usd_per_million): """Return an estimated dollar amount; unreported usage remains unknown.""" if any(v is None for v in (input_tokens, output_tokens, input_usd_per_million, output_usd_per_million)): diff --git a/cookbook/cost-model/notebook.ipynb b/cookbook/cost-model/notebook.ipynb index 44f97a8..cc926a3 100644 --- a/cookbook/cost-model/notebook.ipynb +++ b/cookbook/cost-model/notebook.ipynb @@ -189,7 +189,7 @@ " \"note\": \"Two Jev requests (relevance and support) plus one Grok answer request; complete usage was returned for each stage.\"\n", " }\n", " ],\n", - " \"known_list_estimates_usd\": \"1.04309935\",\n", + " \"known_list_estimates_usd\": \"0.902456134\",\n", " \"unknown_costs\": {\n", " \"terms\": [\n", " {\n", @@ -352,8 +352,8 @@ " \"total_retained_usd\": \"0\"\n", " },\n", " \"cumulative_cap_usd\": \"5\",\n", - " \"known_estimates_plus_retained_usd\": \"1.04309935\",\n", - " \"remaining_cap_usd\": \"3.95690065\"\n", + " \"known_estimates_plus_retained_usd\": \"0.902456134\",\n", + " \"remaining_cap_usd\": \"4.097543866\"\n", "}\n" ] } @@ -397,7 +397,7 @@ "text": [ "{\n", " \"queries\": 1,\n", - " \"known_subtotal_usd\": \"0.71586425\",\n", + " \"known_subtotal_usd\": \"0.575221034\",\n", " \"complete_total_usd\": null,\n", " \"unknown_terms\": [\n", " \"indexing.3.preprocessing\",\n", @@ -408,7 +408,7 @@ "}\n", "{\n", " \"queries\": 100,\n", - " \"known_subtotal_usd\": \"14.81750840\",\n", + " \"known_subtotal_usd\": \"0.753186800\",\n", " \"complete_total_usd\": null,\n", " \"unknown_terms\": [\n", " \"indexing.3.preprocessing\",\n", @@ -419,7 +419,7 @@ "}\n", "{\n", " \"queries\": 1000,\n", - " \"known_subtotal_usd\": \"143.01427340\",\n", + " \"known_subtotal_usd\": \"2.371057400\",\n", " \"complete_total_usd\": null,\n", " \"unknown_terms\": [\n", " \"indexing.3.preprocessing\",\n", @@ -467,19 +467,19 @@ "output_type": "stream", "text": [ "{\n", - " \"cache_policy\": \"disabled for the single recorded baseline request; no reusable cache was explicitly created\",\n", - " \"cache_hit_rate\": \"0\",\n", + " \"cache_policy\": \"No explicit cache was created for the recorded request; implicit cached usage was not reported. The full-rate baseline is an uncached counterfactual, not a measured cache miss.\",\n", + " \"cache_hit_rate\": null,\n", " \"cache_baseline\": {\n", " \"terms\": [\n", " {\n", " \"name\": \"baseline.cache_misses\",\n", - " \"usd\": \"30.807600\",\n", + " \"usd\": null,\n", " \"basis\": \"estimated\",\n", - " \"note\": \"One completed Gemini 3.8 native-video query: measured provider tokens multiplied by recorded upstream list rates; actual billing is unknown.\"\n", + " \"note\": \"One completed Gemini 3.8 native-video query: measured tokens at full published input/output rates. This is an uncached counterfactual estimate; implicit cached usage and actual billing are unknown.\"\n", " },\n", " {\n", " \"name\": \"baseline.cache_hits\",\n", - " \"usd\": \"0\",\n", + " \"usd\": null,\n", " \"basis\": \"unknown\",\n", " \"note\": \"No cache-hit request was measured.\"\n", " },\n", @@ -496,9 +496,12 @@ " \"note\": \"No explicit cache storage was requested.\"\n", " }\n", " ],\n", - " \"known_subtotal_usd\": \"30.807600\",\n", - " \"unknown_terms\": [],\n", - " \"complete_total_usd\": \"30.807600\",\n", + " \"known_subtotal_usd\": \"0\",\n", + " \"unknown_terms\": [\n", + " \"baseline.cache_misses\",\n", + " \"baseline.cache_hits\"\n", + " ],\n", + " \"complete_total_usd\": null,\n", " \"complete_means\": \"all modeled costs supplied; estimated and assumed inputs retain their provenance\"\n", " },\n", " \"fallback\": {\n", diff --git a/cookbook/cost-model/scenario.receipts.json b/cookbook/cost-model/scenario.receipts.json index c42dffe..98c08d0 100644 --- a/cookbook/cost-model/scenario.receipts.json +++ b/cookbook/cost-model/scenario.receipts.json @@ -40,9 +40,9 @@ "note": "The completed Jev-refined query used local FTS retrieval with embeddings disabled; local compute remains unpriced." }, "judge": { - "usd": "0.140784", + "usd": "0.000140784", "basis": "estimated", - "note": "Two completed Jev requests: relevance and support. Measured input/output token usage multiplied by recorded list rates; billed dollars are unknown." + "note": "Two completed Jev requests: 1,888 relevance + 1,464 support input tokens at $42 per billion ($0.042 per million); output is free. Source: https://docs.typesafe.ai/models.md, verified 2026-09-19. Corrected units; billed dollars remain unknown." }, "answer": { "usd": "0.00165685", @@ -75,11 +75,11 @@ "uncached_per_query": { "usd": "0.308076", "basis": "estimated", - "note": "One completed Gemini 3.8 native-video query: measured provider tokens multiplied by recorded upstream list rates; actual billing is unknown." + "note": "One completed Gemini 3.8 native-video query: measured tokens at full published input/output rates. This is an uncached counterfactual estimate; implicit cached usage and actual billing are unknown." }, "cache": { - "policy": "disabled for the single recorded baseline request; no reusable cache was explicitly created", - "hit_rate": 0, + "policy": "No explicit cache was created for the recorded request; implicit cached usage was not reported. The full-rate baseline is an uncached counterfactual, not a measured cache miss.", + "hit_rate": null, "hit_per_query": { "usd": null, "basis": "unknown", @@ -338,13 +338,13 @@ }, { "name": "jev_refined_query", - "usd": "0.14244085", + "usd": "0.001797634", "basis": "estimated", "category": "av_grok_jev_query", "outcome": "completed", "receipt": "../receipts/jev-refined-query.json", "usage_ref": "jev_refined_query", - "note": "Cache-aware upstream list-rate estimate for two Jev requests and one Grok answer request; no retry or fallback occurred." + "note": "Corrected token-derived list-rate estimate for two Jev requests at $0.042/M input with free output, plus one cache-aware Grok answer request; no retry or fallback occurred. Original derived estimates are preserved in the receipt correction record." } ], "unknown_costs": [ diff --git a/cookbook/cost-model/test_model.py b/cookbook/cost-model/test_model.py index a7ced5a..3bb08a1 100644 --- a/cookbook/cost-model/test_model.py +++ b/cookbook/cost-model/test_model.py @@ -7,7 +7,7 @@ from pathlib import Path from build_nb import build_notebook -from model import experiment_report, experiment_spend, scenario, token_cost +from model import experiment_report, experiment_spend, rate_per_million, scenario, token_cost HERE = Path(__file__).resolve().parent RECEIPTS = HERE.parent / "receipts" @@ -100,7 +100,7 @@ def test_receipt_ledger_preserves_cap_and_separates_unknowns(self): data = json.loads((HERE / "scenario.receipts.json").read_text()) report = experiment_report(data) self.assertEqual(report["list_rate_estimates"]["known_subtotal_usd"], - Decimal("1.04309935")) + Decimal("0.902456134")) self.assertEqual(report["reservations"]["total_recorded_usd"], Decimal("6.19945825")) self.assertEqual(report["reservations"]["total_retained_usd"], Decimal("0")) self.assertEqual( @@ -108,8 +108,8 @@ def test_receipt_ledger_preserves_cap_and_separates_unknowns(self): Decimal("4.19945825"), ) self.assertEqual(report["reservations"]["status_totals_usd"]["superseded"], Decimal("2.00")) - self.assertEqual(report["known_estimates_plus_retained_usd"], Decimal("1.04309935")) - self.assertEqual(report["remaining_cap_usd"], Decimal("3.95690065")) + self.assertEqual(report["known_estimates_plus_retained_usd"], Decimal("0.902456134")) + self.assertEqual(report["remaining_cap_usd"], Decimal("4.097543866")) self.assertIsNone(report["unknown_costs"]["complete_total_usd"]) self.assertEqual( [item["outcome"] for item in report["failures"]], @@ -155,6 +155,46 @@ def test_sanitized_receipts_are_present_and_baseline_estimate_recomputes(self): "../receipts/jev-refined-query.json", ) + def test_jev_published_billion_rate_recomputes_receipt_and_ledger(self): + receipt = json.loads((RECEIPTS / "jev-refined-query.json").read_text()) + published = receipt["pricing"]["jev"] + self.assertEqual(published["source_url"], "https://docs.typesafe.ai/models.md") + self.assertEqual(published["published_token_unit"], "billion") + self.assertEqual(published["published_input_usd"], "42") + self.assertEqual(published["published_output_usd"], "0") + input_rate = rate_per_million(published["published_input_usd"], published["published_token_unit"]) + output_rate = rate_per_million(published["published_output_usd"], published["published_token_unit"]) + self.assertEqual(input_rate, Decimal("0.042")) + self.assertEqual(output_rate, Decimal("0")) + self.assertEqual(input_rate, Decimal(published["input_usd_per_million"])) + self.assertEqual(output_rate, Decimal(published["output_usd_per_million"])) + judge = Decimal("0") + for stage in ("relevance", "support"): + usage = receipt["stage_usage"][stage] + estimate = token_cost(usage["input_tokens"], usage["output_tokens"], input_rate, output_rate) + self.assertEqual(estimate, Decimal(usage["estimated_usd"])) + # A nonzero reported output count is still free under the published rate. + self.assertEqual(estimate, token_cost(usage["input_tokens"], 0, input_rate, output_rate)) + judge += estimate + total = judge + Decimal(receipt["stage_usage"]["answer"]["estimated_usd"]) + self.assertEqual(total, Decimal(receipt["cache_aware_list_estimate_usd"])) + data = json.loads((HERE / "scenario.receipts.json").read_text()) + self.assertEqual(judge, Decimal(data["query"]["judge"]["usd"])) + entry = next(row for row in data["experiment_spend"] if row["name"] == "jev_refined_query") + self.assertEqual(total, Decimal(entry["usd"])) + correction = receipt["cost_correction"] + previous = correction["original_derived_estimates"] + self.assertEqual(Decimal(previous["query_estimated_usd"]) - total, Decimal(correction["estimate_reduction_usd"])) + self.assertEqual( + Decimal(previous["known_cumulative_list_estimate_usd"]) - Decimal(correction["estimate_reduction_usd"]), + experiment_report(data)["list_rate_estimates"]["known_subtotal_usd"], + ) + + def test_price_unit_must_be_explicit_and_recognized(self): + self.assertEqual(rate_per_million("42", "million"), Decimal("42")) + with self.assertRaisesRegex(ValueError, "published_token_unit"): + rate_per_million("42", "tokens") + def test_every_receipt_reservation_is_reconciled_with_an_explicit_state(self): data = json.loads((HERE / "scenario.receipts.json").read_text()) reconciled = { @@ -211,10 +251,13 @@ def test_incomplete_av_side_suppresses_baseline_ratio(self): Decimal("0.308076")) self.assertIsNone(result["indexed"]["complete_total_usd"]) self.assertIsNone(result["uncached_to_indexed_before_unknown_costs_ratio"]) + # No explicit cache was created, but missing implicit-cache usage is not zero. + self.assertIsNone(result["cache_hit_rate"]) + self.assertIsNone(result["cache_policy_baseline"]["complete_total_usd"]) def test_cap_rejects_estimates_plus_reservations_above_limit(self): data = json.loads((HERE / "scenario.receipts.json").read_text()) - data["cumulative_cap_usd"] = "1.04309934" + data["cumulative_cap_usd"] = "0.902456133" with self.assertRaisesRegex(ValueError, "exceed cumulative cap"): experiment_report(data) diff --git a/cookbook/receipts/README.md b/cookbook/receipts/README.md index 1fdfbf6..18aeb3f 100644 --- a/cookbook/receipts/README.md +++ b/cookbook/receipts/README.md @@ -22,4 +22,9 @@ because those fields are needed to interpret the recorded result. | [jev-credential-blocked.json](jev-credential-blocked.json) | Jev arm stopped before request because `AV_TYPESAFE_API_KEY` was absent | | [jev-refined-query.json](jev-refined-query.json) | Completed Jev-refined Grok answer at `59a7f73`: 2 Jev requests + 1 Grok answer, no retries/fallbacks, supported evidence | +The Jev-refined receipt labels its factual answer text as an excerpt and records +the 2026-09-19 pricing-unit correction with original derived values and official +rate provenance. Raw provider measurements were preserved. The ingestion timer +excludes prior frame extraction and external ASR. + The receipts are evidence for those individual attempts only. The later Jev-refined query completed the intended AV Grok+Jev path for one question. These receipts do not establish aggregate speed, cost, or quality parity between the direct-video baseline and an AV pipeline. diff --git a/cookbook/receipts/gemini38-baseline.json b/cookbook/receipts/gemini38-baseline.json index 5522923..d251a77 100644 --- a/cookbook/receipts/gemini38-baseline.json +++ b/cookbook/receipts/gemini38-baseline.json @@ -18,14 +18,15 @@ "media_resolution": "low", "media_processing": "static", "input": "native video and audio", - "comparison_note": "This is not sampling-equivalent to the incomplete AV arm." + "comparison_note": "The completed AV arm used 300 sampled frames plus a coarse external transcript sidecar; these inputs are not sampling-equivalent to this native video/audio request." }, "request_policy": { "generation_attempts": 1, "automatic_retries": 0, "max_output_tokens_including_thoughts": 8192, "thinking_level": "high", - "explicit_cache": false + "explicit_cache": false, + "cache_note": "No explicit cache was created. Implicit cached-token usage was not returned and remains unknown." }, "usage": { "input_tokens": 409548, @@ -45,7 +46,8 @@ "worst_case_reserved_list_estimate_usd": 0.37845825, "basis": "Measured tokens multiplied by published paid-tier list rates; not a billing invoice.", "provider_or_proxy_billed_usd": null, - "compute_storage_network_usd": null + "compute_storage_network_usd": null, + "cache_adjusted_list_estimate_usd": null }, "limitations": { "answer_quality_reviewed": false, diff --git a/cookbook/receipts/grok-live-ingestion.json b/cookbook/receipts/grok-live-ingestion.json index 6bfa347..241c635 100644 --- a/cookbook/receipts/grok-live-ingestion.json +++ b/cookbook/receipts/grok-live-ingestion.json @@ -55,5 +55,6 @@ "maximum_completion_tokens": 235, "proxy_billed_usd_unknown": true, "infrastructure_storage_network_unpriced": true - } + }, + "wall_seconds_scope": "Measured AV ingestion run using already extracted frames and an existing transcript sidecar; excludes prior frame extraction and external ASR. Not end-to-end ingestion latency." } diff --git a/cookbook/receipts/jev-refined-query.json b/cookbook/receipts/jev-refined-query.json index b67a263..76c1101 100644 --- a/cookbook/receipts/jev-refined-query.json +++ b/cookbook/receipts/jev-refined-query.json @@ -45,7 +45,7 @@ "requests": 1, "input_tokens": 1888, "output_tokens": 89, - "estimated_usd": "0.079296" + "estimated_usd": "0.000079296" }, "boundary": { "requests": 0, @@ -63,7 +63,7 @@ "requests": 1, "input_tokens": 1464, "output_tokens": 21, - "estimated_usd": "0.061488" + "estimated_usd": "0.000061488" }, "vision": { "requests": 0, @@ -71,9 +71,9 @@ "output_tokens": null } }, - "cache_aware_list_estimate_usd": "0.14244085", + "cache_aware_list_estimate_usd": "0.001797634", "usage_complete": true, - "known_cumulative_list_estimate_usd": "1.04309935", + "known_cumulative_list_estimate_usd": "0.902456134", "cumulative_cap_usd": "5", "proxy_billed_usd": null, "basis": "estimated from measured usage and recorded upstream list rates", @@ -88,5 +88,36 @@ "jev_returned_model_identifier_unavailable": true, "proxy_billed_usd_unknown": true, "infrastructure_storage_network_unpriced": true + }, + "pricing": { + "jev": { + "source_url": "https://docs.typesafe.ai/models.md", + "verified_on": "2026-09-19", + "published_input_usd": "42", + "published_output_usd": "0", + "published_token_unit": "billion", + "input_usd_per_million": "0.042", + "output_usd_per_million": "0", + "source_rate_table": "$42 / $0.042 per Btok / Mtok", + "source_unit_definition": "Price: Charged per input token. Output tokens are free. A Btok is a billion tokens and an Mtok is a million tokens.", + "basis": "Published list price applied to measured tokens; not verified billed spend." + } + }, + "answer_is_excerpt": true, + "answer_note": "Normalized factual excerpt: formatting normalized and additional timestamp/visual commentary omitted. This is not the complete verbatim provider response; the single-question result does not establish the correctness of the omitted claims.", + "cost_correction": { + "corrected_on": "2026-09-19", + "reason": "The observer interpreted $42 per billion input tokens as $42 per million, overstating Jev costs by 1000x. Output tokens are free.", + "original_derived_estimates": { + "jev_input_usd_per_million": "42", + "relevance_estimated_usd": "0.079296", + "support_estimated_usd": "0.061488", + "query_estimated_usd": "0.14244085", + "known_cumulative_list_estimate_usd": "1.04309935" + }, + "estimate_reduction_usd": "0.140643216", + "measured_tokens_answer_and_timing_unchanged": true, + "raw_provider_responses_unchanged": true, + "billing_verification_performed": false } }