Skip to content
This repository was archived by the owner on Oct 11, 2026. It is now read-only.
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 4 additions & 3 deletions calibre_toolkit/commands/audit.py
Original file line number Diff line number Diff line change
Expand Up @@ -86,9 +86,10 @@ def load_audit_records(path: Path) -> list[AuditRecord]:

Builds on the shared `read_audit_entries` reader (one parser for the
whole toolkit), then filters to records that represent AI-applied writes
— those have a non-empty `confidence`, `source`, and `step`. Manual marker
writes (mark_mqg_complete and similar) don't carry these and are correctly
excluded from the calibration pool. Re-grade writes (carrying a `regrade`
— those have a non-empty `confidence`, `source`, and `step`. Flag writes
(mark_mqg_complete / clear_mqg_flag, since v1.10) carry source="flag" and
step="flag:<command>" but no confidence, so they are correctly excluded
from the calibration pool. Re-grade writes (carrying a `regrade`
marker) are also excluded — they re-run a changed rule and would otherwise
mix pre- and post-change accuracy in the same tier stats.
"""
Expand Down
2 changes: 1 addition & 1 deletion calibre_toolkit/commands/clean_titles.py
Original file line number Diff line number Diff line change
Expand Up @@ -397,7 +397,7 @@ def _mark_complete(
if not mqg_column or not book_ids:
return
with console.status(f"[cyan]Marking {len(book_ids)} {label} books as MQG-01 complete…"):
db.mark_mqg_complete(book_ids, mqg_column)
db.mark_mqg_complete(book_ids, mqg_column, audit_step="clean-titles")
console.print(
f"[dim]Marked {len(book_ids)} books as complete in [bold]{mqg_column}[/bold].[/dim]"
)
4 changes: 2 additions & 2 deletions calibre_toolkit/commands/comments_enrich.py
Original file line number Diff line number Diff line change
Expand Up @@ -376,7 +376,7 @@ def run_comments_enrichment(
f"[bold]{mqg_manual_column}[/bold] for manual review."
)
with console.status("Flagging…"):
db.mark_mqg_complete(manual_ids, mqg_manual_column)
db.mark_mqg_complete(manual_ids, mqg_manual_column, audit_step="comments-enrich")

applied_ids_set = set(applied_ids)
applied_suggestions = [s for s in suggestions if s.book_id in applied_ids_set]
Expand Down Expand Up @@ -453,7 +453,7 @@ def _mark_complete(
if not mqg_column or not book_ids:
return
with console.status(f"[cyan]Marking {len(book_ids)} books as {label} complete…"):
db.mark_mqg_complete(book_ids, mqg_column)
db.mark_mqg_complete(book_ids, mqg_column, audit_step="comments-enrich")
console.print(
f"[dim]Marked {len(book_ids)} books complete in [bold]{mqg_column}[/bold].[/dim]"
)
4 changes: 2 additions & 2 deletions calibre_toolkit/commands/enrich_identifiers.py
Original file line number Diff line number Diff line change
Expand Up @@ -553,7 +553,7 @@ def _mark_complete(
if not mqg_column or not book_ids:
return
with console.status(f"[cyan]Marking {len(book_ids)} {label} books as MQG-02 complete…"):
db.mark_mqg_complete(book_ids, mqg_column)
db.mark_mqg_complete(book_ids, mqg_column, audit_step="enrich-identifiers")
console.print(
f"[dim]Marked {len(book_ids)} books as complete in [bold]{mqg_column}[/bold].[/dim]"
)
Expand All @@ -567,4 +567,4 @@ def _mark_manual(
if not book_ids:
return
with console.status(f"[cyan]Flagging {len(book_ids)} books for manual curation…"):
db.mark_mqg_complete(book_ids, mqg_manual_column)
db.mark_mqg_complete(book_ids, mqg_manual_column, audit_step="enrich-identifiers")
4 changes: 2 additions & 2 deletions calibre_toolkit/commands/lcc_enrich.py
Original file line number Diff line number Diff line change
Expand Up @@ -600,7 +600,7 @@ def run_lcc_enrichment(
f"[bold]{mqg_manual_column}[/bold] for manual review."
)
with console.status("Flagging…"):
db.mark_mqg_complete(manual_ids, mqg_manual_column)
db.mark_mqg_complete(manual_ids, mqg_manual_column, audit_step="lcc-enrich")

applied_ids_set = set(applied_ids)
flagged = len(manual_ids) if mqg_manual_column else 0
Expand Down Expand Up @@ -716,5 +716,5 @@ def _mark_complete(db: CalibreDB, mqg_column: str, book_ids: list[int], label: s
if not mqg_column or not book_ids:
return
with console.status(f"[cyan]Marking {len(book_ids)} books as {label} complete…"):
db.mark_mqg_complete(book_ids, mqg_column)
db.mark_mqg_complete(book_ids, mqg_column, audit_step="lcc-enrich")
console.print(f"[dim]Marked {len(book_ids)} books complete in [bold]{mqg_column}[/bold].[/dim]")
4 changes: 2 additions & 2 deletions calibre_toolkit/commands/tags_enrich.py
Original file line number Diff line number Diff line change
Expand Up @@ -310,7 +310,7 @@ def run_tags_enrichment(
f"[bold]{mqg_manual_column}[/bold] for manual review."
)
with console.status("Flagging…"):
db.mark_mqg_complete(manual_ids, mqg_manual_column)
db.mark_mqg_complete(manual_ids, mqg_manual_column, audit_step="tags-enrich")

applied_ids_set = set(applied_ids)
flagged = len(manual_ids) if mqg_manual_column else 0
Expand Down Expand Up @@ -397,7 +397,7 @@ def _mark_complete(
if not mqg_column or not book_ids:
return
with console.status(f"[cyan]Marking {len(book_ids)} books as {label} complete…"):
db.mark_mqg_complete(book_ids, mqg_column)
db.mark_mqg_complete(book_ids, mqg_column, audit_step="tags-enrich")
console.print(
f"[dim]Marked {len(book_ids)} books complete in [bold]{mqg_column}[/bold].[/dim]"
)
4 changes: 2 additions & 2 deletions calibre_toolkit/commands/tags_review.py
Original file line number Diff line number Diff line change
Expand Up @@ -185,9 +185,9 @@ def _prompt_action(has_ai: bool) -> str:

def _lock(db: CalibreDB, book_id: int, reviewed_column: str, mqg_column: str | None) -> None:
"""Mark a book as reviewed and (if configured) as MQG-05 complete."""
db.mark_mqg_complete([book_id], reviewed_column)
db.mark_mqg_complete([book_id], reviewed_column, audit_step="tags-review")
if mqg_column:
db.mark_mqg_complete([book_id], mqg_column)
db.mark_mqg_complete([book_id], mqg_column, audit_step="tags-review")


def _inline_edit(current: list[str], suggested: list[str] | None) -> list[str]:
Expand Down
2 changes: 1 addition & 1 deletion calibre_toolkit/commands/unflag_manual.py
Original file line number Diff line number Diff line change
Expand Up @@ -123,7 +123,7 @@ def run_unflag_manual(
with console.status(f"Clearing flags for {len(books)} book(s)…"):
for book in books:
try:
db.clear_mqg_flag(book.id, mqg_manual_column)
db.clear_mqg_flag(book.id, mqg_manual_column, audit_step="unflag-manual")
cleared += 1
except RuntimeError as e:
console.print(f"[red]Error on book {book.id}: {e}[/red]")
Expand Down
28 changes: 25 additions & 3 deletions calibre_toolkit/db.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,7 +10,7 @@
from pathlib import Path
from dataclasses import dataclass, field

from .logging_config import get_logger
from .logging_config import audit_log, get_logger

_log = get_logger(__name__)

Expand Down Expand Up @@ -291,12 +291,21 @@ def _one(item):
progress_callback(done, len(updates), len(failures))
return applied, failures

def mark_mqg_complete(self, book_ids: list[int], column: str) -> None:
def mark_mqg_complete(
self, book_ids: list[int], column: str, audit_step: str = "",
) -> None:
"""Mark a list of books as complete for a given MQG column.

Writes directly to SQLite in a single transaction — avoids spawning
one calibredb process per book, which is prohibitively slow for
large batches.

`audit_step` names the command performing the flag write; each book
gets one audit entry with source="flag" and step="flag:<audit_step>".
The "flag:" namespace is load-bearing: the regrade staleness selector
filters on exact AI step names ("lcc-enrich" etc.), so a bare command
name here would make a flagged book look freshly enriched. Entries
carry no confidence, which keeps them out of the calibration pool.
"""
label = column.lstrip("#")
with self._connect() as ro:
Expand All @@ -317,7 +326,13 @@ def mark_mqg_complete(self, book_ids: list[int], column: str) -> None:
)
conn.commit()

def clear_mqg_flag(self, book_id: int, column: str) -> None:
step = f"flag:{audit_step}" if audit_step else "flag"
for bid in book_ids:
audit_log(bid, column, True, source="flag", step=step)

def clear_mqg_flag(
self, book_id: int, column: str, audit_step: str = "",
) -> None:
"""Clear a custom boolean MQG column for a single book.

Deletes the row rather than writing value=0: Calibre's bool-column
Expand All @@ -326,6 +341,10 @@ def clear_mqg_flag(self, book_id: int, column: str) -> None:
`not #<manual>:true` exclusion and the book would never be
re-queued. Removing the row restores the undefined state that
`not #col:true` matches.

Audited the same way as mark_mqg_complete (source="flag",
step="flag:<audit_step>"), with new_value=None recording the
cleared state.
"""
label = column.lstrip("#")
with self._connect() as ro:
Expand All @@ -340,6 +359,9 @@ def clear_mqg_flag(self, book_id: int, column: str) -> None:
conn.execute(f"DELETE FROM {table} WHERE book = ?", (book_id,))
conn.commit()

step = f"flag:{audit_step}" if audit_step else "flag"
audit_log(book_id, column, None, source="flag", step=step)

def get_identifiers(self, book_id: int) -> dict[str, str]:
"""Return {type: value} for all identifiers currently on a book."""
query = "SELECT type, val FROM identifiers WHERE book = ?"
Expand Down
12 changes: 12 additions & 0 deletions tests/conftest.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,18 @@
FIXTURES_DIR = Path(__file__).parent / "fixtures"


@pytest.fixture(autouse=True)
def _isolated_audit_log(tmp_path: Path, monkeypatch):
"""Keep every test's audit trail out of ~/.calibre-toolkit/audit.log.

Flag writes audit through db.mark_mqg_complete / clear_mqg_flag since
v1.10, so any test touching those methods would otherwise append to the
user's real audit log. Tests that need a specific path still win — their
own monkeypatch.setenv overwrites this one.
"""
monkeypatch.setenv("CALIBRE_TOOLKIT_AUDIT_LOG", str(tmp_path / "audit.log"))


@pytest.fixture
def fixtures_dir() -> Path:
"""Absolute path to the per-suite fixtures directory."""
Expand Down
170 changes: 170 additions & 0 deletions tests/test_flag_audit.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,170 @@
"""Contract tests for flag-write audit logging (v1.10 item 3).

mark_mqg_complete and clear_mqg_flag write one audit entry per book with
source="flag" and step="flag:<command>". Three reader contracts are
load-bearing:

1. The calibration loader (load_audit_records) excludes flag entries —
they carry no confidence.
2. The regrade staleness selector (find_stale_books) filters on exact AI
step names, so the "flag:" namespace keeps a flag write from making a
book look freshly enriched.
3. A flag write inside a regrade context picks up the regrade marker at
the audit_log choke point — harmless, since contracts 1 and 2 exclude
the entry on other fields.
"""

from __future__ import annotations

import json
import sqlite3
from datetime import datetime, timezone
from pathlib import Path

import pytest

from calibre_toolkit.db import CalibreDB
from calibre_toolkit.logging_config import audit_log, regrade_audit
from calibre_toolkit.commands.audit import load_audit_records
from calibre_toolkit.commands.regrade import find_stale_books


def _build_minimal_calibre_db(library_path: Path) -> None:
library_path.mkdir(parents=True, exist_ok=True)
conn = sqlite3.connect(str(library_path / "metadata.db"))
conn.executescript(
"""
CREATE TABLE books (
id INTEGER PRIMARY KEY,
title TEXT,
sort TEXT,
author_sort TEXT,
pubdate TEXT
);
CREATE TABLE custom_columns (
id INTEGER PRIMARY KEY,
label TEXT,
datatype TEXT
);
CREATE TABLE custom_column_7 (
book INTEGER,
value INTEGER,
UNIQUE(book)
);
"""
)
conn.executemany("INSERT INTO books (id, title) VALUES (?, ?)", [
(1, "Book One"),
(2, "Book Two"),
])
conn.execute(
"INSERT INTO custom_columns (id, label, datatype) VALUES (7, 'mqg_lcc_manual', 'bool')"
)
conn.commit()
conn.close()


@pytest.fixture
def db(tmp_path: Path) -> CalibreDB:
library = tmp_path / "lib"
_build_minimal_calibre_db(library)
return CalibreDB(library_path=str(library), calibredb_path="calibredb")


@pytest.fixture
def audit_path(tmp_path: Path) -> Path:
# The conftest autouse fixture points CALIBRE_TOOLKIT_AUDIT_LOG here.
return tmp_path / "audit.log"


def _entries(path: Path) -> list[dict]:
if not path.exists():
return []
return [json.loads(l) for l in path.read_text(encoding="utf-8").splitlines() if l.strip()]


# ── Writer contract ──────────────────────────────────────────────────────────


def test_mark_writes_one_entry_per_book(db: CalibreDB, audit_path: Path):
db.mark_mqg_complete([1, 2], "#mqg_lcc_manual", audit_step="lcc-enrich")
entries = _entries(audit_path)
assert len(entries) == 2
assert {e["book_id"] for e in entries} == {1, 2}
for e in entries:
assert e["field"] == "#mqg_lcc_manual"
assert e["new_value"] is True
assert e["source"] == "flag"
assert e["step"] == "flag:lcc-enrich"
assert "confidence" not in e


def test_clear_writes_entry_with_null_value(db: CalibreDB, audit_path: Path):
db.mark_mqg_complete([1], "#mqg_lcc_manual", audit_step="lcc-enrich")
db.clear_mqg_flag(1, "#mqg_lcc_manual", audit_step="unflag-manual")
entries = _entries(audit_path)
assert len(entries) == 2
clear = entries[-1]
assert clear["book_id"] == 1
assert clear["new_value"] is None
assert clear["source"] == "flag"
assert clear["step"] == "flag:unflag-manual"


def test_default_step_is_bare_flag(db: CalibreDB, audit_path: Path):
db.mark_mqg_complete([1], "#mqg_lcc_manual")
assert _entries(audit_path)[0]["step"] == "flag"


def test_mark_unknown_column_noop_writes_no_entry(db: CalibreDB, audit_path: Path):
db.mark_mqg_complete([1], "#no_such_column", audit_step="lcc-enrich")
assert _entries(audit_path) == []


def test_clear_unknown_column_raises_and_writes_no_entry(db: CalibreDB, audit_path: Path):
with pytest.raises(RuntimeError, match="not found"):
db.clear_mqg_flag(1, "#no_such_column", audit_step="unflag-manual")
assert _entries(audit_path) == []


# ── Reader contract 1: calibration excludes flag entries ─────────────────────


def test_calibration_loader_excludes_flag_entries(db: CalibreDB, audit_path: Path):
audit_log(1, "#lcc", "PR6059", confidence="high", source="ai", step="lcc-enrich")
db.mark_mqg_complete([1], "#mqg_lcc_manual", audit_step="lcc-enrich")

records = load_audit_records(audit_path)
assert len(records) == 1
assert records[0].field == "#lcc"
assert records[0].source == "ai"


# ── Reader contract 2: regrade staleness ignores flag entries ────────────────


def test_flag_entry_does_not_refresh_regrade_staleness():
entries = [
{"timestamp": "2026-05-01T00:00:00+00:00", "book_id": 1,
"field": "#lcc", "new_value": "PR6059",
"confidence": "high", "source": "ai", "step": "lcc-enrich"},
# Later flag write for the same book — must NOT make it look fresh.
{"timestamp": "2026-06-09T00:00:00+00:00", "book_id": 1,
"field": "#mqg_lcc", "new_value": True,
"source": "flag", "step": "flag:lcc-enrich"},
]
before = datetime(2026, 6, 1, tzinfo=timezone.utc)
assert find_stale_books(entries, "lcc-enrich", before) == [1]


# ── Reader contract 3: regrade marker rides along harmlessly ─────────────────


def test_flag_inside_regrade_context_carries_marker_and_stays_excluded(
db: CalibreDB, audit_path: Path,
):
with regrade_audit("2026-06-01"):
db.mark_mqg_complete([1], "#mqg_lcc_manual", audit_step="lcc-enrich")
entry = _entries(audit_path)[0]
assert entry["regrade"] == "2026-06-01"
assert load_audit_records(audit_path) == []
Loading