diff --git a/README.md b/README.md index 113d669..734a1be 100644 --- a/README.md +++ b/README.md @@ -189,6 +189,8 @@ METAMEND_CACHE_MODE=record METAMEND_FIXTURES=./fixtures ebook-metamend --match " METAMEND_CACHE_MODE=replay METAMEND_FIXTURES=./fixtures ebook-metamend --match "Some Book" ``` +A recording keeps two layers: the record each source parsed, and the raw body it parsed it from (the catalogue's JSON, the OPF a Calibre plugin printed). A replay runs the current parser over the raw bodies when every raw file the record names is present, so a change to a parser shows up in the replay instead of being hidden by it; when any is missing, or the set was recorded before the raw layer existed, it replays the parsed record as before. Recording again over an existing set only touches the network for bodies it lacks. + `tools/snapshot.py` records a library's metadata and content hashes before and after a run and classifies every file as unchanged, meta-changed, content-changed or corrupt, so a change can be proven not to have damaged anything. diff --git a/src/ebook_metamend/sources/cache.py b/src/ebook_metamend/sources/cache.py index 0885835..a2f18c0 100644 --- a/src/ebook_metamend/sources/cache.py +++ b/src/ebook_metamend/sources/cache.py @@ -11,6 +11,13 @@ METAMEND_CACHE_MODE=replay # fixtures only, never touch the network Unset, sources behave normally. + +Two layers are recorded. The parsed record each source returns is what a replay +serves when nothing else is there; the raw body behind it (the JSON a catalogue +sent, the OPF a Calibre plugin printed) is recorded beside it, and a replay +prefers the raw body so that a change to a parser is exercised by the replay +rather than hidden by it. Fixture sets recorded before the raw layer existed +keep working: they simply replay the parsed record. """ from __future__ import annotations @@ -36,6 +43,51 @@ class MissingFixture(RuntimeError): """ +class MissingRaw(MissingFixture): + """Replay needed a raw body that was never recorded. + + Caught by ``wrap``, which then falls back to the parsed record; it only + surfaces when neither layer has the answer. + """ + + +def raw_path(kind: str, key: str) -> Path: + digest = hashlib.sha256(f'{kind}\x00{key}'.encode()).hexdigest()[:16] + return FIXTURES / f'raw-{kind}-{digest}.json' + + +def raw(kind: str, key: str, fetch: Callable[[], str]) -> str: + """Fetch a raw body through the cache, if a mode is set. + + ``kind`` names the layer ("http", "plugin") and ``key`` identifies the call + within it (the URL; the plugin, title and author). Failures are not stored + here: the parsed layer records them, so a failed call simply has no raw file + and replays from that record. + """ + if MODE not in ('record', 'replay'): + return fetch() + path = raw_path(kind, key) + if _consumed is not None: + _consumed.append(path.name) + if path.exists(): + with path.open(encoding='utf8') as fh: + return json.load(fh)['body'] + if MODE == 'replay': + raise MissingRaw(f'{kind} / {key!r} -> {path.name}') + body = fetch() + FIXTURES.mkdir(parents=True, exist_ok=True) + with path.open('w', encoding='utf8') as fh: + json.dump({'kind': kind, 'key': key, 'body': body}, fh, indent=1, ensure_ascii=False) + return body + + +#: The raw files the fetch being recorded has read, so its parsed record can +#: name them. A replay runs the fetch only when every named file is present: +#: that is what keeps a source that bypasses the raw layer, or a set recorded +#: before it existed, from ever reaching the network during a replay. +_consumed: list[str] | None = None + + def fixture_path(source: str, title: str, author: str) -> Path: digest = hashlib.sha256(f'{source}\x00{title}\x00{author}'.encode()).hexdigest()[:16] return FIXTURES / f'{source}-{digest}.json' @@ -58,35 +110,59 @@ def save(path: Path, title: str, author: str, record: dict[str, Any]) -> None: def cached(title: str, author: str) -> Any: path = fixture_path(source, title, author) - if path.exists(): + global _consumed + if MODE == 'replay': + if not path.exists(): + raise MissingFixture(f'{source} / {title!r} / {author!r} -> {path.name}') with path.open(encoding='utf8') as fh: record = json.load(fh) + # The raw layer first, so the parser in the checked-out code runs + # over what the catalogue actually sent. Only when every raw file + # the recording read is present, or a fetch would touch the network. + if record.get('raw') and all((FIXTURES / name).exists() for name in record['raw']): + # A SourceError here can only come from a stored body the + # parser rejects (no network is reached), so the parsed record + # is the better answer. + try: + return fetch(title, author) + except (MissingRaw, SourceError): + pass # A failure is part of what happened and has to replay as one. # Recording only successes meant a run where a source was unreachable # could not be replayed at all: the key was never written, so replay # raised MissingFixture and went back to the network. - # - # Only in replay, though. A failure is by definition transient, so in - # record mode a stored one must be retried and overwritten, or the - # first bad minute would be frozen into the fixtures for good. - if 'error' not in record: - return record['response'] - if MODE == 'replay': + if 'error' in record: raise SourceError(record['error']) - elif MODE == 'replay': - raise MissingFixture(f'{source} / {title!r} / {author!r} -> {path.name}') + return record['response'] + # Record mode always runs the fetch: the raw layer serves a body it + # already has and only goes to the network for one it lacks, so a + # stored failure (transient by definition) is retried rather than + # frozen in, and a set recorded before the raw layer existed fills in. + _consumed = [] try: response = fetch(title, author) except SourceError as exc: - save(path, title, author, {'error': str(exc)}) + # A good record already on disk outlives a bad minute; the failure + # is still raised so the run reports it. + if not _has_response(path): + save(path, title, author, {'error': str(exc)}) raise - save(path, title, author, {'response': response}) + finally: + consumed, _consumed = _consumed, None + save(path, title, author, {'response': response, 'raw': consumed}) return response return cached +def _has_response(path: Path) -> bool: + if not path.exists(): + return False + with path.open(encoding='utf8') as fh: + return 'error' not in json.load(fh) + + def replaying() -> bool: """True when no network will be touched, so pauses are pointless.""" return MODE == 'replay' diff --git a/src/ebook_metamend/sources/calibre_plugin.py b/src/ebook_metamend/sources/calibre_plugin.py index 55322d7..672434f 100644 --- a/src/ebook_metamend/sources/calibre_plugin.py +++ b/src/ebook_metamend/sources/calibre_plugin.py @@ -9,6 +9,7 @@ from typing import Any from .. import calibre, opf +from . import cache from .errors import SourceUnavailable #: Calibre plugin names, as ``fetch-ebook-metadata -p`` expects them. @@ -37,8 +38,11 @@ def require_plugin(plugin: str) -> None: def fetch_plugin( title: str, author: str, plugin: str, timeout: int = DEFAULT_TIMEOUT ) -> dict[str, Any] | None: - require_plugin(plugin) - xml = calibre.fetch_metadata(title, author, plugin, timeout) + def fetch_from_calibre() -> str: + require_plugin(plugin) + return calibre.fetch_metadata(title, author, plugin, timeout) or '' + + xml = cache.raw('plugin', f'{plugin}\x00{title}\x00{author}', fetch_from_calibre) return opf.parse(xml) if xml else None diff --git a/src/ebook_metamend/sources/http.py b/src/ebook_metamend/sources/http.py index d73ab6d..da7f3ea 100644 --- a/src/ebook_metamend/sources/http.py +++ b/src/ebook_metamend/sources/http.py @@ -14,6 +14,7 @@ from collections.abc import Callable from typing import Any +from . import cache from .errors import SourceError #: The contact every catalogue asks for is the project URL, so nothing personal @@ -55,10 +56,24 @@ def get_json( it: a catalogue that is down must not read as "no such book". """ headers = {'User-Agent': USER_AGENT, 'Accept': accept} + + def fetched() -> str: + body = _transport(url, headers, timeout).decode('utf8') + # Parsed before the raw layer may store it: a body that is not JSON is + # a failed attempt to retry, never a fixture to replay. + json.loads(body) + return body + last_error: Exception | None = None + # A replay reads the same file however often it tries; one attempt is enough. + attempts = 1 if cache.replaying() else attempts for attempt in range(attempts): try: - return json.loads(_transport(url, headers, timeout)) + return json.loads(cache.raw('http', url, fetched)) + # A replay with no raw body must reach wrap() untouched, or the retry + # loop would report it as a transport failure. + except cache.MissingRaw: + raise # OSError covers URLError, socket timeouts and TLS errors; ValueError # covers a truncated or non-JSON body. A bare Exception here would also # swallow a programming mistake into retries with sleeps. diff --git a/tests/test_pipeline.py b/tests/test_pipeline.py index bf983c8..e2828e8 100644 --- a/tests/test_pipeline.py +++ b/tests/test_pipeline.py @@ -8,10 +8,10 @@ import pytest -from ebook_metamend import enrich +from ebook_metamend import calibre, enrich from ebook_metamend.library import Book from ebook_metamend.matching import SourceScore -from ebook_metamend.sources import cache +from ebook_metamend.sources import cache, calibre_plugin, http def score(name, title, title_score, author_score): @@ -346,6 +346,152 @@ def test_the_record_is_readable_by_a_human(self, fixtures, monkeypatch): assert saved['response'] == {'title': 'Dune'} +class TestTheRawBodyIsRecordedBesideTheParsedRecord: + """A parsed record replays what an old parser made of the answer; the raw + body lets the parser in the checked-out code run over what the catalogue + actually sent.""" + + @pytest.fixture + def fixtures(self, tmp_path, monkeypatch): + monkeypatch.setattr(cache, 'FIXTURES', tmp_path) + return tmp_path + + def test_a_body_round_trips_through_the_transport(self, fixtures, monkeypatch): + monkeypatch.setattr(cache, 'MODE', 'record') + http.set_transport(lambda url, headers, timeout: b'{"docs": [1]}') + try: + assert http.get_json('https://example.test/search?q=dune') == {'docs': [1]} + monkeypatch.setattr(cache, 'MODE', 'replay') + http.set_transport(lambda *_: pytest.fail('replay hit the network')) + assert http.get_json('https://example.test/search?q=dune') == {'docs': [1]} + finally: + http.set_transport(None) + + def test_a_parser_change_shows_in_replay(self, fixtures, monkeypatch): + monkeypatch.setattr(cache, 'MODE', 'record') + http.set_transport(lambda url, headers, timeout: b'{"title": "Dune", "year": 1965}') + try: + + def old_parser(title, author): + return {'title': http.get_json('https://example.test/' + title)['title']} + + def new_parser(title, author): + answer = http.get_json('https://example.test/' + title) + return {'title': answer['title'], 'year': answer['year']} + + assert cache.wrap('openlib', old_parser)('Dune', 'Herbert') == {'title': 'Dune'} + monkeypatch.setattr(cache, 'MODE', 'replay') + http.set_transport(lambda *_: pytest.fail('replay hit the network')) + assert cache.wrap('openlib', new_parser)('Dune', 'Herbert') == { + 'title': 'Dune', + 'year': 1965, + } + finally: + http.set_transport(None) + + def test_without_the_raw_body_the_parsed_record_still_serves(self, fixtures, monkeypatch): + monkeypatch.setattr(cache, 'MODE', 'record') + http.set_transport(lambda url, headers, timeout: b'{"title": "Dune"}') + try: + fetch = lambda t, a: http.get_json('https://example.test/' + t) # noqa: E731 + cache.wrap('openlib', fetch)('Dune', 'Herbert') + for raw_file in fixtures.glob('raw-*.json'): + raw_file.unlink() + monkeypatch.setattr(cache, 'MODE', 'replay') + replayed = cache.wrap('openlib', lambda t, a: pytest.fail('replay hit the network')) + assert replayed('Dune', 'Herbert') == {'title': 'Dune'} + finally: + http.set_transport(None) + + def test_a_body_that_is_not_json_is_retried_not_recorded(self, fixtures, monkeypatch): + monkeypatch.setattr(cache, 'MODE', 'record') + monkeypatch.setattr(http.time, 'sleep', lambda s: None) + bodies = iter([b'503', b'{"docs": [1]}']) + http.set_transport(lambda url, headers, timeout: next(bodies)) + try: + assert http.get_json('https://example.test/search') == {'docs': [1]} + monkeypatch.setattr(cache, 'MODE', 'replay') + http.set_transport(lambda *_: pytest.fail('replay hit the network')) + assert http.get_json('https://example.test/search') == {'docs': [1]} + finally: + http.set_transport(None) + + def test_a_parser_that_asks_a_new_url_falls_back_to_the_parsed_record( + self, fixtures, monkeypatch + ): + monkeypatch.setattr(cache, 'MODE', 'record') + http.set_transport(lambda url, headers, timeout: b'{"title": "Dune"}') + try: + fetch = lambda t, a: http.get_json('https://example.test/v1/' + t) # noqa: E731 + assert cache.wrap('openlib', fetch)('Dune', 'Herbert') == {'title': 'Dune'} + monkeypatch.setattr(cache, 'MODE', 'replay') + http.set_transport(lambda *_: pytest.fail('replay hit the network')) + moved = lambda t, a: http.get_json('https://example.test/v2/' + t) # noqa: E731 + assert cache.wrap('openlib', moved)('Dune', 'Herbert') == {'title': 'Dune'} + finally: + http.set_transport(None) + + def test_a_stored_body_the_parser_rejects_falls_back_to_the_parsed_record( + self, fixtures, monkeypatch + ): + monkeypatch.setattr(cache, 'MODE', 'record') + http.set_transport(lambda url, headers, timeout: b'{"title": "Dune"}') + try: + fetch = lambda t, a: http.get_json('https://example.test/' + t) # noqa: E731 + cache.wrap('openlib', fetch)('Dune', 'Herbert') + raw_file = cache.raw_path('http', 'https://example.test/Dune') + raw_file.write_text(json.dumps({'kind': 'http', 'key': 'x', 'body': ''})) + monkeypatch.setattr(cache, 'MODE', 'replay') + http.set_transport(lambda *_: pytest.fail('replay hit the network')) + monkeypatch.setattr(http.time, 'sleep', lambda s: pytest.fail('replay slept')) + assert cache.wrap('openlib', fetch)('Dune', 'Herbert') == {'title': 'Dune'} + finally: + http.set_transport(None) + + def test_a_plugin_answer_replays_without_calibre(self, fixtures, monkeypatch): + monkeypatch.setattr(cache, 'MODE', 'record') + monkeypatch.setattr(calibre_plugin, 'require_plugin', lambda plugin: None) + opf_text = ( + '' + 'Dune' + ) + monkeypatch.setattr(calibre, 'fetch_metadata', lambda *a, **k: opf_text) + first = calibre_plugin.fetch_plugin('Dune', 'Herbert', 'Kobo Metadata') + assert first['title'] == 'Dune' + + monkeypatch.setattr(cache, 'MODE', 'replay') + monkeypatch.setattr(calibre, 'fetch_metadata', lambda *a, **k: pytest.fail('ran calibre')) + monkeypatch.setattr( + calibre_plugin, 'require_plugin', lambda plugin: pytest.fail('probed calibre') + ) + assert calibre_plugin.fetch_plugin('Dune', 'Herbert', 'Kobo Metadata') == first + + def test_a_bad_minute_does_not_overwrite_a_good_record(self, fixtures, monkeypatch): + monkeypatch.setattr(cache, 'MODE', 'record') + cache.wrap('openlib', lambda t, a: {'title': 'Dune'})('Dune', 'Herbert') + failing = cache.wrap( + 'openlib', lambda t, a: (_ for _ in ()).throw(enrich.SourceError('TLS timeout')) + ) + with pytest.raises(enrich.SourceError): + failing('Dune', 'Herbert') + + monkeypatch.setattr(cache, 'MODE', 'replay') + replayed = cache.wrap('openlib', lambda t, a: pytest.fail('replay hit the network')) + assert replayed('Dune', 'Herbert') == {'title': 'Dune'} + + def test_the_record_names_the_raw_files_it_read(self, fixtures, monkeypatch): + monkeypatch.setattr(cache, 'MODE', 'record') + http.set_transport(lambda url, headers, timeout: b'{}') + try: + fetch = lambda t, a: http.get_json('https://example.test/' + t) # noqa: E731 + cache.wrap('openlib', fetch)('Dune', 'Herbert') + finally: + http.set_transport(None) + saved = json.loads(cache.fixture_path('openlib', 'Dune', 'Herbert').read_text()) + assert saved['raw'] == [cache.raw_path('http', 'https://example.test/Dune').name] + + class TestAtLowNoSourceIdentifiedTheBook: """LOW means not one source cleared both the title and the author. Under --include-low it may still offer subjects, but not an identifier."""