Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -189,6 +189,8 @@ METAMEND_CACHE_MODE=record METAMEND_FIXTURES=./fixtures ebook-metamend --match "
METAMEND_CACHE_MODE=replay METAMEND_FIXTURES=./fixtures ebook-metamend --match "Some Book"
```

A recording keeps two layers: the record each source parsed, and the raw body it parsed it from (the catalogue's JSON, the OPF a Calibre plugin printed). A replay runs the current parser over the raw bodies when every raw file the record names is present, so a change to a parser shows up in the replay instead of being hidden by it; when any is missing, or the set was recorded before the raw layer existed, it replays the parsed record as before. Recording again over an existing set only touches the network for bodies it lacks.

`tools/snapshot.py` records a library's metadata and content hashes before and
after a run and classifies every file as unchanged, meta-changed, content-changed
or corrupt, so a change can be proven not to have damaged anything.
Expand Down
100 changes: 88 additions & 12 deletions src/ebook_metamend/sources/cache.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,13 @@
METAMEND_CACHE_MODE=replay # fixtures only, never touch the network

Unset, sources behave normally.

Two layers are recorded. The parsed record each source returns is what a replay
serves when nothing else is there; the raw body behind it (the JSON a catalogue
sent, the OPF a Calibre plugin printed) is recorded beside it, and a replay
prefers the raw body so that a change to a parser is exercised by the replay
rather than hidden by it. Fixture sets recorded before the raw layer existed
keep working: they simply replay the parsed record.
"""

from __future__ import annotations
Expand All @@ -36,6 +43,51 @@ class MissingFixture(RuntimeError):
"""


class MissingRaw(MissingFixture):
"""Replay needed a raw body that was never recorded.

Caught by ``wrap``, which then falls back to the parsed record; it only
surfaces when neither layer has the answer.
"""


def raw_path(kind: str, key: str) -> Path:
digest = hashlib.sha256(f'{kind}\x00{key}'.encode()).hexdigest()[:16]
return FIXTURES / f'raw-{kind}-{digest}.json'


def raw(kind: str, key: str, fetch: Callable[[], str]) -> str:
"""Fetch a raw body through the cache, if a mode is set.

``kind`` names the layer ("http", "plugin") and ``key`` identifies the call
within it (the URL; the plugin, title and author). Failures are not stored
here: the parsed layer records them, so a failed call simply has no raw file
and replays from that record.
"""
if MODE not in ('record', 'replay'):
return fetch()
path = raw_path(kind, key)
if _consumed is not None:
_consumed.append(path.name)
if path.exists():
with path.open(encoding='utf8') as fh:
return json.load(fh)['body']
if MODE == 'replay':
raise MissingRaw(f'{kind} / {key!r} -> {path.name}')
body = fetch()
FIXTURES.mkdir(parents=True, exist_ok=True)
with path.open('w', encoding='utf8') as fh:
json.dump({'kind': kind, 'key': key, 'body': body}, fh, indent=1, ensure_ascii=False)
return body


#: The raw files the fetch being recorded has read, so its parsed record can
#: name them. A replay runs the fetch only when every named file is present:
#: that is what keeps a source that bypasses the raw layer, or a set recorded
#: before it existed, from ever reaching the network during a replay.
_consumed: list[str] | None = None


def fixture_path(source: str, title: str, author: str) -> Path:
digest = hashlib.sha256(f'{source}\x00{title}\x00{author}'.encode()).hexdigest()[:16]
return FIXTURES / f'{source}-{digest}.json'
Expand All @@ -58,35 +110,59 @@ def save(path: Path, title: str, author: str, record: dict[str, Any]) -> None:

def cached(title: str, author: str) -> Any:
path = fixture_path(source, title, author)
if path.exists():
global _consumed
if MODE == 'replay':
if not path.exists():
raise MissingFixture(f'{source} / {title!r} / {author!r} -> {path.name}')
with path.open(encoding='utf8') as fh:
record = json.load(fh)
# The raw layer first, so the parser in the checked-out code runs
# over what the catalogue actually sent. Only when every raw file
# the recording read is present, or a fetch would touch the network.
if record.get('raw') and all((FIXTURES / name).exists() for name in record['raw']):
# A SourceError here can only come from a stored body the
# parser rejects (no network is reached), so the parsed record
# is the better answer.
try:
return fetch(title, author)
except (MissingRaw, SourceError):
pass
# A failure is part of what happened and has to replay as one.
# Recording only successes meant a run where a source was unreachable
# could not be replayed at all: the key was never written, so replay
# raised MissingFixture and went back to the network.
#
# Only in replay, though. A failure is by definition transient, so in
# record mode a stored one must be retried and overwritten, or the
# first bad minute would be frozen into the fixtures for good.
if 'error' not in record:
return record['response']
if MODE == 'replay':
if 'error' in record:
raise SourceError(record['error'])
elif MODE == 'replay':
raise MissingFixture(f'{source} / {title!r} / {author!r} -> {path.name}')
return record['response']

# Record mode always runs the fetch: the raw layer serves a body it
# already has and only goes to the network for one it lacks, so a
# stored failure (transient by definition) is retried rather than
# frozen in, and a set recorded before the raw layer existed fills in.
_consumed = []
try:
response = fetch(title, author)
except SourceError as exc:
save(path, title, author, {'error': str(exc)})
# A good record already on disk outlives a bad minute; the failure
# is still raised so the run reports it.
if not _has_response(path):
save(path, title, author, {'error': str(exc)})
raise
save(path, title, author, {'response': response})
finally:
consumed, _consumed = _consumed, None
save(path, title, author, {'response': response, 'raw': consumed})
return response

return cached


def _has_response(path: Path) -> bool:
if not path.exists():
return False
with path.open(encoding='utf8') as fh:
return 'error' not in json.load(fh)


def replaying() -> bool:
"""True when no network will be touched, so pauses are pointless."""
return MODE == 'replay'
8 changes: 6 additions & 2 deletions src/ebook_metamend/sources/calibre_plugin.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@
from typing import Any

from .. import calibre, opf
from . import cache
from .errors import SourceUnavailable

#: Calibre plugin names, as ``fetch-ebook-metadata -p`` expects them.
Expand Down Expand Up @@ -37,8 +38,11 @@ def require_plugin(plugin: str) -> None:
def fetch_plugin(
title: str, author: str, plugin: str, timeout: int = DEFAULT_TIMEOUT
) -> dict[str, Any] | None:
require_plugin(plugin)
xml = calibre.fetch_metadata(title, author, plugin, timeout)
def fetch_from_calibre() -> str:
require_plugin(plugin)
return calibre.fetch_metadata(title, author, plugin, timeout) or ''

xml = cache.raw('plugin', f'{plugin}\x00{title}\x00{author}', fetch_from_calibre)
return opf.parse(xml) if xml else None


Expand Down
17 changes: 16 additions & 1 deletion src/ebook_metamend/sources/http.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@
from collections.abc import Callable
from typing import Any

from . import cache
from .errors import SourceError

#: The contact every catalogue asks for is the project URL, so nothing personal
Expand Down Expand Up @@ -55,10 +56,24 @@ def get_json(
it: a catalogue that is down must not read as "no such book".
"""
headers = {'User-Agent': USER_AGENT, 'Accept': accept}

def fetched() -> str:
body = _transport(url, headers, timeout).decode('utf8')
# Parsed before the raw layer may store it: a body that is not JSON is
# a failed attempt to retry, never a fixture to replay.
json.loads(body)
return body

last_error: Exception | None = None
# A replay reads the same file however often it tries; one attempt is enough.
attempts = 1 if cache.replaying() else attempts
for attempt in range(attempts):
try:
return json.loads(_transport(url, headers, timeout))
return json.loads(cache.raw('http', url, fetched))
# A replay with no raw body must reach wrap() untouched, or the retry
# loop would report it as a transport failure.
except cache.MissingRaw:
raise
# OSError covers URLError, socket timeouts and TLS errors; ValueError
# covers a truncated or non-JSON body. A bare Exception here would also
# swallow a programming mistake into retries with sleeps.
Expand Down
150 changes: 148 additions & 2 deletions tests/test_pipeline.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,10 +8,10 @@

import pytest

from ebook_metamend import enrich
from ebook_metamend import calibre, enrich
from ebook_metamend.library import Book
from ebook_metamend.matching import SourceScore
from ebook_metamend.sources import cache
from ebook_metamend.sources import cache, calibre_plugin, http


def score(name, title, title_score, author_score):
Expand Down Expand Up @@ -346,6 +346,152 @@ def test_the_record_is_readable_by_a_human(self, fixtures, monkeypatch):
assert saved['response'] == {'title': 'Dune'}


class TestTheRawBodyIsRecordedBesideTheParsedRecord:
"""A parsed record replays what an old parser made of the answer; the raw
body lets the parser in the checked-out code run over what the catalogue
actually sent."""

@pytest.fixture
def fixtures(self, tmp_path, monkeypatch):
monkeypatch.setattr(cache, 'FIXTURES', tmp_path)
return tmp_path

def test_a_body_round_trips_through_the_transport(self, fixtures, monkeypatch):
monkeypatch.setattr(cache, 'MODE', 'record')
http.set_transport(lambda url, headers, timeout: b'{"docs": [1]}')
try:
assert http.get_json('https://example.test/search?q=dune') == {'docs': [1]}
monkeypatch.setattr(cache, 'MODE', 'replay')
http.set_transport(lambda *_: pytest.fail('replay hit the network'))
assert http.get_json('https://example.test/search?q=dune') == {'docs': [1]}
finally:
http.set_transport(None)

def test_a_parser_change_shows_in_replay(self, fixtures, monkeypatch):
monkeypatch.setattr(cache, 'MODE', 'record')
http.set_transport(lambda url, headers, timeout: b'{"title": "Dune", "year": 1965}')
try:

def old_parser(title, author):
return {'title': http.get_json('https://example.test/' + title)['title']}

def new_parser(title, author):
answer = http.get_json('https://example.test/' + title)
return {'title': answer['title'], 'year': answer['year']}

assert cache.wrap('openlib', old_parser)('Dune', 'Herbert') == {'title': 'Dune'}
monkeypatch.setattr(cache, 'MODE', 'replay')
http.set_transport(lambda *_: pytest.fail('replay hit the network'))
assert cache.wrap('openlib', new_parser)('Dune', 'Herbert') == {
'title': 'Dune',
'year': 1965,
}
finally:
http.set_transport(None)

def test_without_the_raw_body_the_parsed_record_still_serves(self, fixtures, monkeypatch):
monkeypatch.setattr(cache, 'MODE', 'record')
http.set_transport(lambda url, headers, timeout: b'{"title": "Dune"}')
try:
fetch = lambda t, a: http.get_json('https://example.test/' + t) # noqa: E731
cache.wrap('openlib', fetch)('Dune', 'Herbert')
for raw_file in fixtures.glob('raw-*.json'):
raw_file.unlink()
monkeypatch.setattr(cache, 'MODE', 'replay')
replayed = cache.wrap('openlib', lambda t, a: pytest.fail('replay hit the network'))
assert replayed('Dune', 'Herbert') == {'title': 'Dune'}
finally:
http.set_transport(None)

def test_a_body_that_is_not_json_is_retried_not_recorded(self, fixtures, monkeypatch):
monkeypatch.setattr(cache, 'MODE', 'record')
monkeypatch.setattr(http.time, 'sleep', lambda s: None)
bodies = iter([b'<html>503</html>', b'{"docs": [1]}'])
http.set_transport(lambda url, headers, timeout: next(bodies))
try:
assert http.get_json('https://example.test/search') == {'docs': [1]}
monkeypatch.setattr(cache, 'MODE', 'replay')
http.set_transport(lambda *_: pytest.fail('replay hit the network'))
assert http.get_json('https://example.test/search') == {'docs': [1]}
finally:
http.set_transport(None)

def test_a_parser_that_asks_a_new_url_falls_back_to_the_parsed_record(
self, fixtures, monkeypatch
):
monkeypatch.setattr(cache, 'MODE', 'record')
http.set_transport(lambda url, headers, timeout: b'{"title": "Dune"}')
try:
fetch = lambda t, a: http.get_json('https://example.test/v1/' + t) # noqa: E731
assert cache.wrap('openlib', fetch)('Dune', 'Herbert') == {'title': 'Dune'}
monkeypatch.setattr(cache, 'MODE', 'replay')
http.set_transport(lambda *_: pytest.fail('replay hit the network'))
moved = lambda t, a: http.get_json('https://example.test/v2/' + t) # noqa: E731
assert cache.wrap('openlib', moved)('Dune', 'Herbert') == {'title': 'Dune'}
finally:
http.set_transport(None)

def test_a_stored_body_the_parser_rejects_falls_back_to_the_parsed_record(
self, fixtures, monkeypatch
):
monkeypatch.setattr(cache, 'MODE', 'record')
http.set_transport(lambda url, headers, timeout: b'{"title": "Dune"}')
try:
fetch = lambda t, a: http.get_json('https://example.test/' + t) # noqa: E731
cache.wrap('openlib', fetch)('Dune', 'Herbert')
raw_file = cache.raw_path('http', 'https://example.test/Dune')
raw_file.write_text(json.dumps({'kind': 'http', 'key': 'x', 'body': '<html>'}))
monkeypatch.setattr(cache, 'MODE', 'replay')
http.set_transport(lambda *_: pytest.fail('replay hit the network'))
monkeypatch.setattr(http.time, 'sleep', lambda s: pytest.fail('replay slept'))
assert cache.wrap('openlib', fetch)('Dune', 'Herbert') == {'title': 'Dune'}
finally:
http.set_transport(None)

def test_a_plugin_answer_replays_without_calibre(self, fixtures, monkeypatch):
monkeypatch.setattr(cache, 'MODE', 'record')
monkeypatch.setattr(calibre_plugin, 'require_plugin', lambda plugin: None)
opf_text = (
'<package xmlns="http://www.idpf.org/2007/opf" '
'xmlns:dc="http://purl.org/dc/elements/1.1/">'
'<metadata><dc:title>Dune</dc:title></metadata></package>'
)
monkeypatch.setattr(calibre, 'fetch_metadata', lambda *a, **k: opf_text)
first = calibre_plugin.fetch_plugin('Dune', 'Herbert', 'Kobo Metadata')
assert first['title'] == 'Dune'

monkeypatch.setattr(cache, 'MODE', 'replay')
monkeypatch.setattr(calibre, 'fetch_metadata', lambda *a, **k: pytest.fail('ran calibre'))
monkeypatch.setattr(
calibre_plugin, 'require_plugin', lambda plugin: pytest.fail('probed calibre')
)
assert calibre_plugin.fetch_plugin('Dune', 'Herbert', 'Kobo Metadata') == first

def test_a_bad_minute_does_not_overwrite_a_good_record(self, fixtures, monkeypatch):
monkeypatch.setattr(cache, 'MODE', 'record')
cache.wrap('openlib', lambda t, a: {'title': 'Dune'})('Dune', 'Herbert')
failing = cache.wrap(
'openlib', lambda t, a: (_ for _ in ()).throw(enrich.SourceError('TLS timeout'))
)
with pytest.raises(enrich.SourceError):
failing('Dune', 'Herbert')

monkeypatch.setattr(cache, 'MODE', 'replay')
replayed = cache.wrap('openlib', lambda t, a: pytest.fail('replay hit the network'))
assert replayed('Dune', 'Herbert') == {'title': 'Dune'}

def test_the_record_names_the_raw_files_it_read(self, fixtures, monkeypatch):
monkeypatch.setattr(cache, 'MODE', 'record')
http.set_transport(lambda url, headers, timeout: b'{}')
try:
fetch = lambda t, a: http.get_json('https://example.test/' + t) # noqa: E731
cache.wrap('openlib', fetch)('Dune', 'Herbert')
finally:
http.set_transport(None)
saved = json.loads(cache.fixture_path('openlib', 'Dune', 'Herbert').read_text())
assert saved['raw'] == [cache.raw_path('http', 'https://example.test/Dune').name]


class TestAtLowNoSourceIdentifiedTheBook:
"""LOW means not one source cleared both the title and the author. Under
--include-low it may still offer subjects, but not an identifier."""
Expand Down
Loading