From 98d7de5d7dcec499745b9772bf0ece2e3ec8a8fe Mon Sep 17 00:00:00 2001
From: CrazyFreak <44674613+OffCrazyFreak@users.noreply.github.com>
Date: Wed, 16 Sep 2026 20:37:00 +0200
Subject: [PATCH 1/3] feat(library): Read filenames the way download sites and
library managers write them
Changes:
- Recognise the naming schemes measured in the wild (OceanofPDF, Anna's Archive, Z-Library, Library Genesis, PDFDrive, scene folders, sharing channels, ebook-tools, Standard Ebooks and dokumen.pub slugs, Springer, Calibre and LazyLibrarian's title-first form) and undo each one's encoding before splitting author from title
- Read a plain "A - B" author first as before, unless B reads more like a person, and carry the other reading along as FilenameFacts.alternate when both halves could be a person
- Try the other reading only when no source identifies the book as first read, and keep it only if a source then does; retry the head of a long title once when a subtitle's colon was dropped and the verdict is below HIGH
- Report a name that carries no title (a Gutenberg number, a bare ISBN, an ASIN) as such in the CLI and on the page instead of "no source answered"; print how a non-plain name was read
- Have the page ask the worker how every filename reads, so pending rows use the same parser as the verdicts, and show the reading and scheme in the detail view
- Record each catalogue once per row however many rounds ask it
- Document the audit and the decisions in docs/filenames.md
A file named the OceanofPDF way had no " - " in it, so the whole stem was read as an author with no title, nobody was asked, and the page said "no source answered", which was false. Renamed by hand the same three web sources reached HIGH. The user should not have to rename anything: half the tools out there write the title first and every site adds its own marks, and the parser now knows the shapes rather than one convention.
Notes:
- The safety model is untouched: matching.py, tags.py and the gain rules are unchanged. What changed is how the filename is read and how the catalogues are asked; the bar a source must clear is the same. A wrong reading cannot score, since a source would have to name a book whose title is the author's name and whose author is the title, twice over.
- Checked live against the two OceanofPDF pairs that started this (one HIGH, one LOW, the LOW being a summary edition the model rightly refuses), the same book named the Anna's Archive, Z-Library and Calibre ways (HIGH each), and replayed against the pristine sample with fixtures-v2 (Kobo, Google, Open Library) and fixtures-wide-raw (the web sources): every verdict, source list, gain and figure identical to before, no extra query asked.
- The head retry was narrowed after the replay showed it drawing a single volume out of Open Library for an omnibus, a strict prefix that the pre-existing prefix rule scores 0.95; it no longer cuts mid-phrase. That prefix rule itself is not new and is noted in docs/filenames.md.
---
README.md | 2 +
docs/README.md | 1 +
docs/filenames.md | 69 +++++++
src/ebook_metamend/cli.py | 14 +-
src/ebook_metamend/enrich.py | 86 ++++++++-
src/ebook_metamend/library.py | 306 ++++++++++++++++++++++++++++--
src/ebook_metamend/web.py | 15 +-
tests/test_library_and_opf.py | 227 ++++++++++++++++++++++
tests/test_pipeline.py | 146 +++++++++++++-
tests/test_web.py | 44 ++++-
web/src/App.tsx | 37 +++-
web/src/__tests__/client.test.ts | 14 ++
web/src/__tests__/run.test.ts | 10 +-
web/src/__tests__/use-run.test.ts | 11 +-
web/src/mock/books.ts | 2 +-
web/src/mock/simulate.ts | 5 +-
web/src/run/client.ts | 12 +-
web/src/run/use-run.ts | 19 +-
web/src/styles/blueprint.css | 4 +-
web/src/types.ts | 28 ++-
web/src/worker/metamend.worker.ts | 22 ++-
web/src/worker/protocol.ts | 4 +
22 files changed, 1025 insertions(+), 53 deletions(-)
create mode 100644 docs/filenames.md
diff --git a/README.md b/README.md
index cbf50d5..8940f15 100644
--- a/README.md
+++ b/README.md
@@ -111,6 +111,8 @@ Books are expected to be named `Author - Title - Subtitle.epub`, optionally with
└── Paulo Coelho - The Alchemist.epub
```
+Files that arrived from elsewhere are read the way their source wrote them, so nothing has to be renamed first: Calibre's `Title - Author`, `_OceanofPDF.com_Title_-_Author`, Anna's Archive's `Title -- Author -- ... -- Anna’s Archive`, Z-Library's `Title (Author) (z-lib.org)`, Library Genesis, PDFDrive, Standard Ebooks slugs and the rest of what [docs/filenames.md](docs/filenames.md) lists. A plain `A - B` that could be read either way is read author first, and if no catalogue identifies the book that way the other reading is tried, so the catalogues settle the order rather than a guess. A name that carries no title at all (`pg1342.epub`, a bare ISBN) is reported as such instead of "no source answered".
+
Naive string similarity fails on real book titles, so two cases are handled specially:
- **Prefix containment is legitimate.** "Digital Minimalism" vs "Digital Minimalism: Choosing a Focused Life in a Noisy World" is the same book, main title plus subtitle. Scored 0.95.
diff --git a/docs/README.md b/docs/README.md
index 66083fa..eff553c 100644
--- a/docs/README.md
+++ b/docs/README.md
@@ -3,4 +3,5 @@
Decisions and traps that the code cannot show. Nothing here is live state; anything with a date is a snapshot from that day.
- [sources.md](sources.md): which metadata sources were measured, what each is like, and why the web build uses a different set from the desktop tool.
+- [filenames.md](filenames.md): how stores, library managers and download sites actually name files, which shapes the parser recognises, and why the pipeline may read a name the other way round.
- [web.md](web.md): how the browser version works, why it is hosted the way it is, what each browser can do, and how to check a change.
diff --git a/docs/filenames.md b/docs/filenames.md
new file mode 100644
index 0000000..92cb07f
--- /dev/null
+++ b/docs/filenames.md
@@ -0,0 +1,69 @@
+# Filenames in the wild
+
+What ebook files are actually called when they arrive from a store, a library manager or a download site, measured on 2026-09-16, and how `library.parse_filename` reads each shape. The filename stays the ground truth; this page is about reading it correctly when another tool wrote it. Nothing here changes what a source must prove before a field is written.
+
+## Why
+
+A file named `_OceanofPDF.com_The_Quiet_Orchard_-_Mara_Voss.epub` has no ` - ` in it, so the old parser read the whole stem as an author with no title, asked nobody, and the page reported "no source answered". That report was false: no source was asked. Renamed to `Author - Title` the same three web sources reached HIGH. The user should not have to rename anything, so the parser now recognises the shapes below.
+
+## The shapes, with their sources
+
+Author first:
+
+- This tool's own convention, `Author - Title` and `Author - Series - 02 - Title` (README).
+- Readarr, default `{Author Name} - {Book Title}{ (PartNumber)}` (`NamingConfig.cs` in the Readarr repository).
+- Library Genesis mirrors, `Author - Title (2020, Publisher) - libgen.li`, also dotted `Author.-.Title.2020.Publisher.-.libgen.lc.1`, and `_ ` standing for `: ` inside a title (filenames quoted in GitHub issues; a third-party parser documents the `_ ` convention).
+- Scene release folders, `Author.Name.-.Title.Of.Book.2021.RETAIL.EPUB.eBook-GROUP` (shelfmark pull request 19).
+- IRC sharing channels, `Author - Title [RSC] (retail).epub` (the Shadow Libraries IRC guide).
+- ebook-tools output, `Author - [Series #1] - Title (2008) [ISBN].ext` (its README).
+- Standard Ebooks, `author-name_title-of-book.epub`, `_advanced.epub`, `.kepub.epub` (checked live on standardebooks.org).
+
+Title first:
+
+- Calibre "Save to disk" and "Send to device", defaults `{author_sort}/{title}/{title} - {authors}` and `{author_sort}/{title} - {authors}`, authors joined by ` & ` (`save_to_disk.py`). Calibre's own guess-from-filename regex assumes the same, `(?P
.+) - (?P[^_]+)` (`meta.py`, the manual).
+- Calibre-Web downloads, `Title - FirstAuthor.ext` (`cps/helper.py`).
+- LazyLibrarian, default `$Title - $Author` (`configdefs.py`).
+- Anna's Archive, `Title -- Author -- Edition, Year -- Publisher -- ISBN -- md5 -- Anna’s Archive.ext`, every field at most 60 characters, the whole at most 150, and every `.` in the name turned into `_` so `Mara T. Voss` arrives as `Mara T_ Voss` (`allthethings/page/views.py`). Empty fields are dropped, so the second field is not always the author.
+- Z-Library over the years, `Title (Author) (z-lib.org)`, `Title by Author (z-lib.org)`, `Title (Author)` followed by an em dash and `_Publisher_Language_ISBN (Z-Library)`, `Title (Last, First etc.) (z-library.sk, 1lib.sk, z-lib.sk)`. A colon in the title is dropped and leaves two spaces behind, which is how the subtitle boundary is recovered (filenames quoted in GitHub issues).
+- OceanofPDF, `_OceanofPDF.com_Title_-_Author.ext`, underscores for spaces, the colon dropped without trace, hyphens inside words kept (`Domain-Driven`) (three independent renaming scripts on GitHub and the files that started this).
+- Renaming tools, `Title by Author.ext` (ebook-rename's README).
+
+Title only:
+
+- PDFDrive, `Title ( PDFDrive ).pdf` and `Title ( PDFDrive.com ).pdf`, spaces inside the brackets, `_ ` for `: ` (filenames quoted on GitHub).
+- Kindle "Download & transfer via USB", the title alone; Kindle for PC, `ASIN_EBOK.azw` (DeDRM issues).
+- FanFicFare, default `${title}-${siteabbrev}_${storyId}` (`defaults.ini`).
+- dokumen.pub, vdoc.pub, epdf.pub slugs, `the-title-of-the-book-9780465050659-9780465003945-2013024417`, `-1nbsped-` for "1st ed.", `-4u9bqm2ndpq0` record ids, `epdf-pub-...-pdf` (their page URLs).
+- Springer, `2020_Book_IntroductionToScientificProgra.pdf`, CamelCase and cut at 30 characters (a Springer link in free-programming-books).
+- Publishers and Humble Bundle, `Title_With_Underscores.pdf` (widely seen, not separately sourced).
+- Scribd, `Document Title | PDF | Topic`.
+
+No title at all:
+
+- Project Gutenberg, `pg1342.epub`, `pg1342-images.epub`, `pg1342-images-3.epub` (checked live).
+- Internet Archive, `atomichabitseasy0000clea_lcp.epub` (checked live).
+- A bare ISBN, `978-1-4842-8853-5.pdf` (Springer's DOI links) or `9780465050659.epub`.
+
+Other observations that shaped the rules: a spaced en dash, a spaced em dash, ` -- ` and ` _ ` all appear as the separator between the same two halves; browsers append ` (1)` to a second download; Kobo files carry `.kepub.epub`; Kavita, by contrast, reads the OPF first and uses the filename only as a fallback, the opposite of this tool's premise.
+
+## What the parser does
+
+1. Strips the site's own marks (`_OceanofPDF.com_`, `(z-lib.org)`, `( PDFDrive )`, `- libgen.li`, `-- Anna’s Archive`, `(retail)`, `(v5.0)`, `(epub)`), a trailing `(Year)` or `(Year, Publisher)`, a bracketed ISBN, a duplicate-download counter and `.kepub`.
+2. Undoes the site's encoding: underscores or dots for spaces, `_ ` for `: `, `.-.` for ` - `, Anna's Archive's `_` for `.`, Z-Library's double space for `: `, slugs back into words, Springer's CamelCase into words, `Last, First` into `First Last` (never `Smith, Jr.`).
+3. Recognises the order when the scheme fixes it. A plain `A - B` is read author first, as the README asks, unless B reads more like a person than A (two or three capitalised words, an initial, no digits, no colon). When the name could be read either way and the other half could be a person at all, the other reading travels along as `FilenameFacts.alternate`.
+4. Series shapes from other tools are read too: `[Series #2]` as its own segment and `Title (Series Book 2)`.
+5. A name that carries no title (`pg1342`, an ISBN) is reported as such, in the CLI line and on the page, instead of "no source answered".
+
+`FilenameFacts.scheme` names what was recognised, so the CLI prints `read as Author / Title (scheme name)` and the page can say the same.
+
+## What the pipeline does with the other reading
+
+Only when no source identifies the book as first read does `enrich.propose` ask the catalogues about the alternate reading, and it keeps that reading only if a source then identifies the book. A wrong reading cannot score: a source would have to name a book whose title is the author's name and whose author is the title, and two of them would have to agree. The bar for writing is unchanged; the cost is one extra round of queries for a book that was going to be LOW anyway.
+
+Separately, when a verdict is below HIGH and the title looks like a subtitle glued on without its colon (`Quiet Orchard The Year Of Pruning`), the head of the title (`Quiet Orchard`) is asked once more and each source keeps whichever of its two answers fits the whole filename better. Measured live: Open Library returns nothing for the glued form and finds the book with the head. The cut is made only before a word a subtitle opens with (the, a, an, how, why, what), only when a phrase follows, and never after a preposition or conjunction, because cutting `The Happiest Baby On | The Block And The Happiest Toddler On The Block` drew the single volume out of Open Library, a strict prefix of the omnibus that the prefix rule scores 0.95, and the two-book bundle reached HIGH on its strength. That prefix rule predates this page and still applies to any source that answers with a strict prefix of the filename title; the retry no longer goes looking for one.
+
+## Checked against
+
+- The two OceanofPDF pairs that started this, live with the three web sources: one HIGH (Apple and Open Library agreeing, after the head retry), one LOW. The LOW is correct: that file is a publisher's summary edition of a well-known book, its own metadata names the summary publisher as the first author, and the author score of 0.26 against the original is the safety model refusing to dress a summary up as the book it summarises.
+- The same book renamed the Anna's Archive way, the Z-Library way and the Calibre way (`Title - Author`): HIGH each time, with the reading printed.
+- The pristine ten-book sample replayed against `fixtures-v2` (Kobo, Google, Open Library) and `fixtures-wide-raw` (the web sources): every verdict, source list, gain and figure identical to the code before this change, and no extra query asked.
diff --git a/src/ebook_metamend/cli.py b/src/ebook_metamend/cli.py
index b3e4be1..cf7ffc5 100644
--- a/src/ebook_metamend/cli.py
+++ b/src/ebook_metamend/cli.py
@@ -26,7 +26,13 @@ def _write_json(path: str, payload) -> None:
def _report(index: int, total: int, book: Book, proposal: Proposal | None) -> None:
if proposal is None:
- print(f'[{index}/{total}] {book.stem[:60]:<60} no source answered')
+ facts = book.facts()
+ why = 'no source answered'
+ if not facts.title:
+ why = 'filename names no title, expected "Author - Title"'
+ if facts.scheme:
+ why += f' ({facts.scheme} name)'
+ print(f'[{index}/{total}] {book.stem[:60]:<60} {why}')
return
sources = ','.join(proposal.sources)
@@ -36,6 +42,12 @@ def _report(index: int, total: int, book: Book, proposal: Proposal | None) -> No
f'fn={proposal.fn_score:.2f} au={proposal.au_score:.2f} '
f'src:{sources:<20} gains:{gains}'
)
+ if proposal.facts and (proposal.facts.scheme or proposal.facts.author != book.facts().author):
+ # The name was not the plain "Author - Title": say how it was read.
+ how = f' ({proposal.facts.scheme} name)' if proposal.facts.scheme else ''
+ print(
+ f' read as {proposal.facts.author or "(no author)"} / {proposal.facts.title}{how}'
+ )
if proposal.gains.get('title'):
was = (proposal.current.get('title') or '')[:40]
print(f" title {was} -> {proposal.gains['title'][:60]}")
diff --git a/src/ebook_metamend/enrich.py b/src/ebook_metamend/enrich.py
index cc7fd76..806dfe3 100644
--- a/src/ebook_metamend/enrich.py
+++ b/src/ebook_metamend/enrich.py
@@ -6,6 +6,7 @@
from __future__ import annotations
+import functools
import re
import sys
import time
@@ -14,7 +15,7 @@
from typing import Any
from . import calibre, matching, tags
-from .library import Book, books
+from .library import Book, FilenameFacts, books
from .sources import SOURCES, Pacer, Source, cache
from .sources.errors import SourceError, SourceUnavailable
from .writers import epub, pdf
@@ -56,6 +57,8 @@ class Proposal:
writes: list[tuple[str, bool, str]] = field(default_factory=list, repr=False)
#: Every source's score, including the ones that earned no say. Reporting only.
scores: list[matching.SourceScore] = field(default_factory=list, repr=False)
+ #: The reading of the filename the verdict was scored against. Reporting only.
+ facts: FilenameFacts | None = field(default=None, repr=False)
def to_dict(self) -> dict[str, Any]:
"""The serialised form. Explicit rather than ``asdict`` so that adding a
@@ -387,6 +390,53 @@ def compute_gains(merged: dict[str, Any], current: dict[str, Any], conf: str) ->
return gains
+#: Words a subtitle tends to open with when its colon has been lost.
+_SUBTITLE_STARTERS = frozenset({'the', 'a', 'an', 'how', 'why', 'what'})
+#: Words a title never ends on. A starter after one of these is mid-phrase
+#: ("Baby On | The Block", "Gone with | the Wind"), not a subtitle's first word.
+_PHRASE_WORDS = frozenset(
+ 'and or of on in to for with from at by the a an into onto over under about as'.split()
+)
+
+
+def _head(query: str) -> str:
+ """The title before a subtitle whose colon a download site dropped, or ''.
+
+ Cut only before a word a subtitle opens with, only when a phrase follows
+ it, and never mid-phrase, so "Thinking Fast and Slow", "Gone with the Wind"
+ and an omnibus "A On The Block And B On The Block" are left alone while
+ "Sapiens A Brief History of Humankind" becomes "Sapiens". Never the first
+ word: "The Design of Everyday Things" has no subtitle.
+ """
+ words = query.split()
+ for position, word in enumerate(words[1:], 1):
+ if word.casefold() not in _SUBTITLE_STARTERS:
+ continue
+ # A subtitle is a phrase of its own, and the title before it does not
+ # end mid-phrase. Measured: cutting "The Happiest Baby On The Block And
+ # The Happiest Toddler On The Block" at "On | The" drew the single volume
+ # out of Open Library, a strict prefix of the omnibus that the prefix
+ # rule scores 0.95, and the two-book bundle reached HIGH on its strength.
+ if len(words) - position >= 3 and words[position - 1].casefold() not in _PHRASE_WORDS:
+ return ' '.join(words[:position])
+ return ''
+
+
+def _better_answers(
+ first: dict[str, dict[str, Any]], second: dict[str, dict[str, Any]], facts
+) -> dict[str, dict[str, Any]]:
+ """Per source, whichever of its two answers fits the filename better."""
+ merged = dict(first)
+ for name, answer in second.items():
+ if name not in merged:
+ merged[name] = answer
+ continue
+ old, new = score({name: merged[name]}, facts)[0][0], score({name: answer}, facts)[0][0]
+ if (new.title_score, new.author_score) > (old.title_score, old.author_score):
+ merged[name] = answer
+ return merged
+
+
def propose(
book: Book,
*,
@@ -401,18 +451,37 @@ def propose(
sleep inside Python and waits between books in JavaScript instead.
"""
facts = book.facts()
- # A stem with no " - " parses as all author and no title, so every title
- # score would be 0.0 against an empty string and any answer at all would
- # look equally (un)related. There is nothing to score against, so do not ask.
+ # A stem that names no title (a Gutenberg number, a bare ISBN) leaves nothing
+ # to score against: every title score would be 0.0 against an empty string
+ # and any answer at all would look equally (un)related. So do not ask.
if not facts.query:
return None
- answers = query_sources(
- facts.query, facts.author, sources=sources, pause=pause, on_answer=on_answer
- )
+ ask = functools.partial(query_sources, sources=sources, pause=pause, on_answer=on_answer)
+ answers = ask(facts.query, facts.author)
+ scores, conf = score(answers, facts) if answers else ([], 'LOW')
+ if facts.alternate is not None and not any(s.strong for s in scores):
+ # Nothing identified the book as first read, and the name could be read
+ # the other way round: half the tools out there write the title first.
+ # The catalogues settle it. A wrong reading cannot score: a source would
+ # have to name a book whose title is the author's name and whose author
+ # is the title, twice over, so the bar for writing is unchanged.
+ other = facts.alternate
+ if other.query:
+ other_answers = ask(other.query, other.author)
+ other_scores, other_conf = score(other_answers, other) if other_answers else ([], 'LOW')
+ if any(s.strong for s in other_scores):
+ facts, answers, scores, conf = other, other_answers, other_scores, other_conf
+ if conf != 'HIGH' and (head := _head(facts.query)):
+ # A subtitle glued on without its colon ("Quiet Orchard The Year Of
+ # Pruning", as OceanofPDF writes it) is a query Open Library answers with
+ # nothing at all, measured live, while the head alone finds the book. The
+ # answers are still scored against the whole filename reading; a source
+ # only trades its answer for one that fits the filename better.
+ answers = _better_answers(answers, ask(head, facts.author), facts)
+ scores, conf = score(answers, facts) if answers else ([], 'LOW')
if not answers:
return None
- scores, conf = score(answers, facts)
trusted = trusted_names(scores)
title_score, author_score = reported_scores(scores, trusted)
surviving = {name: answers[name] for name in trusted}
@@ -437,6 +506,7 @@ def propose(
current=current or {},
unreadable=unreadable,
scores=scores,
+ facts=facts,
)
diff --git a/src/ebook_metamend/library.py b/src/ebook_metamend/library.py
index 5d7c6df..6887afe 100644
--- a/src/ebook_metamend/library.py
+++ b/src/ebook_metamend/library.py
@@ -9,7 +9,7 @@
import os
import re
-from dataclasses import dataclass, field
+from dataclasses import dataclass, field, replace
from .config import LIBRARY
@@ -18,8 +18,12 @@
#: "Series Name - 02.5 - Book Title". The number is the boundary: everything
#: before it names the series, everything after it is the book's own title.
_SERIES_PART = re.compile(r'^(?P.+?)\s*-\s*(?P\d+(?:\.\d+)?)\s*-\s*(?P.+)$')
+#: "[Series Name #2]" as its own segment, the shape ebook-tools writes.
+_SERIES_SEGMENT = re.compile(r'^\[(?P.+?)\s*#(?P\d+(?:\.\d+)?)\]$')
+#: "Title (Series Name Book 2)", the shape Amazon gives a title in a series.
+_SERIES_SUFFIX = re.compile(r'^(?P.+?)\s*\((?P[^()]+?)\s+Book\s+(?P\d+)\)$')
#: Where a search query should stop: a subtitle separator or a parenthesis.
-_QUERY_TAIL = re.compile(r'\s+-\s+|\s*:\s*|\s*\(')
+_QUERY_TAIL = re.compile(r'\s+-\s+|\s*:\s*|\s*[(\[]')
@dataclass(frozen=True)
@@ -37,18 +41,228 @@ class FilenameFacts:
series: str | None = None
#: Its position in that series, as written.
series_index: str | None = None
+ #: The naming scheme the stem was recognised as (see docs/filenames.md),
+ #: '' for the plain "Author - Title" this tool asks for.
+ scheme: str = ''
+ #: The same name read the other way round, present only when nothing in the
+ #: name says which side is the author. The pipeline falls back to it when no
+ #: source identifies the book as first read.
+ alternate: FilenameFacts | None = None
-def parse_filename(stem: str) -> FilenameFacts:
- """Split "Author - Series - 02 - Title" into the parts that mean something.
+# The shapes below were measured on real download sites and library managers,
+# not guessed; each rule names the scheme it exists for. docs/filenames.md has
+# the sources and the examples.
+
+#: "_OceanofPDF.com_Title_-_Author": underscores for spaces, title first.
+_OCEANOFPDF = re.compile(r'^_?oceanofpdf\.com_(?P.+)$', re.I)
+#: Z-Library over the years: "(z-lib.org)", "(Z-Library)", "(z-library.sk, 1lib.sk, z-lib.sk)".
+_ZLIBRARY = re.compile(r'\s*\((?:z-?lib(?:rary)?\b[^)]*|1lib\b[^)]*)\)\s*$', re.I)
+#: One Z-Library form glues publisher, language and ISBN on after an em dash.
+_ZLIBRARY_TAIL = re.compile(r'\u2014_.*$')
+#: "( PDFDrive )" and "( PDFDrive.com )", spaces inside the brackets and all.
+_PDFDRIVE = re.compile(r'\s*\(\s*pdfdrive(?:\.com)?\s*\)\s*$', re.I)
+#: "- libgen.li", ".-.libgen.lc", also with a duplicate counter: "libgen.lc.1".
+_LIBGEN = re.compile(r'(?:\s+-\s+|\.-\.|\s+)libgen\.(?:li|lc|rs|is|st|gs)(?:\.\d+)?$', re.I)
+#: "(2020)", "(2020, Manning Publications)", "[9780765333698]": trailing edition
+#: facts that name the printing, not the book.
+_TRAILING_YEAR = re.compile(r'\s*\((?:19|20)\d\d(?:\s*,\s*[^)]*)?\)\s*$')
+_TRAILING_ISBN = re.compile(r'\s*\[(?:97[89])?\d{9}[\dXx]\]\s*$')
+#: Format and quality tags from sharing channels: "(retail)", "(v5.0)", "(epub)".
+_TRAILING_TAG = re.compile(
+ r'(?:\s*[(\[](?:retail|v\d+(?:\.\d+)?|epub|mobi|azw3?|pdf|kindle|ebook)[)\]])+$', re.I
+)
+#: A browser's duplicate-download counter, "Name (1)"; also Readarr's part number.
+_DUPLICATE_COUNTER = re.compile(r'\s*\((?:[1-9]|1\d)\)$')
+#: Names that carry no title at all: a Project Gutenberg number, an Internet
+#: Archive identifier, an ISBN, a Kindle ASIN. Reported rather than searched.
+_NO_TITLE = (
+ ('gutenberg', re.compile(r'^pg\d+(?:-images)?(?:-\d)?$', re.I)),
+ ('internet-archive', re.compile(r'^[a-z0-9]+0000[a-z]{4}(?:_[a-z]+)?$')),
+ ('isbn', re.compile(r'^(?:97[89][- ]?)?\d(?:[- ]?\d){8}[- ]?[\dXx]$')),
+ ('kindle', re.compile(r'^B0[A-Z0-9]{8}_EBOK$')),
+)
+#: Springer: "2020_Book_IntroductionToScientificProgra", CamelCase and cut short.
+_SPRINGER = re.compile(r'^(?:19|20)\d\d_Book_(?P[A-Za-z0-9]+)$')
+_CAMEL_BOUNDARY = re.compile(r'(?<=[a-z0-9])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])')
+#: Slug tails that identify a listing rather than the book: ISBNs and record ids
+#: (dokumen.pub, vdoc.pub), "1nbsped" for "1st ed.", the format the site served.
+_SLUG_ID = re.compile(r'^(?:\d{7,}|(?=.*\d)[a-z0-9]{10,}|\d*nbsped|pdf|epub)$')
+#: Separators other tools write between the same two halves.
+_OTHER_SEPARATORS = re.compile(r'\s+(?:--|\u2013|\u2014|_)\s+')
+#: A comma-first person, "Zimmermann, Reinhard", as opposed to "Smith, Jr.".
+_NAME_SUFFIXES = frozenset({'jr', 'sr', 'ii', 'iii', 'iv', 'phd', 'md', 'esq'})
+#: Words that make a segment read as a title rather than a person.
+_TITLE_WORDS = frozenset(
+ 'the a an of to in for on from at by how why what your you my is are '
+ 'guide edition handbook introduction book vol volume'.split()
+)
+#: How a filename names several authors; commas are absent, "Smith, Jr." is one person.
+_CO_AUTHORS = re.compile(r'\s+(?:and|&)\s+')
+_LOWER_NAME_PARTS = frozenset(
+ {'van', 'von', 'de', 'da', 'del', 'di', 'la', 'le', 'du', 'der', 'den'}
+)
+
+
+def _person_first(author: str) -> str:
+ """ "Last, First" written the way the catalogues answer, "First Last".
+
+ Only the one-comma, short shape is turned round. "Smith, Jr." keeps its
+ comma, and so does anything long enough to be a list rather than a person.
+ """
+ author = re.sub(r'\s+(?:etc\.?|et al\.?)$', '', author.strip(), flags=re.I)
+ if author.count(',') != 1:
+ return author
+ last, first = (part.strip() for part in author.split(','))
+ if not last or not first or first.rstrip('.').casefold() in _NAME_SUFFIXES:
+ return author
+ if len(last.split()) > 2 or len(first.split()) > 3:
+ return author
+ return f'{first} {last}'
+
+
+def _name_likeness(segment: str) -> int:
+ """How much a segment reads like a person's name rather than a title.
+
+ A rough score, used only to pick which reading of "A - B" to try first; the
+ sources decide in the end. Two or three capitalised words, an initial, no
+ digits and no punctuation is what a name looks like on disk.
+ """
+ halves = _CO_AUTHORS.split(segment)
+ if len(halves) in (2, 3) and all(1 < len(h.split()) < 4 and h[:1].isupper() for h in halves):
+ # "Colin Bryar and Bill Carr": two names, not a five-word title.
+ return 4
+ words = segment.split()
+ score = 0
+ if len(words) in (2, 3):
+ score += 2
+ elif len(words) > 4:
+ score -= 2
+ if re.search(r'\b[A-Z]\.', segment):
+ score += 2
+ if re.search(r'\d', segment):
+ score -= 3
+ if re.search(r'[:?!]', segment):
+ score -= 3
+ lowered = {w.strip(".,'\"").casefold() for w in words}
+ if len(words) > 1 and lowered & _TITLE_WORDS:
+ score -= 2
+ if lowered & _NAME_SUFFIXES:
+ score += 1
+ if words and all(w[:1].isupper() or w.casefold() in _LOWER_NAME_PARTS for w in words):
+ score += 1
+ else:
+ score -= 1
+ return score
- The series is kept apart from the title rather than folded into it. Glued
- together they read "The Ravenhood Flock", which no catalogue has ever
- returned, so every book named this way scored 0.69 on the title and could
- never reach HIGH however exactly the sources agreed. Roughly one book in ten
- here is named that way.
+
+def _split(text: str) -> list[str]:
+ return [part.strip() for part in text.split(' - ') if part.strip()]
+
+
+def _words(slug: str) -> str:
+ return ' '.join(w.capitalize() for w in slug.split('-') if w)
+
+
+def _read(stem: str) -> tuple[list[str], str, str | None]:
+ """Undo what a download site or library manager did to the name.
+
+ Returns the name's segments, the scheme recognised, and which side is the
+ author when the scheme says so ('author-first', 'title-first') or None when
+ the segments could be read either way.
"""
- author, _, rest = stem.partition(' - ')
+ s = stem.strip()
+ # Kobo's "Title.kepub.epub" leaves ".kepub" on the stem.
+ s = re.sub(r'\.kepub$', '', s, flags=re.I)
+ s = _DUPLICATE_COUNTER.sub('', s)
+
+ for scheme, pattern in _NO_TITLE:
+ if pattern.match(s):
+ return [], scheme, None
+ match = _SPRINGER.match(s)
+ if match:
+ return [_CAMEL_BOUNDARY.sub(' ', match.group('title'))], 'springer', 'title-first'
+
+ match = _OCEANOFPDF.match(s)
+ if match:
+ text = match.group('rest').replace('_-_', ' - ').replace('_', ' ')
+ return _split(text), 'oceanofpdf', 'title-first'
+
+ if ' -- ' in s:
+ parts = [p.strip() for p in s.split(' -- ')]
+ if re.search(r"anna.s\s+archive", parts[-1], re.I):
+ parts.pop()
+ parts = [p.replace('_', '.') for p in parts if not re.fullmatch(r'[0-9a-f]{32}', p)]
+ title = parts[0] if parts else ''
+ # The author is the second field only if there was one: the site drops
+ # empty fields, so an edition or a publisher can move up into its place.
+ author = parts[1] if len(parts) > 1 else ''
+ if re.search(r'\b(?:19|20)\d\d\b|\bed(?:ition|\.)', author, re.I):
+ author = ''
+ return [title, author], 'annas-archive', 'title-first'
+
+ if _ZLIBRARY.search(s):
+ s = _ZLIBRARY_TAIL.sub('', _ZLIBRARY.sub('', s))
+ # A dropped colon leaves two spaces behind, so the subtitle boundary survives.
+ s = re.sub(r'(?<=\S) (?=\S)', ': ', s).strip()
+ match = re.match(r'^(?P.+?)\s+\((?P[^()]+)\)$', s)
+ if match and not re.search(r'\d|\bed(?:ition|\.)', match.group('author'), re.I):
+ return (
+ [match.group('title'), _person_first(match.group('author'))],
+ 'z-library',
+ 'title-first',
+ )
+ if ' by ' in s:
+ title, _, author = s.rpartition(' by ')
+ return [title, _person_first(author)], 'z-library', 'title-first'
+ return [s], 'z-library', 'title-first'
+
+ scheme, order = '', None
+ if _PDFDRIVE.search(s):
+ s, scheme, order = _PDFDRIVE.sub('', s), 'pdfdrive', 'title-first'
+ if _LIBGEN.search(s):
+ s, scheme, order = _LIBGEN.sub('', s), 'libgen', 'author-first'
+
+ if ' ' not in s and s.count('.') >= 3:
+ # Scene and dotted libgen names: "Author.Name.-.Title.Of.Book.2021.RETAIL.EPUB.eBook-GRP".
+ s = s.replace('.-.', ' - ').replace('.', ' ')
+ parts = _split(s)
+ if parts:
+ trimmed = re.sub(r'\s+(?:19|20)\d\d(?:\s+.*)?$', '', parts[-1])
+ parts[-1] = trimmed or parts[-1]
+ return parts, scheme or 'dotted', order or 'author-first'
+
+ if ' ' not in s and '-' in s and s == s.lower():
+ # Slugs: "author-name_title-of-book" (Standard Ebooks), "title-of-book-9780000000000" (dokumen.pub).
+ s = re.sub(r'^epdf-pub-', '', s)
+ s = re.sub(r'_advanced$', '', s)
+ if s.count('_') == 1 and all('-' in half or half.isalpha() for half in s.split('_')):
+ author, title = s.split('_')
+ return [_words(author), _words(title)], 'slug', 'author-first'
+ tokens = s.split('-')
+ while tokens and _SLUG_ID.match(tokens[-1]):
+ tokens.pop()
+ return [_words('-'.join(tokens))] if tokens else [], 'slug', 'title-first'
+
+ if ' ' not in s and '_' in s:
+ s = s.replace('_', ' ')
+ s = _OTHER_SEPARATORS.sub(' - ', s)
+ # libgen and PDFDrive write "_ " where the title had ": ".
+ s = re.sub(r'_\s', ': ', s).replace('_', ' ')
+ # Scribd: "Title | PDF | Topic".
+ s = s.split(' | ', 1)[0]
+ s = _TRAILING_TAG.sub('', s)
+ s = _TRAILING_ISBN.sub('', s)
+ s = _TRAILING_YEAR.sub('', s)
+ s = re.sub(r'\s+', ' ', s).strip(' -')
+
+ if ' - ' not in s and ' by ' in s:
+ title, _, author = s.rpartition(' by ')
+ return [title, author], scheme or 'title-by-author', 'title-first'
+ return _split(s), scheme, order
+
+
+def _facts(stem: str, author: str, rest: str, scheme: str) -> FilenameFacts:
series = index = None
match = _SERIES_PART.match(rest)
if match:
@@ -56,12 +270,82 @@ def parse_filename(stem: str) -> FilenameFacts:
index = match.group('index')
rest = match.group('title')
title = rest.strip()
+ match = _SERIES_SUFFIX.match(title)
+ if match and series is None:
+ series, index, title = match.group('series'), match.group('index'), match.group('title')
query = _QUERY_TAIL.split(title)[0].strip() or title
return FilenameFacts(
- stem=stem, author=author, title=title, query=query, series=series, series_index=index
+ stem=stem,
+ author=_person_first(author),
+ title=title,
+ query=query,
+ series=series,
+ series_index=index,
+ scheme=scheme,
)
+def _author_first(stem: str, parts: list[str], scheme: str) -> FilenameFacts:
+ author, rest = parts[0], parts[1:]
+ series = index = None
+ if rest and (match := _SERIES_SEGMENT.match(rest[0])):
+ series, index, rest = match.group('series'), match.group('index'), rest[1:]
+ facts = _facts(stem, author, ' - '.join(rest), scheme)
+ if series is not None and facts.series is None:
+ facts = replace(facts, series=series, series_index=index)
+ return facts
+
+
+def _title_first(stem: str, parts: list[str], scheme: str) -> FilenameFacts:
+ return _facts(stem, parts[-1], ' - '.join(parts[:-1]), scheme or 'title-first')
+
+
+def parse_filename(stem: str) -> FilenameFacts:
+ """Read "Author - Title" out of a stem, however a download site mangled it.
+
+ "Author - Series - 02 - Title" keeps the series apart from the title rather
+ than folding it in. Glued together they read "The Ravenhood Flock", which no
+ catalogue has ever returned, so every book named this way scored 0.69 on the
+ title and could never reach HIGH however exactly the sources agreed.
+
+ Half the tools out there write the title first (Calibre, Calibre-Web,
+ LazyLibrarian, Anna's Archive, Z-Library, OceanofPDF) and half the author
+ first (this tool, Readarr, libgen, the sharing channels). Where the scheme is
+ recognisable the order is known. A plain "A - B" is read author first, as
+ the README asks, unless B reads more like a person than A; the other reading
+ travels along as ``alternate`` for the pipeline to try when the first finds
+ nothing, so the catalogues settle it rather than a guess.
+ """
+ parts, scheme, order = _read(stem)
+ if not parts:
+ return FilenameFacts(stem=stem, author='', title='', query='', scheme=scheme)
+ if len(parts) == 1:
+ # A title and nobody's name: nothing for a source's author to be scored
+ # against, so it can never reach HIGH, but it is still worth reporting.
+ return _facts(stem, '', parts[0], scheme)
+ if scheme == 'title-by-author':
+ # "Death by Black Hole" is a title with nobody's name in it.
+ whole = _facts(stem, '', ' by '.join(parts), '')
+ return replace(_title_first(stem, parts, scheme), alternate=whole)
+ if order == 'title-first':
+ return _title_first(stem, parts, scheme)
+ if order == 'author-first':
+ return _author_first(stem, parts, scheme)
+ author_first = _author_first(stem, parts, scheme)
+ if author_first.series is not None:
+ # "Author - Series - 02 - Title" is this tool's own convention; no other reading fits.
+ return author_first
+ head, tail = _name_likeness(parts[0]), _name_likeness(parts[-1])
+ # The other reading travels along only when the half it would make the
+ # author could be a person at all; a subtitle or a product name cannot, and
+ # asking the catalogues about it would cost a round for nothing.
+ if tail > head:
+ other = author_first if head >= 0 else None
+ return replace(_title_first(stem, parts, scheme), alternate=other)
+ other = _title_first(stem, parts, scheme) if tail >= 0 else None
+ return replace(author_first, alternate=other)
+
+
@dataclass
class Book:
stem: str
diff --git a/src/ebook_metamend/web.py b/src/ebook_metamend/web.py
index 066d48f..c3e21c4 100644
--- a/src/ebook_metamend/web.py
+++ b/src/ebook_metamend/web.py
@@ -13,7 +13,7 @@
from typing import Any
from . import enrich
-from .library import Book, parse_filename
+from .library import Book, FilenameFacts, parse_filename
from .sources import WEB_SOURCE_NAMES, http, select
WEB_SOURCES = select(WEB_SOURCE_NAMES)
@@ -79,17 +79,21 @@ def _remove(book: Book, root: str) -> None:
folder = os.path.dirname(folder)
-def facts(stem: str) -> dict[str, Any]:
- """The filename's claims, in the shape the page shows."""
- f = parse_filename(os.path.basename(stem))
+def _facts(f: FilenameFacts) -> dict[str, Any]:
return {
'author': f.author,
'title': f.title,
'series': f.series,
'series_index': f.series_index,
+ 'scheme': f.scheme,
}
+def facts(stem: str) -> dict[str, Any]:
+ """The filename's claims, in the shape the page shows."""
+ return _facts(parse_filename(os.path.basename(stem)))
+
+
def propose(
stem: str,
files: dict[str, bytes],
@@ -116,6 +120,9 @@ def propose(
'files': {ext: os.path.basename(path) for ext, path in proposal.files.items()},
'current': proposal.current,
'unreadable': proposal.unreadable,
+ # The reading the verdict was scored against, which may be the name the
+ # other way round from what the page showed while it waited.
+ 'facts': _facts(proposal.facts) if proposal.facts else facts(stem),
'scores': [
{
'name': s.name,
diff --git a/tests/test_library_and_opf.py b/tests/test_library_and_opf.py
index 1c97539..01bdcd7 100644
--- a/tests/test_library_and_opf.py
+++ b/tests/test_library_and_opf.py
@@ -49,6 +49,233 @@ def test_query_never_ends_up_empty(self):
assert facts.query
+class TestNamesFromTheWild:
+ """How download sites and library managers actually name files, measured
+ (docs/filenames.md). Each shape is read the way its own tool wrote it, and
+ the invented book behind every example is Mara Voss's "The Quiet Orchard".
+ """
+
+ @pytest.mark.parametrize(
+ 'stem, scheme, author, title',
+ [
+ # OceanofPDF: underscores for spaces, title first, "_-_" between them.
+ (
+ '_OceanofPDF.com_The_Quiet_Orchard_-_Mara_Voss',
+ 'oceanofpdf',
+ 'Mara Voss',
+ 'The Quiet Orchard',
+ ),
+ (
+ 'OceanofPDF.com_Slow-Grown_Fruit_-_Mara_Voss',
+ 'oceanofpdf',
+ 'Mara Voss',
+ 'Slow-Grown Fruit',
+ ),
+ # Anna's Archive: " -- " between fields, every "." turned into "_".
+ (
+ 'The Quiet Orchard -- Mara T_ Voss -- Hill Press, 2011 -- Hill Press -- 9781594488849 -- '
+ '0123456789abcdef0123456789abcdef -- Anna’s Archive',
+ 'annas-archive',
+ 'Mara T. Voss',
+ 'The Quiet Orchard',
+ ),
+ # Z-Library over the years, the dropped colon leaving two spaces behind.
+ (
+ 'The Quiet Orchard a year of pruning (Voss, Mara) (z-lib.org)',
+ 'z-library',
+ 'Mara Voss',
+ 'The Quiet Orchard: a year of pruning',
+ ),
+ (
+ 'The Quiet Orchard (Mara Voss) (Z-Library)',
+ 'z-library',
+ 'Mara Voss',
+ 'The Quiet Orchard',
+ ),
+ (
+ 'The Quiet Orchard (Voss, Mara etc.) (z-library.sk, 1lib.sk, z-lib.sk)',
+ 'z-library',
+ 'Mara Voss',
+ 'The Quiet Orchard',
+ ),
+ (
+ 'The Quiet Orchard by Mara Voss (z-lib.org)',
+ 'z-library',
+ 'Mara Voss',
+ 'The Quiet Orchard',
+ ),
+ (
+ 'The Quiet Orchard (Mara Voss)\u2014_Hill Press_English_9781594488849 (Z-Library)',
+ 'z-library',
+ 'Mara Voss',
+ 'The Quiet Orchard',
+ ),
+ # PDFDrive: title only, "_ " where a colon was.
+ ('The Quiet Orchard ( PDFDrive )', 'pdfdrive', '', 'The Quiet Orchard'),
+ (
+ 'Orchards_ The Quiet Ones ( PDFDrive.com )',
+ 'pdfdrive',
+ '',
+ 'Orchards: The Quiet Ones',
+ ),
+ # Library Genesis, spaced and dotted.
+ (
+ 'Mara Voss - The Quiet Orchard (2011, Hill Press) - libgen.li',
+ 'libgen',
+ 'Mara Voss',
+ 'The Quiet Orchard',
+ ),
+ (
+ 'Mara.Voss.-.The.Quiet.Orchard.2011.Hill.Press.-.libgen.lc.1',
+ 'libgen',
+ 'Mara Voss',
+ 'The Quiet Orchard',
+ ),
+ ('Mara Voss - The Quiet Orch - libgen.li', 'libgen', 'Mara Voss', 'The Quiet Orch'),
+ # Scene release folders.
+ (
+ 'Mara.Voss.-.The.Quiet.Orchard.2011.RETAIL.EPUB.eBook-GRP',
+ 'dotted',
+ 'Mara Voss',
+ 'The Quiet Orchard',
+ ),
+ # Sharing channels and ebook-tools output.
+ (
+ 'Mara Voss - The Quiet Orchard [RSC] (retail)',
+ '',
+ 'Mara Voss',
+ 'The Quiet Orchard [RSC]',
+ ),
+ (
+ 'Mara Voss - The Quiet Orchard (2011) [9781594488849]',
+ '',
+ 'Mara Voss',
+ 'The Quiet Orchard',
+ ),
+ # Slugs: Standard Ebooks, dokumen.pub, vdoc.pub, epdf.pub.
+ ('mara-voss_the-quiet-orchard', 'slug', 'Mara Voss', 'The Quiet Orchard'),
+ ('mara-voss_the-quiet-orchard_advanced', 'slug', 'Mara Voss', 'The Quiet Orchard'),
+ (
+ 'the-quiet-orchard-9781594488849-9781594488856-2011024417',
+ 'slug',
+ '',
+ 'The Quiet Orchard',
+ ),
+ ('the-quiet-orchard-1nbsped-9781594488849', 'slug', '', 'The Quiet Orchard'),
+ ('the-quiet-orchard-4u9bqm2ndpq0', 'slug', '', 'The Quiet Orchard'),
+ ('epdf-pub-the-quiet-orchard-pdf', 'slug', '', 'The Quiet Orchard'),
+ # Springer: CamelCase after the year, cut short by the site.
+ (
+ '2011_Book_TheQuietOrchardAYearOfPrun',
+ 'springer',
+ '',
+ 'The Quiet Orchard A Year Of Prun',
+ ),
+ # Underscores for spaces and nothing else (Humble Bundle, publishers).
+ ('The_Quiet_Orchard_2nd_Edition', '', '', 'The Quiet Orchard 2nd Edition'),
+ # "Title by Author" (Z-Library once, renaming tools still).
+ ('The Quiet Orchard by Mara Voss', 'title-by-author', 'Mara Voss', 'The Quiet Orchard'),
+ # Other separators between the same two halves.
+ ('Mara Voss \u2013 The Quiet Orchard', '', 'Mara Voss', 'The Quiet Orchard'),
+ ('Mara Voss \u2014 The Quiet Orchard', '', 'Mara Voss', 'The Quiet Orchard'),
+ ('Mara Voss _ The Quiet Orchard', '', 'Mara Voss', 'The Quiet Orchard'),
+ # Kobo's double extension, a browser's duplicate counter, Scribd's title bar.
+ ('Mara Voss - The Quiet Orchard.kepub', '', 'Mara Voss', 'The Quiet Orchard'),
+ ('Mara Voss - The Quiet Orchard (1)', '', 'Mara Voss', 'The Quiet Orchard'),
+ ('The Quiet Orchard | PDF | Gardening', '', '', 'The Quiet Orchard'),
+ # A title and nobody's name.
+ ('The Quiet Orchard', '', '', 'The Quiet Orchard'),
+ ],
+ )
+ def test_is_read_the_way_its_tool_wrote_it(self, stem, scheme, author, title):
+ facts = parse_filename(stem)
+ assert (facts.scheme, facts.author, facts.title) == (scheme, author, title)
+
+ @pytest.mark.parametrize(
+ 'stem, scheme',
+ [
+ ('pg1342-images-3', 'gutenberg'),
+ ('quietorchardyea0000voss_lcp', 'internet-archive'),
+ ('978-1-59448-884-9', 'isbn'),
+ ('9781594488849', 'isbn'),
+ ('B00KYB2XAA_EBOK', 'kindle'),
+ ],
+ )
+ def test_a_name_that_carries_no_title_says_so(self, stem, scheme):
+ facts = parse_filename(stem)
+ assert facts.title == ''
+ assert facts.query == ''
+ assert facts.scheme == scheme
+ assert facts.alternate is None
+
+ def test_a_plain_name_is_read_author_first_with_the_other_way_kept(self):
+ facts = parse_filename('Mara Voss - Quiet Orchard')
+ assert (facts.author, facts.title) == ('Mara Voss', 'Quiet Orchard')
+ assert facts.alternate is not None
+ assert (facts.alternate.author, facts.alternate.title) == ('Quiet Orchard', 'Mara Voss')
+ assert facts.alternate.alternate is None
+
+ @pytest.mark.parametrize(
+ 'stem',
+ [
+ 'Mara Voss - The Quiet Orchard: A Year of Pruning',
+ 'Mara Voss - Some Product Guide for Version 4 Cloud and Beyond',
+ 'The Quiet Orchard: A Year of Pruning - Mara Voss',
+ ],
+ )
+ def test_a_half_that_cannot_be_a_person_leaves_no_other_reading(self, stem):
+ facts = parse_filename(stem)
+ assert facts.author == 'Mara Voss'
+ assert facts.alternate is None
+
+ @pytest.mark.parametrize(
+ 'stem',
+ [
+ # A subtitle, an article or a long run of words reads as a title.
+ 'The Quiet Orchard: A Year of Pruning - Mara T. Voss',
+ 'The Quiet Orchard - Mara T. Voss',
+ 'Quiet Orchards of the North - Mara Voss',
+ 'Orchard - Mara Voss',
+ ],
+ )
+ def test_a_name_whose_last_half_reads_as_a_person_is_read_title_first(self, stem):
+ facts = parse_filename(stem)
+ assert facts.author == facts.stem.split(' - ')[-1]
+ assert facts.scheme == 'title-first'
+
+ def test_two_names_joined_by_and_are_one_author(self):
+ facts = parse_filename('Colin Bryar and Bill Carr - Working Backwards')
+ assert facts.author == 'Colin Bryar and Bill Carr'
+
+ def test_a_series_name_has_no_other_reading(self):
+ assert parse_filename('Mara Voss - Hill Country - 02 - The Quiet Orchard').alternate is None
+
+ def test_series_shapes_from_other_tools(self):
+ facts = parse_filename(
+ 'Mara Voss - [Hill Country #2] - The Quiet Orchard (2011) [9781594488849]'
+ )
+ assert (facts.series, facts.series_index, facts.title) == (
+ 'Hill Country',
+ '2',
+ 'The Quiet Orchard',
+ )
+ facts = parse_filename('Mara Voss - The Quiet Orchard (Hill Country Book 2)')
+ assert (facts.series, facts.series_index, facts.title) == (
+ 'Hill Country',
+ '2',
+ 'The Quiet Orchard',
+ )
+
+ def test_a_title_containing_by_keeps_the_whole_as_a_second_reading(self):
+ facts = parse_filename('Death by Black Hole')
+ assert facts.alternate is not None
+ assert (facts.alternate.author, facts.alternate.title) == ('', 'Death by Black Hole')
+
+ def test_last_comma_first_is_turned_round_but_a_suffix_is_not(self):
+ assert parse_filename('Voss, Mara - The Quiet Orchard').author == 'Mara Voss'
+ assert parse_filename('Smith, Jr. - The Quiet Orchard').author == 'Smith, Jr.'
+
+
class TestWalk:
@pytest.fixture
def library(self, tmp_path):
diff --git a/tests/test_pipeline.py b/tests/test_pipeline.py
index e2828e8..fea4aa1 100644
--- a/tests/test_pipeline.py
+++ b/tests/test_pipeline.py
@@ -176,12 +176,150 @@ def test_and_is_replaced_at_high(self):
class TestNothingIsWrittenWithoutSomethingToScoreAgainst:
- def test_a_filename_with_no_author_separator_is_skipped(self, tmp_path):
- """ "Dune" parses as all author and no title, so every score would be 0.0
- against an empty string and any answer would look equally related."""
+ def test_a_filename_with_no_title_is_skipped(self, tmp_path):
+ """A bare ISBN names no title, so every score would be 0.0 against an
+ empty string and any answer would look equally related."""
+ book = tmp_path / '9780465050659.epub'
+ book.write_bytes(b'')
+ assert enrich.propose(Book(stem='9780465050659', formats={'.epub': str(book)})) is None
+
+ def test_a_title_with_no_author_can_never_be_strong(self, tmp_path, monkeypatch):
+ """ "Dune" is a title and nobody's name. A source naming the book exactly
+ still has no author to be checked against, so it cannot vouch for the
+ book on its own and the verdict stays LOW."""
book = tmp_path / 'Dune.epub'
book.write_bytes(b'')
- assert enrich.propose(Book(stem='Dune', formats={'.epub': str(book)})) is None
+ exact = enrich.SOURCES[0].__class__(
+ name='exact',
+ fetch=lambda t, a: {'title': 'Dune', 'authors': ['Frank Herbert']},
+ pause=0,
+ )
+ monkeypatch.setattr(enrich, 'SOURCES', (exact,))
+ monkeypatch.setattr(calibre, 'read_book_metadata', lambda path: {})
+ enrich.reset_run_state()
+ proposal = enrich.propose(Book(stem='Dune', formats={'.epub': str(book)}), pause=False)
+ assert proposal is not None
+ assert proposal.conf == 'LOW'
+ assert proposal.au_score == 0.0
+ enrich.reset_run_state()
+
+
+class TestTheOtherReadingOfANameIsTriedWhenTheFirstFindsNothing:
+ """Half the tools out there write the title first (Calibre, Anna's Archive,
+ Z-Library), half the author first (this tool, Readarr, libgen). The
+ catalogues settle it, not a guess."""
+
+ @pytest.fixture
+ def catalogue(self, monkeypatch, tmp_path):
+ asked = []
+
+ def fetch(title, author):
+ asked.append((title, author))
+ # The catalogue knows one book and answers only a query that names it.
+ if title in ('The Quiet Orchard', 'Quiet Orchard') and author == 'Mara Voss':
+ return {'title': title, 'authors': ['Mara Voss']}
+ return None
+
+ one = enrich.SOURCES[0].__class__(name='one', fetch=fetch, pause=0)
+ two = enrich.SOURCES[0].__class__(name='two', fetch=fetch, pause=0)
+ monkeypatch.setattr(enrich, 'SOURCES', (one, two))
+ monkeypatch.setattr(calibre, 'read_book_metadata', lambda path: {})
+ enrich.reset_run_state()
+ yield asked
+ enrich.reset_run_state()
+
+ def _book(self, tmp_path, stem):
+ path = tmp_path / f'{stem}.epub'
+ path.write_bytes(b'')
+ return Book(stem=stem, formats={'.epub': str(path)})
+
+ def test_a_title_first_name_reaches_high_on_the_second_reading(self, catalogue, tmp_path):
+ """ "Quiet Orchard - Mara Voss" reads author first by the tool's own
+ convention; the catalogues know it the other way round."""
+ proposal = enrich.propose(self._book(tmp_path, 'Quiet Orchard - Mara Voss'), pause=False)
+ assert proposal is not None
+ assert proposal.conf == 'HIGH'
+ assert proposal.facts.author == 'Mara Voss'
+ assert proposal.facts.title == 'Quiet Orchard'
+ assert catalogue[:2] == [('Mara Voss', 'Quiet Orchard')] * 2
+
+ def test_an_author_first_name_costs_one_round(self, catalogue, tmp_path):
+ proposal = enrich.propose(
+ self._book(tmp_path, 'Mara Voss - The Quiet Orchard'), pause=False
+ )
+ assert proposal is not None
+ assert proposal.conf == 'HIGH'
+ assert catalogue == [('The Quiet Orchard', 'Mara Voss')] * 2
+
+ def test_the_second_reading_is_not_taken_when_it_finds_nothing_either(
+ self, catalogue, tmp_path
+ ):
+ proposal = enrich.propose(self._book(tmp_path, 'Some Other Book - Ann Person'), pause=False)
+ assert proposal is None
+ # Both readings were tried, once per source.
+ assert len(catalogue) == 4
+
+ def test_a_half_that_cannot_be_a_person_is_never_asked_as_one(self, catalogue, tmp_path):
+ """A subtitle or a product name makes no author, so the other reading is
+ not worth a round of queries."""
+ stem = 'Ann Person - Some Product Guide for Version 4 Cloud and Beyond'
+ enrich.propose(self._book(tmp_path, stem), pause=False)
+ assert len(catalogue) == 2
+
+ def test_a_lost_subtitle_colon_is_worked_around(self, monkeypatch, tmp_path):
+ """OceanofPDF drops the colon, so the whole subtitle rides along in the
+ query and Open Library finds nothing; the head of the title alone does.
+ Measured live on a real file."""
+ asked = []
+
+ def fussy(title, author):
+ asked.append(title)
+ return (
+ {'title': 'Quiet Orchard', 'authors': ['Mara Voss']}
+ if title == 'Quiet Orchard'
+ else None
+ )
+
+ def easy(title, author):
+ return {'title': 'Quiet Orchard: The Year Of Pruning', 'authors': ['Mara Voss']}
+
+ one = enrich.SOURCES[0].__class__(name='fussy', fetch=fussy, pause=0)
+ two = enrich.SOURCES[0].__class__(name='easy', fetch=easy, pause=0)
+ monkeypatch.setattr(enrich, 'SOURCES', (one, two))
+ monkeypatch.setattr(calibre, 'read_book_metadata', lambda path: {})
+ enrich.reset_run_state()
+ stem = '_OceanofPDF.com_Quiet_Orchard_The_Year_Of_Pruning_-_Mara_Voss'
+ proposal = enrich.propose(self._book(tmp_path, stem), pause=False)
+ assert proposal is not None
+ assert proposal.conf == 'HIGH'
+ assert sorted(proposal.sources) == ['easy', 'fussy']
+ assert asked == ['Quiet Orchard The Year Of Pruning', 'Quiet Orchard']
+ enrich.reset_run_state()
+
+ @pytest.mark.parametrize(
+ 'query, head',
+ [
+ ('Sapiens A Brief History of Humankind', 'Sapiens'),
+ ('The Happiest Baby On The Block', ''),
+ ('The Happiest Baby On The Block And The Happiest Toddler On The Block', ''),
+ ('Gone with the Wind and Other Stories', ''),
+ ('The Power of Now A Guide to Spiritual Enlightenment', 'The Power of Now'),
+ ('Atomic Habits An Easy And Proven Way', 'Atomic Habits'),
+ ('The Design of Everyday Things', ''),
+ ('Thinking Fast and Slow', ''),
+ ('Zero to One', ''),
+ ],
+ )
+ def test_the_head_of_a_title_is_cut_only_before_a_subtitle(self, query, head):
+ assert enrich._head(query) == head
+
+ def test_a_series_name_has_one_reading(self, catalogue, tmp_path):
+ """ "Author - Series - 02 - Title" is this tool's own convention; the
+ other way round fits nothing, so it is never asked."""
+ enrich.propose(
+ self._book(tmp_path, 'Ann Person - Hill Country - 02 - Some Book'), pause=False
+ )
+ assert catalogue == [('Some Book', 'Ann Person')] * 2
class TestTheApplyGate:
diff --git a/tests/test_web.py b/tests/test_web.py
index 9d96d63..930bb08 100644
--- a/tests/test_web.py
+++ b/tests/test_web.py
@@ -108,10 +108,29 @@ def test_a_folder_in_the_stem_is_kept_apart(self, tmp_path):
assert web.facts('Fiction/Mara Voss - The Quiet Orchard')['author'] == 'Mara Voss'
def test_a_stem_without_a_title_asks_nobody(self, tmp_path, _fake_catalogues):
- assert (
- web.propose('Just A Name', {'.epub': epub_bytes(tmp_path)}, str(tmp_path / 'l')) is None
- )
+ """A Project Gutenberg number names no title, so there is nothing to score
+ a catalogue's answer against."""
+ assert web.propose('pg1342', {'.epub': epub_bytes(tmp_path)}, str(tmp_path / 'l')) is None
assert _fake_catalogues == []
+ assert web.facts('pg1342') == {
+ 'author': '',
+ 'title': '',
+ 'series': None,
+ 'series_index': None,
+ 'scheme': 'gutenberg',
+ }
+
+ def test_a_title_with_no_author_is_asked_but_cannot_be_written(
+ self, tmp_path, _fake_catalogues
+ ):
+ """Nothing in the name vouches for an author, so no source can be strong
+ and the verdict stays LOW; the visitor still sees what was found."""
+ result = web.propose(
+ 'The Quiet Orchard', {'.epub': epub_bytes(tmp_path)}, str(tmp_path / 'l')
+ )
+ assert result is not None
+ assert result['conf'] == 'LOW'
+ assert result['facts']['author'] == ''
def test_facts_follow_the_filename_parser(self):
assert web.facts('Mara Voss - Hill Country - 02 - The Quiet Orchard') == {
@@ -119,6 +138,25 @@ def test_facts_follow_the_filename_parser(self):
'title': 'The Quiet Orchard',
'series': 'Hill Country',
'series_index': '02',
+ 'scheme': '',
+ }
+
+ def test_the_facts_returned_are_the_reading_that_earned_the_verdict(
+ self, tmp_path, _fake_catalogues
+ ):
+ """Calibre writes the title first. The page showed the name read author
+ first while it waited; the verdict comes back with the reading that fit."""
+ result = web.propose(
+ 'The Quiet Orchard - Mara Voss', {'.epub': epub_bytes(tmp_path)}, str(tmp_path / 'l')
+ )
+ assert result is not None
+ assert result['conf'] == 'HIGH'
+ assert result['facts'] == {
+ 'author': 'Mara Voss',
+ 'title': 'The Quiet Orchard',
+ 'series': None,
+ 'series_index': None,
+ 'scheme': 'title-first',
}
diff --git a/web/src/App.tsx b/web/src/App.tsx
index 59b174f..488136f 100644
--- a/web/src/App.tsx
+++ b/web/src/App.tsx
@@ -34,25 +34,30 @@ import {
type Selection,
} from '@/selection'
import {
+ SCHEME_LABEL,
SOURCE_LABEL,
verdict,
willWrite,
type BookResult,
- type Confidence,
type Metadata,
type SourceName,
+ type Verdict,
} from '@/types'
-type Verdict = Confidence | 'NONE' | 'UNREADABLE'
-
const VERDICT_LABEL: Record = {
HIGH: 'HIGH',
MED: 'MED',
LOW: 'LOW',
NONE: 'NO ANSWER',
+ UNTITLED: 'NO TITLE',
UNREADABLE: 'UNREADABLE',
}
+// A name that carries no title is shown as the file it is.
+function shown(book: BookResult): string {
+ return book.facts.title || book.stem.slice(book.stem.lastIndexOf('/') + 1)
+}
+
const EASE = [0.2, 0.8, 0.2, 1] as const
export function App() {
@@ -915,7 +920,7 @@ function Summary({
type="button"
className="bp-segment block h-full w-full"
data-verdict={b.status === 'done' ? verdict(b) : 'PENDING'}
- aria-label={`${b.facts.title}: ${b.status === 'done' ? VERDICT_LABEL[verdict(b)] : 'not checked yet'}`}
+ aria-label={`${shown(b)}: ${b.status === 'done' ? VERDICT_LABEL[verdict(b)] : 'not checked yet'}`}
onMouseEnter={() => setHover(i)}
onMouseLeave={() => setHover(null)}
onFocus={() => setHover(i)}
@@ -938,7 +943,7 @@ function Summary({
: 'translateX(-50%)',
}}
>
- {books[hover].facts.title}
+ {shown(books[hover])}
)}
@@ -1020,7 +1025,7 @@ function Results({
{done ? (
onToggle(book)}
@@ -1038,7 +1043,7 @@ function Results({
{String(i + 1).padStart(2, '0')}
- {book.facts.title}
+ {shown(book)}
{book.facts.author}
{book.facts.series && ` · ${book.facts.series} ${book.facts.series_index}`}
@@ -1386,14 +1391,28 @@ function Detail({
sheet detail
- {book.facts.title}
+ {shown(book)}
{book.facts.author}
{book.facts.series && ` · ${book.facts.series} ${book.facts.series_index}`}
+ {book.facts.scheme && book.facts.title && (
+
+ Read as {SCHEME_LABEL[book.facts.scheme] ?? book.facts.scheme}:{' '}
+ {book.facts.author || 'no author'} /{' '}
+ {book.facts.title}
+
+ )}
- {p === null ? (
+ {p === null && !book.facts.title ? (
+
+ The file name carries no title to look up
+ {book.facts.scheme && ` (${SCHEME_LABEL[book.facts.scheme] ?? book.facts.scheme})`},
+ so no catalogue was asked. Name the file{' '}
+ Author - Title and check it again.
+
+ ) : p === null ? (
No catalogue answered for this filename. Nothing is proposed.
diff --git a/web/src/__tests__/client.test.ts b/web/src/__tests__/client.test.ts
index c8d5e1b..f422fd6 100644
--- a/web/src/__tests__/client.test.ts
+++ b/web/src/__tests__/client.test.ts
@@ -115,6 +115,20 @@ describe('Client', () => {
await expect(pending).resolves.toMatchObject({ id: 1, pause: 3, unavailable: ['openlib'] })
})
+ it('asks the worker how each filename reads', async () => {
+ const { client, worker } = make()
+ const pending = client.facts(['A - B', 'pg1342'])
+ expect(worker.posted[1].message).toMatchObject({
+ type: 'facts',
+ id: 1,
+ stems: ['A - B', 'pg1342'],
+ })
+ const read = { author: 'A', title: 'B', series: null, series_index: null, scheme: '' }
+ const bare = { author: '', title: '', series: null, series_index: null, scheme: 'gutenberg' }
+ worker.reply({ type: 'facts', id: 1, facts: [read, bare] })
+ await expect(pending).resolves.toEqual([read, bare])
+ })
+
it('gives each request a fresh id', () => {
const { client, worker } = make()
void client.propose('one', {})
diff --git a/web/src/__tests__/run.test.ts b/web/src/__tests__/run.test.ts
index 8b8cb20..3441ab6 100644
--- a/web/src/__tests__/run.test.ts
+++ b/web/src/__tests__/run.test.ts
@@ -6,7 +6,7 @@ import type { BookResult } from '@/types'
const row = (stem: string): BookResult => ({
stem,
files: ['.epub'],
- facts: { author: 'A', title: stem, series: null, series_index: null },
+ facts: { author: 'A', title: stem, series: null, series_index: null, scheme: '' },
status: 'pending',
proposal: null,
})
@@ -22,6 +22,14 @@ describe('applyEvent', () => {
expect(list[1].answered).toBeUndefined()
})
+ it('records a catalogue once however many rounds ask it', () => {
+ let list = [row('one')]
+ for (const source of ['openlib', 'apple', 'apple', 'openlib'] as const) {
+ list = applyEvent(list, { type: 'answer', stem: 'one', source })
+ }
+ expect(list[0].answered).toEqual(['openlib', 'apple'])
+ })
+
it('replaces a row with its result without reordering', () => {
const list = applyEvent([row('one'), row('two')], {
type: 'book',
diff --git a/web/src/__tests__/use-run.test.ts b/web/src/__tests__/use-run.test.ts
index 09d0c33..22898ef 100644
--- a/web/src/__tests__/use-run.test.ts
+++ b/web/src/__tests__/use-run.test.ts
@@ -13,7 +13,13 @@ type Proposed = Extract
// Hoisted with the mock below: vi.mock runs before any import, so the fake
// must exist before use-run.ts asks for its Client.
const { FakeClient, script, FACTS } = vi.hoisted(() => {
- const FACTS = { author: 'Ada Example', title: 'Sample', series: null, series_index: null }
+ const FACTS = {
+ author: 'Ada Example',
+ title: 'Sample',
+ series: null,
+ series_index: null,
+ scheme: '',
+ }
// What the fake worker answers, set per test. A propose that is never
// answered leaves the run mid-book, which is how stop is exercised.
const script = {
@@ -44,6 +50,9 @@ const { FakeClient, script, FACTS } = vi.hoisted(() => {
reset() {
this.resets++
}
+ facts(stems: string[]) {
+ return Promise.resolve(stems.map(() => FACTS))
+ }
async propose(stem: string, files: FileBytes) {
const id = this.nextId++
this.live = id
diff --git a/web/src/mock/books.ts b/web/src/mock/books.ts
index bb5d2c6..8a410f8 100644
--- a/web/src/mock/books.ts
+++ b/web/src/mock/books.ts
@@ -27,7 +27,7 @@ export function parseStem(stem: string): FilenameFacts {
index = match.groups.index
remainder = match.groups.title
}
- return { author, title: remainder.trim(), series, series_index: index }
+ return { author, title: remainder.trim(), series, series_index: index, scheme: '' }
}
function meta(partial: Partial = {}): Metadata {
diff --git a/web/src/mock/simulate.ts b/web/src/mock/simulate.ts
index a8239fd..542b5d2 100644
--- a/web/src/mock/simulate.ts
+++ b/web/src/mock/simulate.ts
@@ -78,8 +78,11 @@ export function applyEvent(list: BookResult[], event: RunEvent): BookResult[] {
case 'querying':
return list.map((b) => (b.stem === event.stem ? { ...b, status: 'querying' } : b))
case 'answer':
+ // A catalogue asked again in a retry round is the same witness, not a new one.
return list.map((b) =>
- b.stem === event.stem ? { ...b, answered: [...(b.answered ?? []), event.source] } : b,
+ b.stem !== event.stem || b.answered?.includes(event.source)
+ ? b
+ : { ...b, answered: [...(b.answered ?? []), event.source] },
)
case 'book':
return list.map((b) => (b.stem === event.result.stem ? event.result : b))
diff --git a/web/src/run/client.ts b/web/src/run/client.ts
index b519959..43a1e4a 100644
--- a/web/src/run/client.ts
+++ b/web/src/run/client.ts
@@ -1,7 +1,7 @@
// The page's side of the worker protocol: one worker, requests matched to
// replies by id, and broadcast events (loading, answers) handed to a listener.
-import type { Proposal } from '@/types'
+import type { FilenameFacts, Proposal } from '@/types'
import type { FileBytes, FromWorker, ToWorker, WriteOutcome } from '@/worker/protocol'
@@ -81,6 +81,16 @@ export class Client {
})
}
+ // What each filename claims, from the parser inside the worker.
+ async facts(stems: string[]): Promise {
+ const reply = await this.request>((id) => ({
+ type: 'facts',
+ id,
+ stems,
+ }))
+ return reply.facts
+ }
+
// The id of the propose request in flight, for telling its answers apart.
live = 0
diff --git a/web/src/run/use-run.ts b/web/src/run/use-run.ts
index 6823d6d..fbbda79 100644
--- a/web/src/run/use-run.ts
+++ b/web/src/run/use-run.ts
@@ -2,7 +2,6 @@ import { downloadZip } from 'client-zip'
import { useCallback, useEffect, useRef, useState } from 'react'
import { groupBooks, type Intake, type IntakeBook } from '@/intake'
-import { parseStem } from '@/mock/books'
import { applyEvent, simulateRun, type Simulation } from '@/mock/simulate'
import type { BookResult, Extension, RunEvent, SourceName } from '@/types'
import type { FileBytes } from '@/worker/protocol'
@@ -109,10 +108,17 @@ export function useRun(options: { failLoad?: boolean } = {}) {
const books = groupBooks(chosen.books)
intake.current = new Map(books.map((b) => [b.stem, b]))
folders.current = chosen.folders
+ // The file's own name stands in until the worker has read it.
const rows: BookResult[] = books.map((b) => ({
stem: b.stem,
files: Object.keys(b.files).sort() as Extension[],
- facts: parseStem(b.stem.slice(b.stem.lastIndexOf('/') + 1)),
+ facts: {
+ author: '',
+ title: b.stem.slice(b.stem.lastIndexOf('/') + 1),
+ series: null,
+ series_index: null,
+ scheme: '',
+ },
status: 'pending',
proposal: null,
}))
@@ -126,6 +132,15 @@ export function useRun(options: { failLoad?: boolean } = {}) {
}
if (token.current !== mine) return
py.reset()
+ try {
+ const facts = await py.facts(rows.map((r) => r.stem))
+ rows.forEach((row, i) => {
+ if (facts[i]) row.facts = facts[i]
+ })
+ } catch {
+ // A row keeps its file name; the verdict brings the reading with it.
+ }
+ if (token.current !== mine) return
handle({ type: 'ready', books: rows })
for (const row of rows) {
if (token.current !== mine) return
diff --git a/web/src/styles/blueprint.css b/web/src/styles/blueprint.css
index 5537a68..880a5e5 100644
--- a/web/src/styles/blueprint.css
+++ b/web/src/styles/blueprint.css
@@ -208,7 +208,8 @@ html {
color: var(--bp-steel);
border-style: dashed;
}
-.bp-stamp[data-verdict='NONE'] {
+.bp-stamp[data-verdict='NONE'],
+.bp-stamp[data-verdict='UNTITLED'] {
border-style: dotted;
}
.bp-stamp[data-verdict='UNREADABLE'] {
@@ -231,6 +232,7 @@ html {
background: rgba(103, 128, 176, 0.45);
}
.bp-segment[data-verdict='NONE'],
+.bp-segment[data-verdict='UNTITLED'],
.bp-segment[data-verdict='UNREADABLE'] {
background: rgba(167, 182, 217, 0.25);
}
diff --git a/web/src/types.ts b/web/src/types.ts
index 526148f..296b9d9 100644
--- a/web/src/types.ts
+++ b/web/src/types.ts
@@ -62,6 +62,27 @@ export interface FilenameFacts {
title: string
series: string | null
series_index: string | null
+ // The naming scheme the file arrived in (docs/filenames.md); '' for the
+ // plain "Author - Title".
+ scheme: string
+}
+
+// How the page names a recognised scheme. A scheme with no title says so.
+export const SCHEME_LABEL: Record = {
+ 'title-first': 'title first',
+ 'title-by-author': 'title by author',
+ oceanofpdf: 'an OceanofPDF name',
+ 'annas-archive': 'an Anna\u2019s Archive name',
+ 'z-library': 'a Z-Library name',
+ libgen: 'a Library Genesis name',
+ pdfdrive: 'a PDFDrive name',
+ dotted: 'a dotted release name',
+ slug: 'a slug',
+ springer: 'a Springer name',
+ gutenberg: 'a Project Gutenberg number',
+ 'internet-archive': 'an Internet Archive identifier',
+ isbn: 'a bare ISBN',
+ kindle: 'a Kindle ASIN',
}
// skipped: the visitor stopped the run before this book was checked.
@@ -87,8 +108,11 @@ export type RunEvent =
| { type: 'book'; result: BookResult }
| { type: 'done'; elapsedMs: number }
-export function verdict(result: BookResult): Confidence | 'NONE' | 'UNREADABLE' {
- if (result.proposal === null) return 'NONE'
+export type Verdict = Confidence | 'NONE' | 'UNTITLED' | 'UNREADABLE'
+
+export function verdict(result: BookResult): Verdict {
+ // A name with no title in it was never asked about; that is not "no answer".
+ if (result.proposal === null) return result.facts.title ? 'NONE' : 'UNTITLED'
if (result.proposal.unreadable) return 'UNREADABLE'
return result.proposal.conf
}
diff --git a/web/src/worker/metamend.worker.ts b/web/src/worker/metamend.worker.ts
index e712f50..6953dda 100644
--- a/web/src/worker/metamend.worker.ts
+++ b/web/src/worker/metamend.worker.ts
@@ -119,9 +119,15 @@ scope.onmessage = async (event: MessageEvent) => {
fn.destroy()
const proposal = result === undefined ? null : result.toJs(toPlain)
result?.destroy?.()
- const factsProxy = web.facts(message.stem)
- const facts = factsProxy.toJs(toPlain)
- factsProxy.destroy()
+ // The reading the verdict was scored against travels with the proposal;
+ // a name nobody answered for is read the same way it was shown.
+ let facts = proposal?.facts
+ if (facts) delete proposal.facts
+ else {
+ const factsProxy = web.facts(message.stem)
+ facts = factsProxy.toJs(toPlain)
+ factsProxy.destroy()
+ }
const pause: number = web.pause_after()
const shelved = web.unavailable()
const unavailable = shelved.toJs() as SourceName[]
@@ -129,6 +135,16 @@ scope.onmessage = async (event: MessageEvent) => {
post({ type: 'proposed', id: message.id, facts, proposal, pause, unavailable })
return
}
+ case 'facts': {
+ const facts = message.stems.map((stem) => {
+ const proxy = web!.facts(stem)
+ const plain = proxy.toJs(toPlain)
+ proxy.destroy()
+ return plain
+ })
+ post({ type: 'facts', id: message.id, facts })
+ return
+ }
case 'apply': {
const fn = py.globals.get('apply')
const result = fn(message.stem, views(message.files), message.proposal)
diff --git a/web/src/worker/protocol.ts b/web/src/worker/protocol.ts
index 51f14a7..2c7f204 100644
--- a/web/src/worker/protocol.ts
+++ b/web/src/worker/protocol.ts
@@ -15,6 +15,9 @@ export type ToWorker =
| { type: 'init'; base: string }
// A new run: shelved catalogues and back-off from the last one are forgotten.
| { type: 'reset' }
+ // What each filename claims, read by the same parser the verdicts use, so the
+ // rows never show a second parser's reading while they wait.
+ | { type: 'facts'; id: number; stems: string[] }
| { type: 'propose'; id: number; stem: string; files: FileBytes }
| { type: 'apply'; id: number; stem: string; files: FileBytes; proposal: Proposal }
@@ -24,6 +27,7 @@ export type FromWorker =
| { type: 'failed'; message: string }
// id names the propose request, so an answer for a stopped one can be told apart.
| { type: 'answer'; id: number; stem: string; source: SourceName }
+ | { type: 'facts'; id: number; facts: FilenameFacts[] }
| {
type: 'proposed'
id: number
From d57dd752ff931cce1064cb32fb224fb7785d8075 Mon Sep 17 00:00:00 2001
From: CrazyFreak <44674613+OffCrazyFreak@users.noreply.github.com>
Date: Wed, 16 Sep 2026 20:45:16 +0200
Subject: [PATCH 2/3] fix(library): Read OceanofPDF's double underscore as the
lost colon
Changes:
- Turn "Title__Subtitle" into "Title: Subtitle" in an OceanofPDF name, so the query stops at the main title
- Pin it with a parser test and note the shape in docs/filenames.md
A newly downloaded pair showed that the site keeps the colon as a double underscore and cuts the title at about forty characters. Read as a colon, the main title is queried directly and all three web sources answer at once instead of one source needing the head retry.
---
docs/filenames.md | 4 ++--
src/ebook_metamend/library.py | 4 +++-
tests/test_library_and_opf.py | 7 +++++++
3 files changed, 12 insertions(+), 3 deletions(-)
diff --git a/docs/filenames.md b/docs/filenames.md
index 92cb07f..fad3585 100644
--- a/docs/filenames.md
+++ b/docs/filenames.md
@@ -25,7 +25,7 @@ Title first:
- LazyLibrarian, default `$Title - $Author` (`configdefs.py`).
- Anna's Archive, `Title -- Author -- Edition, Year -- Publisher -- ISBN -- md5 -- Anna’s Archive.ext`, every field at most 60 characters, the whole at most 150, and every `.` in the name turned into `_` so `Mara T. Voss` arrives as `Mara T_ Voss` (`allthethings/page/views.py`). Empty fields are dropped, so the second field is not always the author.
- Z-Library over the years, `Title (Author) (z-lib.org)`, `Title by Author (z-lib.org)`, `Title (Author)` followed by an em dash and `_Publisher_Language_ISBN (Z-Library)`, `Title (Last, First etc.) (z-library.sk, 1lib.sk, z-lib.sk)`. A colon in the title is dropped and leaves two spaces behind, which is how the subtitle boundary is recovered (filenames quoted in GitHub issues).
-- OceanofPDF, `_OceanofPDF.com_Title_-_Author.ext`, underscores for spaces, the colon dropped without trace, hyphens inside words kept (`Domain-Driven`) (three independent renaming scripts on GitHub and the files that started this).
+- OceanofPDF, `_OceanofPDF.com_Title_-_Author.ext`, underscores for spaces, a colon left as a double underscore (`Title__Subtitle`), the title cut at about 40 characters, dots dropped from initials, hyphens inside words kept (`Domain-Driven`) (three independent renaming scripts on GitHub and the files that started this).
- Renaming tools, `Title by Author.ext` (ebook-rename's README).
Title only:
@@ -49,7 +49,7 @@ Other observations that shaped the rules: a spaced en dash, a spaced em dash, `
## What the parser does
1. Strips the site's own marks (`_OceanofPDF.com_`, `(z-lib.org)`, `( PDFDrive )`, `- libgen.li`, `-- Anna’s Archive`, `(retail)`, `(v5.0)`, `(epub)`), a trailing `(Year)` or `(Year, Publisher)`, a bracketed ISBN, a duplicate-download counter and `.kepub`.
-2. Undoes the site's encoding: underscores or dots for spaces, `_ ` for `: `, `.-.` for ` - `, Anna's Archive's `_` for `.`, Z-Library's double space for `: `, slugs back into words, Springer's CamelCase into words, `Last, First` into `First Last` (never `Smith, Jr.`).
+2. Undoes the site's encoding: underscores or dots for spaces, `_ ` for `: `, `.-.` for ` - `, Anna's Archive's `_` for `.`, Z-Library's double space and OceanofPDF's double underscore for `: `, slugs back into words, Springer's CamelCase into words, `Last, First` into `First Last` (never `Smith, Jr.`).
3. Recognises the order when the scheme fixes it. A plain `A - B` is read author first, as the README asks, unless B reads more like a person than A (two or three capitalised words, an initial, no digits, no colon). When the name could be read either way and the other half could be a person at all, the other reading travels along as `FilenameFacts.alternate`.
4. Series shapes from other tools are read too: `[Series #2]` as its own segment and `Title (Series Book 2)`.
5. A name that carries no title (`pg1342`, an ISBN) is reported as such, in the CLI line and on the page, instead of "no source answered".
diff --git a/src/ebook_metamend/library.py b/src/ebook_metamend/library.py
index 6887afe..48a9c02 100644
--- a/src/ebook_metamend/library.py
+++ b/src/ebook_metamend/library.py
@@ -185,7 +185,9 @@ def _read(stem: str) -> tuple[list[str], str, str | None]:
match = _OCEANOFPDF.match(s)
if match:
- text = match.group('rest').replace('_-_', ' - ').replace('_', ' ')
+ text = match.group('rest').replace('_-_', ' - ')
+ # A dropped colon leaves a double underscore, so the subtitle boundary survives.
+ text = re.sub(r'(?<=\w)__(?=\w)', ': ', text).replace('_', ' ')
return _split(text), 'oceanofpdf', 'title-first'
if ' -- ' in s:
diff --git a/tests/test_library_and_opf.py b/tests/test_library_and_opf.py
index 01bdcd7..2697c0d 100644
--- a/tests/test_library_and_opf.py
+++ b/tests/test_library_and_opf.py
@@ -71,6 +71,13 @@ class TestNamesFromTheWild:
'Mara Voss',
'Slow-Grown Fruit',
),
+ # The colon survives as a double underscore, the site cuts the title short.
+ (
+ '_OceanofPDF.com_The_Quiet_Orchard__A_Year_Of_Prun_-_Mara_T_Voss',
+ 'oceanofpdf',
+ 'Mara T Voss',
+ 'The Quiet Orchard: A Year Of Prun',
+ ),
# Anna's Archive: " -- " between fields, every "." turned into "_".
(
'The Quiet Orchard -- Mara T_ Voss -- Hill Press, 2011 -- Hill Press -- 9781594488849 -- '
From 7fb32ea97a4a19ae2c09473206b1b80d379ced90 Mon Sep 17 00:00:00 2001
From: CrazyFreak <44674613+OffCrazyFreak@users.noreply.github.com>
Date: Wed, 16 Sep 2026 21:09:02 +0200
Subject: [PATCH 3/3] fix(enrich): Drop the head retry and settle the review
findings
Changes:
- Remove the retry with the head of a long title; only the whole title is ever asked for, pinned by a test
- Multiply the page's pause between books by the rounds the last book took, so Apple's rate holds when a name is read both ways
- Read a PDFDrive name as a title alone; a dash inside it is a subtitle, never an author
- Turn "Last, First" round only when a bare surname stands before the comma, so a comma-separated pair of authors is left as it is
- Drop the title-only second reading of "Title by Author": it names nobody and could never be strong
- Skip the second reading in replay when the recording lacks its key, reporting it, instead of aborting the run
Review of the pull request reproduced the head retry proposing a different volume: a series name glued to a title drew the volume of that name out of two sources, a strict prefix of the filename title that the prefix rule scores 0.95, and its ISBN and series index were proposed at HIGH where the code before stopped at MED. Anything that makes the tool more willing to write is a change to the safety model, so the retry is gone; a glued subtitle now stays at what the full query earns.
Notes:
- Replayed the pristine sample against fixtures-v2 and fixtures-wide-raw: every verdict, source list, gain and figure identical to main
- Live on the four wild names: the Z-Library and double-underscore OceanofPDF forms HIGH with all three sources, the summary edition LOW, the OceanofPDF name with no colon trace MED on Apple alone, which is the honest result
---
docs/filenames.md | 10 +++--
src/ebook_metamend/enrich.py | 79 +++++++++--------------------------
src/ebook_metamend/library.py | 18 ++++----
src/ebook_metamend/web.py | 9 +++-
tests/test_library_and_opf.py | 16 +++++--
tests/test_pipeline.py | 75 +++++++++++++++------------------
6 files changed, 87 insertions(+), 120 deletions(-)
diff --git a/docs/filenames.md b/docs/filenames.md
index fad3585..b21b14e 100644
--- a/docs/filenames.md
+++ b/docs/filenames.md
@@ -30,7 +30,7 @@ Title first:
Title only:
-- PDFDrive, `Title ( PDFDrive ).pdf` and `Title ( PDFDrive.com ).pdf`, spaces inside the brackets, `_ ` for `: ` (filenames quoted on GitHub).
+- PDFDrive, `Title ( PDFDrive ).pdf` and `Title ( PDFDrive.com ).pdf`, spaces inside the brackets, `_ ` for `: `, a dash inside is a subtitle and never an author (filenames quoted on GitHub).
- Kindle "Download & transfer via USB", the title alone; Kindle for PC, `ASIN_EBOK.azw` (DeDRM issues).
- FanFicFare, default `${title}-${siteabbrev}_${storyId}` (`defaults.ini`).
- dokumen.pub, vdoc.pub, epdf.pub slugs, `the-title-of-the-book-9780465050659-9780465003945-2013024417`, `-1nbsped-` for "1st ed.", `-4u9bqm2ndpq0` record ids, `epdf-pub-...-pdf` (their page URLs).
@@ -49,7 +49,7 @@ Other observations that shaped the rules: a spaced en dash, a spaced em dash, `
## What the parser does
1. Strips the site's own marks (`_OceanofPDF.com_`, `(z-lib.org)`, `( PDFDrive )`, `- libgen.li`, `-- Anna’s Archive`, `(retail)`, `(v5.0)`, `(epub)`), a trailing `(Year)` or `(Year, Publisher)`, a bracketed ISBN, a duplicate-download counter and `.kepub`.
-2. Undoes the site's encoding: underscores or dots for spaces, `_ ` for `: `, `.-.` for ` - `, Anna's Archive's `_` for `.`, Z-Library's double space and OceanofPDF's double underscore for `: `, slugs back into words, Springer's CamelCase into words, `Last, First` into `First Last` (never `Smith, Jr.`).
+2. Undoes the site's encoding: underscores or dots for spaces, `_ ` for `: `, `.-.` for ` - `, Anna's Archive's `_` for `.`, Z-Library's double space and OceanofPDF's double underscore for `: `, slugs back into words, Springer's CamelCase into words, `Last, First` into `First Last` (never `Smith, Jr.`, never `Mara Voss, Ann Person`).
3. Recognises the order when the scheme fixes it. A plain `A - B` is read author first, as the README asks, unless B reads more like a person than A (two or three capitalised words, an initial, no digits, no colon). When the name could be read either way and the other half could be a person at all, the other reading travels along as `FilenameFacts.alternate`.
4. Series shapes from other tools are read too: `[Series #2]` as its own segment and `Title (Series Book 2)`.
5. A name that carries no title (`pg1342`, an ISBN) is reported as such, in the CLI line and on the page, instead of "no source answered".
@@ -60,10 +60,12 @@ Other observations that shaped the rules: a spaced en dash, a spaced em dash, `
Only when no source identifies the book as first read does `enrich.propose` ask the catalogues about the alternate reading, and it keeps that reading only if a source then identifies the book. A wrong reading cannot score: a source would have to name a book whose title is the author's name and whose author is the title, and two of them would have to agree. The bar for writing is unchanged; the cost is one extra round of queries for a book that was going to be LOW anyway.
-Separately, when a verdict is below HIGH and the title looks like a subtitle glued on without its colon (`Quiet Orchard The Year Of Pruning`), the head of the title (`Quiet Orchard`) is asked once more and each source keeps whichever of its two answers fits the whole filename better. Measured live: Open Library returns nothing for the glued form and finds the book with the head. The cut is made only before a word a subtitle opens with (the, a, an, how, why, what), only when a phrase follows, and never after a preposition or conjunction, because cutting `The Happiest Baby On | The Block And The Happiest Toddler On The Block` drew the single volume out of Open Library, a strict prefix of the omnibus that the prefix rule scores 0.95, and the two-book bundle reached HIGH on its strength. That prefix rule predates this page and still applies to any source that answers with a strict prefix of the filename title; the retry no longer goes looking for one.
+A retry with the head of a long title (asking for `Quiet Orchard` when the name says `Quiet Orchard The Year Of Pruning`) was built, measured and removed. It did rescue a name whose colon had been dropped, because Open Library answers nothing for the glued form. It also manufactured a HIGH for the wrong book: a series name glued to a title (`The Dark Tower The Waste Lands`) drew the volume called `The Dark Tower` out of two sources, and a source title that is a strict prefix of the filename title scores 0.95 under the prefix rule, so volume VII's ISBN and series index were proposed for volume III. Anything that makes the tool more willing to write is a change to the safety model, so the retry is gone; a glued subtitle now stays at whatever the full query earns, usually MED, which is honest. The prefix rule itself predates this page and still applies to any source that answers with a strict prefix of the filename title.
+
+The page paces between books, not inside one, so `web.pause_after` multiplies the wait by the rounds the last book took (`enrich.last_rounds`); Apple's twenty calls a minute hold even when every book is read both ways. In replay mode a fixture set recorded before names had two readings has no key for the second one; that round reports the missing recording and is skipped, while a missing key in the first round stays loud as before.
## Checked against
-- The two OceanofPDF pairs that started this, live with the three web sources: one HIGH (Apple and Open Library agreeing, after the head retry), one LOW. The LOW is correct: that file is a publisher's summary edition of a well-known book, its own metadata names the summary publisher as the first author, and the author score of 0.26 against the original is the safety model refusing to dress a summary up as the book it summarises.
+- The two OceanofPDF pairs that started this, live with the three web sources: one HIGH (Apple and Open Library agreeing), one LOW. The LOW is correct: that file is a publisher's summary edition of a well-known book, its own metadata names the summary publisher as the first author, and the author score of 0.26 against the original is the safety model refusing to dress a summary up as the book it summarises.
- The same book renamed the Anna's Archive way, the Z-Library way and the Calibre way (`Title - Author`): HIGH each time, with the reading printed.
- The pristine ten-book sample replayed against `fixtures-v2` (Kobo, Google, Open Library) and `fixtures-wide-raw` (the web sources): every verdict, source list, gain and figure identical to the code before this change, and no extra query asked.
diff --git a/src/ebook_metamend/enrich.py b/src/ebook_metamend/enrich.py
index 806dfe3..eddbfef 100644
--- a/src/ebook_metamend/enrich.py
+++ b/src/ebook_metamend/enrich.py
@@ -84,6 +84,10 @@ def to_dict(self) -> dict[str, Any]:
_pacer = Pacer()
#: Consecutive transport failures per source, for the shelving rule above.
_failures: dict[str, int] = {}
+#: How many rounds of queries the last propose() made: two when a name was
+#: read both ways. A caller pacing itself between books waits that many times
+#: longer, so a source's rate holds even when a book cost two rounds.
+last_rounds = 1
def reset_run_state() -> None:
@@ -93,10 +97,11 @@ def reset_run_state() -> None:
once is never retried, and back-off from a previous run still applies. Fine
for a single CLI invocation, wrong for anything longer lived.
"""
- global _pacer
+ global _pacer, last_rounds
unavailable_sources.clear()
_failures.clear()
_pacer = Pacer()
+ last_rounds = 1
def query_sources(
@@ -390,53 +395,6 @@ def compute_gains(merged: dict[str, Any], current: dict[str, Any], conf: str) ->
return gains
-#: Words a subtitle tends to open with when its colon has been lost.
-_SUBTITLE_STARTERS = frozenset({'the', 'a', 'an', 'how', 'why', 'what'})
-#: Words a title never ends on. A starter after one of these is mid-phrase
-#: ("Baby On | The Block", "Gone with | the Wind"), not a subtitle's first word.
-_PHRASE_WORDS = frozenset(
- 'and or of on in to for with from at by the a an into onto over under about as'.split()
-)
-
-
-def _head(query: str) -> str:
- """The title before a subtitle whose colon a download site dropped, or ''.
-
- Cut only before a word a subtitle opens with, only when a phrase follows
- it, and never mid-phrase, so "Thinking Fast and Slow", "Gone with the Wind"
- and an omnibus "A On The Block And B On The Block" are left alone while
- "Sapiens A Brief History of Humankind" becomes "Sapiens". Never the first
- word: "The Design of Everyday Things" has no subtitle.
- """
- words = query.split()
- for position, word in enumerate(words[1:], 1):
- if word.casefold() not in _SUBTITLE_STARTERS:
- continue
- # A subtitle is a phrase of its own, and the title before it does not
- # end mid-phrase. Measured: cutting "The Happiest Baby On The Block And
- # The Happiest Toddler On The Block" at "On | The" drew the single volume
- # out of Open Library, a strict prefix of the omnibus that the prefix
- # rule scores 0.95, and the two-book bundle reached HIGH on its strength.
- if len(words) - position >= 3 and words[position - 1].casefold() not in _PHRASE_WORDS:
- return ' '.join(words[:position])
- return ''
-
-
-def _better_answers(
- first: dict[str, dict[str, Any]], second: dict[str, dict[str, Any]], facts
-) -> dict[str, dict[str, Any]]:
- """Per source, whichever of its two answers fits the filename better."""
- merged = dict(first)
- for name, answer in second.items():
- if name not in merged:
- merged[name] = answer
- continue
- old, new = score({name: merged[name]}, facts)[0][0], score({name: answer}, facts)[0][0]
- if (new.title_score, new.author_score) > (old.title_score, old.author_score):
- merged[name] = answer
- return merged
-
-
def propose(
book: Book,
*,
@@ -456,6 +414,8 @@ def propose(
# and any answer at all would look equally (un)related. So do not ask.
if not facts.query:
return None
+ global last_rounds
+ last_rounds = 1
ask = functools.partial(query_sources, sources=sources, pause=pause, on_answer=on_answer)
answers = ask(facts.query, facts.author)
scores, conf = score(answers, facts) if answers else ([], 'LOW')
@@ -466,19 +426,18 @@ def propose(
# have to name a book whose title is the author's name and whose author
# is the title, twice over, so the bar for writing is unchanged.
other = facts.alternate
- if other.query:
+ try:
other_answers = ask(other.query, other.author)
- other_scores, other_conf = score(other_answers, other) if other_answers else ([], 'LOW')
- if any(s.strong for s in other_scores):
- facts, answers, scores, conf = other, other_answers, other_scores, other_conf
- if conf != 'HIGH' and (head := _head(facts.query)):
- # A subtitle glued on without its colon ("Quiet Orchard The Year Of
- # Pruning", as OceanofPDF writes it) is a query Open Library answers with
- # nothing at all, measured live, while the head alone finds the book. The
- # answers are still scored against the whole filename reading; a source
- # only trades its answer for one that fits the filename better.
- answers = _better_answers(answers, ask(head, facts.author), facts)
- scores, conf = score(answers, facts) if answers else ([], 'LOW')
+ except cache.MissingFixture as exc:
+ # A set recorded before names had two readings has no key for the
+ # second one. The first round stays loud about a missing key; this
+ # round is opportunistic, so it is reported and skipped.
+ print(f'replay has no recording for the other reading: {exc}', file=sys.stderr)
+ other_answers = {}
+ last_rounds = 2
+ other_scores, other_conf = score(other_answers, other) if other_answers else ([], 'LOW')
+ if any(s.strong for s in other_scores):
+ facts, answers, scores, conf = other, other_answers, other_scores, other_conf
if not answers:
return None
diff --git a/src/ebook_metamend/library.py b/src/ebook_metamend/library.py
index 48a9c02..69b5d97 100644
--- a/src/ebook_metamend/library.py
+++ b/src/ebook_metamend/library.py
@@ -107,8 +107,9 @@ class FilenameFacts:
def _person_first(author: str) -> str:
""" "Last, First" written the way the catalogues answer, "First Last".
- Only the one-comma, short shape is turned round. "Smith, Jr." keeps its
- comma, and so does anything long enough to be a list rather than a person.
+ Only the one-comma shape with a bare surname in front is turned round.
+ "Smith, Jr." keeps its comma, and so does "Mara Voss, Ann Person", which
+ is two people, not one written backwards.
"""
author = re.sub(r'\s+(?:etc\.?|et al\.?)$', '', author.strip(), flags=re.I)
if author.count(',') != 1:
@@ -116,7 +117,10 @@ def _person_first(author: str) -> str:
last, first = (part.strip() for part in author.split(','))
if not last or not first or first.rstrip('.').casefold() in _NAME_SUFFIXES:
return author
- if len(last.split()) > 2 or len(first.split()) > 3:
+ surname = last.split()
+ if len(surname) > 2 or (len(surname) == 2 and surname[0].casefold() not in _LOWER_NAME_PARTS):
+ return author
+ if len(first.split()) > 3:
return author
return f'{first} {last}'
@@ -221,7 +225,9 @@ def _read(stem: str) -> tuple[list[str], str, str | None]:
scheme, order = '', None
if _PDFDRIVE.search(s):
- s, scheme, order = _PDFDRIVE.sub('', s), 'pdfdrive', 'title-first'
+ # The site names a file by its title alone; a dash inside is a subtitle.
+ s = _PDFDRIVE.sub('', s)
+ return [re.sub(r'_\s', ': ', s).replace('_', ' ').strip()], 'pdfdrive', 'title-first'
if _LIBGEN.search(s):
s, scheme, order = _LIBGEN.sub('', s), 'libgen', 'author-first'
@@ -325,10 +331,6 @@ def parse_filename(stem: str) -> FilenameFacts:
# A title and nobody's name: nothing for a source's author to be scored
# against, so it can never reach HIGH, but it is still worth reporting.
return _facts(stem, '', parts[0], scheme)
- if scheme == 'title-by-author':
- # "Death by Black Hole" is a title with nobody's name in it.
- whole = _facts(stem, '', ' by '.join(parts), '')
- return replace(_title_first(stem, parts, scheme), alternate=whole)
if order == 'title-first':
return _title_first(stem, parts, scheme)
if order == 'author-first':
diff --git a/src/ebook_metamend/web.py b/src/ebook_metamend/web.py
index c3e21c4..0441a7b 100644
--- a/src/ebook_metamend/web.py
+++ b/src/ebook_metamend/web.py
@@ -170,5 +170,10 @@ def unavailable() -> list[str]:
def pause_after() -> float:
- """Seconds the worker should wait before the next book."""
- return enrich.pause_after(WEB_SOURCES)
+ """Seconds the worker should wait before the next book.
+
+ The page cannot pause inside a book, so a book that took two rounds of
+ queries is paid for here: twice the wait, and Apple's twenty calls a
+ minute hold.
+ """
+ return enrich.pause_after(WEB_SOURCES) * enrich.last_rounds
diff --git a/tests/test_library_and_opf.py b/tests/test_library_and_opf.py
index 2697c0d..18db968 100644
--- a/tests/test_library_and_opf.py
+++ b/tests/test_library_and_opf.py
@@ -273,14 +273,22 @@ def test_series_shapes_from_other_tools(self):
'The Quiet Orchard',
)
- def test_a_title_containing_by_keeps_the_whole_as_a_second_reading(self):
+ def test_a_title_containing_by_has_no_second_reading(self):
+ """ "Death by Black Hole" reads as a title and an author, wrongly, but the
+ whole as a title names nobody and could never be strong, so it is not
+ worth a round of queries."""
facts = parse_filename('Death by Black Hole')
- assert facts.alternate is not None
- assert (facts.alternate.author, facts.alternate.title) == ('', 'Death by Black Hole')
+ assert (facts.author, facts.title) == ('Black Hole', 'Death')
+ assert facts.alternate is None
- def test_last_comma_first_is_turned_round_but_a_suffix_is_not(self):
+ def test_last_comma_first_is_turned_round_but_a_suffix_or_a_pair_is_not(self):
assert parse_filename('Voss, Mara - The Quiet Orchard').author == 'Mara Voss'
+ assert parse_filename('van Voss, Mara - The Quiet Orchard').author == 'Mara van Voss'
assert parse_filename('Smith, Jr. - The Quiet Orchard').author == 'Smith, Jr.'
+ assert (
+ parse_filename('Mara Voss, Ann Person - The Quiet Orchard').author
+ == 'Mara Voss, Ann Person'
+ )
class TestWalk:
diff --git a/tests/test_pipeline.py b/tests/test_pipeline.py
index fea4aa1..aec57b9 100644
--- a/tests/test_pipeline.py
+++ b/tests/test_pipeline.py
@@ -266,52 +266,43 @@ def test_a_half_that_cannot_be_a_person_is_never_asked_as_one(self, catalogue, t
enrich.propose(self._book(tmp_path, stem), pause=False)
assert len(catalogue) == 2
- def test_a_lost_subtitle_colon_is_worked_around(self, monkeypatch, tmp_path):
- """OceanofPDF drops the colon, so the whole subtitle rides along in the
- query and Open Library finds nothing; the head of the title alone does.
- Measured live on a real file."""
- asked = []
-
- def fussy(title, author):
- asked.append(title)
- return (
- {'title': 'Quiet Orchard', 'authors': ['Mara Voss']}
- if title == 'Quiet Orchard'
- else None
- )
-
- def easy(title, author):
- return {'title': 'Quiet Orchard: The Year Of Pruning', 'authors': ['Mara Voss']}
-
- one = enrich.SOURCES[0].__class__(name='fussy', fetch=fussy, pause=0)
- two = enrich.SOURCES[0].__class__(name='easy', fetch=easy, pause=0)
- monkeypatch.setattr(enrich, 'SOURCES', (one, two))
+ def test_a_missing_recording_for_the_other_reading_is_skipped(self, monkeypatch, tmp_path):
+ """A fixture set recorded before names had two readings has no key for
+ the second one; the run reports it and carries on rather than aborting."""
+ from ebook_metamend.sources import cache
+
+ def strict(title, author):
+ if title == 'The Quiet Orchard':
+ return None
+ raise cache.MissingFixture(f'{title!r} / {author!r}')
+
+ one = enrich.SOURCES[0].__class__(name='one', fetch=strict, pause=0)
+ monkeypatch.setattr(enrich, 'SOURCES', (one,))
monkeypatch.setattr(calibre, 'read_book_metadata', lambda path: {})
enrich.reset_run_state()
- stem = '_OceanofPDF.com_Quiet_Orchard_The_Year_Of_Pruning_-_Mara_Voss'
- proposal = enrich.propose(self._book(tmp_path, stem), pause=False)
- assert proposal is not None
- assert proposal.conf == 'HIGH'
- assert sorted(proposal.sources) == ['easy', 'fussy']
- assert asked == ['Quiet Orchard The Year Of Pruning', 'Quiet Orchard']
+ assert (
+ enrich.propose(self._book(tmp_path, 'Ann Person - The Quiet Orchard'), pause=False)
+ is None
+ )
+ assert enrich.last_rounds == 2
enrich.reset_run_state()
- @pytest.mark.parametrize(
- 'query, head',
- [
- ('Sapiens A Brief History of Humankind', 'Sapiens'),
- ('The Happiest Baby On The Block', ''),
- ('The Happiest Baby On The Block And The Happiest Toddler On The Block', ''),
- ('Gone with the Wind and Other Stories', ''),
- ('The Power of Now A Guide to Spiritual Enlightenment', 'The Power of Now'),
- ('Atomic Habits An Easy And Proven Way', 'Atomic Habits'),
- ('The Design of Everyday Things', ''),
- ('Thinking Fast and Slow', ''),
- ('Zero to One', ''),
- ],
- )
- def test_the_head_of_a_title_is_cut_only_before_a_subtitle(self, query, head):
- assert enrich._head(query) == head
+ def test_a_book_read_both_ways_reports_two_rounds(self, catalogue, tmp_path):
+ """The page paces between books, not inside one, so it must know a book
+ cost two rounds to keep a source under its rate."""
+ enrich.propose(self._book(tmp_path, 'Some Other Book - Ann Person'), pause=False)
+ assert enrich.last_rounds == 2
+ enrich.propose(self._book(tmp_path, 'Mara Voss - The Quiet Orchard'), pause=False)
+ assert enrich.last_rounds == 1
+
+ def test_only_the_whole_title_is_ever_asked_for(self, catalogue, tmp_path):
+ """A retry with the head of a glued title once drew a different volume out
+ of the catalogues, a strict prefix that the prefix rule scores 0.95, and
+ proposed its ISBN. The catalogues are asked about the whole title only."""
+ enrich.propose(
+ self._book(tmp_path, 'Ann Person - The Quiet Orchard The Year Of Pruning'), pause=False
+ )
+ assert {title for title, _ in catalogue} == {'The Quiet Orchard The Year Of Pruning'}
def test_a_series_name_has_one_reading(self, catalogue, tmp_path):
""" "Author - Series - 02 - Title" is this tool's own convention; the