diff --git a/README.md b/README.md index 2a31bd590..9189a5667 100644 --- a/README.md +++ b/README.md @@ -152,6 +152,7 @@ Clean-room Go rewrite, modern React UI, MIT-licensed, actively developed. | [Google Books](https://developers.google.com/books) | API key (free) | Enrichment: descriptions, ratings | | [Hardcover.app](https://hardcover.app) | API token (free) | Search enrichment + community ratings, series, wishlist; can be promoted to **primary** for a curated, translation-free catalogue. Token required for **all** queries — without it Hardcover is silently skipped ([troubleshooting](docs/Troubleshooting-Wiki.md#a-book-is-on-hardcoverapp-but-doesnt-show-up-in-the-add-to-library-search)) | | [DNB](https://www.dnb.de/) | None (public SRU) | German-language descriptions, language, year, publisher; can be promoted to **primary** | +| [Nasjonalbiblioteket](https://www.nb.no/) | None (public API) | Norwegian catalogues under their original titles (legal deposit), with series and series numbers; opt-in **primary** only, never queried otherwise | | [Audnex](https://api.audnex.us) | None | Audiobook narrator, duration, cover by ASIN | | [Audible](https://audible.com) | None | Supplemental audiobook author lookup — pulls ASINs OL/Hardcover miss | diff --git a/changelog.d/2979-metadata-nasjonalbiblioteket.md b/changelog.d/2979-metadata-nasjonalbiblioteket.md new file mode 100644 index 000000000..f31971358 --- /dev/null +++ b/changelog.d/2979-metadata-nasjonalbiblioteket.md @@ -0,0 +1,2 @@ +### Added +- **Nasjonalbiblioteket metadata provider** (#2979). The National Library of Norway can be chosen as the primary metadata provider in Settings → Metadata Profiles. It covers Norwegian publications by legal deposit under their original Norwegian titles, so a Norwegian library's files match the author's catalogue instead of the English translations other providers list. Books carry the author's series and their number in it, and the library's genres. Audiobooks carry their narrator and running time where the library records them. An author's other spellings, such as the name without its Norwegian letters, become aliases so release names still match. It is opt in: an install that does not select it never contacts the National Library. diff --git a/cmd/bindery/main.go b/cmd/bindery/main.go index b1715e575..9d8970020 100644 --- a/cmd/bindery/main.go +++ b/cmd/bindery/main.go @@ -40,6 +40,7 @@ import ( "github.com/vavallee/bindery/internal/metadata/dnb" "github.com/vavallee/bindery/internal/metadata/googlebooks" "github.com/vavallee/bindery/internal/metadata/hardcover" + "github.com/vavallee/bindery/internal/metadata/nb" "github.com/vavallee/bindery/internal/metadata/openlibrary" "github.com/vavallee/bindery/internal/metrics" "github.com/vavallee/bindery/internal/models" @@ -306,6 +307,8 @@ func main() { // // - "dnb" is the recommended choice for German/Austrian/Swiss catalogues, // where OpenLibrary coverage is too thin for German-language books. + // - "nb" is the same for Norwegian ones: original titles where + // OpenLibrary has the English translation. // - "hardcover" trades breadth for a cleaner, editorially curated // catalogue: no translation editions masquerading as separate works, no // omnibus bundles, no non-book merchandise rows (#2040). @@ -318,6 +321,11 @@ func main() { switch primaryName { case "dnb": primaryProvider = dnbClient + case "nb": + // Opt-in only: unlike the others, NB is never added as an enricher, + // so an install that did not choose it sends NB no traffic. NB + // publishes no rate limits for a fleet of independent installs. + primaryProvider = nb.New() case "hardcover": primaryProvider = hcClient default: diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 0843d0a29..7e275b582 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -22,8 +22,11 @@ OpenLibrary Google Books, Hardcover.app, Audnex, Audible ``` Exactly one provider is *primary* — it defines what an author's catalogue is. -`metadata.primary_provider` selects OpenLibrary (default), DNB, or Hardcover -(token required); every provider that isn't primary is wired as an enricher. +`metadata.primary_provider` selects OpenLibrary (default), DNB, Nasjonalbiblioteket +(`nb`), or Hardcover (token required); every provider that isn't primary is wired +as an enricher, except Nasjonalbiblioteket, which is wired only when it is primary. +Switching the primary away from it therefore leaves `nb:` authors with no provider +to sync from until they are relinked. ## Components @@ -48,7 +51,7 @@ The `internal/` tree is organised by domain, not by layer: | `db` | Connection pooling, transaction helpers, repository interfaces, and the embedded schema migrations under `db/migrations/` (`NNN_description.sql`, applied idempotently at boot). | | `migrate` | Bulk-import of authors and related records from a `readarr.db` or a Goodreads CSV export. | | `models` | Domain types (Author, Book, Edition, Series, Indexer, etc.) shared across handlers, repos, and pipelines. | -| `metadata` | OpenLibrary, Google Books, Hardcover, DNB, Audnex, Audible — fetchers and unifying interfaces. | +| `metadata` | OpenLibrary, Google Books, Hardcover, DNB, Nasjonalbiblioteket, Audnex, Audible — fetchers and unifying interfaces. | | `indexer` | Newznab/Torznab clients, query builder, four-tier fallback, per-indexer query deduplication, result deduplication, ranking. | | `decision` | Quality profiles, language filter, custom formats, delay profiles, blocklist consultation. | | `downloader` | SABnzbd, NZBGet, qBittorrent, Transmission, Deluge, rTorrent clients (queue/history polling, submission, deletion). | diff --git a/docs/User-Guide-Wiki.md b/docs/User-Guide-Wiki.md index 519d59cc3..fc3714756 100644 --- a/docs/User-Guide-Wiki.md +++ b/docs/User-Guide-Wiki.md @@ -766,7 +766,11 @@ like everything else on that tab that names server paths. author's catalogue looks like. It is community data: expect occasional duplicates, language mix-ups, and box-set entries. The "primary" selector (Settings → Metadata Profiles → Library Defaults) offers OpenLibrary, - **DNB** (German National Library), and **Hardcover**. + **DNB** (German National Library), **Nasjonalbiblioteket** (National Library + of Norway: Norwegian books under their original titles, only used when chosen + as primary), and **Hardcover**. Switching the primary away from + Nasjonalbiblioteket switches it off entirely, so authors linked to it stop + syncing until you relink each one with "Link metadata" on their page. - **Hardcover** is an enricher by default — it improves search results, ratings, and series data, and powers import lists and the Discover wishlist row. **Without an API token (Settings → API Keys) Hardcover is silently @@ -839,8 +843,8 @@ bound to with a copy button, and lists any other provider ids the same book is known by. That is the thing to check before deciding a book needs re-binding, and the id is what to quote in a bug report. Hover or activate **Links** while confirming a book in the Add to library dialog or in the book header to open -trustworthy upstream pages for OpenLibrary, Google Books, Hardcover, and DNB -records. +trustworthy upstream pages for OpenLibrary, Google Books, Hardcover, DNB and +Nasjonalbiblioteket records. Calibre and Audiobookshelf ids remain visible only under **Metadata source** because they do not map to stable public pages. diff --git a/docs/third-party-data.md b/docs/third-party-data.md index 73ce50600..9e8574a36 100644 --- a/docs/third-party-data.md +++ b/docs/third-party-data.md @@ -77,6 +77,39 @@ Public domain catalogue data (CC0). Cover images come from `covers.openlibrary.org` and are subject to their rate limits; Bindery serves them through the same `/api/v1/images` cache. +## Nasjonalbiblioteket + +Sources: (the catalogue search +API) and (the Norwegian authority file). Reviewed +2026-10-04. Only contacted when Nasjonalbiblioteket is the primary provider. + +- **The delivered records are CC0; the search API states no licence.** The + National Library publishes its catalogue records under CC0 through its + metadata delivery over OAI-PMH and SRU + (). Bindery does + not use that delivery: it reads the catalogue search API, which publishes no + licence, terms or rate limits of its own. What Bindery stores from it is the + same bibliographic data (titles, authors, ISBNs, years, languages, series, + genres, audiobook narrators and running times). Bindery keeps NB opt in so + installs that do not choose it send no traffic. +- **Summaries are publisher copy.** A record's summary is usually the + publisher's own description of the book, not the library's cataloguing, so + CC0 does not reach it. Bindery stores it as the book description, the way it + stores descriptions from the other providers. +- **Covers are excluded.** NB's image service refuses in-copyright books, and + the cover images some records link to are licensed to library catalogues + only. Bindery fetches no cover from NB or from those links. +- **The authority file is Sikt's, under NLOD 2.0.** Bindery reads an author's + name heading from it to look up their catalogue, and the name's other forms, + which may become aliases. NLOD asks for attribution when the data is + redistributed, for example published as a dataset: + + > Contains data under the Norwegian licence for Open Government data (NLOD) + > distributed by Sikt. + + A self-hosted install that shows the names to its own users does not + redistribute them. + ## Audible `internal/metadata/audible` calls an unpublished Amazon endpoint, and Amazon's diff --git a/internal/api/catalogue_reconciliation.go b/internal/api/catalogue_reconciliation.go index 70ce6f332..f6f17ee92 100644 --- a/internal/api/catalogue_reconciliation.go +++ b/internal/api/catalogue_reconciliation.go @@ -275,6 +275,20 @@ func (h *AuthorHandler) buildCatalogueReconciliation(ctx context.Context, author if provider == "dnb" { return } + // NB works come from the author catalogue with every edition + // already attached, and its per-work editions lookup is a full + // GetBook, so the work's own editions are the evidence. None + // attached is not evidence of no ISBN or too few pages, and on a + // partial catalogue a work may be missing editions, so then + // nothing is judged on them. + if provider == "nb" { + if snapshot.Complete && len(work.Editions) > 0 { + mu.Lock() + editions[work.ForeignID] = editionEvidence{editions: work.Editions, known: true} + mu.Unlock() + } + return + } found, err := h.meta.GetEditionsFromProvider(ctx, provider, work.ForeignID) mu.Lock() defer mu.Unlock() diff --git a/internal/api/catalogue_reconciliation_test.go b/internal/api/catalogue_reconciliation_test.go index f601a785a..7d7351d88 100644 --- a/internal/api/catalogue_reconciliation_test.go +++ b/internal/api/catalogue_reconciliation_test.go @@ -847,6 +847,69 @@ func TestBuildCatalogueReconciliation_UsesConservativeEditionAndIdentityEvidence } } +// NB works arrive from the author catalogue with every edition attached, so +// the per-work editions lookup (a full GetBook each) is skipped and the +// work's own editions are the evidence. A work that somehow has none stays +// indeterminate: no editions is not proof of no ISBN or too few pages. On a +// partial catalogue (page cap, failed series recall) a work may be missing +// editions, so none of them is judged on that evidence. +func TestBuildCatalogueReconciliation_NBUsesTheWorksOwnEditions(t *testing.T) { + for _, complete := range []bool{true, false} { + pages100, pages300 := 100, 300 + isbn := "9788200000011" + lookupErr := errors.New("per-work editions lookup must not run for nb works") + provider := &editionResultsProvider{ + partialSnapshotProvider: partialSnapshotProvider{ + stubMetaProvider: stubMetaProvider{name: "nb", works: []models.Book{ + {ForeignID: "nb:long", Title: "Long Work", Language: "eng", MetadataProvider: "nb", + Editions: []models.Edition{{NumPages: &pages300, ISBN13: &isbn}}}, + {ForeignID: "nb:short", Title: "Short Work", Language: "eng", MetadataProvider: "nb", + Editions: []models.Edition{{NumPages: &pages100, ISBN13: &isbn}}}, + {ForeignID: "nb:bare", Title: "Bare Work", Language: "eng", MetadataProvider: "nb"}, + }}, + complete: complete, + }, + errors: map[string]error{"nb:long": lookupErr, "nb:short": lookupErr, "nb:bare": lookupErr}, + } + f := newReconciliationFixture(t, provider) + profile, err := f.profiles.GetByID(context.Background(), models.DefaultMetadataProfileID) + if err != nil || profile == nil { + t.Fatalf("load profile: profile=%+v err=%v", profile, err) + } + profile.MinPages = 200 + if err := f.profiles.Update(context.Background(), profile); err != nil { + t.Fatal(err) + } + long := f.createBook(t, "nb:long", "Long Work", "nb", models.BookStatusWanted) + short := f.createBook(t, "nb:short", "Short Work", "nb", models.BookStatusWanted) + bare := f.createBook(t, "nb:bare", "Bare Work", "nb", models.BookStatusWanted) + f.author.ForeignID = "nb:author:10000001" // an NB author: the snapshot routes by prefix + + ctx := auth.WithUserRole(auth.WithUserID(context.Background(), 7), "user") + got, err := f.handler.buildCatalogueReconciliation(ctx, f.author) + if err != nil { + t.Fatal(err) + } + assertIndeterminateRow(t, got, bare, reconcileIndeterminateReasonEditionUnavailable) + if !complete { + if len(got.Candidates) != 0 { + t.Errorf("partial catalogue: candidates = %+v, want none judged on its editions", got.Candidates) + } + assertIndeterminateRow(t, got, short, reconcileIndeterminateReasonEditionUnavailable) + assertIndeterminateRow(t, got, long, reconcileIndeterminateReasonEditionUnavailable) + continue + } + if len(got.Candidates) != 1 || got.Candidates[0].BookID != short.ID || got.Candidates[0].Reason != reconcileReasonPages { + t.Errorf("candidates = %+v, want only the short work, for pages", got.Candidates) + } + for _, row := range got.IndeterminateRows { + if row.BookID == long.ID || row.BookID == short.ID { + t.Errorf("%q is indeterminate (%s): its own editions were not used", row.Title, row.Reason) + } + } + } +} + func TestReconciliationRejectReason_ProfileReasonsAndIndeterminateEvidence(t *testing.T) { pages100 := 100 pages250 := 250 diff --git a/internal/api/settings_handler.go b/internal/api/settings_handler.go index 529beef91..e8151f409 100644 --- a/internal/api/settings_handler.go +++ b/internal/api/settings_handler.go @@ -72,7 +72,7 @@ const SettingDefaultAudiobookRootFolderID = "library.defaultAudiobookRootFolderI // SettingMetadataPrimaryProvider is the KV key that selects the primary // metadata provider used for author/book search and lookup. Valid values are -// "openlibrary" (default), "dnb", and "hardcover". Empty or unset falls back to +// "openlibrary" (default), "dnb", "nb", and "hardcover". Empty or unset falls back to // "openlibrary" for backwards compatibility. // // "hardcover" requires a Hardcover API token (SettingHardcoverAPIToken) — @@ -85,7 +85,7 @@ const SettingMetadataPrimaryProvider = "metadata.primary_provider" // MetadataPrimaryProviders lists the accepted values of // SettingMetadataPrimaryProvider, in the order the UI presents them. The first // entry is the default used when the setting is empty or unset. -var MetadataPrimaryProviders = []string{"openlibrary", "dnb", "hardcover"} +var MetadataPrimaryProviders = []string{"openlibrary", "dnb", "nb", "hardcover"} // IsMetadataPrimaryProviderValid reports whether value names a provider that // may be promoted to primary. The empty string is valid and means "default". diff --git a/internal/metadata/nb/client.go b/internal/metadata/nb/client.go new file mode 100644 index 000000000..8aa390c14 --- /dev/null +++ b/internal/metadata/nb/client.go @@ -0,0 +1,553 @@ +// Package nb provides a read-only client for Nasjonalbiblioteket, the +// National Library of Norway, via its public catalogue search API. No API key +// is required. +// +// Role: opt-in primary. NB holds every Norwegian publication by legal deposit +// under its original title, where OpenLibrary tends to catalogue Norwegian +// books under their English translation. It is wired only when selected as the +// primary provider, so installs that did not choose it never call NB. +// +// Endpoints: +// - https://api.nb.no/catalog/v1/items — bibliographic search. The API +// states no licence of its own; NB's metadata delivery (OAI-PMH, SRU) +// publishes the same records under CC0. See docs/third-party-data.md. +// - https://authority.bibsys.no — the Norwegian authority file, used only to +// turn an author's authority ID back into a name, because NB's search has +// no field that accepts the ID. +// +// NB catalogues editions, not works: every printing, translation and +// audiobook is its own record. groupWorks folds them into works. +package nb + +import ( + "context" + "encoding/json" + "encoding/xml" + "errors" + "fmt" + "io" + "log/slog" + "net/http" + "net/url" + "regexp" + "sort" + "strconv" + "strings" + "time" + + "github.com/vavallee/bindery/internal/httpsec" + "github.com/vavallee/bindery/internal/isbnutil" + "github.com/vavallee/bindery/internal/metadata" + "github.com/vavallee/bindery/internal/metadata/providererr" + "github.com/vavallee/bindery/internal/models" + "github.com/vavallee/bindery/internal/useragent" +) + +const ( + itemsBase = "https://api.nb.no/catalog/v1/items" + metadataBase = "https://api.nb.no/catalog/v1/metadata/" + authorityBase = "https://authority.bibsys.no/authority/rest/authorities/v2/" + idPrefix = "nb:" + authorPrefix = "nb:author:" + // authorityIDPrefix is how NB records name a person's authority record. + authorityIDPrefix = "bibsys.no:authority:" + // pageSize is the API's largest accepted page; 101 is rejected with 400. + pageSize = 100 + // maxWorksPages caps an author catalogue at 2000 edition records. A larger + // author is reported as a partial snapshot rather than paged without end. + // The creator index counts every translation, so a widely translated + // author runs to well over a thousand. + maxWorksPages = 20 + // maxResponseBytes bounds what a misbehaving host can make Bindery decode + // (#2357). A full 100-record page with expanded metadata is ~1 MiB. + maxResponseBytes = 32 << 20 +) + +// sesamIDRe matches NB's record identifier, a 32-char hex string. Validated +// before it is put in a request path. +var sesamIDRe = regexp.MustCompile(`^[0-9a-f]{32}$`) + +// authorityIDRe matches a Norwegian authority file system control number. +var authorityIDRe = regexp.MustCompile(`^[0-9]+$`) + +// Client implements metadata.Provider for Nasjonalbiblioteket. +type Client struct { + http *http.Client +} + +// New creates a new NB client. +func New() *Client { + return &Client{ + http: &http.Client{Timeout: 15 * time.Second, Transport: httpsec.DefaultProxyTransport()}, + } +} + +func (c *Client) Name() string { return "nb" } + +// ready reports ErrProviderNotConfigured for a client that cannot make +// requests. NB needs no credentials, so New always returns a usable client; +// this guards the zero value so it is skipped rather than panicking. +func (c *Client) ready() error { + if c == nil || c.http == nil { + return metadata.ErrProviderNotConfigured + } + return nil +} + +// SearchAuthors finds authors whose name matches every word of query, from the +// author credits of matching records. Each person is keyed by authority ID, so +// two people sharing a name stay apart. +func (c *Client) SearchAuthors(ctx context.Context, query string) ([]models.Author, error) { + if err := c.ready(); err != nil { + return nil, err + } + words := strings.Fields(query) + if len(words) == 0 { + return nil, nil + } + params := url.Values{"q": {"*"}} + for _, w := range words { + params.Add("filter", "api_nameauthor:"+escapeQuery(w)) + } + params.Add("filter", "mediatype:bøker") + page, err := c.search(ctx, params, 0) + if err != nil { + return nil, fmt.Errorf("nb search authors: %w", err) + } + + counts := make(map[string]int) + byID := make(map[string]models.Author) + var order []string + for _, it := range page.Embedded.Items { + for _, p := range it.Metadata.People { + id := p.authorityID() + if id == "" || !p.isAuthor() || !nameMatches(p.Name, words) { + continue + } + if _, ok := byID[id]; !ok { + byID[id] = personToAuthor(p) + order = append(order, id) + } + counts[id]++ + } + } + authors := make([]models.Author, 0, len(order)) + for _, id := range order { + a := byID[id] + // A count of matching records, not works: editions are not grouped + // here. It only ranks same-name candidates against each other. + a.Statistics = &models.AuthorStats{BookCount: counts[id]} + authors = append(authors, a) + } + return authors, nil +} + +// SearchBooks searches catalogue metadata (not digitised full text) and folds +// the matching editions into works. An ISBN query is answered by +// GetBookByISBN. +func (c *Client) SearchBooks(ctx context.Context, query string) ([]models.Book, error) { + if err := c.ready(); err != nil { + return nil, err + } + query = strings.TrimSpace(query) + if query == "" { + return nil, nil + } + // Only a query that is nothing but an ISBN, optionally in the "isbn:" + // form the aggregator's canonical lookup sends; a title containing digits + // stays a text search. + bare := query + if len(bare) > 5 && strings.EqualFold(bare[:5], "isbn:") { + bare = strings.TrimSpace(bare[5:]) + } + if isbn13, isbn10 := isbnutil.Extract(bare); isbnutil.Normalize(bare) == firstNonEmpty(isbn13, isbn10) { + b, err := c.GetBookByISBN(ctx, bare) + if err != nil || b == nil { + return nil, err + } + return []models.Book{*b}, nil + } + params := url.Values{ + "q": {escapeQuery(query)}, + "searchType": {"FIELD_RESTRICTED_SEARCH"}, + // The author catalogue's media types, so a work found here has the + // same editions as in the catalogue. + "filter": {"mediatype:(bøker OR lydopptak)"}, + } + page, err := c.search(ctx, params, 0) + if err != nil { + return nil, fmt.Errorf("nb search books: %w", err) + } + return groupWorks(page.Embedded.Items, ""), nil +} + +// GetAuthor resolves an "nb:author:" through the Norwegian +// authority file. Returns (nil, nil) when the authority record does not exist. +func (c *Client) GetAuthor(ctx context.Context, foreignID string) (*models.Author, error) { + if err := c.ready(); err != nil { + return nil, err + } + id, err := authorityIDFromForeignID(foreignID) + if err != nil { + return nil, err + } + rec, err := c.authority(ctx, id) + if err != nil { + return nil, fmt.Errorf("nb get author %s: %w", foreignID, err) + } + if rec == nil { + return nil, nil + } + a := personToAuthor(person{Name: rec.heading(), Identifier: authorityIDPrefix + id}) + // Bindery saves the ones that are another spelling of the same name as + // aliases, so release names without the diacritics still match. + a.AlternateNames = rec.variants() + return &a, nil +} + +// GetAuthorWorks returns the author's catalogue, folded into works. +func (c *Client) GetAuthorWorks(ctx context.Context, authorForeignID string) ([]models.Book, error) { + books, _, err := c.GetAuthorWorksSnapshot(ctx, authorForeignID) + return books, err +} + +// GetAuthorWorksSnapshot is GetAuthorWorks plus whether the catalogue is +// complete. It is not when the author has more records than maxWorksPages +// covers, so catalogue reconciliation must not read a missing work as removed. +// +// The records are fetched by the authority file's name heading and then kept +// only when they credit this authority ID as author, which drops homonyms and +// records where the person is translator or narrator. A failed name lookup is +// an error, never an empty catalogue: an empty result from a primary is taken +// as fact. +func (c *Client) GetAuthorWorksSnapshot(ctx context.Context, authorForeignID string) ([]models.Book, bool, error) { + if err := c.ready(); err != nil { + return nil, false, err + } + id, err := authorityIDFromForeignID(authorForeignID) + if err != nil { + return nil, false, err + } + name, found, err := c.authorityName(ctx, id) + if err != nil { + return nil, false, fmt.Errorf("nb get author works %s: %w", authorForeignID, err) + } + if !found { + // A record that is gone (Sikt merges duplicate records) says + // nothing about the author's books, so it is not a complete, + // empty catalogue that reconciliation could act on. + return nil, false, nil + } + + params := url.Values{ + "q": {"*"}, + // Audiobooks are included so their ISBNs land on the work: a + // library file is as likely to be the audiobook edition. + // namecreators, not nameauthor: NB leaves many records crediting + // the author out of the author index, some authors' most of them. + // The authority-ID check in groupWorks drops the creator index's + // translator and narrator credits. + "filter": {`namecreators:"` + escapeQuery(name) + `"`, "mediatype:(bøker OR lydopptak)"}, + } + var items []item + complete := false + for p := 0; p < maxWorksPages; p++ { + page, err := c.search(ctx, params, p) + if err != nil { + return nil, false, fmt.Errorf("nb get author works %s: %w", authorForeignID, err) + } + items = append(items, page.Embedded.Items...) + if p+1 >= page.Page.TotalPages { + complete = true + break + } + } + if len(items) == 0 { + // The authority record exists, so the author does too; finding no + // records is a gap in NB's name index (see recallSeriesVolumes), + // not proof of an empty catalogue. + complete = false + } + books := groupWorks(items, id) + memo := &seriesMemo{} + c.fillSeries(ctx, books, items, id, memo) + recalled, recallComplete := c.recallSeriesVolumes(ctx, books, items, id) + if !recallComplete { + complete = false + } + if len(recalled) > 0 { + items = append(items, recalled...) + books = groupWorks(items, id) + c.fillSeries(ctx, books, items, id, memo) + } + return books, complete, nil +} + +// recallSeriesVolumes finds the author's records that the author query +// missed. NB leaves some records out of every name index although they +// credit the author, so no name search returns them; a search on the series +// name does. One search per series the catalogue links the author to, +// keeping only records crediting the author's authority ID that are not +// already in items. +// +// complete is false when a search failed, so a caller reconciling the +// catalogue does not read a volume found this way last time as removed. +// ponytail: first result page only (100 records); a series name matching +// more than that across all authors misses the rest. +func (c *Client) recallSeriesVolumes(ctx context.Context, books []models.Book, items []item, authorID string) (recalled []item, complete bool) { + seen := make(map[string]bool, len(items)) + for _, it := range items { + seen[it.ID] = true + } + var titles []string + titleSeen := make(map[string]bool) + for _, b := range books { + for _, ref := range b.SeriesRefs { + if !titleSeen[ref.Title] { + titleSeen[ref.Title] = true + titles = append(titles, ref.Title) + } + } + } + sort.Strings(titles) + complete = true + for _, title := range titles { + params := url.Values{ + "q": {`"` + escapeQuery(title) + `"`}, + "searchType": {"FIELD_RESTRICTED_SEARCH"}, + "filter": {"mediatype:(bøker OR lydopptak)"}, + } + page, err := c.search(ctx, params, 0) + if err != nil { + slog.Debug("nb: series recall search failed", "series", title, "error", err) + complete = false + continue + } + for _, it := range page.Embedded.Items { + if seen[it.ID] || primaryAuthor(it.Metadata, authorID) == nil { + continue + } + seen[it.ID] = true + recalled = append(recalled, it) + } + } + return recalled, complete +} + +// GetBook fetches the edition record "nb:" and returns the work it +// belongs to. NB has no work record, so the record's siblings are found the +// way the author catalogue finds them: same authority-file author, folded by +// title. The aggregator refreshes ISBN matches through here, so a book built +// from one record would lose its other editions' ISBNs. +func (c *Client) GetBook(ctx context.Context, foreignID string) (*models.Book, error) { + if err := c.ready(); err != nil { + return nil, err + } + id := strings.TrimPrefix(foreignID, idPrefix) + if id == foreignID || !sesamIDRe.MatchString(id) { + return nil, fmt.Errorf("nb: not a book id: %q", foreignID) + } + var it item + found, err := c.getJSON(ctx, itemsBase+"/"+id, &it) + if err != nil { + return nil, fmt.Errorf("nb get book %s: %w", foreignID, err) + } + if !found { + return nil, nil + } + if b := c.workOf(ctx, it); b != nil { + return b, nil + } + books := groupWorks([]item{it}, "") + if len(books) == 0 { + return nil, nil + } + // A rebind replaces the book's series with these, so they are filled + // here as well as in the catalogue. + if a := primaryAuthor(it.Metadata, ""); a != nil && a.authorityID() != "" { + c.fillSeries(ctx, books, []item{it}, a.authorityID(), nil) + } + return &books[0], nil +} + +// workOf returns the work containing it, built from a search for the record's +// author and title, or nil when that cannot be done. Best-effort: the caller +// already holds the record and falls back to it alone. +func (c *Client) workOf(ctx context.Context, it item) *models.Book { + author := primaryAuthor(it.Metadata, "") + if author == nil || author.authorityID() == "" { + return nil + } + params := url.Values{ + "q": {escapeQuery(recordTitle(it.Metadata))}, + "searchType": {"FIELD_RESTRICTED_SEARCH"}, + // No name-index filter: NB leaves some records out of it (see + // recallSeriesVolumes). groupWorks keeps only the author's records. + "filter": {"mediatype:(bøker OR lydopptak)"}, + } + page, err := c.search(ctx, params, 0) + if err != nil { + return nil + } + want := idPrefix + it.ID + books := groupWorks(page.Embedded.Items, author.authorityID()) + for i := range books { + for _, ed := range books[i].Editions { + if ed.ForeignID != want { + continue + } + // Keep the requested ID: callers look the book up by it. + books[i].ForeignID = want + c.fillSeries(ctx, books[i:i+1], page.Embedded.Items, author.authorityID(), nil) + return &books[i] + } + } + return nil +} + +// GetEditions returns the editions of the work bookForeignID belongs to, as +// GetBook assembles it: the record and its siblings, or the record alone when +// the sibling search fails. Profile checks on ISBN and page count then have +// evidence instead of nothing. +func (c *Client) GetEditions(ctx context.Context, bookForeignID string) ([]models.Edition, error) { + b, err := c.GetBook(ctx, bookForeignID) + if err != nil || b == nil { + return nil, err + } + return b.Editions, nil +} + +// GetBookByISBN looks up an edition by ISBN-13 or ISBN-10, any media type, so +// an audiobook ISBN resolves as well as a print one. +func (c *Client) GetBookByISBN(ctx context.Context, isbn string) (*models.Book, error) { + if err := c.ready(); err != nil { + return nil, err + } + isbn13, isbn10 := isbnutil.Extract(isbn) + want := firstNonEmpty(isbn13, isbn10) + if want == "" { + return nil, nil + } + page, err := c.search(ctx, url.Values{"q": {"isbn:" + want}}, 0) + if err != nil { + return nil, fmt.Errorf("nb get book by ISBN: %w", err) + } + books := groupWorks(page.Embedded.Items, "") + if len(books) == 0 { + return nil, nil + } + return &books[0], nil +} + +// search runs one page of an items query with expanded metadata, which is +// what carries the author credits and uniform titles. +func (c *Client) search(ctx context.Context, params url.Values, page int) (*searchResponse, error) { + params.Set("size", strconv.Itoa(pageSize)) + params.Set("page", strconv.Itoa(page)) + params.Set("expand", "metadata") + var out searchResponse + if _, err := c.getJSON(ctx, itemsBase+"?"+params.Encode(), &out); err != nil { + return nil, err + } + return &out, nil +} + +// authority fetches an authority record, or nil when it is missing, deleted, +// or has no name heading. +func (c *Client) authority(ctx context.Context, id string) (*authorityRecord, error) { + var rec authorityRecord + found, err := c.getJSON(ctx, authorityBase+id+"?format=json", &rec) + if err != nil || !found || rec.Deleted || rec.heading() == "" { + return nil, err + } + return &rec, nil +} + +// authorityName returns the authorised name heading ("Last, First") for an +// authority record. found is false when the record is missing or deleted. +func (c *Client) authorityName(ctx context.Context, id string) (string, bool, error) { + rec, err := c.authority(ctx, id) + if err != nil || rec == nil { + return "", false, err + } + return rec.heading(), true, nil +} + +// getJSON GETs endpoint and decodes the body into out. found is false on 404. +func (c *Client) getJSON(ctx context.Context, endpoint string, out any) (bool, error) { + return c.get(ctx, endpoint, "application/json", func(r io.Reader) error { + return json.NewDecoder(r).Decode(out) + }) +} + +// getXML is getJSON for the MODS endpoint. +func (c *Client) getXML(ctx context.Context, endpoint string, out any) (bool, error) { + return c.get(ctx, endpoint, "application/xml", func(r io.Reader) error { + return xml.NewDecoder(r).Decode(out) + }) +} + +// get GETs endpoint and decodes the size-limited body. found is false on 404. +func (c *Client) get(ctx context.Context, endpoint, accept string, decode func(io.Reader) error) (bool, error) { + req, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil) + if err != nil { + return false, err + } + req.Header.Set("User-Agent", useragent.Get()) + req.Header.Set("Accept", accept) + + resp, err := c.http.Do(req) + if err != nil { + return false, err + } + defer resp.Body.Close() + + if resp.StatusCode == http.StatusNotFound { + return false, nil + } + if resp.StatusCode != http.StatusOK { + body, _ := io.ReadAll(io.LimitReader(resp.Body, 512)) + // A refusal or outage is the provider's state, not this request's: + // marked so scheduled discovery backs off. NB publishes no rate + // limits, so its refusals are taken at their word. + switch { + case resp.StatusCode == http.StatusTooManyRequests: + return false, fmt.Errorf("%w: HTTP %d: %s", providererr.ErrRateLimited, resp.StatusCode, string(body)) + case resp.StatusCode >= 500: + return false, fmt.Errorf("%w: HTTP %d: %s", providererr.ErrUnavailable, resp.StatusCode, string(body)) + } + return false, fmt.Errorf("HTTP %d: %s", resp.StatusCode, string(body)) + } + if err := decode(io.LimitReader(resp.Body, maxResponseBytes)); err != nil { + return false, fmt.Errorf("decode response: %w", err) + } + return true, nil +} + +func authorityIDFromForeignID(foreignID string) (string, error) { + id := strings.TrimPrefix(foreignID, authorPrefix) + if id == foreignID || !authorityIDRe.MatchString(id) { + return "", errors.New("nb: not an author id: " + strconv.Quote(foreignID)) + } + return id, nil +} + +// escapeQuery backslash-escapes the query_string syntax characters so user +// input is searched as text instead of being parsed as query operators. +func escapeQuery(s string) string { + var b strings.Builder + for _, r := range s { + if strings.ContainsRune(`+-=&|>`) + for _, e := range entries { // title, number, "linked" or "" + href := "" + if e[2] == "linked" { + href = ` xlink:href="(NO-TrBIB)10000001"` + } + b.WriteString(`` + e[0] + `` + e[1] + ``) + } + b.WriteString(``) + return b.String() +} + +// Some series are linked only on a translation's record. Such a series is a +// last resort: a book takes it only when its Norwegian and Scandinavian +// records give it none, not even an unlinked entry of a known series. It +// still counts as known for other volumes' unlinked Norwegian entries. +func TestFillSeries_TranslationFallback(t *testing.T) { + id := func(n int) string { return fmt.Sprintf("a00000000000000000000000000000%02d", n) } + mods := map[string]string{ + // Book 1: the Norwegian record names only an imprint; the series is + // linked on the Finnish record alone. + id(1): modsSeriesXML([3]string{"Eksempelkrim", "7", ""}), + id(2): modsSeriesXML([3]string{"Tunturisarja", "1", "linked"}), + // Book 2: an unlinked Norwegian entry of a series another book links, + // and a linked Finnish one. The Norwegian one wins. + id(3): modsSeriesXML([3]string{"Fjellserien", "2", ""}), + id(4): modsSeriesXML([3]string{"Tunturisarja", "2", "linked"}), + // Book 3: links the Norwegian series. + id(5): modsSeriesXML([3]string{"Fjellserien", "3", "linked"}), + // Book 4: an unlinked Norwegian entry of a series only a translation + // links (book 1's). + id(6): modsSeriesXML([3]string{"Tunturisarja", "4", ""}), + } + c := &Client{http: &http.Client{Transport: roundTripFunc(func(r *http.Request) (*http.Response, error) { + key := strings.TrimSuffix(strings.TrimPrefix(r.URL.Path, "/catalog/v1/metadata/"), "/mods") + body, ok := mods[key] + status := 200 + if !ok { + status = 404 + } + return &http.Response{StatusCode: status, Body: io.NopCloser(strings.NewReader(body)), Header: make(http.Header)}, nil + })}} + var items []item + for n := 1; n <= 6; n++ { + var m itemMetadata + m.Identifiers.SesamID, m.Series = id(n), []string{"x"} + items = append(items, item{ID: id(n), Metadata: m}) + } + book := func(rep int, eds ...models.Edition) models.Book { + return models.Book{ForeignID: idPrefix + id(rep), Language: "nob", Editions: eds} + } + fin := func(n int) models.Edition { return models.Edition{ForeignID: idPrefix + id(n), Language: "fin"} } + books := []models.Book{book(1, fin(2)), book(3, fin(4)), book(5), book(6)} + c.fillSeries(context.Background(), books, items, "10000001", nil) + + for i, want := range []string{"Tunturisarja 1", "Fjellserien 2", "Fjellserien 3", "Tunturisarja 4"} { + got := "" + if len(books[i].SeriesRefs) == 1 { + got = books[i].SeriesRefs[0].Title + " " + books[i].SeriesRefs[0].Position + } + if got != want { + t.Errorf("book %d series = %q, want %q", i+1, got, want) + } + } +} diff --git a/internal/metadata/nb/client_test.go b/internal/metadata/nb/client_test.go new file mode 100644 index 000000000..2e7ec097c --- /dev/null +++ b/internal/metadata/nb/client_test.go @@ -0,0 +1,763 @@ +package nb + +import ( + "context" + "errors" + "io" + "net/http" + "os" + "strings" + "sync" + "testing" + + "github.com/vavallee/bindery/internal/metadata" + "github.com/vavallee/bindery/internal/metadata/providererr" + "github.com/vavallee/bindery/internal/models" +) + +// The fixtures in testdata/ have the exact shape of api.nb.no and +// authority.bibsys.no responses, with an invented author, titles and ISBNs. + +type roundTripFunc func(*http.Request) (*http.Response, error) + +func (f roundTripFunc) RoundTrip(r *http.Request) (*http.Response, error) { return f(r) } + +// fakeNB answers each request from route, which maps it to (fixture file, +// status), and records every request it saw. +type fakeNB struct { + t *testing.T + mu sync.Mutex + reqs []*http.Request + route func(*http.Request) (string, int) +} + +func (f *fakeNB) client() *Client { + return &Client{http: &http.Client{Transport: roundTripFunc(func(r *http.Request) (*http.Response, error) { + f.mu.Lock() + f.reqs = append(f.reqs, r) + f.mu.Unlock() + file, status := f.route(r) + body := "" + if file != "" { + b, err := os.ReadFile("testdata/" + file) + if err != nil { + f.t.Fatalf("fixture: %v", err) + } + body = string(b) + } + return &http.Response{StatusCode: status, Body: io.NopCloser(strings.NewReader(body)), Header: make(http.Header)}, nil + })}} +} + +func isAuthority(r *http.Request) bool { return r.URL.Host == "authority.bibsys.no" } + +func TestSearchAuthors(t *testing.T) { + f := &fakeNB{t: t, route: func(*http.Request) (string, int) { return "author_works.json", 200 }} + authors, err := f.client().SearchAuthors(context.Background(), "kari nordmann") + if err != nil { + t.Fatal(err) + } + // Two people share the name; the authority ID keeps them apart. The + // co-credited author and translators are dropped by the name match. + if len(authors) != 2 { + t.Fatalf("got %d authors, want 2: %+v", len(authors), authors) + } + a := authors[0] + if a.ForeignID != "nb:author:10000001" || a.Name != "Kari Nordmann" || a.SortName != "Nordmann, Kari" || a.MetadataProvider != "nb" { + t.Errorf("first author = %+v", a) + } + if a.Statistics == nil || a.Statistics.BookCount != 6 { + t.Errorf("record count = %+v, want 6 (author credits only, not the translator credit)", a.Statistics) + } + if authors[1].ForeignID != "nb:author:10000002" { + t.Errorf("second author = %q", authors[1].ForeignID) + } + + q := f.reqs[0].URL.Query() + if got := q["filter"]; strings.Join(got, ",") != "api_nameauthor:kari,api_nameauthor:nordmann,mediatype:bøker" { + t.Errorf("filters = %v", got) + } + if q.Get("expand") != "metadata" { + t.Errorf("expand = %q; without it search hits carry no author credits", q.Get("expand")) + } + if ua := f.reqs[0].Header.Get("User-Agent"); !strings.HasPrefix(ua, "bindery/") { + t.Errorf("User-Agent = %q", ua) + } +} + +func TestGetBookByISBN_Audiobook(t *testing.T) { + f := &fakeNB{t: t, route: func(*http.Request) (string, int) { return "isbn_audiobook.json", 200 }} + b, err := f.client().GetBookByISBN(context.Background(), "978-82-00-00002-8") + if err != nil || b == nil { + t.Fatalf("book=%v err=%v", b, err) + } + if got := f.reqs[0].URL.Query().Get("q"); got != "isbn:9788200000028" { + t.Errorf("q = %q", got) + } + if got := f.reqs[0].URL.Query()["filter"]; len(got) != 0 { + t.Errorf("ISBN lookup must not filter by media type, got %v", got) + } + if b.ForeignID != "nb:a0000000000000000000000000000002" || b.Title != "Fjellvinden" || b.Language != "nob" { + t.Errorf("book = %q %q %q", b.ForeignID, b.Title, b.Language) + } + if b.Author == nil || b.Author.ForeignID != "nb:author:10000001" { + t.Errorf("author = %+v; the narrator must not be taken for the author", b.Author) + } + if len(b.Editions) != 1 || b.Editions[0].Format != "audiobook" || *b.Editions[0].ISBN13 != "9788200000028" { + t.Errorf("editions = %+v", b.Editions) + } +} + +func TestSearchBooks_ISBNQueryUsesISBNLookup(t *testing.T) { + f := &fakeNB{t: t, route: func(*http.Request) (string, int) { return "isbn_audiobook.json", 200 }} + books, err := f.client().SearchBooks(context.Background(), "9788200000028") + if err != nil || len(books) != 1 { + t.Fatalf("books=%v err=%v", books, err) + } + if got := f.reqs[0].URL.Query().Get("q"); got != "isbn:9788200000028" { + t.Errorf("q = %q", got) + } +} + +// The aggregator's canonical lookup searches the primary with "isbn:". +func TestSearchBooks_PrefixedISBNQueryUsesISBNLookup(t *testing.T) { + f := &fakeNB{t: t, route: func(*http.Request) (string, int) { return "isbn_audiobook.json", 200 }} + books, err := f.client().SearchBooks(context.Background(), "isbn:9788200000028") + if err != nil || len(books) != 1 { + t.Fatalf("books=%v err=%v", books, err) + } + if got := f.reqs[0].URL.Query().Get("q"); got != "isbn:9788200000028" { + t.Errorf("q = %q", got) + } +} + +func TestSearchBooks_TextSearchIsMetadataOnly(t *testing.T) { + f := &fakeNB{t: t, route: func(*http.Request) (string, int) { return "author_works.json", 200 }} + if _, err := f.client().SearchBooks(context.Background(), "fjellvinden: roman"); err != nil { + t.Fatal(err) + } + q := f.reqs[0].URL.Query() + // The default searchType includes OCR'd full text of digitised books. + if q.Get("searchType") != "FIELD_RESTRICTED_SEARCH" || q.Get("q") != `fjellvinden\: roman` { + t.Errorf("searchType=%q q=%q", q.Get("searchType"), q.Get("q")) + } + // Same media types as the author catalogue, so a work found by search + // carries the same editions (audiobook ISBNs included) as in the catalogue. + if got := q.Get("filter"); got != "mediatype:(bøker OR lydopptak)" { + t.Errorf("filter = %q", got) + } +} + +// The translation case this provider exists for: the author's catalogue +// carries the Norwegian original title, with the English and French +// translations folded into the same work instead of listed beside it. +func TestGetAuthorWorks_TranslationJoinsOriginal(t *testing.T) { + f := &fakeNB{t: t, route: func(r *http.Request) (string, int) { + if isAuthority(r) { + return "authority.json", 200 + } + return "author_works.json", 200 + }} + books, complete, err := f.client().GetAuthorWorksSnapshot(context.Background(), "nb:author:10000001") + if err != nil { + t.Fatal(err) + } + if !complete { + t.Error("a single-page catalogue must be complete") + } + // Homonym's book (10000002) and the book she only translated are excluded. + if len(books) != 3 { + titles := make([]string, len(books)) + for i, b := range books { + titles[i] = b.Title + } + t.Fatalf("got %d works %v, want 3", len(books), titles) + } + w := books[0] + if w.Title != "Fjellvinden" || w.ForeignID != "nb:a0000000000000000000000000000001" || w.Language != "nob" { + t.Errorf("work = %q %q %q, want the Norwegian print edition as representative", w.Title, w.ForeignID, w.Language) + } + if len(w.Editions) != 4 { + t.Errorf("editions = %d, want 4 (print, audio, English, French)", len(w.Editions)) + } + if w.ReleaseDate == nil || w.ReleaseDate.Year() != 2019 { + t.Errorf("release = %v, want the earliest edition's year", w.ReleaseDate) + } + if w.Description == "" { + t.Error("description missing") + } + if strings.Join(w.ProviderISBNs, ",") != "9788200000011,9788200000028,9781000000012,9782000000013" { + t.Errorf("ISBNs = %v", w.ProviderISBNs) + } + // An original's noisy uniform title ("Noveller Utvalg") must not rename it. + if books[1].Title != "Havets stemme" || books[1].ReleaseDate.Year() != 2015 { + t.Errorf("second work = %q %v", books[1].Title, books[1].ReleaseDate) + } + + q := f.reqs[1].URL.Query() + if got := strings.Join(q["filter"], ","); got != `namecreators:"Nordmann, Kari",mediatype:(bøker OR lydopptak)` { + t.Errorf("filters = %s", got) + } +} + +func TestGetAuthorWorks_PartialWhenCapped(t *testing.T) { + b, err := os.ReadFile("testdata/author_works.json") + if err != nil { + t.Fatal(err) + } + many := strings.Replace(string(b), `"totalPages": 1`, `"totalPages": 99`, 1) + calls := 0 + c := &Client{http: &http.Client{Transport: roundTripFunc(func(r *http.Request) (*http.Response, error) { + body := many + if isAuthority(r) { + a, _ := os.ReadFile("testdata/authority.json") + body = string(a) + } else if strings.HasSuffix(r.URL.Path, "/items") { + calls++ + } + return &http.Response{StatusCode: 200, Body: io.NopCloser(strings.NewReader(body)), Header: make(http.Header)}, nil + })}} + _, complete, err := c.GetAuthorWorksSnapshot(context.Background(), "nb:author:10000001") + if err != nil { + t.Fatal(err) + } + if complete || calls != maxWorksPages { + t.Errorf("complete=%v pages=%d; a capped catalogue must report partial", complete, calls) + } +} + +// An upstream failure must surface as an error, never as an empty catalogue: +// the aggregator treats an empty answer from the primary as fact (#2332). +func TestGetAuthorWorks_ErrorsAreNotEmpty(t *testing.T) { + for name, route := range map[string]func(*http.Request) (string, int){ + "authority down": func(r *http.Request) (string, int) { + if isAuthority(r) { + return "", 503 + } + return "author_works.json", 200 + }, + "search down": func(r *http.Request) (string, int) { + if isAuthority(r) { + return "authority.json", 200 + } + return "", 500 + }, + } { + f := &fakeNB{t: t, route: route} + if books, err := f.client().GetAuthorWorks(context.Background(), "nb:author:10000001"); err == nil { + t.Errorf("%s: got %d books and no error", name, len(books)) + } + } +} + +// routeWithMODS serves the search, authority and MODS endpoints from the +// fixtures, MODS by record ID. +func routeWithMODS(mods map[string]string) func(*http.Request) (string, int) { + return func(r *http.Request) (string, int) { + switch { + case isAuthority(r): + return "authority.json", 200 + case strings.HasSuffix(r.URL.Path, "/mods"): + id := strings.TrimSuffix(strings.TrimPrefix(r.URL.Path, "/catalog/v1/metadata/"), "/mods") + if f, ok := mods[id]; ok { + return f, 200 + } + return "", 404 + case strings.HasSuffix(r.URL.Path, "/items/a0000000000000000000000000000001"): + return "item_print.json", 200 + case strings.Contains(r.URL.Query().Get("q"), "Fjellserien"): + return "series_search.json", 200 + } + return "author_works.json", 200 + } +} + +func modsRequests(f *fakeNB) []string { + var ids []string + for _, r := range f.reqs { + if strings.HasSuffix(r.URL.Path, "/mods") { + ids = append(ids, r.URL.Path) + } + } + return ids +} + +// The series number is only in the per-record MODS. Only the author's own +// series counts: the one linked to their authority ID, not a publisher's +// imprint series, which NB also records as a series. +func TestGetAuthorWorks_SeriesFromMODS(t *testing.T) { + f := &fakeNB{t: t, route: routeWithMODS(map[string]string{ + "a0000000000000000000000000000001": "mods_author_series.xml", + "a0000000000000000000000000000005": "mods_publisher_series.xml", + })} + books, err := f.client().GetAuthorWorks(context.Background(), "nb:author:10000001") + if err != nil || len(books) != 4 { + t.Fatalf("books=%d err=%v", len(books), err) + } + want := models.SeriesRef{ForeignID: "nb-series:10000001:fjellserien", Title: "Fjellserien", Position: "2", Primary: true} + if len(books[0].SeriesRefs) != 1 || books[0].SeriesRefs[0] != want { + t.Errorf("series = %+v, want %+v", books[0].SeriesRefs, want) + } + if len(books[1].SeriesRefs) != 0 { + t.Errorf("publisher imprint taken as a series: %+v", books[1].SeriesRefs) + } + // Only records whose search hit lists a series are fetched: two from the + // catalogue and the one the series recall adds. + if got := modsRequests(f); len(got) != 3 { + t.Errorf("MODS requests = %v, want 3", got) + } + if ua := f.reqs[len(f.reqs)-1].Header.Get("User-Agent"); !strings.HasPrefix(ua, "bindery/") { + t.Errorf("MODS User-Agent = %q", ua) + } +} + +// NB sometimes catalogues a volume under its series' name, with the volume's +// own title and number only in the part fields: " : ", the +// number on the uniform title. The book must carry the volume's title, and +// the number counts once the author's catalogue links that series elsewhere. +func TestGetAuthorWorks_VolumeTitledBySeries(t *testing.T) { + f := &fakeNB{t: t, route: routeWithMODS(map[string]string{ + "a0000000000000000000000000000001": "mods_author_series.xml", + })} + books, err := f.client().GetAuthorWorks(context.Background(), "nb:author:10000001") + if err != nil { + t.Fatal(err) + } + var vol *models.Book + for i := range books { + if books[i].ForeignID == "nb:a0000000000000000000000000000008" { + vol = &books[i] + } + if books[i].Title == "Fjellserien" { + t.Errorf("a volume took the series name as its title: %+v", books[i]) + } + } + if vol == nil { + t.Fatalf("volume missing from %d works", len(books)) + } + if vol.Title != "Siste vinter" || vol.Editions[0].Title != "Siste vinter" { + t.Errorf("title = %q, edition title = %q, want the part name", vol.Title, vol.Editions[0].Title) + } + want := models.SeriesRef{ForeignID: "nb-series:10000001:fjellserien", Title: "Fjellserien", Position: "3", Primary: true} + if len(vol.SeriesRefs) != 1 || vol.SeriesRefs[0] != want { + t.Errorf("series = %+v, want %+v", vol.SeriesRefs, want) + } +} + +func TestPartPosition(t *testing.T) { + for in, want := range map[string]string{"6": "6", "[5]": "5", "3.": "3", " [12]. ": "12", "": ""} { + if got := partPosition(in); got != want { + t.Errorf("partPosition(%q) = %q, want %q", in, got, want) + } + } +} + +// A cataloguer sometimes records the author's series without the authority +// link. That entry is accepted only when the author's catalogue links the same +// series elsewhere; an unlinked series nobody links is still a publisher's. +func TestGetAuthorWorks_UnlinkedEntryOfKnownSeries(t *testing.T) { + f := &fakeNB{t: t, route: routeWithMODS(map[string]string{ + "a0000000000000000000000000000001": "mods_author_series.xml", + "a0000000000000000000000000000005": "mods_unlinked_series.xml", + })} + books, err := f.client().GetAuthorWorks(context.Background(), "nb:author:10000001") + if err != nil || len(books) != 4 { + t.Fatalf("books=%d err=%v", len(books), err) + } + want := models.SeriesRef{ForeignID: "nb-series:10000001:fjellserien", Title: "Fjellserien", Position: "3", Primary: true} + if len(books[1].SeriesRefs) != 1 || books[1].SeriesRefs[0] != want { + t.Errorf("series = %+v, want %+v", books[1].SeriesRefs, want) + } +} + +// NB leaves some records out of every name index, so the author query never +// returns them, though they credit the author. A search on each of the +// author's series names recovers them, keeping only records that credit the +// author's authority ID. +func TestGetAuthorWorks_RecallsVolumesMissingFromNameIndex(t *testing.T) { + f := &fakeNB{t: t, route: routeWithMODS(map[string]string{ + "a0000000000000000000000000000001": "mods_author_series.xml", + "a0000000000000000000000000000009": "mods_series_4.xml", + })} + books, complete, err := f.client().GetAuthorWorksSnapshot(context.Background(), "nb:author:10000001") + if err != nil || !complete { + t.Fatalf("complete=%v err=%v", complete, err) + } + var recalled *models.Book + for i := range books { + switch books[i].ForeignID { + case "nb:a0000000000000000000000000000009": + recalled = &books[i] + case "nb:a0000000000000000000000000000010": + t.Errorf("another author's record was kept: %+v", books[i]) + } + } + if recalled == nil { + t.Fatalf("volume missing from the name index was not recovered (%d works)", len(books)) + } + want := models.SeriesRef{ForeignID: "nb-series:10000001:fjellserien", Title: "Fjellserien", Position: "4", Primary: true} + if len(recalled.SeriesRefs) != 1 || recalled.SeriesRefs[0] != want { + t.Errorf("series = %+v, want %+v", recalled.SeriesRefs, want) + } + if len(books) != 4 { + t.Errorf("works = %d, want 4 (the catalogue's 3 plus the recalled one, no duplicates)", len(books)) + } + + var recall *http.Request + mods := map[string]int{} + for _, r := range f.reqs { + if strings.Contains(r.URL.Query().Get("q"), "Fjellserien") { + recall = r + } + if strings.HasSuffix(r.URL.Path, "/mods") { + mods[r.URL.Path]++ + } + } + if recall == nil { + t.Fatal("no series-name search was made") + } + q := recall.URL.Query() + if q.Get("q") != `"Fjellserien"` || q.Get("searchType") != "FIELD_RESTRICTED_SEARCH" || q.Get("filter") != "mediatype:(bøker OR lydopptak)" { + t.Errorf("recall query q=%q searchType=%q filter=%v", q.Get("q"), q.Get("searchType"), q["filter"]) + } + // Regrouping after the recall must not fetch a record's MODS twice. + for path, n := range mods { + if n > 1 { + t.Errorf("MODS %s fetched %d times", path, n) + } + } +} + +// The authority record exists but the name search finds nothing: that is a +// gap in NB's name index, not proof the author has no books, so it must not +// pass as a complete, empty catalogue that reconciliation could act on. +func TestGetAuthorWorks_EmptySearchIsPartial(t *testing.T) { + f := &fakeNB{t: t, route: func(r *http.Request) (string, int) { + if isAuthority(r) { + return "authority.json", 200 + } + return "empty_search.json", 200 + }} + books, complete, err := f.client().GetAuthorWorksSnapshot(context.Background(), "nb:author:10000001") + if err != nil || complete || len(books) != 0 { + t.Errorf("books=%d complete=%v err=%v, want none, partial, no error", len(books), complete, err) + } +} + +// An authority record that is gone (404, or deleted when Sikt merges +// duplicate records) says nothing about the author's books: an empty, +// complete catalogue would have reconciliation offer to remove them all. +func TestGetAuthorWorks_MissingAuthorityIsPartial(t *testing.T) { + for name, authority := range map[string]func() (string, int){ + "404": func() (string, int) { return "", 404 }, + "deleted": func() (string, int) { return "authority_deleted.json", 200 }, + } { + f := &fakeNB{t: t, route: func(r *http.Request) (string, int) { + if isAuthority(r) { + return authority() + } + return "author_works.json", 200 + }} + books, complete, err := f.client().GetAuthorWorksSnapshot(context.Background(), "nb:author:10000001") + if err != nil || complete || len(books) != 0 { + t.Errorf("%s: books=%d complete=%v err=%v, want none, partial, no error", name, len(books), complete, err) + } + } +} + +// A failed recall search must not pass as a complete catalogue: a recovered +// volume missing from it would read as removed upstream. +func TestGetAuthorWorks_RecallFailureIsPartial(t *testing.T) { + f := &fakeNB{t: t, route: func(r *http.Request) (string, int) { + if strings.Contains(r.URL.Query().Get("q"), "Fjellserien") { + return "", 503 + } + return routeWithMODS(map[string]string{"a0000000000000000000000000000001": "mods_author_series.xml"})(r) + }} + books, complete, err := f.client().GetAuthorWorksSnapshot(context.Background(), "nb:author:10000001") + if err != nil || complete || len(books) != 3 { + t.Errorf("books=%d complete=%v err=%v, want the catalogue, partial, no error", len(books), complete, err) + } +} + +// Series is enrichment: a MODS failure leaves the work without a series but +// must not fail or empty the catalogue. +func TestGetAuthorWorks_SeriesFailureIsNotFatal(t *testing.T) { + f := &fakeNB{t: t, route: func(r *http.Request) (string, int) { + if strings.HasSuffix(r.URL.Path, "/mods") { + return "", 503 + } + return routeWithMODS(nil)(r) + }} + books, err := f.client().GetAuthorWorks(context.Background(), "nb:author:10000001") + if err != nil || len(books) != 3 || len(books[0].SeriesRefs) != 0 { + t.Errorf("books=%d err=%v series=%v", len(books), err, books[0].SeriesRefs) + } +} + +// A rebind replaces a book's series with what GetBook returns, so GetBook +// must carry the series too. +func TestGetBook_CarriesSeries(t *testing.T) { + f := &fakeNB{t: t, route: routeWithMODS(map[string]string{ + "a0000000000000000000000000000001": "mods_author_series.xml", + })} + b, err := f.client().GetBook(context.Background(), "nb:a0000000000000000000000000000001") + if err != nil || b == nil { + t.Fatalf("book=%v err=%v", b, err) + } + if len(b.SeriesRefs) != 1 || b.SeriesRefs[0].Position != "2" { + t.Errorf("series = %+v", b.SeriesRefs) + } +} + +func TestGetAuthor(t *testing.T) { + f := &fakeNB{t: t, route: func(*http.Request) (string, int) { return "authority.json", 200 }} + a, err := f.client().GetAuthor(context.Background(), "nb:author:10000001") + if err != nil || a == nil { + t.Fatalf("author=%v err=%v", a, err) + } + if a.Name != "Kari Nordmann" || a.ForeignID != "nb:author:10000001" { + t.Errorf("author = %+v", a) + } + // The authority record's name variants, in display form. Bindery decides + // which become aliases (textutil.LatinAliasBinds). + if got := strings.Join(a.AlternateNames, "|"); got != "Kari Nordman|Кари Нордманн" { + t.Errorf("alternate names = %q", got) + } + if got := f.reqs[0].URL.String(); got != authorityBase+"10000001?format=json" { + t.Errorf("url = %s", got) + } + + missing := &fakeNB{t: t, route: func(*http.Request) (string, int) { return "", 404 }} + if a, err := missing.client().GetAuthor(context.Background(), "nb:author:99999999"); a != nil || err != nil { + t.Errorf("unknown authority: author=%v err=%v, want nil, nil", a, err) + } + if _, err := missing.client().GetAuthor(context.Background(), "nb:author:../x"); err == nil { + t.Error("malformed id must be rejected before any request") + } +} + +// GetBook must return the whole work, not one record: the aggregator refreshes +// an ISBN match through it, and a print record alone would drop the +// audiobook's ISBN from the book. +func TestGetBook_ReturnsWorkEditions(t *testing.T) { + f := &fakeNB{t: t, route: func(r *http.Request) (string, int) { + if strings.HasSuffix(r.URL.Path, "/items/a0000000000000000000000000000001") { + return "item_print.json", 200 + } + return "author_works.json", 200 + }} + b, err := f.client().GetBook(context.Background(), "nb:a0000000000000000000000000000001") + if err != nil || b == nil { + t.Fatalf("book=%v err=%v", b, err) + } + if b.ForeignID != "nb:a0000000000000000000000000000001" || len(b.Editions) != 4 { + t.Errorf("book %s has %d editions, want the work's 4", b.ForeignID, len(b.Editions)) + } + q := f.reqs[1].URL.Query() + // Not filtered on NB's name index, which misses some records; groupWorks + // keeps only records crediting the author's authority ID. + if got := strings.Join(q["filter"], ","); got != `mediatype:(bøker OR lydopptak)` || q.Get("q") != "Fjellvinden" { + t.Errorf("sibling search q=%q filters=%s", q.Get("q"), got) + } + + // The sibling search is best-effort: the record itself is still returned. + down := &fakeNB{t: t, route: func(r *http.Request) (string, int) { + if strings.Contains(r.URL.Path, "/items/") { + return "item_print.json", 200 + } + return "", 503 + }} + b, err = down.client().GetBook(context.Background(), "nb:a0000000000000000000000000000001") + if err != nil || b == nil || len(b.Editions) != 1 { + t.Errorf("sibling search down: book=%v err=%v", b, err) + } +} + +func TestGetBook_RejectsMalformedID(t *testing.T) { + f := &fakeNB{t: t, route: func(*http.Request) (string, int) { return "", 200 }} + for _, id := range []string{"OL1W", "nb:author:10000001", "nb:../../x"} { + if _, err := f.client().GetBook(context.Background(), id); err == nil { + t.Errorf("GetBook(%q) succeeded", id) + } + } + if len(f.reqs) != 0 { + t.Errorf("made %d requests for malformed ids", len(f.reqs)) + } +} + +// A refusal or outage is marked with the shared provider errors, so +// scheduled discovery backs off instead of walking on through its queue. +func TestProviderErrors(t *testing.T) { + for status, want := range map[int]error{ + 429: providererr.ErrRateLimited, + 500: providererr.ErrUnavailable, + 502: providererr.ErrUnavailable, + 503: providererr.ErrUnavailable, + } { + f := &fakeNB{t: t, route: func(*http.Request) (string, int) { return "", status }} + _, err := f.client().GetBookByISBN(context.Background(), "9788200000028") + if !errors.Is(err, want) { + t.Errorf("HTTP %d: err = %v, want %v", status, err, want) + } + } + // A bad request is about this request, not the provider being down. + f := &fakeNB{t: t, route: func(*http.Request) (string, int) { return "", 400 }} + _, err := f.client().GetBookByISBN(context.Background(), "9788200000028") + if err == nil || errors.Is(err, providererr.ErrRateLimited) || errors.Is(err, providererr.ErrUnavailable) { + t.Errorf("HTTP 400: err = %v, want a plain error", err) + } +} + +// Narrator, audiobook duration and genres come from the same catalogue +// records the work is built from: no extra requests. +func TestGetAuthorWorks_NarratorDurationGenres(t *testing.T) { + f := &fakeNB{t: t, route: routeWithMODS(nil)} + books, err := f.client().GetAuthorWorks(context.Background(), "nb:author:10000001") + if err != nil { + t.Fatal(err) + } + var work, part *models.Book + for i := range books { + switch books[i].ForeignID { + case "nb:a0000000000000000000000000000001": + work = &books[i] + case "nb:a0000000000000000000000000000008": + part = &books[i] + } + } + if work == nil || part == nil { + t.Fatalf("works missing from %d", len(books)) + } + if work.Narrator != "Ola Leser" { + t.Errorf("narrator = %q, want the audiobook edition's narrator credit", work.Narrator) + } + const want = 11*3600 + 16*60 + if work.DurationSeconds != want { + t.Errorf("duration = %d, want %d from the audiobook edition", work.DurationSeconds, want) + } + for _, ed := range work.Editions { + if ed.Format == models.MediaTypeAudiobook && ed.DurationSeconds != want { + t.Errorf("audiobook edition duration = %d, want %d", ed.DurationSeconds, want) + } + if ed.Format != models.MediaTypeAudiobook && ed.DurationSeconds != 0 { + t.Errorf("print edition has a duration: %+v", ed) + } + } + if got := strings.Join(work.Genres, ","); got != "Romaner,Krim,Politi og detektiver" { + t.Errorf("genres = %q, want the subject genres without the format term or Nynorsk twins", got) + } + if part.DurationSeconds != 17*3600+42*60 { + t.Errorf("hh:mm:ss duration = %d", part.DurationSeconds) + } +} + +func TestExtentDuration(t *testing.T) { + for in, want := range map[string]int{ + "1 lydfil (11 t, 16 min)": 11*3600 + 16*60, + "21:34:00": 21*3600 + 34*60, + "3 plater (CD)(3 t, 7 min) digital 12 cm, i eske": 3*3600 + 7*60, + "1 lydfil (45 min)": 45 * 60, + "312 s.": 0, + "": 0, + } { + if got := extentDuration(in); got != want { + t.Errorf("extentDuration(%q) = %d, want %d", in, got, want) + } + } +} + +func TestNotConfigured(t *testing.T) { + ctx := context.Background() + var c Client + checks := map[string]error{} + _, checks["SearchAuthors"] = c.SearchAuthors(ctx, "x") + _, checks["SearchBooks"] = c.SearchBooks(ctx, "x") + _, checks["GetAuthor"] = c.GetAuthor(ctx, "nb:author:1") + _, checks["GetAuthorWorks"] = c.GetAuthorWorks(ctx, "nb:author:1") + _, checks["GetBook"] = c.GetBook(ctx, "nb:a0000000000000000000000000000001") + _, checks["GetEditions"] = c.GetEditions(ctx, "nb:a0000000000000000000000000000001") + _, checks["GetBookByISBN"] = c.GetBookByISBN(ctx, "9788200000028") + for name, err := range checks { + if !errors.Is(err, metadata.ErrProviderNotConfigured) { + t.Errorf("%s: err = %v, want ErrProviderNotConfigured", name, err) + } + } +} + +func TestStripLanguageQualifier(t *testing.T) { + for in, want := range map[string]string{ + "Fjellvinden Fransk": "Fjellvinden", + "The quiet one Norsk": "The quiet one", + "Fjellvinden": "Fjellvinden", + "Havet og Os": "Havet og Os", + "Mot øst": "Mot øst", + // A name ending like a language word is not a language qualifier. + "Inspektør Brask": "Inspektør Brask", + "Kiosk": "Kiosk", + } { + if got := stripLanguageQualifier(in); got != want { + t.Errorf("%q: got %q, want %q", in, got, want) + } + } +} + +// A catalogue slip that splits a title with a space ("Fjel lbyen") must not +// make a second book, nor name the book: the title most editions carry wins. +func TestGroupWorks_FoldsSpacingSlip(t *testing.T) { + kari := person{Name: "Nordmann, Kari", Identifier: "bibsys.no:authority:10000001", Roles: []struct { + Name string `json:"name"` + }{{Name: "aut"}}} + rec := func(id, title string, tis []titleInfo, media string) item { + var m itemMetadata + m.Title, m.TitleInfos, m.People, m.MediaTypes = title, tis, []person{kari}, []string{media} + m.Identifiers.SesamID = id + m.Languages = []struct { + Code string `json:"code"` + }{{Code: "nob"}} + return item{ID: id, Metadata: m} + } + books := groupWorks([]item{ + rec("b0000000000000000000000000000001", "Fjel lbyen : roman", []titleInfo{{Title: "lbyen "}}, "bøker"), + rec("b0000000000000000000000000000002", "Fjellbyen : roman", []titleInfo{{Title: "Fjellbyen"}}, "bøker"), + rec("b0000000000000000000000000000003", "Fjellserien. [5] : Fjellbyen", []titleInfo{{Title: "jellserien", PartName: "Fjellbyen", PartNumber: "[5]"}}, "lydopptak"), + rec("b0000000000000000000000000000004", "Fjell byen og havet", []titleInfo{{Title: "Fjell byen og havet"}}, "bøker"), + }, "10000001") + if len(books) != 2 { + t.Fatalf("works = %d, want 2", len(books)) + } + if books[0].Title != "Fjellbyen" || len(books[0].Editions) != 3 { + t.Errorf("work = %q with %d editions, want Fjellbyen with 3", books[0].Title, len(books[0].Editions)) + } + if books[1].Title != "Fjell byen og havet" { + t.Errorf("a different title was folded in: %q", books[1].Title) + } +} + +// NB can name one series two ways, linking both to the author. Names a +// work's records give the same number are one series across the catalogue, +// under the name most books carry. An unlinked imprint sharing the number is +// not part of it. +func TestGetAuthorWorks_MergesSeriesNamedTwoWays(t *testing.T) { + f := &fakeNB{t: t, route: routeWithMODS(map[string]string{ + "a0000000000000000000000000000001": "mods_author_series.xml", + "a0000000000000000000000000000005": "mods_series_alias.xml", + "a0000000000000000000000000000009": "mods_series_4.xml", + })} + books, err := f.client().GetAuthorWorks(context.Background(), "nb:author:10000001") + if err != nil { + t.Fatal(err) + } + positions := map[string]string{} + for _, b := range books { + for _, ref := range b.SeriesRefs { + if ref.ForeignID != "nb-series:10000001:fjellserien" || ref.Title != "Fjellserien" { + t.Errorf("%s: series = %+v, want the one Fjellserien", b.Title, ref) + } + positions[b.ForeignID] = ref.Position + } + } + if positions["nb:a0000000000000000000000000000005"] != "3" || len(positions) < 3 { + t.Errorf("positions = %v, want the aliased volume at 3 among the series", positions) + } +} diff --git a/internal/metadata/nb/series.go b/internal/metadata/nb/series.go new file mode 100644 index 000000000..b02c7d514 --- /dev/null +++ b/internal/metadata/nb/series.go @@ -0,0 +1,390 @@ +package nb + +import ( + "context" + "log/slog" + "strings" + "sync" + "unicode" + + "github.com/vavallee/bindery/internal/concurrency" + "github.com/vavallee/bindery/internal/models" + "github.com/vavallee/bindery/internal/textutil" +) + +const ( + seriesIDPrefix = "nb-series:" + // seriesConcurrency bounds MODS requests in flight for one catalogue. + seriesConcurrency = 4 + // maxSeriesProbes caps MODS requests per work. Not every edition records + // the author's series (a reprint may list only the publisher's imprint), + // so a few are tried before giving up. + maxSeriesProbes = 3 +) + +// modsRecord is the part of a MODS record that carries series membership. +type modsRecord struct { + RelatedItems []struct { + Type string `xml:"type,attr"` + Href string `xml:"http://www.w3.org/1999/xlink href,attr"` + Title string `xml:"titleInfo>title"` + Part string `xml:"titleInfo>partNumber"` + } `xml:"relatedItem"` +} + +// seriesRecords returns the IDs of the records whose search hit names any +// series. Only those are worth a MODS request; the search JSON has the name +// but not the number. +func seriesRecords(items []item) map[string]bool { + ids := make(map[string]bool) + for _, it := range items { + if len(it.Metadata.Series) > 0 { + ids[firstNonEmpty(it.Metadata.Identifiers.SesamID, it.ID)] = true + } + } + return ids +} + +// fillSeries sets each book's series and position from the MODS record of +// one of its editions. Best-effort: series is enrichment, so a failed lookup +// leaves the book without one rather than failing the catalogue. +// +// Two passes. The author's own series are the entries linked to their +// authority record. A cataloguer occasionally records one without the link; +// such an entry is accepted only when its series is linked elsewhere in the +// same books, because an unlinked series nobody links is a publisher imprint. +// +// The second pass also takes the number a volume carries in its own title +// fields (see partSeries), on the same condition. +// +// Both passes read only the book's Norwegian and other Scandinavian records +// (see seriesLanguage). A series linked on a translation's record alone is +// the last resort, for books both passes leave without one. Finally, names +// one work's records give the same number are merged into one series (see +// mergeSeriesNames). +func (c *Client) fillSeries(ctx context.Context, books []models.Book, items []item, authorID string, memo *seriesMemo) { + if memo == nil { + memo = &seriesMemo{} + } + withSeries := seriesRecords(items) + parts := partSeries(items, authorID) + var todo []int + for i := range books { + if p, f := seriesCandidates(books[i], withSeries); len(p)+len(f) > 0 { + todo = append(todo, i) + } + } + unlinked := make([][]models.SeriesRef, len(books)) + same := make([][]string, len(books)) // series IDs naming the book's series + // Each goroutine writes only its own books[i], unlinked[i] and same[i]. + concurrency.RunBounded(ctx, todo, seriesConcurrency, func(ctx context.Context, i int) { + var linked []models.SeriesRef + preferred, _ := seriesCandidates(books[i], withSeries) + for _, id := range preferred { + l, other, err := memo.recordSeries(ctx, c, id, authorID) + if err != nil { + slog.Debug("nb: series lookup failed", "record", id, "error", err) + break + } + linked = append(linked, l...) + unlinked[i] = append(unlinked[i], other...) + } + if len(linked) == 0 { + return + } + // Not every record numbers the volume; take a numbered entry when + // one exists. + for j, ref := range linked { + if ref.Position != "" { + linked[0], linked[j] = linked[j], linked[0] + break + } + } + books[i].SeriesRefs = []models.SeriesRef{linked[0]} + // NB names one series differently across records ("<A> og <B>", + // "<full name A> og <full name B>"). Names linked to the author that + // the work's records give the same number are one series. Unlinked + // names are not evidence: a publisher's imprint can share the number. + for _, ref := range linked { + if ref.Position == linked[0].Position { + same[i] = append(same[i], ref.ForeignID) + } + } + }) + + // Last resort: a series linked only on a translation's record. A foreign + // name beats none, but never one the Norwegian records supply, so it is + // applied after the second pass. It still counts as known there: another + // volume's Norwegian record may name the same series unlinked. + fallback := make([]*models.SeriesRef, len(books)) + var rest []int + for i := range books { + if _, f := seriesCandidates(books[i], withSeries); len(books[i].SeriesRefs) == 0 && len(f) > 0 { + rest = append(rest, i) + } + } + concurrency.RunBounded(ctx, rest, seriesConcurrency, func(ctx context.Context, i int) { + _, ids := seriesCandidates(books[i], withSeries) + for _, id := range ids { + linked, _, err := memo.recordSeries(ctx, c, id, authorID) + if err != nil { + slog.Debug("nb: series lookup failed", "record", id, "error", err) + return + } + if len(linked) > 0 { + fallback[i] = &linked[0] + return + } + } + }) + + known := make(map[string]bool) + for i, b := range books { + if fallback[i] != nil { + known[fallback[i].ForeignID] = true + } + for _, ref := range b.SeriesRefs { + known[ref.ForeignID] = true + } + for _, id := range same[i] { + known[id] = true + } + } + for i := range books { + if len(books[i].SeriesRefs) > 0 { + continue + } + candidates := unlinked[i] + for _, ed := range books[i].Editions { + if !seriesLanguage(books[i].Language, ed.Language) { + continue + } + if ref, ok := parts[strings.TrimPrefix(ed.ForeignID, idPrefix)]; ok { + candidates = append(candidates, ref) + } + } + for _, ref := range candidates { + if known[ref.ForeignID] { + books[i].SeriesRefs = []models.SeriesRef{ref} + break + } + } + } + + for i := range books { + if len(books[i].SeriesRefs) == 0 && fallback[i] != nil { + books[i].SeriesRefs = []models.SeriesRef{*fallback[i]} + } + } + mergeSeriesNames(books, same) +} + +// mergeSeriesNames gives every name of one series the same ID and title: the +// name most of the books carry, the longer on a tie. same[i] lists the series +// IDs book i's records give its number; names linked through any book are +// one series. +func mergeSeriesNames(books []models.Book, same [][]string) { + parent := make(map[string]string) + var find func(string) string + find = func(id string) string { + if p, ok := parent[id]; ok && p != id { + parent[id] = find(p) + return parent[id] + } + parent[id] = id + return id + } + for _, ids := range same { + for _, id := range ids[min(1, len(ids)):] { + parent[find(id)] = find(ids[0]) + } + } + if len(parent) == 0 { + return + } + count := make(map[string]int) + title := make(map[string]string) + for _, b := range books { + for _, ref := range b.SeriesRefs { + count[ref.ForeignID]++ + title[ref.ForeignID] = ref.Title + } + } + best := make(map[string]string) // group root -> chosen series ID + for id := range count { + root := find(id) + cur, ok := best[root] + if !ok || count[id] > count[cur] || (count[id] == count[cur] && (len(title[id]) > len(title[cur]) || len(title[id]) == len(title[cur]) && id < cur)) { + best[root] = id + } + } + for i := range books { + for j, ref := range books[i].SeriesRefs { + if id := best[find(ref.ForeignID)]; id != "" && id != ref.ForeignID { + books[i].SeriesRefs[j].ForeignID, books[i].SeriesRefs[j].Title = id, title[id] + } + } + } +} + +// partSeries returns, per record ID, the series a record names in its own +// title fields: a record catalogued as part n of a larger work carries that +// work's name and the number in a titleInfo (the uniform title preferred, +// since the title proper may have its leading article split off). It is a +// candidate only: the work may be an omnibus rather than a series, so +// fillSeries takes it only when the author's catalogue links that series. +func partSeries(items []item, authorID string) map[string]models.SeriesRef { + refs := make(map[string]models.SeriesRef) + for _, it := range items { + var found *titleInfo + for i, ti := range it.Metadata.TitleInfos { + if strings.TrimSpace(ti.PartName) == "" || partPosition(ti.PartNumber) == "" { + continue + } + if found == nil || ti.Type == "uniform" { + found = &it.Metadata.TitleInfos[i] + } + } + if found == nil { + continue + } + title := stripLanguageQualifier(strings.Join(strings.Fields(found.Title), " ")) + slug := seriesSlug(title) + if slug == "" { + continue + } + refs[firstNonEmpty(it.Metadata.Identifiers.SesamID, it.ID)] = models.SeriesRef{ + ForeignID: seriesIDPrefix + authorID + ":" + slug, + Title: title, + Position: partPosition(found.PartNumber), + Primary: true, + } + } + return refs +} + +// partPosition cleans a catalogue part number: "[5]" (supplied by the +// cataloguer) and "3." both mean the plain number. +func partPosition(s string) string { + return strings.Trim(strings.TrimSpace(s), "[]. ") +} + +// seriesCandidates lists the book's record IDs that name a series: preferred +// are the representative, then editions in the book's own language, then +// other Scandinavian ones (see seriesLanguage), at most maxSeriesProbes. +// fallback are the other translations, at most maxSeriesProbes, tried only +// when no preferred record links a series: some series are linked on a +// translation's record alone, and a foreign name beats none. +func seriesCandidates(b models.Book, withSeries map[string]bool) (preferred, fallback []string) { + seen := make(map[string]bool) + add := func(ids *[]string, foreignID string) { + id := strings.TrimPrefix(foreignID, idPrefix) + if withSeries[id] && !seen[id] && len(*ids) < maxSeriesProbes { + seen[id] = true + *ids = append(*ids, id) + } + } + add(&preferred, b.ForeignID) + for _, own := range []bool{true, false} { + for _, ed := range b.Editions { + if (ed.Language == b.Language) == own && seriesLanguage(b.Language, ed.Language) { + add(&preferred, ed.ForeignID) + } + } + } + for _, ed := range b.Editions { + if !seriesLanguage(b.Language, ed.Language) { + add(&fallback, ed.ForeignID) + } + } + return preferred, fallback +} + +// scandinavian are the languages whose editions name an author's series as +// the Norwegian ones do ("Min kamp", "Barrøy"), often when the Norwegian +// records name none. +var scandinavian = map[string]bool{"nob": true, "nno": true, "nor": true, "dan": true, "swe": true} + +// seriesLanguage reports whether an edition in language ed may supply the +// series of a book in language book: its own language, or between +// Scandinavian languages. Other translations name the series in their own +// language ("<series>-romaani"), which is not the series the book belongs to. +func seriesLanguage(book, ed string) bool { + return ed == book || scandinavian[book] && scandinavian[ed] +} + +// seriesMemo caches recordSeries results for one catalogue fetch, which may +// fill series twice (see recallSeriesVolumes). The zero value is ready. +type seriesMemo struct { + mu sync.Mutex + seen map[string]modsSeries +} + +type modsSeries struct { + linked []models.SeriesRef + unlinked []models.SeriesRef + err error +} + +func (m *seriesMemo) recordSeries(ctx context.Context, c *Client, sesamID, authorID string) ([]models.SeriesRef, []models.SeriesRef, error) { + m.mu.Lock() + r, ok := m.seen[sesamID] + m.mu.Unlock() + if !ok { + r.linked, r.unlinked, r.err = c.recordSeries(ctx, sesamID, authorID) + m.mu.Lock() + if m.seen == nil { + m.seen = make(map[string]modsSeries) + } + m.seen[sesamID] = r + m.mu.Unlock() + } + return r.linked, r.unlinked, r.err +} + +// recordSeries reads a record's MODS. linked is the author's own series: the +// entries linked to their authority record, usually one, more when the record +// names the series twice. Publisher imprint series ("<publisher> +// krim") are recorded as series too but carry no such link; those, and any +// author series recorded without the link, come back in unlinked when they +// have a number, for fillSeries to decide on. +func (c *Client) recordSeries(ctx context.Context, sesamID, authorID string) (linked, unlinked []models.SeriesRef, err error) { + var rec modsRecord + found, err := c.getXML(ctx, metadataBase+sesamID+"/mods", &rec) + if err != nil || !found { + return nil, nil, err + } + for _, ri := range rec.RelatedItems { + if ri.Type != "series" { + continue + } + title := stripLanguageQualifier(strings.Join(strings.Fields(ri.Title), " ")) + slug := seriesSlug(title) + if slug == "" { + continue + } + ref := models.SeriesRef{ + // Scoped to the author: the series is the author's authority- + // linked one, and two authors can name a series alike. + ForeignID: seriesIDPrefix + authorID + ":" + slug, + Title: title, + Position: strings.TrimRight(strings.TrimSpace(ri.Part), "."), + Primary: true, + } + if strings.TrimSpace(ri.Href) == "(NO-TrBIB)"+authorID { + linked = append(linked, ref) + } else if ref.Position != "" { + unlinked = append(unlinked, ref) + } + } + return linked, unlinked, nil +} + +// seriesSlug is the stable part of a series ID: folded, script-preserving +// (textutil.FoldForSlug), with each run of other characters collapsed to "-". +func seriesSlug(title string) string { + words := strings.FieldsFunc(textutil.FoldForSlug(title), func(r rune) bool { + return !unicode.IsLetter(r) && !unicode.IsNumber(r) && !unicode.Is(unicode.Mn, r) && !unicode.Is(unicode.Mc, r) + }) + return strings.Join(words, "-") +} diff --git a/internal/metadata/nb/testdata/author_works.json b/internal/metadata/nb/testdata/author_works.json new file mode 100644 index 000000000..6fe684360 --- /dev/null +++ b/internal/metadata/nb/testdata/author_works.json @@ -0,0 +1,455 @@ +{ + "_embedded": { + "items": [ + { + "id": "a0000000000000000000000000000001", + "metadata": { + "title": "Fjellvinden : roman", + "titleInfos": [ + { + "title": "Fjellvinden", + "subTitle": "roman" + }, + { + "title": "Fjellvinden Norsk", + "type": "uniform" + } + ], + "people": [ + { + "name": "Nordmann, Kari", + "identifier": "bibsys.no:authority:10000001", + "roles": [ + { + "name": "aut", + "description": "Forfatter" + } + ], + "usage": "primary" + } + ], + "identifiers": { + "isbn13": [ + "9788200000011" + ], + "sesamId": "a0000000000000000000000000000001" + }, + "languages": [ + { + "code": "nob" + } + ], + "mediaTypes": [ + "bøker" + ], + "originInfo": { + "publisher": "Eksempelforlaget", + "issued": "2019" + }, + "summary": "En oppdiktet roman brukt som testdata.", + "pageCount": 312, + "series": [ + "Fjellserien" + ], + "subject": { + "genres": [ + "Romaner", + "Romanar", + "Lydbøker", + "Lydbøker", + "Krim", + "Krim", + "Politi og detektiver", + "Politi og detektivar", + "Krim", + "Romaner" + ] + }, + "physicalDescription": { + "extent": "312 s." + } + } + }, + { + "id": "a0000000000000000000000000000002", + "metadata": { + "title": "Fjellvinden : roman", + "titleInfos": [ + { + "title": "Fjellvinden", + "subTitle": "roman" + }, + { + "title": "Fjellvinden", + "type": "uniform" + } + ], + "people": [ + { + "name": "Nordmann, Kari", + "identifier": "bibsys.no:authority:10000001", + "roles": [ + { + "name": "aut" + } + ], + "usage": "primary" + }, + { + "name": "Leser, Ola", + "identifier": "bibsys.no:authority:30000001", + "roles": [ + { + "name": "nrt" + } + ] + } + ], + "identifiers": { + "isbn13": [ + "9788200000028" + ], + "sesamId": "a0000000000000000000000000000002" + }, + "languages": [ + { + "code": "nob" + } + ], + "mediaTypes": [ + "lydopptak" + ], + "originInfo": { + "publisher": "Eksempelforlaget", + "issued": "2020" + }, + "pageCount": 0, + "physicalDescription": { + "extent": "1 lydfil (11 t, 16 min)" + } + } + }, + { + "id": "a0000000000000000000000000000003", + "metadata": { + "title": "The mountain wind", + "titleInfos": [ + { + "title": "mountain wind" + }, + { + "title": "Fjellvinden", + "type": "alternative", + "displayLabel": "Originaltittel:" + } + ], + "people": [ + { + "name": "Nordmann, Kari", + "identifier": "bibsys.no:authority:10000001", + "roles": [ + { + "name": "aut" + } + ], + "usage": "primary" + }, + { + "name": "Smith, Alex", + "identifier": "bibsys.no:authority:20000001", + "roles": [ + { + "name": "trl" + } + ] + } + ], + "identifiers": { + "isbn13": [ + "9781000000012" + ], + "sesamId": "a0000000000000000000000000000003" + }, + "languages": [ + { + "code": "eng" + } + ], + "mediaTypes": [ + "bøker" + ], + "originInfo": { + "publisher": "Example Press", + "issued": "2021" + }, + "pageCount": 330 + } + }, + { + "id": "a0000000000000000000000000000004", + "metadata": { + "title": "Le vent de la montagne", + "titleInfos": [ + { + "title": "vent de la montagne" + }, + { + "title": "Fjellvinden Fransk", + "type": "uniform" + } + ], + "people": [ + { + "name": "Nordmann, Kari", + "identifier": "bibsys.no:authority:10000001", + "roles": [ + { + "name": "aut" + } + ], + "usage": "primary" + }, + { + "name": "Martin, Claude", + "identifier": "bibsys.no:authority:20000002", + "roles": [ + { + "name": "trl" + } + ] + } + ], + "identifiers": { + "isbn13": [ + "9782000000013" + ], + "sesamId": "a0000000000000000000000000000004" + }, + "languages": [ + { + "code": "fre" + } + ], + "mediaTypes": [ + "bøker" + ], + "originInfo": { + "publisher": "Éditions Exemple", + "issued": "2022" + } + } + }, + { + "id": "a0000000000000000000000000000005", + "metadata": { + "title": "Havets stemme : noveller", + "titleInfos": [ + { + "title": "Havets stemme", + "subTitle": "noveller" + }, + { + "title": "Noveller Utvalg", + "type": "uniform" + } + ], + "people": [ + { + "name": "Nordmann, Kari", + "identifier": "bibsys.no:authority:10000001", + "roles": [ + { + "name": "aut" + } + ], + "usage": "primary" + } + ], + "identifiers": { + "isbn13": [ + "9788200000035" + ], + "sesamId": "a0000000000000000000000000000005" + }, + "languages": [ + { + "code": "nno" + } + ], + "mediaTypes": [ + "bøker" + ], + "originInfo": { + "publisher": "Eksempelforlaget", + "issued": "[2015]" + }, + "pageCount": 180, + "series": [ + "Eksempelkrim" + ] + } + }, + { + "id": "a0000000000000000000000000000006", + "metadata": { + "title": "Kokebok for alle", + "titleInfos": [ + { + "title": "Kokebok for alle" + } + ], + "people": [ + { + "name": "Nordmann, Kari", + "identifier": "bibsys.no:authority:10000002", + "roles": [ + { + "name": "aut" + } + ], + "usage": "primary" + } + ], + "identifiers": { + "isbn13": [ + "9788200000042" + ], + "sesamId": "a0000000000000000000000000000006" + }, + "languages": [ + { + "code": "nob" + } + ], + "mediaTypes": [ + "bøker" + ], + "originInfo": { + "publisher": "Matforlaget", + "issued": "2018" + } + } + }, + { + "id": "a0000000000000000000000000000007", + "metadata": { + "title": "Stille vann", + "titleInfos": [ + { + "title": "Stille vann" + }, + { + "title": "Quiet water Norsk", + "type": "uniform" + } + ], + "people": [ + { + "name": "Writer, Pat", + "identifier": "bibsys.no:authority:40000001", + "roles": [ + { + "name": "aut" + } + ], + "usage": "primary" + }, + { + "name": "Nordmann, Kari", + "identifier": "bibsys.no:authority:10000001", + "roles": [ + { + "name": "trl" + } + ] + } + ], + "identifiers": { + "isbn13": [ + "9788200000059" + ], + "sesamId": "a0000000000000000000000000000007" + }, + "languages": [ + { + "code": "nob" + } + ], + "mediaTypes": [ + "bøker" + ], + "originInfo": { + "publisher": "Eksempelforlaget", + "issued": "2017" + } + } + }, + { + "id": "a0000000000000000000000000000008", + "metadata": { + "title": "Fjellserien : Siste vinter", + "titleInfos": [ + { + "title": "Fjellserien", + "partName": "Siste vinter" + }, + { + "title": "Fjellserien", + "partNumber": "3", + "partName": "Siste vinter", + "type": "uniform" + } + ], + "people": [ + { + "name": "Nordmann, Kari", + "identifier": "bibsys.no:authority:10000001", + "roles": [ + { + "name": "aut" + } + ], + "usage": "primary" + }, + { + "name": "Leser, Ola", + "identifier": "bibsys.no:authority:30000001", + "roles": [ + { + "name": "nrt" + } + ] + } + ], + "identifiers": { + "isbn13": [ + "9788200000066" + ], + "sesamId": "a0000000000000000000000000000008" + }, + "languages": [ + { + "code": "nob" + } + ], + "mediaTypes": [ + "lydopptak" + ], + "originInfo": { + "publisher": "Eksempelforlaget", + "issued": "2018" + }, + "physicalDescription": { + "extent": "17:42:00" + } + } + } + ] + }, + "page": { + "number": 0, + "size": 100, + "totalElements": 8, + "totalPages": 1 + } +} diff --git a/internal/metadata/nb/testdata/authority.json b/internal/metadata/nb/testdata/authority.json new file mode 100644 index 000000000..f86250cea --- /dev/null +++ b/internal/metadata/nb/testdata/authority.json @@ -0,0 +1,59 @@ +{ + "authorityType": "PERSON", + "deleted": false, + "systemControlNumber": "10000001", + "marcdata": [ + { + "tag": "040", + "ind1": " ", + "ind2": " ", + "subfields": [ + { + "subcode": "a", + "value": "NO-TrBIB" + } + ] + }, + { + "tag": "100", + "ind1": "1", + "ind2": " ", + "subfields": [ + { + "subcode": "a", + "value": "Nordmann, Kari" + }, + { + "subcode": "d", + "value": "1970-" + } + ] + }, + { + "tag": "400", + "ind1": "1", + "ind2": " ", + "subfields": [ + { + "subcode": "a", + "value": "Nordman, Kari" + } + ] + }, + { + "tag": "400", + "ind1": "1", + "ind2": " ", + "subfields": [ + { + "subcode": "a", + "value": "Нордманн, Кари" + }, + { + "subcode": "d", + "value": "1970-" + } + ] + } + ] +} diff --git a/internal/metadata/nb/testdata/authority_deleted.json b/internal/metadata/nb/testdata/authority_deleted.json new file mode 100644 index 000000000..9cff10e6c --- /dev/null +++ b/internal/metadata/nb/testdata/authority_deleted.json @@ -0,0 +1,44 @@ +{ + "authorityType": "PERSON", + "deleted": true, + "systemControlNumber": "10000001", + "marcdata": [ + { + "tag": "040", + "ind1": " ", + "ind2": " ", + "subfields": [ + { + "subcode": "a", + "value": "NO-TrBIB" + } + ] + }, + { + "tag": "100", + "ind1": "1", + "ind2": " ", + "subfields": [ + { + "subcode": "a", + "value": "Nordmann, Kari" + }, + { + "subcode": "d", + "value": "1970-" + } + ] + }, + { + "tag": "400", + "ind1": "1", + "ind2": " ", + "subfields": [ + { + "subcode": "a", + "value": "Nordman, Kari" + } + ] + } + ] +} diff --git a/internal/metadata/nb/testdata/empty_search.json b/internal/metadata/nb/testdata/empty_search.json new file mode 100644 index 000000000..78b133b07 --- /dev/null +++ b/internal/metadata/nb/testdata/empty_search.json @@ -0,0 +1 @@ +{"_links": {}, "page": {"number": 0, "size": 100, "totalElements": 0, "totalPages": 0}} diff --git a/internal/metadata/nb/testdata/isbn_audiobook.json b/internal/metadata/nb/testdata/isbn_audiobook.json new file mode 100644 index 000000000..aaf907f2f --- /dev/null +++ b/internal/metadata/nb/testdata/isbn_audiobook.json @@ -0,0 +1,24 @@ +{ + "_embedded": { + "items": [ + { + "id": "a0000000000000000000000000000002", + "metadata": { + "title": "Fjellvinden : roman", + "titleInfos": [{"title": "Fjellvinden", "subTitle": "roman"}, {"title": "Fjellvinden", "type": "uniform"}], + "people": [ + {"name": "Nordmann, Kari", "identifier": "bibsys.no:authority:10000001", "roles": [{"name": "aut"}], "usage": "primary"}, + {"name": "Leser, Ola", "identifier": "bibsys.no:authority:30000001", "roles": [{"name": "nrt"}]} + ], + "identifiers": {"isbn13": ["9788200000028"], "sesamId": "a0000000000000000000000000000002"}, + "languages": [{"code": "nob"}], + "mediaTypes": ["lydopptak"], + "originInfo": {"publisher": "Eksempelforlaget", "issued": "2020"}, + "summary": "En oppdiktet roman brukt som testdata.", + "pageCount": 0 + } + } + ] + }, + "page": {"number": 0, "size": 100, "totalElements": 1, "totalPages": 1} +} diff --git a/internal/metadata/nb/testdata/item_print.json b/internal/metadata/nb/testdata/item_print.json new file mode 100644 index 000000000..f809a0b99 --- /dev/null +++ b/internal/metadata/nb/testdata/item_print.json @@ -0,0 +1,52 @@ +{ + "id": "a0000000000000000000000000000001", + "metadata": { + "title": "Fjellvinden : roman", + "titleInfos": [ + { + "title": "Fjellvinden", + "subTitle": "roman" + }, + { + "title": "Fjellvinden Norsk", + "type": "uniform" + } + ], + "people": [ + { + "name": "Nordmann, Kari", + "identifier": "bibsys.no:authority:10000001", + "roles": [ + { + "name": "aut", + "description": "Forfatter" + } + ], + "usage": "primary" + } + ], + "identifiers": { + "isbn13": [ + "9788200000011" + ], + "sesamId": "a0000000000000000000000000000001" + }, + "languages": [ + { + "code": "nob" + } + ], + "mediaTypes": [ + "bøker" + ], + "originInfo": { + "publisher": "Eksempelforlaget", + "issued": "2019" + }, + "summary": "En oppdiktet roman brukt som testdata.", + "pageCount": 312, + "series": [ + "Fjellserien" + ] + } +} diff --git a/internal/metadata/nb/testdata/mods_author_series.xml b/internal/metadata/nb/testdata/mods_author_series.xml new file mode 100644 index 000000000..a8630c546 --- /dev/null +++ b/internal/metadata/nb/testdata/mods_author_series.xml @@ -0,0 +1,28 @@ +<?xml version="1.0" encoding="UTF-8"?><mods xmlns="http://www.loc.gov/mods/v3" xmlns:xlink="http://www.w3.org/1999/xlink" version="3.8"> + <titleInfo> + <title>Fjellvinden + roman + + + + Fjellserien Norsk + 2 + + + Nordmann, Kari + 1970- + + + + + Eksempelkrim + 41 + + + + + https://example.invalid/content + + + 9788200000011 + diff --git a/internal/metadata/nb/testdata/mods_publisher_series.xml b/internal/metadata/nb/testdata/mods_publisher_series.xml new file mode 100644 index 000000000..a11cce1a3 --- /dev/null +++ b/internal/metadata/nb/testdata/mods_publisher_series.xml @@ -0,0 +1,12 @@ + + + Havets stemme + + + + Eksempelkrim + 12 + + + 9788200000035 + diff --git a/internal/metadata/nb/testdata/mods_series_4.xml b/internal/metadata/nb/testdata/mods_series_4.xml new file mode 100644 index 000000000..e4c911d3a --- /dev/null +++ b/internal/metadata/nb/testdata/mods_series_4.xml @@ -0,0 +1,28 @@ + + + Tåkeheim + roman + + + + Fjellserien Norsk + 4 + + + Nordmann, Kari + 1970- + + + + + Eksempelkrim + 41 + + + + + https://example.invalid/content + + + 9788200000073 + diff --git a/internal/metadata/nb/testdata/mods_series_alias.xml b/internal/metadata/nb/testdata/mods_series_alias.xml new file mode 100644 index 000000000..b64b8fa6a --- /dev/null +++ b/internal/metadata/nb/testdata/mods_series_alias.xml @@ -0,0 +1,24 @@ + + + Havets stemme + + + + Serien om fjellet + 3 + + + + + Fjellserien + 3 + + + + + Eksempelkrim + 3 + + + 9788200000035 + diff --git a/internal/metadata/nb/testdata/mods_unlinked_series.xml b/internal/metadata/nb/testdata/mods_unlinked_series.xml new file mode 100644 index 000000000..73d68a022 --- /dev/null +++ b/internal/metadata/nb/testdata/mods_unlinked_series.xml @@ -0,0 +1,18 @@ + + + Havets stemme + + + + Eksempelkrim + 12 + + + + + Fjellserien + 3 + + + 9788200000035 + diff --git a/internal/metadata/nb/testdata/series_search.json b/internal/metadata/nb/testdata/series_search.json new file mode 100644 index 000000000..f66e178c0 --- /dev/null +++ b/internal/metadata/nb/testdata/series_search.json @@ -0,0 +1,153 @@ +{ + "_embedded": { + "items": [ + { + "id": "a0000000000000000000000000000001", + "metadata": { + "title": "Fjellvinden : roman", + "titleInfos": [ + { + "title": "Fjellvinden", + "subTitle": "roman" + }, + { + "title": "Fjellvinden Norsk", + "type": "uniform" + } + ], + "people": [ + { + "name": "Nordmann, Kari", + "identifier": "bibsys.no:authority:10000001", + "roles": [ + { + "name": "aut", + "description": "Forfatter" + } + ], + "usage": "primary" + } + ], + "identifiers": { + "isbn13": [ + "9788200000011" + ], + "sesamId": "a0000000000000000000000000000001" + }, + "languages": [ + { + "code": "nob" + } + ], + "mediaTypes": [ + "bøker" + ], + "originInfo": { + "publisher": "Eksempelforlaget", + "issued": "2019" + }, + "summary": "En oppdiktet roman brukt som testdata.", + "pageCount": 312, + "series": [ + "Fjellserien" + ] + } + }, + { + "id": "a0000000000000000000000000000009", + "metadata": { + "title": "Tåkeheim : roman", + "titleInfos": [ + { + "title": "Tåkeheim", + "subTitle": "roman" + } + ], + "people": [ + { + "name": "Nordmann, Kari", + "identifier": "bibsys.no:authority:10000001", + "roles": [ + { + "name": "aut" + } + ], + "usage": "primary" + } + ], + "identifiers": { + "isbn13": [ + "9788200000073" + ], + "sesamId": "a0000000000000000000000000000009" + }, + "languages": [ + { + "code": "nob" + } + ], + "mediaTypes": [ + "bøker" + ], + "originInfo": { + "publisher": "Eksempelforlaget", + "issued": "2017" + }, + "series": [ + "Fjellserien" + ] + } + }, + { + "id": "a0000000000000000000000000000010", + "metadata": { + "title": "Fjellserien", + "titleInfos": [ + { + "title": "Fjellserien" + } + ], + "people": [ + { + "name": "Annen, Per", + "identifier": "bibsys.no:authority:40000002", + "roles": [ + { + "name": "aut" + } + ], + "usage": "primary" + } + ], + "identifiers": { + "isbn13": [ + "9788200000080" + ], + "sesamId": "a0000000000000000000000000000010" + }, + "languages": [ + { + "code": "nob" + } + ], + "mediaTypes": [ + "bøker" + ], + "originInfo": { + "publisher": "Eksempelforlaget", + "issued": "2001" + }, + "series": [ + "Fjellserien" + ] + } + } + ] + }, + "page": { + "number": 0, + "size": 100, + "totalElements": 3, + "totalPages": 1 + } +} diff --git a/internal/metadata/nb/types.go b/internal/metadata/nb/types.go new file mode 100644 index 000000000..39106fd15 --- /dev/null +++ b/internal/metadata/nb/types.go @@ -0,0 +1,188 @@ +package nb + +import ( + "strings" + + "github.com/vavallee/bindery/internal/models" + "github.com/vavallee/bindery/internal/textutil" +) + +// searchResponse is the HAL page returned by GET /catalog/v1/items. +type searchResponse struct { + Embedded struct { + Items []item `json:"items"` + } `json:"_embedded"` + Page struct { + Number int `json:"number"` + TotalElements int `json:"totalElements"` + TotalPages int `json:"totalPages"` + } `json:"page"` +} + +// item is one catalogue record: a single edition (print, e-book, audiobook or +// translation), never a work. +type item struct { + ID string `json:"id"` + Metadata itemMetadata `json:"metadata"` +} + +type itemMetadata struct { + Title string `json:"title"` + TitleInfos []titleInfo `json:"titleInfos"` + People []person `json:"people"` + Identifiers struct { + ISBN13 []string `json:"isbn13"` + ISBN10 []string `json:"isbn10"` + SesamID string `json:"sesamId"` + } `json:"identifiers"` + Languages []struct { + Code string `json:"code"` + } `json:"languages"` + MediaTypes []string `json:"mediaTypes"` + OriginInfo struct { + Publisher string `json:"publisher"` + Issued string `json:"issued"` + } `json:"originInfo"` + Summary string `json:"summary"` + PageCount int `json:"pageCount"` + // Series names the record's series, author's and publisher's alike, + // without the number. See authorSeries. + Series []string `json:"series"` + // Subject.Genres is the cataloguer's genre list, Bokmål and Nynorsk + // forms side by side. See workGenres. + Subject struct { + Genres []string `json:"genres"` + } `json:"subject"` + // PhysicalDescription.Extent is free text: page count for print, the + // running time for an audiobook. See extentDuration. + PhysicalDescription struct { + Extent string `json:"extent"` + } `json:"physicalDescription"` +} + +// titleInfo is one title of a record. Type is "" for the title proper, +// "uniform" for the cataloguer's work title, and "alternative" for others, +// including the original title of a translation. +type titleInfo struct { + Title string `json:"title"` + Type string `json:"type"` + DisplayLabel string `json:"displayLabel"` + // PartName and PartNumber are set when the record is catalogued as one + // numbered part of a larger work: Title is then that work's name (often + // the series), PartName the volume's own title. See recordTitle. + PartName string `json:"partName"` + PartNumber string `json:"partNumber"` +} + +// person is a credited name. Identifier is "bibsys.no:authority:" when the +// name is linked to the Norwegian authority file. Roles are MARC relator codes. +type person struct { + Name string `json:"name"` + Identifier string `json:"identifier"` + Roles []struct { + Name string `json:"name"` + } `json:"roles"` +} + +func (p person) hasRole(code string) bool { + for _, r := range p.Roles { + if r.Name == code { + return true + } + } + return false +} + +// isAuthor reports an author credit. NB catalogues one as "aut" or, about +// as often, as the generic "cre" (creator), occasionally spelled out. +func (p person) isAuthor() bool { + return p.hasRole("aut") || p.hasRole("cre") || p.hasRole("creator") +} + +func (p person) authorityID() string { + id, ok := strings.CutPrefix(p.Identifier, authorityIDPrefix) + if !ok || !authorityIDRe.MatchString(id) { + return "" + } + return id +} + +// authorityRecord is the subset of an authority.bibsys.no record used here. +// The authorised name heading is MARC field 100 $a. +type authorityRecord struct { + Deleted bool `json:"deleted"` + MarcData []struct { + Tag string `json:"tag"` + Subfields []struct { + Code string `json:"subcode"` + Value string `json:"value"` + } `json:"subfields"` + } `json:"marcdata"` +} + +// variants returns the record's other name forms (MARC 400 $a) in display +// form: spellings without diacritics, transliterations, earlier names. +func (r authorityRecord) variants() []string { + var out []string + seen := map[string]bool{invertName(r.heading()): true} + for _, f := range r.MarcData { + if f.Tag != "400" { + continue + } + for _, sf := range f.Subfields { + if sf.Code != "a" { + continue + } + if name := invertName(strings.TrimSpace(sf.Value)); name != "" && !seen[name] { + seen[name] = true + out = append(out, name) + } + } + } + return out +} + +func (r authorityRecord) heading() string { + for _, f := range r.MarcData { + if f.Tag != "100" { + continue + } + for _, sf := range f.Subfields { + if sf.Code == "a" { + return strings.TrimSpace(sf.Value) + } + } + } + return "" +} + +func personToAuthor(p person) models.Author { + return models.Author{ + ForeignID: authorPrefix + p.authorityID(), + Name: invertName(p.Name), + SortName: p.Name, + MetadataProvider: "nb", + } +} + +// invertName turns the catalogue form "Last, First" into "First Last". +func invertName(name string) string { + last, first, ok := strings.Cut(name, ", ") + if !ok || strings.TrimSpace(first) == "" { + return name + } + return strings.TrimSpace(first) + " " + last +} + +// nameMatches reports whether every query word appears in the name, ignoring +// case and diacritics. The search filter already matched the record; this +// drops the record's other credited authors. +func nameMatches(name string, words []string) bool { + folded := textutil.FoldForSearch(name) + for _, w := range words { + if !strings.Contains(folded, textutil.FoldForSearch(w)) { + return false + } + } + return true +} diff --git a/internal/metadata/nb/works.go b/internal/metadata/nb/works.go new file mode 100644 index 000000000..d2103a88d --- /dev/null +++ b/internal/metadata/nb/works.go @@ -0,0 +1,386 @@ +package nb + +import ( + "regexp" + "sort" + "strconv" + "strings" + "time" + + "github.com/vavallee/bindery/internal/indexer" + "github.com/vavallee/bindery/internal/models" +) + +// groupWorks folds edition records into one book per (author, work title), in +// first-appearance order so the API's relevance ordering survives. When +// authorID is set, only records crediting that authority ID as author are +// kept: the name filter that fetched them also matches homonyms, and records +// where the person is only translator or narrator. +// +// Translations join the original's work through the title the cataloguer +// recorded for it (see workTitle). A translation with no such title stays its +// own book; the metadata profile's language filter then decides whether it is +// wanted, the same as for any other provider. +func groupWorks(items []item, authorID string) []models.Book { + var order []string + groups := make(map[string][]item) + for _, it := range items { + m := it.Metadata + if m.Identifiers.SesamID == "" { + m.Identifiers.SesamID = it.ID + } + if !sesamIDRe.MatchString(m.Identifiers.SesamID) || recordTitle(m) == "" { + continue + } + author := primaryAuthor(m, authorID) + if authorID != "" && author == nil { + continue + } + authorKey := "" + if author != nil { + authorKey = author.Identifier + author.Name + } + // Spaces are ignored so a catalogue slip that splits a word + // ("Fjel lbyen") stays with its work; the key is per author, so two + // titles differing only in spacing are taken as one work. + key := authorKey + "|" + strings.ReplaceAll(indexer.CanonicalDedupKey(workTitle(m)), " ", "") + if _, ok := groups[key]; !ok { + order = append(order, key) + } + groups[key] = append(groups[key], item{ID: it.ID, Metadata: m}) + } + books := make([]models.Book, 0, len(order)) + for _, key := range order { + books = append(books, buildWork(groups[key], authorID)) + } + return books +} + +// buildWork turns one group of editions into a book. The representative +// record, which supplies the ID and title, is the best Norwegian edition: this +// provider exists for original-language Norwegian titles, and for a foreign +// author it yields the Norwegian translation's title, which is what a +// Norwegian library's files are named after. +func buildWork(group []item, authorID string) models.Book { + rep := group[0] + for _, it := range group[1:] { + if repScore(it.Metadata) > repScore(rep.Metadata) { + rep = it + } + } + m := rep.Metadata + title := commonTitle(group, recordTitle(m)) + b := models.Book{ + ForeignID: idPrefix + m.Identifiers.SesamID, + Title: title, + SortTitle: title, + Description: m.Summary, + Language: language(m), + MetadataProvider: "nb", + Monitored: true, + Status: models.BookStatusWanted, + } + if p := primaryAuthor(m, authorID); p != nil && p.authorityID() != "" { + a := personToAuthor(*p) + b.Author = &a + } + for _, p := range m.People { + if p.isAuthor() && p.authorityID() != "" { + b.CreditedAuthorForeignIDs = append(b.CreditedAuthorForeignIDs, authorPrefix+p.authorityID()) + } + } + b.Genres = workGenres(group, rep) + for _, it := range group { + em := it.Metadata + if b.Narrator == "" && isAudio(em) { + b.Narrator = narrators(em) + } + if b.Description == "" { + b.Description = em.Summary + } + date := parseYear(em.OriginInfo.Issued) + if date != nil && (b.ReleaseDate == nil || date.Before(*b.ReleaseDate)) { + b.ReleaseDate = date + } + ed := toEdition(em, date) + if b.DurationSeconds == 0 { + b.DurationSeconds = ed.DurationSeconds + } + b.Editions = append(b.Editions, ed) + b.ProviderISBNs = append(b.ProviderISBNs, em.Identifiers.ISBN13...) + b.ProviderISBNs = append(b.ProviderISBNs, em.Identifiers.ISBN10...) + } + return b +} + +// commonTitle is the title most of a work's editions carry, so one record's +// cataloguing slip cannot name the book. Ties keep fallback, the +// representative's own title; among other equal counts the first in sort +// order wins, so the result does not depend on map order. +func commonTitle(group []item, fallback string) string { + counts := make(map[string]int, len(group)) + for _, it := range group { + counts[recordTitle(it.Metadata)]++ + } + titles := make([]string, 0, len(counts)) + for t := range counts { + titles = append(titles, t) + } + sort.Strings(titles) // a deterministic winner among equal counts + best := fallback + for _, t := range titles { + if counts[t] > counts[best] { + best = t + } + } + return best +} + +// narrators lists a record's narrator credits (relator "nrt") in display +// form, comma separated, as the Audible provider writes them. +func narrators(m itemMetadata) string { + var names []string + for _, p := range m.People { + if p.hasRole("nrt") { + names = append(names, invertName(p.Name)) + } + } + return strings.Join(names, ", ") +} + +// genreFormatTerms are entries in NB's genre list that name the format, not +// a genre. +var genreFormatTerms = map[string]bool{"lydbøker": true, "lydbok": true, "e-bøker": true, "e-bok": true} + +// workGenres returns the work's genres from the representative record's +// subject genres, or the first edition that has any. NB lists each term in +// both written standards ("Romaner", "Romanar"), so a Nynorsk "-ar" form is +// dropped when its Bokmål "-er" form is present. +// ponytail: plural-ending rule, not a Bokmål/Nynorsk vocabulary; a twin +// differing in more than the ending stays as a second genre. +func workGenres(group []item, rep item) []string { + terms := rep.Metadata.Subject.Genres + for _, it := range group { + if len(terms) > 0 { + break + } + terms = it.Metadata.Subject.Genres + } + present := make(map[string]bool, len(terms)) + for _, t := range terms { + present[strings.ToLower(strings.TrimSpace(t))] = true + } + genres := []string{} + seen := make(map[string]bool, len(terms)) + for _, t := range terms { + t = strings.TrimSpace(t) + key := strings.ToLower(t) + if t == "" || seen[key] || genreFormatTerms[key] { + continue + } + if stem, ok := strings.CutSuffix(key, "ar"); ok && present[stem+"er"] { + continue + } + seen[key] = true + genres = append(genres, t) + } + return genres +} + +var ( + extentClockRe = regexp.MustCompile(`^\s*(\d+):(\d{2}):(\d{2})\s*$`) + extentHoursRe = regexp.MustCompile(`(\d+)\s*t\b`) + extentMinsRe = regexp.MustCompile(`(\d+)\s*min\b`) +) + +// extentDuration reads an audiobook's running time, in seconds, from NB's +// extent text: "1 lydfil (11 t, 16 min)", "21:34:00", or a CD set's +// "3 plater (CD)(3 t, 7 min) …". 0 when it states none. +func extentDuration(extent string) int { + if m := extentClockRe.FindStringSubmatch(extent); m != nil { + h, _ := strconv.Atoi(m[1]) + mins, _ := strconv.Atoi(m[2]) + secs, _ := strconv.Atoi(m[3]) + return h*3600 + mins*60 + secs + } + total := 0 + if m := extentHoursRe.FindStringSubmatch(extent); m != nil { + h, _ := strconv.Atoi(m[1]) + total += h * 3600 + } + if m := extentMinsRe.FindStringSubmatch(extent); m != nil { + mins, _ := strconv.Atoi(m[1]) + total += mins * 60 + } + return total +} + +func toEdition(m itemMetadata, date *time.Time) models.Edition { + ed := models.Edition{ + ForeignID: idPrefix + m.Identifiers.SesamID, + Title: recordTitle(m), + Publisher: m.OriginInfo.Publisher, + PublishDate: date, + Language: language(m), + } + if len(m.Identifiers.ISBN13) > 0 { + ed.ISBN13 = &m.Identifiers.ISBN13[0] + } + if len(m.Identifiers.ISBN10) > 0 { + ed.ISBN10 = &m.Identifiers.ISBN10[0] + } + if isAudio(m) { + ed.Format = models.MediaTypeAudiobook + ed.DurationSeconds = extentDuration(m.PhysicalDescription.Extent) + } else if m.PageCount > 0 { + pages := m.PageCount + ed.NumPages = &pages + } + return ed +} + +// repScore ranks a record as the representative of its work: Norwegian beats +// other languages, print beats audio. +func repScore(m itemMetadata) int { + score := 0 + switch language(m) { + case "nob", "nno", "nor": + score += 2 + } + if !isAudio(m) { + score++ + } + return score +} + +// primaryAuthor picks the credited author: the one with authorID when given, +// otherwise the first author credit. +func primaryAuthor(m itemMetadata, authorID string) *person { + for i := range m.People { + p := &m.People[i] + if !p.isAuthor() { + continue + } + if authorID == "" || p.authorityID() == authorID { + return p + } + } + return nil +} + +// workTitle is the title editions of one work share. A translation records +// its original title either as an "Originaltittel" alternative or as a +// uniform title of the form " " (" Fransk"). +// Anything else groups by its own title: on originals the uniform title is +// cataloguing noise (" Norsk", "<title> 2023", a collection heading) +// that would split one work's printings apart. +func workTitle(m itemMetadata) string { + for _, ti := range m.TitleInfos { + if ti.Type == "alternative" && strings.HasPrefix(ti.DisplayLabel, "Originaltittel") { + return ti.Title + } + } + if translated(m) { + for _, ti := range m.TitleInfos { + if ti.Type == "uniform" { + return stripLanguageQualifier(ti.Title) + } + } + } + return recordTitle(m) +} + +// languageQualifiers are the Norwegian language names the catalogue appends +// to a uniform or series title ("<title> Fransk", "<series> Norsk"). A list, +// not a pattern: a pattern also matched names ("Inspektør Brask"). A language +// missing here leaves that translation as its own book and its series as its +// own row, which the language filter still handles. +var languageQualifiers = map[string]bool{ + "norsk": true, "nynorsk": true, "engelsk": true, "fransk": true, "tysk": true, + "svensk": true, "dansk": true, "islandsk": true, "finsk": true, "spansk": true, + "italiensk": true, "portugisisk": true, "nederlandsk": true, "russisk": true, + "ukrainsk": true, "polsk": true, "tsjekkisk": true, "slovakisk": true, + "slovensk": true, "kroatisk": true, "serbisk": true, "bulgarsk": true, + "rumensk": true, "ungarsk": true, "gresk": true, "tyrkisk": true, + "estisk": true, "latvisk": true, "litauisk": true, "hebraisk": true, + "arabisk": true, "persisk": true, "kinesisk": true, "japansk": true, + "koreansk": true, "vietnamesisk": true, +} + +// stripLanguageQualifier drops a trailing language name from a catalogue +// title. See languageQualifiers. +func stripLanguageQualifier(title string) string { + words := strings.Fields(title) + if len(words) < 2 || !languageQualifiers[strings.ToLower(words[len(words)-1])] { + return title + } + return strings.Join(words[:len(words)-1], " ") +} + +func translated(m itemMetadata) bool { + for _, p := range m.People { + if p.hasRole("trl") { + return true + } + } + return false +} + +// recordTitle is the record's own title. A volume catalogued as a part of a +// larger work ("<series> : <volume>", "<series>. [5] : <volume>") has the +// series name as its title proper and its own title only as the part name, +// so the part name is used; otherwise the title proper. Taking the title +// proper there turned every such volume into a book named after its series. +func recordTitle(m itemMetadata) string { + for _, ti := range m.TitleInfos { + if ti.Type == "" && strings.TrimSpace(ti.PartName) != "" { + return strings.Join(strings.Fields(ti.PartName), " ") + } + } + return mainTitle(m.Title) +} + +// mainTitle is the title proper without its subtitle ("<title> : roman"), with +// the double space NB leaves after a non-sorting article collapsed. +func mainTitle(title string) string { + title, _, _ = strings.Cut(title, " : ") + return strings.Join(strings.Fields(title), " ") +} + +func language(m itemMetadata) string { + if len(m.Languages) == 0 { + return "" + } + return m.Languages[0].Code +} + +func isAudio(m itemMetadata) bool { + for _, t := range m.MediaTypes { + if t == "lydopptak" { + return true + } + } + return false +} + +// parseYear reads the first four-digit run of a publication date such as +// "2019", "[2019]" or "cop. 2019". +func parseYear(s string) *time.Time { + run := 0 + for i, r := range s { + if r < '0' || r > '9' { + run = 0 + continue + } + run++ + if run == 4 { + year, _ := strconv.Atoi(s[i-3 : i+1]) + if year < 1400 || year > 2100 { + return nil + } + t := time.Date(year, 1, 1, 0, 0, 0, 0, time.UTC) + return &t + } + } + return nil +} diff --git a/internal/models/author.go b/internal/models/author.go index 8b3903d46..4cf557ed5 100644 --- a/internal/models/author.go +++ b/internal/models/author.go @@ -291,6 +291,8 @@ func AuthorProviderFromForeignID(foreignID string) string { return "hardcover" case strings.HasPrefix(foreignID, "dnb:"): return "dnb" + case strings.HasPrefix(foreignID, "nb:"): + return "nb" case strings.HasPrefix(foreignID, "calibre:"): return "calibre" case strings.HasPrefix(foreignID, "abs:"): diff --git a/internal/models/author_test.go b/internal/models/author_test.go index 8d78d8c45..eb9c15171 100644 --- a/internal/models/author_test.go +++ b/internal/models/author_test.go @@ -47,6 +47,7 @@ func TestAuthorProviderFromForeignID(t *testing.T) { {id: "OL13200512A", want: "openlibrary"}, {id: "hc:emilia-jae", want: "hardcover"}, {id: "dnb:123456789", want: "dnb"}, + {id: "nb:author:10000001", want: "nb"}, {id: "gb:volume", want: "googlebooks"}, {id: "calibre:author:1", want: "calibre"}, {id: "abs:author:lib:author", want: "audiobookshelf"}, diff --git a/internal/models/book.go b/internal/models/book.go index dc62fa694..3e84964d2 100644 --- a/internal/models/book.go +++ b/internal/models/book.go @@ -321,6 +321,8 @@ func BookProviderFromForeignID(foreignID string) string { return "hardcover" case strings.HasPrefix(foreignID, "dnb:"): return "dnb" + case strings.HasPrefix(foreignID, "nb:"): + return "nb" case strings.HasPrefix(foreignID, "calibre:"): return "calibre" case strings.HasPrefix(foreignID, "abs:"): diff --git a/internal/models/book_identity_test.go b/internal/models/book_identity_test.go index fba9144b8..e79482623 100644 --- a/internal/models/book_identity_test.go +++ b/internal/models/book_identity_test.go @@ -7,16 +7,17 @@ import "testing" // OpenLibrary, matching the long-standing books.foreign_id convention. func TestBookProviderFromForeignID(t *testing.T) { cases := map[string]string{ - "hc:volume-1": "hardcover", - "HC:VOLUME-1": "hardcover", - "OL1W": "openlibrary", - "calibre:12": "calibre", - "abs:lib:item": "audiobookshelf", - "gb:xyz": "googlebooks", - "dnb:123": "dnb", - "": "openlibrary", - " hc:spaced ": "hardcover", - "something-random": "openlibrary", + "hc:volume-1": "hardcover", + "HC:VOLUME-1": "hardcover", + "OL1W": "openlibrary", + "calibre:12": "calibre", + "abs:lib:item": "audiobookshelf", + "gb:xyz": "googlebooks", + "dnb:123": "dnb", + "nb:a0000000000000000000000000000001": "nb", + "": "openlibrary", + " hc:spaced ": "hardcover", + "something-random": "openlibrary", } for id, want := range cases { if got := BookProviderFromForeignID(id); got != want { diff --git a/web/src/components/AddToLibraryModal.tsx b/web/src/components/AddToLibraryModal.tsx index e16d2486d..4e986c616 100644 --- a/web/src/components/AddToLibraryModal.tsx +++ b/web/src/components/AddToLibraryModal.tsx @@ -90,7 +90,7 @@ export default function AddToLibraryModal({ onClose, onAdded, initialQuery, mode const value = (s.value || '').trim().toLowerCase() // Mirror MetadataPrimaryProviders on the backend; anything else means // no explicit choice, so no notice. - if (value === 'openlibrary' || value === 'dnb' || value === 'hardcover') setPrimaryProvider(value) + if (value === 'openlibrary' || value === 'dnb' || value === 'nb' || value === 'hardcover') setPrimaryProvider(value) }) .catch(() => { /* unset; no provider notice needed */ }) }, [isRequester]) diff --git a/web/src/components/AuthorMetadataLinkModal.tsx b/web/src/components/AuthorMetadataLinkModal.tsx index 9403052bc..f07b00710 100644 --- a/web/src/components/AuthorMetadataLinkModal.tsx +++ b/web/src/components/AuthorMetadataLinkModal.tsx @@ -14,6 +14,7 @@ function providerLabel(author: Author): string { if (provider === 'hardcover' || author.foreignAuthorId.startsWith('hc:')) return 'Hardcover' if (provider === 'openlibrary' || author.foreignAuthorId.startsWith('OL')) return 'OpenLibrary' if (provider === 'dnb' || author.foreignAuthorId.startsWith('dnb:')) return 'DNB' + if (provider === 'nb' || author.foreignAuthorId.startsWith('nb:')) return 'Nasjonalbiblioteket' if (provider) return provider return 'Metadata' } diff --git a/web/src/i18n/locales/en.json b/web/src/i18n/locales/en.json index b0cf655d8..228c5d566 100644 --- a/web/src/i18n/locales/en.json +++ b/web/src/i18n/locales/en.json @@ -1395,9 +1395,10 @@ "telemetryDetail": "Sends install_id (random UUID), version, OS, architecture, deploy method, feature-usage counts, and coarse error counters (ERROR/WARN totals over the last 24h plus the 5 most frequent fixed error-message strings — never log details, titles, paths, or URLs) to getbindery.dev once per day. Disable at any time.", "metadataProvider": "Metadata Provider", "metadataProviderLabel": "Primary metadata provider", - "metadataProviderHint": "Selects the source used for author and book search, and decides what an author's catalogue looks like. DNB (Deutsche Nationalbibliothek) is recommended for German, Austrian, and Swiss catalogues — it covers German-language publications since 1913 where OpenLibrary coverage is thin. Hardcover is a curated catalogue: fewer translation editions, omnibus bundles, and non-book entries than OpenLibrary, at the cost of a thinner long tail. OpenLibrary remains the default. Whichever you pick, the other providers are still used as enrichers.", + "metadataProviderHint": "Selects the source used for author and book search, and decides what an author's catalogue looks like. DNB (Deutsche Nationalbibliothek) is recommended for German, Austrian, and Swiss catalogues — it covers German-language publications since 1913 where OpenLibrary coverage is thin. Nasjonalbiblioteket (the National Library of Norway) is recommended for Norwegian catalogues — it holds every Norwegian publication by legal deposit, under its original Norwegian title where OpenLibrary often has only the English translation. Hardcover is a curated catalogue: fewer translation editions, omnibus bundles, and non-book entries than OpenLibrary, at the cost of a thinner long tail. OpenLibrary remains the default. Whichever you pick, the other providers are still used as enrichers, except Nasjonalbiblioteket, which is only queried when it is the primary.", "metadataProviderOpenlibrary": "OpenLibrary (default)", "metadataProviderDnb": "DNB — Deutsche Nationalbibliothek (German/DACH)", + "metadataProviderNb": "Nasjonalbiblioteket — National Library of Norway (Norwegian)", "metadataProviderHardcover": "Hardcover (curated catalogue)", "metadataProviderHardcoverNoToken": "Hardcover — requires an API token (Settings → API Keys)", "metadataProviderRestartHint": "Takes effect after the next Bindery restart — the provider is wired at startup, so nothing changes until then. Authors already in your library stay linked to the provider they were added from; the new primary applies to authors you add afterwards. To move an existing author, use \"Link metadata\" on their page.", diff --git a/web/src/pages/settings/MetadataTab.tsx b/web/src/pages/settings/MetadataTab.tsx index 6cd2a4a01..4f6c69f2a 100644 --- a/web/src/pages/settings/MetadataTab.tsx +++ b/web/src/pages/settings/MetadataTab.tsx @@ -89,7 +89,7 @@ export default function MetadataTab() { {t('settings.general.metadataProviderLabel', 'Primary metadata provider')} </label> <p className="text-xs text-slate-600 dark:text-zinc-500 mb-2"> - {t('settings.general.metadataProviderHint', 'Selects the source used for author and book search, and decides what an author\'s catalogue looks like. DNB (Deutsche Nationalbibliothek) is recommended for German, Austrian, and Swiss catalogues — it covers German-language publications since 1913 where OpenLibrary coverage is thin. Hardcover is a curated catalogue: fewer translation editions, omnibus bundles, and non-book entries than OpenLibrary, at the cost of a thinner long tail. OpenLibrary remains the default. Whichever you pick, the other providers are still used as enrichers.')} + {t('settings.general.metadataProviderHint', 'Selects the source used for author and book search, and decides what an author\'s catalogue looks like. DNB (Deutsche Nationalbibliothek) is recommended for German, Austrian, and Swiss catalogues — it covers German-language publications since 1913 where OpenLibrary coverage is thin. Nasjonalbiblioteket (the National Library of Norway) is recommended for Norwegian catalogues — it holds every Norwegian publication by legal deposit, under its original Norwegian title where OpenLibrary often has only the English translation. Hardcover is a curated catalogue: fewer translation editions, omnibus bundles, and non-book entries than OpenLibrary, at the cost of a thinner long tail. OpenLibrary remains the default. Whichever you pick, the other providers are still used as enrichers, except Nasjonalbiblioteket, which is only queried when it is the primary.')} </p> <select value={settings['metadata.primary_provider'] ?? 'openlibrary'} @@ -102,6 +102,7 @@ export default function MetadataTab() { > <option value="openlibrary">{t('settings.general.metadataProviderOpenlibrary', 'OpenLibrary (default)')}</option> <option value="dnb">{t('settings.general.metadataProviderDnb', 'DNB — Deutsche Nationalbibliothek (German/DACH)')}</option> + <option value="nb">{t('settings.general.metadataProviderNb', 'Nasjonalbiblioteket — National Library of Norway (Norwegian)')}</option> <option value="hardcover" disabled={!hardcoverTokenConfigured}> {hardcoverTokenConfigured ? t('settings.general.metadataProviderHardcover', 'Hardcover (curated catalogue)') diff --git a/web/src/util/authorMetadata.ts b/web/src/util/authorMetadata.ts index 941bb4666..fbe00a018 100644 --- a/web/src/util/authorMetadata.ts +++ b/web/src/util/authorMetadata.ts @@ -23,6 +23,7 @@ export function authorProviderKey(author?: Pick<Author, 'foreignAuthorId' | 'met if (id.startsWith('gb:')) return 'googlebooks' if (id.startsWith('hc:')) return 'hardcover' if (id.startsWith('dnb:')) return 'dnb' + if (id.startsWith('nb:')) return 'nb' if (id.startsWith('calibre:')) return 'calibre' if (id.startsWith('abs:')) return 'audiobookshelf' if (id !== '') return 'openlibrary' diff --git a/web/src/util/metadataSource.test.ts b/web/src/util/metadataSource.test.ts index 06519059e..0d2c60d06 100644 --- a/web/src/util/metadataSource.test.ts +++ b/web/src/util/metadataSource.test.ts @@ -54,6 +54,15 @@ describe('metadataSourceLink', () => { }) }) + it('links NB book records but not authority-file authors', () => { + expect(metadataSourceLink('nb:a0000000000000000000000000000001', 'book')).toEqual({ + url: 'https://www.nb.no/items/a0000000000000000000000000000001', + label: 'Nasjonalbiblioteket', + }) + expect(metadataSourceLink('nb:author:10000001', 'author')).toBeNull() + expect(metadataSourceLink('nb:../x', 'book')).toBeNull() + }) + it('returns null for local or malformed provider ids', () => { expect(metadataSourceLink('abs:abc', 'book')).toBeNull() expect(metadataSourceLink('calibre:7', 'book')).toBeNull() @@ -76,6 +85,7 @@ describe('providerDisplayName', () => { expect(providerDisplayName('hardcover')).toBe('Hardcover') expect(providerDisplayName('googlebooks')).toBe('Google Books') expect(providerDisplayName('dnb')).toBe('DNB') + expect(providerDisplayName('nb')).toBe('Nasjonalbiblioteket') expect(providerDisplayName('calibre')).toBe('Calibre') expect(providerDisplayName('audiobookshelf')).toBe('Audiobookshelf') }) @@ -101,6 +111,7 @@ describe('providerFromBookForeignId', () => { expect(providerFromBookForeignId('hc:123')).toBe('hardcover') expect(providerFromBookForeignId('HC:123')).toBe('hardcover') expect(providerFromBookForeignId('dnb:123')).toBe('dnb') + expect(providerFromBookForeignId('nb:a0000000000000000000000000000001')).toBe('nb') expect(providerFromBookForeignId('calibre:7')).toBe('calibre') expect(providerFromBookForeignId('abs:lib:item')).toBe('audiobookshelf') }) diff --git a/web/src/util/metadataSource.ts b/web/src/util/metadataSource.ts index d4f80c3fb..5fce9b15c 100644 --- a/web/src/util/metadataSource.ts +++ b/web/src/util/metadataSource.ts @@ -38,8 +38,16 @@ export function metadataSourceLink( return { url: `https://d-nb.info/${controlNumber}`, label: 'DNB' } } + // Book IDs are NB record ids; author IDs are authority-file ids with no + // public NB page. + if (kind === 'book' && lowerID.startsWith('nb:')) { + const record = lowerID.slice(3) + if (!/^[0-9a-f]{32}$/.test(record)) return null + return { url: `https://www.nb.no/items/${record}`, label: 'Nasjonalbiblioteket' } + } + // No reliable public URL for these providers. - if (lowerID.startsWith('hc:') || lowerID.startsWith('dnb:') || lowerID.startsWith('abs:') || lowerID.startsWith('calibre:')) { + if (lowerID.startsWith('hc:') || lowerID.startsWith('dnb:') || lowerID.startsWith('nb:') || lowerID.startsWith('abs:') || lowerID.startsWith('calibre:')) { return null } @@ -63,6 +71,7 @@ const PROVIDER_NAMES: Record<string, string> = { hardcover: 'Hardcover', googlebooks: 'Google Books', dnb: 'DNB', + nb: 'Nasjonalbiblioteket', calibre: 'Calibre', audiobookshelf: 'Audiobookshelf', } @@ -82,6 +91,7 @@ export function providerFromBookForeignId(foreignId: string | undefined | null): if (id.startsWith('gb:')) return 'googlebooks' if (id.startsWith('hc:')) return 'hardcover' if (id.startsWith('dnb:')) return 'dnb' + if (id.startsWith('nb:')) return 'nb' if (id.startsWith('calibre:')) return 'calibre' if (id.startsWith('abs:')) return 'audiobookshelf' return 'openlibrary'