From a0a2a18362cd702a20472eb0fce85fe1e46d393d Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Thu, 27 Aug 2026 08:22:19 +0200 Subject: [PATCH 1/3] fix(pipeline): lowercase-lien IGP heading, is_wine categories fallback, corpus-walk widening MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 02: `lien\b` carve-out in IGP_SECTION_HDR_RE's uppercase title guard — Côte Vermeille's "10- lien avec la zone géographique" was never sliced, so its lien (7,110 chars) and terroir facts were lost; verified byte-identical lien lengths corpus-wide on re-extraction. - 04: fall back to the SIQO `categories` list when the manifest `categorie` is empty — Côte roannaise + Muscat du Cap Corse were is_wine=0 (mirror/override-path records) and off the wine facet. - 02g + grape_corpus: walk mt/cy/nl/be/lu/ch/hu extracted dirs and the BG/GR/SK/RO national-spec sidecars (register-fiche dirs deliberately excluded — stage 04 reads only their terroir text). Surfaced 141 VIVC-invisible corpus slugs, since swept. - lexicon: `ryzlink-buketovy` own slug (blanc) — CZ legacy zemské-víno variety from the fiche §6 legacy block; identity contested (Goldriesling #4884 vs Bukettraube #1611), no fold. Co-Authored-By: Claude Fable 5 --- scripts/02_extract_cahiers.py | 6 +++++- scripts/02g_fetch_vivc.py | 21 ++++++++++++++++++++- scripts/04_build_maps.py | 6 ++++++ scripts/_lib/grape_corpus.py | 19 +++++++++++++++++++ scripts/_lib/grape_lexicon.py | 5 +++++ 5 files changed, 55 insertions(+), 2 deletions(-) diff --git a/scripts/02_extract_cahiers.py b/scripts/02_extract_cahiers.py index 47d1cca..f1c6445 100644 --- a/scripts/02_extract_cahiers.py +++ b/scripts/02_extract_cahiers.py @@ -343,8 +343,12 @@ def find_segment(segments: dict[str, str], target: str) -> str | None: # postal codes ("11010 Vitoria-Gasteiz") as phantom section markers # whenever a misrouted cahier (e.g. an EU letter-section AOP) flowed # through this parser. Subsection depth stays unbounded (`4.1.2.3`). + # Title must start uppercase (a lowercase start would re-admit the + # analytic-norm lines like "125 mg/l …" as phantom titles) — with one + # carve-out: a literal lowercase "lien" start, seen in the Côte + # Vermeille cahier ("10- lien avec la zone géographique"). rf"^[ \t]*(\d{{1,2}}(?:[.\-]\d+)*)[\.\-:\)]*[ \t]*(?:{DASH}[ \t]*)?" - rf"([A-ZÉÈÀÂÔÎÏÛŸ][\wÀ-ÿ '’\-]{{3,80}})[ \t]*[:.]?[ \t]*$", + rf"((?:[A-ZÉÈÀÂÔÎÏÛŸ]|lien\b)[\wÀ-ÿ '’\-]{{3,80}})[ \t]*[:.]?[ \t]*$", re.MULTILINE, ) # Match `Chapitre 1 :` (legacy) and `CHAPITRE 1 – DENOMINATION` (post-2020 diff --git a/scripts/02g_fetch_vivc.py b/scripts/02g_fetch_vivc.py index 6246514..704daee 100644 --- a/scripts/02g_fetch_vivc.py +++ b/scripts/02g_fetch_vivc.py @@ -76,6 +76,22 @@ GR_EXTRACTED = ROOT / "raw" / "gr" / "dokumenti-extracted" SK_EXTRACTED = ROOT / "raw" / "sk" / "dokumenty-extracted" CZ_EXTRACTED = ROOT / "raw" / "cz" / "dokumenty-extracted" +CH_EXTRACTED = ROOT / "raw" / "ch" / "dokumente-extracted" +MT_EXTRACTED = ROOT / "raw" / "mt" / "dokumente-extracted" +CY_EXTRACTED = ROOT / "raw" / "cy" / "dokumenti-extracted" +NL_EXTRACTED = ROOT / "raw" / "nl" / "dokumenten-extracted" +BE_EXTRACTED = ROOT / "raw" / "be" / "dokumenten-extracted" +LU_EXTRACTED = ROOT / "raw" / "lu" / "cahier-extracted" +# National-spec sidecars whose grape rosters stage 04 merges into the map +# (grandfathered wines with no EU-OJ document). The cz/sk/gr/ro +# register-fiches-extracted dirs are deliberately absent: stage 04 reads +# only their terroir text, so their rosters must not weigh the +# corpus-slug frequency tiers. +CY_SPEC_EXTRACTED = ROOT / "raw" / "cy" / "national-specs-extracted" +BG_SPEC_EXTRACTED = ROOT / "raw" / "bg" / "national-specs-extracted" +GR_SPEC_EXTRACTED = ROOT / "raw" / "gr" / "national-specs-extracted" +SK_SPEC_EXTRACTED = ROOT / "raw" / "sk" / "national-specs-extracted" +RO_SPEC_EXTRACTED = ROOT / "raw" / "ro" / "national-specs-extracted" OUT_DIR = ROOT / "raw" / "vivc" SEARCH_DIR = OUT_DIR / "search" @@ -96,7 +112,10 @@ def _record_files() -> list[Path]: for d in (EXTRACTED, ES_EXTRACTED, PT_EXTRACTED, IT_EXTRACTED, IT_MASAF_EXTRACTED, AT_EXTRACTED, DE_EXTRACTED, SI_EXTRACTED, SI_SPEC_EXTRACTED, HR_EXTRACTED, HR_SPEC_EXTRACTED, RO_EXTRACTED, - HU_EXTRACTED, BG_EXTRACTED, GR_EXTRACTED, SK_EXTRACTED, CZ_EXTRACTED): + HU_EXTRACTED, BG_EXTRACTED, GR_EXTRACTED, SK_EXTRACTED, CZ_EXTRACTED, + CH_EXTRACTED, MT_EXTRACTED, CY_EXTRACTED, NL_EXTRACTED, BE_EXTRACTED, + LU_EXTRACTED, CY_SPEC_EXTRACTED, BG_SPEC_EXTRACTED, GR_SPEC_EXTRACTED, + SK_SPEC_EXTRACTED, RO_SPEC_EXTRACTED): if not d.exists(): continue out.extend(jp for jp in d.glob("*.json") if not jp.name.startswith("_")) diff --git a/scripts/04_build_maps.py b/scripts/04_build_maps.py index 555e565..d5083db 100644 --- a/scripts/04_build_maps.py +++ b/scripts/04_build_maps.py @@ -2216,6 +2216,12 @@ def main() -> int: classifications = sorted(_aging_tiers_from_text(record)) categories = record.get("categories") or [] categorie = record.get("categorie", "") or "" + # Records extracted via the manual-override / mirror path carry an + # empty manifest `categorie` while the SIQO-derived `categories` + # list is populated (Côte roannaise, Muscat du Cap Corse) — fall + # back so the wine/non-wine split doesn't mis-flag them. + if not categorie and categories: + categorie = categories[0] # Wine vs. non-wine split: every INAO `categorie` value beginning with # "Vin" (Vin tranquille, Vin mousseux, Vin de liqueur, Vin doux # naturel) is a wine. Spirits (Eaux-de-vie, Rhum, Calvados, diff --git a/scripts/_lib/grape_corpus.py b/scripts/_lib/grape_corpus.py index bd759d0..56de2c6 100644 --- a/scripts/_lib/grape_corpus.py +++ b/scripts/_lib/grape_corpus.py @@ -50,6 +50,25 @@ ("ch", ROOT / "raw" / "ch" / "dokumente-extracted"), # Malta — source language is English (EU single documents are EN). ("en", ROOT / "raw" / "mt" / "dokumente-extracted"), + ("hu", ROOT / "raw" / "hu" / "dokumentumok-extracted"), + ("nl", ROOT / "raw" / "nl" / "dokumenten-extracted"), + # Belgium — per-record source_lang (nl Flemish / fr Walloon); nl covers + # the majority incl. the cross-border Maasvallei. + ("nl", ROOT / "raw" / "be" / "dokumenten-extracted"), + ("fr", ROOT / "raw" / "lu" / "cahier-extracted"), + ("el", ROOT / "raw" / "cy" / "dokumenti-extracted"), + # National-spec sidecars whose rosters stage 04 merges into the map + # (the SI/HR precedent above). The cz/sk/gr/ro register-fiches-extracted + # dirs stay out: stage 04 reads only their terroir text. + ("el", ROOT / "raw" / "cy" / "national-specs-extracted"), + ("bg", ROOT / "raw" / "bg" / "national-specs-extracted"), + # "gr" matches the primary gr row above so the per-slug dominant-lang + # vote doesn't split (the whole column mixes country codes and locales + # — el/de/sl/cs would be the locale-correct values; unknown codes fall + # through the 02b-translate source chain harmlessly). + ("gr", ROOT / "raw" / "gr" / "national-specs-extracted"), + ("sk", ROOT / "raw" / "sk" / "national-specs-extracted"), + ("ro", ROOT / "raw" / "ro" / "national-specs-extracted"), ) diff --git a/scripts/_lib/grape_lexicon.py b/scripts/_lib/grape_lexicon.py index c524b0f..2f11e46 100644 --- a/scripts/_lib/grape_lexicon.py +++ b/scripts/_lib/grape_lexicon.py @@ -1575,6 +1575,10 @@ def _ends_with_colour_word(name: str) -> bool: "ranuse-muskatova": "ranuse-muskatova", # CZ aromatic "sedy-portugal": "sedy-portugal", # CZ "grey Portugal" "tramin-zluty": "tramin-zluty", # CZ "yellow Traminer" — kept distinct from gewurztraminer + # CZ legacy white, in the fiche §6 "**/OTHER" block (old Vyhláška 323/2004 + # list, dropped by 88/2017; not in the Státní odrůdová kniha). Identity + # contested (Goldriesling #4884 vs Bukettraube #1611) — own slug, no fold. + "ryzlink-buketovy": "ryzlink-buketovy", # Croatia autochthonous varieties (MPS specifikacija proizvoda, 2026-05-29). # Self-maps register a distinct native variety; folds collapse a @@ -2447,6 +2451,7 @@ def _ends_with_colour_word(name: str) -> bool: "bily-portugal": "blanc", "modry-janek": "noir", "ranuse-muskatova": "blanc", + "ryzlink-buketovy": "blanc", "sedy-portugal": "gris", "tramin-zluty": "blanc", From 7e814cd49d0602ae98a0ad432271665654c8e5bc Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Thu, 27 Aug 2026 08:22:19 +0200 Subject: [PATCH 2/3] curate(vivc,fr,si): 42-slug VIVC pin pass; SI OJ-C promotions; retire 2 FR SIQO ghosts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - VIVC: 40 ambiguous slugs pinned (research-verified against live passports + NN 25/2020 / NN 81/2022 / NFJ 2024 / Genes 2020 DNA); korithi + schiava pinned vivc_id:false; cristina (RO) brand suspicion refuted — real SCDVV Murfatlar crossing #21045. Pins recorded in the CURATOR_TODO restore block (slug_overrides.json is gitignored). - SI: belokranjec + metliska-crnina promoted to full EU-OJ extractions (OJ C/2026/3572 + C/2026/3598, the due re-check found both). - FR: Cabernet de Saumur + Côtes de Blaye verified RETIRED (2016 arrêté / 2026 EU cancellation) — intentionally absent from SIQO, no pinning. - Reconciliation log: stale-audit pass + action pass + pin-pass entries. Co-Authored-By: Claude Fable 5 --- CURATOR_TODO.md | 209 +++++++++++++++++++++++++++---------- VERIFICATION.md | 2 +- docs/reconciliation-log.md | 95 +++++++++++++++++ 3 files changed, 251 insertions(+), 55 deletions(-) diff --git a/CURATOR_TODO.md b/CURATOR_TODO.md index 2e23320..8d0eeb5 100644 --- a/CURATOR_TODO.md +++ b/CURATOR_TODO.md @@ -81,14 +81,19 @@ DGC cascading unlock realised in this round: **+106 DGCs** (Beaune climats, Chas **To retry the cookie-expired ones:** refresh `cf_clearance` in your browser (open , copy fresh cookie), update `~/.config/openwinemap/legifrance.json`, then `.venv/bin/python scripts/01b_solve_legifrance.py --refresh --only 71 --only 134 --only 211 --only 230 --only 247`. -### SIQO referentiel — 2 wines missing (eAmbrosia has them, INAO doesn't) +### SIQO referentiel — 2 wines missing (eAmbrosia has them, INAO doesn't) — ✅ both RETIRED (2026-08-26) -❌ Surfaced by 2026-05-17 eAmbrosia FR-wine reconciliation in [VERIFICATION.md](VERIFICATION.md). Both exist in the EU register but not in `raw/inao/siqo-referentiel.csv` — likely retired/merged on the INAO side without flowing through to the EU register. +✅ Web-research pass confirmed both are intentionally absent — no pinning needed; +the eAmbrosia `registered` rows are stale-register artifacts (Austrian-PDO precedent). +Full evidence in [VERIFICATION.md](VERIFICATION.md) (2026-05-17 entry, finding #4). -| eAmbrosia file_number | Name | Verification needed | +| eAmbrosia file_number | Name | Verdict | |---|---|---| -| PDO-FR-A0257 | Cabernet de Saumur | Confirm via INAO product page or Légifrance whether still in force; if active, pin via `manual_overrides.json` | -| PDO-FR-A0271 | Côtes de Blaye | Often considered merged into the Blaye / Premières Côtes de Blaye family. Verify status. | +| PDO-FR-A0257 | Cabernet de Saumur | RETIRED-MERGED into AOC Saumur (rosé) — Arrêté du 19 juillet 2016, art. 2 abrogates Décret 2011-1360 | +| PDO-FR-A0271 | Côtes de Blaye | RETIRED — not claimed since 2015, off INAO's list; EU cancellation PDO-FR-A0271-CANCEL filed 13/01/2026 ("Applied") | + +🟡 Loose end: confirm the OJ cancellation notice (likely C/2026/2994) in a +browser once the EU procedure terminates — purely for the provenance note. ### Geometry — Comté Tolosan cluster — ✅ resolved @@ -112,9 +117,17 @@ Reconciled 2026-08-26 against `raw/terroir-facts/`: 7 of the 8 now carry bullets (cotes-de-thau 4 · calvados-vin 2 · cotes-catalanes 5 · thezac-perricard 5 · vicomte-d-aumelas 4 · vallee-du-torgan 3 · pays-d-herault 2). -⏳ **`cote-vermeille` still at 0 facts** (record is otherwise healthy — 85 grape -slugs, `aires-csv` geometry). Re-run [scripts/02d_extract_terroir_facts.py](scripts/02d_extract_terroir_facts.py) -on it with `--verbose` to diagnose the fuzzy-coverage drop. +✅ **`cote-vermeille` fixed later the same day — 8/8 closed.** Root cause was +upstream of 02d: its cahier's lien heading is `10- lien avec la zone +géographique` with a **lowercase** title start, which `IGP_SECTION_HDR_RE` +rejected (the uppercase requirement that keeps "125 mg/l …" lines from +becoming phantom titles), so section 10 was never sliced and 02d grounded on +the 1,141-char section 8 (élevage rules) — every candidate failed the fuzzy +filter. Fix: a `lien\b` lowercase carve-out in the title class +([scripts/02_extract_cahiers.py](scripts/02_extract_cahiers.py)); full FR +re-extraction verified byte-identical lien lengths corpus-wide (agenais 9190 +· maures 8523 · pays-d-oc 11546) with cote-vermeille going 0 → **7,110 +chars** → 5 terroir facts, translated en/es/nl. ### PNOCDC draft PDFs — section X missing or template-only — ✅ complete @@ -758,13 +771,19 @@ semidano, moscatello-selvatico, schiava-grigia, francavilla, pelaverga-piccolo) gain only the VIVC# citation — no tooltip text exists to fetch. -### `ortrugo-dei-colli-piacentini` — ❌ no DOCUMENTO UNICO anchor +### `ortrugo-dei-colli-piacentini` — ◑ investigated 2026-08-26: old table-template; content covered by MASAF -One wine (PDO-IT-A0350) whose EUR-Lex HTML doesn't have the standard -`

DOCUMENTO UNICO

` anchor — likely an older -template. Investigate the raw HTML at -`raw/it/oj-pages/ortrugo-dei-colli-piacentini.html` and either extend -the anchor regex or pin a working override URL. +Investigation result: the cached EUR-Lex HTML is the **pre-2016 +table-based OJ layout** (958 `class="table"`/`tbl-txt` cells, a single +`ti-grseq-1` occurrence) — a different parser family entirely, not an +anchor-regex tweak. Meanwhile the wine is content-complete through the +MASAF sidecar (grapes, articles 1/2/3/9, regione Emilia-Romagna, +`figshare-pdo` polygon), so nothing is missing on the map. Remaining +value of a real EU-OJ extraction is provenance polish only. Two Phase-2 +options if ever wanted: (a) add IT to +[scripts/extract_register_fiches.py](scripts/extract_register_fiches.py) +`COUNTRY_CONFIG` and pull its register fiche (uniform template), or +(b) write a table-template slicer. Low priority. ### Complete-coverage pass residuals — ⏳ (2026-05-30; re-verified still open 2026-08-26) @@ -1099,14 +1118,26 @@ an Austria clone, but only 1 wine has a fetchable EU single document. ✅ `cvicek` (PDO-SI-A1561) — full extract from its EUR-Lex ENOTNI DOKUMENT (OJ C/2026/256), 17 grape varieties. -✅ The 16 former content-stubs are all augmented by the MKGP/Uradni-list -national-spec layer (stage 01c/02f, shipped 2026-05-29 — see below), and -`bela-krajina` + `belokranjec` additionally carry per-DOP terroir from -the eAmbrosia register fiche (2026-06-02 pass). 🟡 The "re-check in 3–6 -months" window for an OJ-C ENOTNI DOKUMENT landing for `belokranjec` -(PDO-SI-A1576) + `metliska-crnina` (PDO-SI-A1579) — set 2026-05-23 — **is -now due**; an EU-OJ publication would still upgrade them from -national-spec to full EU-OJ extraction. +✅ **Promoted 2026-08-26: `belokranjec` + `metliska-crnina` are now full +EU-OJ extractions — SI is 3 EU-OJ + 14 national-spec.** The due re-check +found both OJ-C publications (the Cviček path, exactly as predicted): +Belokranjec **OJ C/2026/3572** (6.7.2026, PDO-SI-A1576-AM01 approved +20.4.2026) and Metliška črnina **OJ C/2026/3598** (13.7.2026, +PDO-SI-A1579-AM01 approved 17.4.2026) — modernised consolidated ENOTNI +DOKUMENT texts. EUR-Lex URLs pinned in the overrides (WAF-free mirror: +Publications Office Cellar, `publications.europa.eu/resource/oj/C_2026035xx` +with `Accept: application/xhtml+xml` + `Accept-Language: slv`); fetched +via si/01 + the 01b Chromium bootstrap; si/02 extracted both (18 + 10 +grapes, 2.7/2.4 KB lien); 02d re-grounded their terroir on the EU-OJ text +(10 + 8 facts) and 02e translated en/fr/es/nl. NB the two override +entries keep `specifikacija_url` for the old national-spec source — do +NOT run `si/01c --refresh` for these slugs (it would clobber the spec +cache with the EU-OJ page). + +The remaining 14 content-stubs stay augmented by the MKGP/Uradni-list +national-spec layer (stage 01c/02f, shipped 2026-05-29 — see below); +`bela-krajina` additionally carries per-DOP terroir from the eAmbrosia +register fiche (2026-06-02 pass). Historical detail (13 grandfathered DOPs + the 3 region IGPs had no public single-document URL in eAmbrosia — only a non-fetchable @@ -1548,13 +1579,15 @@ Rkatsiteli) pinned in [raw/vivc/slug_overrides.json](raw/vivc/slug_overrides.jso After re-extraction one survivor remains, needing a curator look at the source EU-OJ HTML: -- 🟡 **Colinele Dobrogei — `Cristina N`**. No VIVC entry, no - wein.plus / SCDVV reference, no Romanian viticulture-press - mention. Suspected wine **brand/cuvée name** mis-parsed by stage 02 - as a variety. Verify against - [raw/ro/oj-pages/colinele-dobrogei.html](raw/ro/oj-pages/colinele-dobrogei.html) - section 7 — if it's a brand, add it to `GRAPE_BLOCKLIST`; if a - real variety, mint a new slug. +- ✅ **Colinele Dobrogei — `Cristina N` — real variety (2026-08-26).** + The brand suspicion is refuted: VIVC **#21045 CRISTINA** is a + registered Romanian wine grape (noir, Chardonnay × Băbească Neagră + marker-confirmed, bred at SCDVV Murfatlar by Ionescu/Oslobeanu, + European Catalogue), and the citing document lists "Cristina N" + inside its variety roster beside the sibling Murfatlar crossings + Columna and Mamaia, plus in a per-variety yield row. Slug existed + (`cristina`, noir, self-map); VIVC pin added in the 2026-08-26 + `/research-gaps vivc-ambiguous` pass. - 🟡 **Dealurile Moldovei — `Zghihară neagră`**. VIVC's Zghihară de Huși #20281 is firmly white; no documented red biotype in wein.plus, Indigene, or Crameromania. Likely a typo for plain @@ -1874,16 +1907,24 @@ oblasť under different brand registrations. Country #14 (added 2026-05-24). 13 wine GIs (11 DOP + 2 PGI), all 13 on the map. -### Register-fiche variety `Ryzlink buketový` — ⏳ verify before minting - -The EU-register fiche §6 for some CZ wines lists `Ryzlink buketový` -("bouquet Riesling"). Research (2026-06) could not ground it in VIVC / -wein.plus / the Czech Státní odrůdová kniha; the name is ambiguous -(Bukettriesling is a documented synonym of BOTH Riesling and the distinct -German Bukettraube). Left UNFOLDED (own queue entry) pending a check -against **Vyhláška 88/2017 Sb. Příloha 2 / ÚKZÚZ register** — it may be a -label term rather than a registered variety. All other CZ/GR/SI/BG/HU/HR -fiche natives resolved + folded into `grape_lexicon.py`. +### Register-fiche variety `Ryzlink buketový` — ✅ resolved 2026-08-26 (own slug, no fold, no VIVC) + +Verification ran both checks: **absent** from Vyhláška 88/2017 Sb. +Příloha 2 (confirmed against the local cache — 67 varieties, no buket-*) +and **absent** from the ÚKZÚZ Státní odrůdová kniha (Přehled odrůd révy +2020: zero hits; eAGRI/trade sources cite it as the canonical example of +a *non-registered* zemské-víno variety). It IS a real legacy variety — +item ~20 of the old Vyhláška 323/2004 Příloha 15 list, **dropped by the +2017 decree** — surviving only in the register fiche §6 `**`/OTHER +legacy block (ceske + moravske). Identity is contested (cs.wikipedia → +Goldriesling VIVC #4884; the German Bukettriesling synonym chain → +Bukettraube VIVC #1611; NEITHER passport carries the Czech name), so per +the Cornalin/Humagne precedent it got its **own slug** +`ryzlink-buketovy` (blanc) in `grape_lexicon.py`, a `vivc_id: false` +pin in `raw/vivc/slug_overrides.json`, and the CZ register-fiche +re-extraction now resolves it (unknowns queue cleared). Note: CZ fiche +grapes feed only the terroir-text layer, not the map roster, so this is +queue hygiene + future-proofing, not a visible-pill change. ### JEDNOTNÝ DOKUMENT — ❌ 0 / 13 extracted @@ -2108,14 +2149,16 @@ style descriptions + variety/yield rules to replace the amendment- boilerplate summary and add regulator-grounded data. Licence: Maltese legislation © Govt of Malta — verify reuse terms before ingesting. -### Indigenous varieties — ✅ +### Indigenous varieties — ✅ fully enriched (2026-08-26) Ġellewża (red) + Girgentina (white) folded into `grape_lexicon.py` -(`DEFAULT_COLOUR` + `GRAPE_ALIAS`). VIVC IDs + grape-Wikipedia tooltips -not yet resolved (02g/02b not re-run for the 2 new slugs) — optional -enrichment; pills render with colour but no tooltip card. Run -`scripts/02g_fetch_vivc.py` + `scripts/02b_fetch_grape_lexicon.py --only -gellewza --only girgentina` to add them. +(`DEFAULT_COLOUR` + `GRAPE_ALIAS`). Enrichment ran 2026-08-26: VIVC +resolved exact-cultivar for both — **Ġellewża #14174**, **Girgentina +#17787** — en.wikipedia cards fetched (both exist) and translated into +fr/es/nl via the 02b translate sidecar. Pills now carry colour, VIVC +bracket/link, and tooltip in all four locales after the next stage-04 +rebuild. (Prerequisite fix: `raw/mt/dokumente-extracted` was missing +from the 02g corpus walk — see the cross-country corpus-walk note.) ## Cyprus @@ -2496,13 +2539,43 @@ Deferred (per-record or parser-level, not slug-level): to Piave. Left as-is. - Bare "piquepoul" is colour-mixed (Saint-Chinian N vs La Clape B) — a plain alias can't split it; needs colour-aware alias support. +- Bare "korithi" (GR) is the same case (2026-08-26): Zakynthos "Korithi + B" = #6414 KORITHI ASPRO vs Mantzavinata "Korithi N" = #6415 KORITHI + MAVRO; pinned `vivc_id: false` until colour-aware aliasing lands — + then split B→6414 / N→6415. Loose ends: -- `grape_corpus._SOURCES` misses `hu/dokumentumok-extracted` plus the - BG/GR/SK/RO national-spec + register-fiche sidecar dirs — HU-only slugs - (e.g. muscat-hambourg before its fold) are invisible to 02g/02b and to - the collision audit. Adding them will surface a batch of unresolved - slugs for 02g. +- ✅ **Corpus-walk gap closed 2026-08-26.** `grape_corpus._SOURCES` gained + hu / nl / be / lu / cy (+ cy national-specs) and the BG/GR/SK/RO + national-specs-extracted sidecar dirs; `02g_fetch_vivc.py`'s own walk + gained ch / mt / cy / nl / be / lu + the same sidecar dirs. The + cz/sk/gr/ro **register-fiches-extracted dirs stay deliberately + excluded** — stage 04 reads only their terroir text, so their rosters + must not weigh the corpus-slug frequency tiers (comments at both + sites). As predicted this surfaced **141 corpus slugs without a VIVC + by-slug record** (GR/CY/BG native tails, CZ registry crossings…); a + full incremental 02g sweep was run over them 2026-08-26 — final + buckets: 852 exact-cultivar + 5 exact-prime + 387 override, 120 miss + (no VIVC candidate — obscure natives, ship without bracket, by + design), and 42 `ambiguous-cultivar` slugs queued — **✅ all 42 + resolved the same day** via a `/research-gaps vivc-ambiguous` pass + (5 parallel research agents; evidence table in + [tmp/vivc-ambiguous-research-results.md](tmp/vivc-ambiguous-research-results.md)): + 40 pinned against live VIVC passports + national registers (NN + 25/2020 + NN 81/2022 for HR, NFJ 2024 for HU, Bundessortenamt-class + evidence for DE, Genes 2020 DNA anchors), 2 pinned `vivc_id: false` + (korithi — a two-variety colour-split case; schiava — the standing + bare-family-name ruling). Three new pins join existing same-id + groups, handled by the facet tier: tribidrag → primitivo+zinfandel + (#9703), muskat-zuti → moscato-giallo (#8056), bratkovina → maresco + (#1660). Notable: `cristina` (RO) was NOT a brand — it is the SCDVV + Murfatlar crossing VIVC #21045 (closes the 2026-05-23 🟡 below). + Residual + cosmetic muddle: the `_SOURCES` lang column mixes country codes and + locales (gr/at/si/cz where el/de/sl/cs would be locale-correct); + unknown codes fall through the 02b-translate source chain harmlessly, + but normalising them (+ invalidating the dominant-lang cache) is a + clean-up candidate. - **Stage-02 runs across countries must not run concurrently.** The vocabulary scans the FR/ES/PT extracted dirs and silently skips mid-write files (`json.JSONDecodeError → continue`), so a parallel @@ -2593,10 +2666,10 @@ Remaining loose ends: 2026-08-21: all six (plus budai-zold 881, verduzzo-trevigiano 12977 and the corrected forastera-blanca 24859) re-fetched against live vivc.de via `02g --refresh --only`; every prime/colour matches the pin. -- 36 RISKY ambiguous synonyms after the 2026-08-21 re-extractions — none - known to bind a wrong cultivar (Brachetto, the worst offender found - since, now has its own slug), but re-run the audit after any vocab - change. +- 38 RISKY ambiguous synonyms after the 2026-08-26 corpus-walk widening + (was 36 on 2026-08-21; +2 from the newly-walked dirs) — none known to + bind a wrong cultivar (Brachetto, the worst offender found since, now + has its own slug), but re-run the audit after any vocab change. ### Note — recently added VIVC pins live only in gitignored `raw/` @@ -2631,5 +2704,33 @@ Added in the 2026-08-21 collision pass (see OQ3 above for the evidence): `jurancon-noir` via GRAPE_ALIAS; `vivc_id: false` = deliberately absent from VIVC, the bianchello mechanism.) +Added in the 2026-08-26 pass: + +```json +"ryzlink-buketovy": {"vivc_id": false} +``` + +(CZ legacy zemské-víno white from the fiche §6 `**`/OTHER block; identity +contested Goldriesling #4884 vs Bukettraube #1611, neither VIVC-grounded — +see the CZ section for the evidence. The MT natives resolved WITHOUT pins: +gellewza #14174 + girgentina #17787 are plain `exact-cultivar` by-slug +records, no override needed.) + +Added in the 2026-08-26 `/research-gaps vivc-ambiguous` pass (42 entries; +full evidence in [tmp/vivc-ambiguous-research-results.md](tmp/vivc-ambiguous-research-results.md)): + +``` +avgoustiatis→801 kanella→16124 kontokladi→6395 kotsifali→6446 +koutsoubeli→6463 mavrotragano→40210 skiadopoulo→11849 thrapsathiri→12428 +vertzami-lefko→13013 bratkovina→1660 debit→10423 draganela→21070 +grk→5066 vugava→13184 zadarka→13365 zlahtina→22843 modra-kosovina→24493 +muskat-zuti→8056 svrdlovina-crna→15638 trbljan→8075 zumic→24915 +zametovka→6047 vitovska-grganja→16017 harslevelu→5314 goher→767 +csomor→3281 nektar→16179 rozalia→23930 zierfandler→13443 tribidrag→9703 +negroamaro→8456 andre→456 helios→17133 juwel→13212 orion→8802 +orangentraube→16645 tauberschwarz→16156 weisser-lagler→24537 +busuioaca-de-bohotin→8248 cristina→21045 korithi→false schiava→false +``` + Same applies to the other ~450 pins already in that file; the deployed site is built from the curator's machine, so production is unaffected. diff --git a/VERIFICATION.md b/VERIFICATION.md index a4dd775..dac523c 100644 --- a/VERIFICATION.md +++ b/VERIFICATION.md @@ -87,7 +87,7 @@ upstream gap. Findings: | 1 | **Synonym / composite-name fold** | Bourg ↔ "Côtes de Bourg, Bourg et Bourgeais"; Corse ↔ "Vin de Corse ou Corse" | None — same wines, INAO uses long composite, eAmbrosia uses short form. Improve matcher to compare across all `protectedNames` not just `protectedNames[0]`. | | 2 | **Bordeaux umbrella rollup** | Blaye, Sainte-Foy-Bordeaux | None — eAmbrosia tracks Blaye / Sainte-Foy-Bordeaux as separate PDOs (PDO-FR-A0712, A0407); INAO rolls them under id_appellation=685 "Côtes de Bordeaux" as DGCs. Same wines, modeling difference. | | 3 | ~~**Parent-detection bug (4 AOCs)** ⚠️ Alsace (id=1), Blagny (id=136), Comté Tolosan (id=861), Fiefs Vendéens (id=1028)~~ | **Resolved 2026-05-17** | Stage 02's parent-detection used strict `denomination == appellation` equality and fell back to `denoms[0]` when it failed — the fallback row was then re-emitted as a DGC, clobbering the parent's index entry. Replaced with a `_is_parent_denom` helper that folds synonym order ("Alsace" vs "Alsace ou Vin d'Alsace"), case ("Comté Tolosan" vs "Comté tolosan"), and composite forms ("Blagny" vs "Blagny ou Blagny Côte de Beaune") via the existing `candidate_keys()` normaliser. When no SIQO row matches at all (Fiefs Vendéens — all 5 rows are DGCs), a parent is synthesized from the cahier header. DGC loop now skips by chosen parent's `id_denomination_geo`, not by strict equality. The bug also hit 3 cider/spirit AOCs (id=335 Calvados Domfontais, id=553 Cidre de Bretagne, id=1268 Euskal Sagardoa) — 7 orphans total, now 0. Also fixed a latent key-collision bug where `index[id_app]` fallback could clash with a same-numeric `id_denomination_geo` from another appellation (Fiefs Vendéens id_app=1028 vs Pommard DGC "Clos de la Commaraine" id_denom=1028) — synthetic-parent index entries now use `f"app:{id_app}"` keys. See `scripts/02_extract_cahiers.py:_is_parent_denom`. | -| 4 | **Missing from INAO SIQO entirely** | Cabernet de Saumur (PDO-FR-A0257), Côtes de Blaye (PDO-FR-A0271) | Upstream gap. Both exist in eAmbrosia but not in `raw/inao/siqo-referentiel.csv` (likely retired/merged on the INAO side without flowing through to the EU register). Curator follow-up: confirm via INAO product pages whether these are still in force, then either pin via `manual_overrides.json` or annotate. | +| 4 | **Missing from INAO SIQO entirely** | Cabernet de Saumur (PDO-FR-A0257), Côtes de Blaye (PDO-FR-A0271) | ✅ **Resolved 2026-08-26 (web-research pass): both RETIRED — intentionally absent, no pinning.** *Cabernet de Saumur*: suppressed July 2016, folded into AOC Saumur as Saumur rosé — Arrêté du 19 juillet 2016 (JORF 29/07/2016), art. 2 abrogates Décret n° 2011-1360 (abrogation banner at legifrance.gouv.fr/loda/id/JORFTEXT000024717966); INAO product page 8125 is 404. *Côtes de Blaye*: not claimed since 2015, removed from INAO's AOC list (national retirement by cessation, no abrogation décret found); EU cancellation application **PDO-FR-A0271-CANCEL** filed 13/01/2026, status "Applied" in eAmbrosia as of 2026-08-26 (likely OJ notice C/2026/2994 — EUR-Lex WAF-blocked, confirm in a browser). Both eAmbrosia `registered` rows are stale-register artifacts (the Austrian-PDO precedent). | Two wines exist in our pipeline as `status=registered` parents but in eAmbrosia as `status=applied` (not yet registered): diff --git a/docs/reconciliation-log.md b/docs/reconciliation-log.md index 58a8a27..e7e6e62 100644 --- a/docs/reconciliation-log.md +++ b/docs/reconciliation-log.md @@ -8,6 +8,101 @@ Newest first. --- +## 2026-08-26 — /research-gaps vivc-ambiguous: 42-slug pin pass + +The queue surfaced by the corpus-walk widening (same-day action pass below), +researched by 5 parallel agents against live VIVC passports + national +registers (HR NN 25/2020 + NN 81/2022, HU NFJ 2024, Genes 2020 DNA anchors, +wein.plus, Wine-Grapes-adjacent literature). User-confirmed and applied: + +- **40 PINNED** into `raw/vivc/slug_overrides.json` (now 499 entries) — + full per-slug evidence in `tmp/vivc-ambiguous-research-results.md` and + the pin table in CURATOR_TODO's restore-note block. Highlights: + harslevelu→5314, zierfandler→13443 (Spätrot), tribidrag→9703 (joins the + primitivo+zinfandel same-id group), negroamaro→8456, zametovka→6047 + (Kavčina črna / Stara trta), grk→5066, debit→10423 (RUZEVINA), + trbljan→8075 (MOSTOSA, with a caveat on VIVC's new 0-holding #27344), + andre→456 (the Czech crossing), goher→767 (white conculta member), + busuioaca-de-bohotin→8248 (Muscat rouge à petits grains). +- **`cristina` (RO) brand suspicion REFUTED** — VIVC #21045, a real SCDVV + Murfatlar crossing (Chardonnay × Băbească Neagră); closes the + 2026-05-23 Colinele-Dobrogei 🟡. +- **2 × `vivc_id: false`**: `korithi` (two distinct varieties — B=#6414 / + N=#6415; queued next to bare-piquepoul for colour-aware alias support) + and `schiava` (standing bare-family-name ruling). +- Three new legitimate same-id groups, auto-unified by the facet tier: + tribidrag/primitivo/zinfandel #9703, muskat-zuti/moscato-giallo #8056, + bratkovina/maresco #1660. +- Applied via `02g --refresh --only <40 slugs>` (all resolve `override`); + a scoped 02b sweep over the 33 card-less pins recovered **12 en + Wikipedia cards** via the VIVC-synonym chain (Grk, Kotsifali, Debit, + Vugava, Žlahtina, Mavrotragano, Thrapsathiri, Trbljan, Muškat žuti, + Vitovska, Busuioacă de Bohotin, Tauberschwarz) + 29 fr/es/nl + translations; the other 21 are genuinely article-less natives. +- Provenance: `tmp/vivc-ambiguous-research-prompt.md` + + `tmp/vivc-ambiguous-research-results.md`. + +--- + +## 2026-08-26 — open-todo action pass (post-reconciliation) + +Worked the reconciled queue top-down; full pipeline runs, every result +verified in the rebuilt `wiki/` (stage 04 clean, 2,908 records, 0 stale +overrides). Landed: + +- **FR `cote-vermeille` terroir gap closed (8/8 zero-bullet parents done).** + Root cause was a stage-02 slicing miss, not 02d: the cahier's heading + `10- lien avec la zone géographique` starts lowercase, which + `IGP_SECTION_HDR_RE`'s uppercase-title requirement rejected. Added a + `lien\b` lowercase carve-out; full FR re-extraction (466 + 1,074 DGCs, 0 + errors) verified byte-identical lien lengths corpus-wide (agenais 9190 / + maures 8523 / pays-d-oc 11546); cote-vermeille 0 → 7,110 chars → 5 facts + (02d anthropic) → en/es/nl (02e; exactly 3 jobs = proof nothing else + drifted). +- **SI: belokranjec + metliska-crnina promoted to full EU-OJ extractions.** + The due OJ-C re-check found both: C/2026/3572 (6.7.2026) + C/2026/3598 + (13.7.2026), consolidated ENOTNI DOKUMENT each (WAF-free mirror: Cellar + XHTML). Pinned in the SI overrides (old spec URL kept as + `specifikacija_url`; si/01c --refresh for these slugs is now forbidden), + fetched via si/01 + 01b Chromium, extracted (18 + 10 grapes, 2.7/2.4 KB + lien), terroir re-grounded on the EU-OJ text (10 + 8 facts, 02e ×4 + locales). SI = 3 EU-OJ + 14 national-spec. +- **FR SIQO 2 missing wines: both RETIRED (research-verified).** Cabernet + de Saumur folded into AOC Saumur rosé (Arrêté 19.7.2016 art. 2 abrogates + Décret 2011-1360); Côtes de Blaye unclaimed since 2015 + EU cancellation + PDO-FR-A0271-CANCEL filed 13.1.2026. Recorded in VERIFICATION.md; no + pinning — eAmbrosia rows are stale-register artifacts. +- **`is_wine` mis-flag fixed** for Côte roannaise + Muscat du Cap Corse: + stage 04 now falls back to the SIQO `categories` list when the manifest + `categorie` is empty (both records came via the mirror/override path). + Verified True in the rebuilt blob. +- **CZ `Ryzlink buketový` resolved**: absent from Vyhláška 88/2017 Příloha + 2 (local check) AND the ÚKZÚZ Státní odrůdová kniha (web research); a + real legacy variety from the repealed 323/2004 list with contested + identity (Goldriesling #4884 vs Bukettraube #1611, neither VIVC-grounded) + → own slug `ryzlink-buketovy` (blanc), `vivc_id: false` pin, CZ fiche + re-extraction cleared the unknowns queue. +- **MT grape enrichment complete**: Ġellewża VIVC #14174 + Girgentina + #17787 (exact-cultivar), en.wikipedia cards fetched, fr/es/nl translated + (a 19-job corpus-wide 02b-translate pass also drained the HU/ES tail). +- **Corpus-walk gap closed (OQ3 loose end)**: 02g + + `grape_corpus._SOURCES` now walk mt/cy/nl/be/lu/ch/hu + the BG/GR/SK/RO + national-specs-extracted sidecars (register-fiche dirs deliberately + excluded — terroir-only). Full incremental 02g sweep over the 141 + newly-visible slugs: 852 exact + 5 prime + 387 override / 120 miss / 42 + ambiguous → queued in `slug_overrides.example.json` for a + `/research-gaps` pin pass (search caches already on disk). + `audit_ambiguous_synonyms`: 36 → 38 RISKY (+2, none known-wrong). +- **ortrugo-dei-colli-piacentini investigated**: pre-2016 table-based OJ + template (a different parser family, not an anchor tweak); wine is + content-complete via MASAF — downgraded to low-priority with two Phase-2 + options. +- Wiki regenerated (1,540 FR + 17 SI pages); stage-04 rebuild + facet + canonical roll-ups verified intact (malbec→cot, araignan→picardan, + s-saul→cinsault). + +--- + ## 2026-08-26 — interprofession / organisation URL sweep The parallel curation the stale-audit entry below defers to. Scope: the From b77677ab68eae27b857673be161c3f7c4f09ef4a Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Sun, 6 Sep 2026 18:33:24 +0200 Subject: [PATCH 3/3] =?UTF-8?q?feat(gb):=20United=20Kingdom=20pipeline=20?= =?UTF-8?q?=E2=80=94=20country=20#19,=206/6=20extracted,=200=20stubs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The first post-Brexit register in the corpus, and the first where BOTH of the usual spines are unavailable: the UK is no longer in the EU GI scheme (Sussex, registered 2022, was never an EU registration), and Bétard 2022 is an EU PDO layer carrying no PDO-GB-* geometry. Both are replaced by UK-domestic sources. Corpus: 6 registered wine GIs — 4 PDO (English, Welsh, Sussex, Darnibole) + 2 PGI (English Regional, Welsh Regional). Every one ships a public product specification, so the UK is the only country with no stub tier, no national-spec fallback layer and no WAF bootstrap. Spine (gb/00): the GOV.UK "protected food and drink names" register, via the site's own search + content API. The register publishes no GI file number, so stage 00 bridges slug → PDO-GB-* against the EU register. A 7th name (The Crouch Valley) is still an application and is filtered out. Specs (gb/01 → gb/02): 5 PDF + Sussex's .docx, three unrelated layouts handled by _lib/gb/spec.py — defra-pfn-2011 (PART blocks + upper-case VINE VARIETIES), defra-pfn-application (Darnibole, no link section so terroir falls back to the demarcated-area definition) and uk-gi-single-document (Sussex). Roster parsing rejoins PDF line-wraps by run (so "Blau\nPortugueser" stays one variety) but splits bulleted one-per-line rosters, and classifies prose out by item shape. Two source typos are repaired structurally because they split a name before any alias could apply. Geometry: ONS Open Geography (OGL v3.0) — England/Wales country polygons and the East Sussex ∪ West Sussex ∪ Brighton and Hove union. Areas verified against published figures (England 130,522 vs 130,279 km²; Sussex 3,788 vs 3,782). Darnibole, a 5 ha single vineyard with no published polygon, is reconstructed from the OS/RPA parcel references on its own specification plan — 6.1 ha against a declared 5 ha, anchored three ways (internal consistency, an EU-DEM slope transect, the ONS postcode centroid) and disclosed in the panel as approximate. Terroir: 02d/02e ship 51 cahier-primary bullets across all 6 wines, translated to fr/es/nl. Unlike Malta (the other English-source corpus) the UK specs carry a real link narrative, so a missing Wikipedia article is not disqualifying. Also in this change: - grape lexicon: 7 UK varieties (Cascade, Madeleine Angevine/Sylvaner, Roter Veltliner — distinct from Frühroter — Frühgipfler via its synonym Senator, Triomphe d'Alsace, Gagarin Blue) + Blau Portugueser / Elbling White aliases. VIVC ids verified 2026-09. - fix(map): grape pills showed another country's spelling. emit_html dropped a record's own name whenever it matched its slug, and the client fell back to the most frequent spelling corpus-wide — so English "Optima" rendered as Germany's "Optima 113", German "Müller Thurgau" as Hungarian "Rizlingszilváni", Spanish "godello" as Portuguese "Gouveio". 1,047 substantive cases across 10 countries. - fix(it): tarantino and rotae were bound to the wrong appellation's Wikipedia article (Trentino and Roma), and rotae shipped a wiki-sourced terroir bullet about the Roma DOC in all four locales. Both pinned missing; 02d/02e re-run for the two slugs. - audit_terroir_facts: wire GB in, and make unsupported countries an explicit counted skip instead of a KeyError swallowed as a corrupt cache (it silently covered only FR/ES — 1,030 records across 18 countries). - geometry-outlier whitelist for the four national GB records (Isles of Scilly, Anglesey). - WineGB as the England/Wales trade body. Co-Authored-By: Claude Opus 5 (1M context) --- CLAUDE.md | 244 +++++++ CURATOR_TODO.md | 288 ++++++++ VERIFICATION.md | 68 ++ raw/wikipedia/aoc_overrides.json | 344 +++++----- raw/wikipedia/grape_overrides.json | 4 +- scripts/04_build_maps.py | 92 ++- scripts/_lib/appellation_urls.json | 8 + scripts/_lib/assets/app.js | 11 + scripts/_lib/content_block.py | 6 + scripts/_lib/gb/__init__.py | 0 scripts/_lib/gb/darnibole.py | 103 +++ scripts/_lib/gb/geometry.py | 179 +++++ scripts/_lib/gb/region.py | 59 ++ scripts/_lib/gb/spec.py | 456 +++++++++++++ scripts/_lib/geometry_outlier_overrides.json | 58 +- scripts/_lib/grape_corpus.py | 1 + scripts/_lib/grape_lexicon.py | 36 + scripts/_lib/map_template.py | 9 + scripts/_lib/summaries.py | 7 +- scripts/audit_gb_coverage.py | 134 ++++ scripts/audit_terroir_facts.py | 101 ++- scripts/gb/00_fetch_data.py | 364 +++++++++++ scripts/gb/01_fetch_specs.py | 168 +++++ scripts/gb/02_extract_specs.py | 296 +++++++++ scripts/gb/02d_extract_terroir_facts.py | 617 ++++++++++++++++++ scripts/gb/02e_translate_terroir_facts.py | 363 +++++++++++ scripts/gb/03_generate_wiki.py | 265 ++++++++ .../gb_defra_application_darnibole.txt | 31 + tests/fixtures/gb_defra_pfn_2011_english.txt | 73 +++ .../fixtures/gb_uk_single_document_sussex.txt | 36 + tests/test_gb_parser.py | 241 +++++++ 31 files changed, 4459 insertions(+), 203 deletions(-) create mode 100644 scripts/_lib/gb/__init__.py create mode 100644 scripts/_lib/gb/darnibole.py create mode 100644 scripts/_lib/gb/geometry.py create mode 100644 scripts/_lib/gb/region.py create mode 100644 scripts/_lib/gb/spec.py create mode 100644 scripts/audit_gb_coverage.py create mode 100644 scripts/gb/00_fetch_data.py create mode 100644 scripts/gb/01_fetch_specs.py create mode 100644 scripts/gb/02_extract_specs.py create mode 100644 scripts/gb/02d_extract_terroir_facts.py create mode 100644 scripts/gb/02e_translate_terroir_facts.py create mode 100644 scripts/gb/03_generate_wiki.py create mode 100644 tests/fixtures/gb_defra_application_darnibole.txt create mode 100644 tests/fixtures/gb_defra_pfn_2011_english.txt create mode 100644 tests/fixtures/gb_uk_single_document_sussex.txt create mode 100644 tests/test_gb_parser.py diff --git a/CLAUDE.md b/CLAUDE.md index 140aed8..5577ea6 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -4102,6 +4102,250 @@ template.py` writes a 1-entry queue for it — the curator's only job is to pin a public, licence-clear EU-OJ English SINGLE-DOCUMENT page if the Commission ever publishes one, then re-run mt/01 → mt/02 → stage 04. +## United Kingdom pipeline (`scripts/gb/`) + +Country #19, and the first **post-Brexit** register in the corpus: the +UK is no longer part of the EU GI scheme, so *both* of the spines every +other country leans on are unavailable. eAmbrosia still lists the UK's +pre-2021 names as legacy rows, but Sussex — registered in 2022 under the +UK scheme — is not an EU registration at all; and Bétard 2022 is an **EU** +PDO layer that carries no `PDO-GB-*` geometry. Both are replaced by +UK-domestic sources. + +Corpus: **6 registered wine GIs** — 4 PDO (English, Welsh, Sussex, +Darnibole) + 2 PGI (English Regional, Welsh Regional). The country code +is `gb` (ISO 3166-1 alpha-2, matching the register's own `PDO-GB-*` +identifiers); `source_lang` is `"en"`, so GB joins Malta as an +English-source corpus needing no translation for the canonical `/` +surface. + +**The UK has no stub tier.** Every registered wine ships a public +product specification, which makes it the only country in the corpus at +6/6 extraction with no national-spec fallback layer, no curator queue +for missing documents, and no `01b_solve_waf.py` — GOV.UK serves the +register and its attachments without a WAF, a cookie gate or a +JavaScript challenge. + +Spine: the **GOV.UK "protected food and drink names" register**. DEFRA +publishes the UK GI schemes as a structured GOV.UK *finder*, queryable +through the site's own search API: + + https://www.gov.uk/api/search.json + ?filter_format=protected_food_drink_name + &filter_register=wines + &filter_country_of_origin=united-kingdom + +Each hit resolves to a per-GI content-API document +(`/api/content/protected-food-drink-names/`) carrying typed +metadata — protection type, status, application and UK/EU registration +dates, reason for protection — plus the **product specification** as an +attachment. A 7th name, *The Crouch Valley* (PDO, applied 2023-03-06), +is still in assessment and is filtered out by `status=registered` the +way the eAmbrosia countries filter theirs; `audit_gb_coverage.py` +reports it so the queue stays visible. + +The register publishes **no GI file number** (only Sussex's +specification states one, in its own text: "PDO GB number: W0006"). +Since the rest of the corpus keys on `PDO-xx-*` / `PGI-xx-*`, stage 00 +bridges slug → file number via `_FILE_NUMBER_BY_SLUG`, verified against +`raw/eambrosia-register/gi-index.json` (all six UK wines are also listed +in the EU register — the five pre-Brexit names under the Withdrawal +Agreement, Sussex added 2025-01-31 under the UK-EU agreement). + +| Script | Reads | Writes | +|---|---|---| +| gb/00_fetch_data.py | (network: GOV.UK search + content API, ONS Open Geography) | raw/gb/gov-uk/{index,manifest}.json, raw/gb/ons/{countries,counties}.geojson + manifest.json | +| gb/01_fetch_specs.py | raw/gb/gov-uk/index.json | raw/gb/specs/*.{pdf,docx} + manifest.json | +| gb/02_extract_specs.py | raw/gb/specs/*.{pdf,docx} | raw/gb/specs-extracted/*.json + _index.json + raw/gb/extraction-unknowns.json | +| gb/02d_extract_terroir_facts.py | raw/gb/specs-extracted/*.json + raw/wikipedia/aocs/en/ | raw/terroir-facts/*.json (country="gb") + manifest-gb.json | +| gb/02e_translate_terroir_facts.py | raw/terroir-facts/*.json (country="gb") | raw/translations/terroir-facts//*.json | +| gb/03_generate_wiki.py | raw/gb/specs-extracted/*.json | wiki/.md (per GB record) + merges GB entries into wiki/_index.json | +| audit_gb_coverage.py | raw/gb/{gov-uk,specs,specs-extracted}/ + raw/terroir-facts/ | (stdout — coverage table + curator queue) | + +GB-specific notes: + +- `kind` is `"DOP"` / `"IGP"`. The UK register spells these out in + English ("Protected Designation of Origin (PDO)"), but the corpus + convention is the DOP/IGP pair — and concretely, the map's + polygon-colour expression keys on `kind == 'IGP'`, so a literal + "PGI" would silently mis-colour both UK PGIs as PDOs. +- **Three parser templates**, because the six specifications were + written under three regimes + ([scripts/_lib/gb/spec.py](scripts/_lib/gb/spec.py)): + - **`defra-pfn-2011`** (4 records) — the December-2011 DEFRA + specifications. A `PROTECTED NAME:` / `DEMARCATION:` header block, + then one `PART n: ` block per grapevine category + (`STILL WINE`, `QUALITY SPARKLING WINE`, plus a closing `GENERAL + PROVISIONS` part that carries no wine). Each wine part opens with + the link narrative, then `SPECIFICATION` and a run of upper-case + subsections, of which `VINE VARIETIES` and `MAXIMUM YIELDS` are + read. These carry the corpus's longest single variety rosters — + 81-85 resolved slugs each. + - **`defra-pfn-application`** (Darnibole) — the 2017 EU application + form: a numbered outline (`1. Details of protection` … `7. + Demarcated area`) with lettered sub-items. It has **no link + section** — the document stops after the demarcated area and its + plan — so the terroir narrative is taken from `7 b) Definition of + the demarcated area`, which is where the slate subsoil, the slope + and the aspect are actually described. + - **`uk-gi-single-document`** (Sussex) — the only post-Brexit + UK-scheme registration, and the only `.docx`. A numbered EU-style + single document with a proper `9. Link` section (9.1 natural + human + factors, 9.2 characteristics, 9.3 causal link); varieties live in + `7.2 Viticulture practices`, listed separately for sparkling and + still. Parsed via the stdlib zip → `word/document.xml` route the HR + pipeline already uses. +- **Roster parsing.** The still-wine rosters are semicolon-separated and + wrap across PDF lines; the sparkling ones are bulleted one-per-line + (`pdftotext` renders the Wingdings bullet as U+F0B7); Sussex's are + comma-separated with a trailing "X and Y". `grape_candidates` groups + contiguous roster lines into *runs* and picks the join per run — a + space where the lines already carry separators (so "Black Hamburg; + Blau\nPortugueser" is not split into "Blau" + "Portugueser"), a + separator where they do not. Prose lines are classified out by the + shape of their split items, so Sussex's record-keeping and + yield-dispensation paragraphs never reach the matcher. +- **Two source typos are repaired structurally**, not via `GRAPE_ALIAS`, + because a stray comma and a line break split one variety name into two + before any alias could apply: Sussex's roster reads "… Pinot Noir, + Pinot Noir, Précoce, Regent …" for *Pinot Noir Précoce*, and the DEFRA + rosters drop the semicolon between *Gamay* and *Garanoir* (their own + alphabetical order — Gamaret, Gamay, Garanoir, Gewurztraminer — proves + these are two entries). +- **New lexicon entries.** The UK rosters spell several varieties in + forms no other country uses; VIVC ids verified 2026-09: Cascade + (#2139, noir), Madeleine Angevine (#7062, blanc), Madeleine Sylvaner + (#7070, blanc), Roter Veltliner (#12931 VELTLINER ROT, rose — a + *distinct* cultivar from Frühroter Veltliner, which the fuzzy matcher + scores at 88), Frühgipfler (#4269, blanc — the UK lists it under its + synonym *Senator*), Triomphe d'Alsace (#12650, noir), plus aliases for + "Blau Portugueser" (#9620) and "Elbling White" (#3865). *Gagarin Blue* + has no VIVC accession — a black Russian/Caucasus cultivar grown + outdoors in the UK, colour from the RHS plant register. +- Grape roles: no UK specification splits principal from accessory, so + every match resolves as `principal` (the PT/IT/HR/BG/SK convention). +- v1 models the 6 wine GIs as a **flat corpus**. Sussex and Darnibole + sit geographically inside the English PDO's territory but are + first-class PDOs on the register rather than sub-denominations of it — + Darnibole's own specification makes the point explicitly ("It + qualified under the 'English' PDO last year, but is considered unique + and sufficiently different so as to merit its own PDO") — so they are + siblings, the way the CZ podoblasti are siblings of Čechy / Morava. +- Region facet = the **home nation** the specification demarcates the GI + to (England / Wales), which is the register's own `DEMARCATION` tier + ([scripts/_lib/gb/region.py](scripts/_lib/gb/region.py)). Native form, + not gettext-translated. +- Stage 02d/02e wire terroir-fact extraction + translation for GB. Unlike + Malta — the other English-source corpus, whose amendment + communications carry no link section — every UK specification carries a + real terroir narrative, so GB uses the **standard cahier-primary** + dual-source model rather than the Wikipedia-primary CH/MT one. A + missing Wikipedia article is not disqualifying: the specification's own + link text grounds the record. 02e targets fr/es/nl. + +### GB geometry resolution chain (stage 04) + +Bétard 2022 carries no `PDO-GB-*` rows, so GB resolves entirely against +**ONS Open Geography** boundaries (Open Government Licence v3.0; "Source: +Office for National Statistics"; "Contains OS data © Crown copyright and +database right"). That is a better fit than Bétard would have been +anyway: every UK wine GI is demarcated to whole administrative units +named in its own specification. BGC ("generalised, clipped to the +coastline") is the generalisation fetched — full-resolution BFC is +several times larger with no visible gain at the zooms this corpus +renders. + +Per GB record, in priority order +([scripts/_lib/gb/geometry.py](scripts/_lib/gb/geometry.py); `geom_source` +records the choice): + +1. **`ons-country`** — the England / Wales whole-country polygon, for the + four national GIs (English + Welsh PDO, English + Welsh Regional PGI), + whose `DEMARCATION` field is literally `ENGLAND` / `WALES`. +2. **`ons-county-union`** — Sussex, whose specification demarcates "the + administrative boundaries of the counties of East and West Sussex". + Brighton and Hove is a unitary authority carved out of East Sussex in + 1997 and sits between the two, so the union of the three CTYUAs + reconstructs the ceremonial Sussex the specification's own map shows. +3. **`pdo-plan-parcel-hull-approx`** — Darnibole (see below). +4. **`stub-no-geometry`** — last resort; not hit in v1, all 6 resolve. + +### Darnibole — a boundary reconstructed from the specification's plan + +`PDO-GB-N1636` Darnibole is a single-vineyard PDO at Camel Valley, +Nanstallon (Cornwall). Its specification defines the area **narratively** +— "bordered to the West by a marked soil change to alluvial sand (the old +River Camel river bed) … The disused railway (now the Camel Trail) +demarcates the Southern boundary" — with no coordinates, and no public +polygon of it exists anywhere. + +What the specification *does* carry is a **"Plan of demarcated area"** +(page 5): an OS-based Rural Payments Agency land-parcel map with the PDO +boundary drawn in red over numbered field parcels, each labelled "Vines +planted" / "Not yet planted". Those parcel numbers are not arbitrary ids +— under the OS/RPA convention a field parcel is identified by the +**four-figure National Grid reference of its centroid within its 1 km +grid square** (two digits of easting then two of northing, each in units +of 10 m), so parcel `7985` sits at 790 m E, 850 m N inside its square. + +[scripts/_lib/gb/darnibole.py](scripts/_lib/gb/darnibole.py) reconstructs +an approximate boundary from that: the convex hull of the centroids of +the seven parcels the red line encloses. Anchoring the square (SX 03 67 → +BNG base E 203000 / N 67000) was verified three ways: + +1. **Internal consistency.** The decoded north→south and west→east order + reproduces the plan's layout exactly, including the 1 km grid line + visible on the plan between parcels 5107/8101 (N 68xxx) and 5095 + (N 67950). +2. **Terrain.** An elevation transect north from the block (EU-DEM 25 m) + puts the valley floor at 11–14 m over N 67300–67400 and climbs + steadily to 96 m by N 68100. The seven parcels sit at 47–80 m on a + ~13 % south-facing slope immediately above the old river bed — + precisely the specification's "steep south facing slope", bounded + south by the River Camel's old bed and north by land "above the + optimum thermal band". +3. **Address.** The ONS centroid of Camel Valley's postcode (PL30 5LG) is + E 203164 N 67751 — on the same slope, ~330 m west of the block. + +The hull spans E 203490–203890 / N 67680–67950 and measures **6.1 ha** +against the specification's declared "whole 5 hectare area" — the right +size in the right place, with no arbitrary buffer parameter. It remains a +reconstruction: it interpolates between parcel centroids rather than +tracing the red line, so its edges fall inside the true boundary by up to +roughly half a field. The record carries `geom_approximate: True`, and +the map panel discloses it ("Aire approchée — reconstituée à partir des +références parcellaires du plan de l'aire délimitée annexé au cahier des +charges ; ce n'est pas une limite officielle.") through the same +`approx-line` mechanism the FR cadastre-lieu-dit DGCs use. + +### Curator workflow for the UK + +There is no missing-document queue. The two things that can need +attention: + +1. **A new registration.** When a pending application is granted (The + Crouch Valley is the one in flight), re-running stage 00 picks it up + automatically — except for its file number, which must be added to + `_FILE_NUMBER_BY_SLUG` in + [scripts/gb/00_fetch_data.py](scripts/gb/00_fetch_data.py); stage 00 + warns loudly when a registered wine has none, because region and + geometry both key on it. A new PDO will also need a geometry entry in + `GB_COUNTRY_TERRITORY` / `GB_COUNTY_TERRITORY`. +2. **A rotated attachment URL.** GOV.UK asset URLs carry an opaque media + id; stage 00 re-reads them from the register on every run, so a + rotation is picked up by re-running 00 → 01 → 02. + +``` +.venv/bin/python scripts/gb/00_fetch_data.py +.venv/bin/python scripts/gb/01_fetch_specs.py +.venv/bin/python scripts/gb/02_extract_specs.py +.venv/bin/python scripts/02b_fetch_aoc_lexicon.py --lang en --source raw/gb/specs-extracted +.venv/bin/python scripts/gb/02d_extract_terroir_facts.py --batch --provider anthropic +.venv/bin/python scripts/gb/02e_translate_terroir_facts.py --batch --provider anthropic +.venv/bin/python scripts/gb/03_generate_wiki.py +.venv/bin/python scripts/04_build_maps.py +``` + ## Batch API (02b-grapes / 02c / 02d / 02e) The LLM stages — `02b_translate_grapes` (grape-tooltip translation), diff --git a/CURATOR_TODO.md b/CURATOR_TODO.md index 8d0eeb5..45521c2 100644 --- a/CURATOR_TODO.md +++ b/CURATOR_TODO.md @@ -2217,6 +2217,294 @@ still point at `moa.gov.cy`; the PDFs are cached, so nothing is broken today, but a `--refresh` would fail. The other 3 already moved to eAmbrosia attachments. +## United Kingdom + +Added 2026-09-06 (country #19). The UK is the corpus's cleanest register: +**6 / 6 registered wine GIs extract**, all with a public product +specification, all on the map. There is no missing-document queue. + +### Open — pending application + +| name | kind | applied | status | +|---|---|---|---| +| The Crouch Valley | PDO | 2023-03-06 | ❌ still in assessment on the GOV.UK register | + +When it is granted: re-run `scripts/gb/00_fetch_data.py` (it picks the +name up automatically), then add its file number to +`_FILE_NUMBER_BY_SLUG` in [scripts/gb/00_fetch_data.py](scripts/gb/00_fetch_data.py) +and a geometry entry to `GB_COUNTY_TERRITORY` in +[scripts/_lib/gb/geometry.py](scripts/_lib/gb/geometry.py) (the Crouch +Valley is in Essex). Stage 00 warns when a registered wine has no file +number, because region + geometry both key on it. + +### Open — Darnibole boundary (approximate by construction) + +`PDO-GB-N1636` renders as the convex hull of the seven parcel centroids +decoded from its specification's own plan — 6.1 ha against a declared +5 ha, correctly sited on the verified south-facing slope, and disclosed +in the panel as approximate. It interpolates between centroids rather +than tracing the red boundary line, so its edges sit inside the true +boundary by up to ~half a field. To improve it, a curator would need +either (a) the RPA/OS parcel polygons for those seven ids, or (b) a +georeferenced trace of the plan's red line. Neither is currently +available under a public licence. Full derivation + the three-way anchor +check: [scripts/_lib/gb/darnibole.py](scripts/_lib/gb/darnibole.py). + +### Curator inputs recorded 2026-09-06 (files under `raw/` are gitignored) + +VIVC pins added to `raw/vivc/slug_overrides.json` for the seven varieties the +UK rosters introduced (3 resolved automatically — cascade #2139, +roter-veltliner #12931, fruehgipfler #4269): + +``` +madeleine-angevine -> 7062 madeleine-sylvaner -> 7070 +triomphe-dalsace -> 12650 gagarin-blue -> false +``` + +`madeleine-angevine` and `madeleine-sylvaner` are ambiguous in VIVC's +cultivarname search (Oberlin / 4N forms, and GEILWEILERHOF 3-28-51 +respectively); `triomphe-dalsace` never auto-matches because the slug strips +the apostrophe from TRIOMPHE D'ALSACE, and VIVC also binds the bare surface +"Triomphe" to DODRELYABI #3616; `gagarin-blue` is a verified absence (no VIVC +accession — a black Russian/Caucasus cultivar documented by the RHS plant +register). + +Wikipedia grape-tooltip pin in `raw/wikipedia/grape_overrides.json`: +`triomphe-dalsace` -> "Triomphe d'Alsace" (en + fr), same apostrophe cause. +5 of the 7 now have tooltips; `madeleine-sylvaner` and `gagarin-blue` have no +article in any locale (verified) and are deliberately left as misses. + +Wikidata suppressions in `raw/wikidata/slug_overrides.json`: `english-wine` +and `welsh-wine`. en.wikipedia redirects both "English wine" and "Welsh wine" +to the umbrella article *Wine from the United Kingdom* (Q1467810), so the +sitelink path resolved BOTH PDOs to that one QID. Q1467810 is the topic, not +either PDO, and `sameAs` asserts identity — so it is suppressed until Wikidata +has per-PDO items. Sussex keeps its own Q39056976 ("Sussex wine"). + +Geometry-outlier whitelist (checked in, `scripts/_lib/geometry_outlier_overrides.json`): +all four national GB records, whose detached parts are the Isles of Scilly +(England) and the Anglesey islands (Wales) in the ONS country polygons. + +### Open — "Findling" binds to Bouvier via VIVC + +The English + Welsh rosters list *Findling*, which `match_variety` +resolves to `bouvier` on an **exact** VIVC synonym (VIVC #1625 BOUVIER +carries FINDLING). In a UK context Findling is far more likely the +Müller-Thurgau seedling grown in England. VIVC is the project's taxonomy +authority so the current binding stands, but it wants a curator ruling — +and, if VIVC is wrong for the UK reading, a `GRAPE_ALIAS` pin. + +--- + +## Italy — two Wikipedia articles were bound to the wrong appellation ✅ fixed + +Found 2026-09-06 by grouping the emitted JSON-LD `sameAs` links (see the +cross-country section below). The `02b_fetch_aoc_lexicon` title cascade bound +two IT records to a *different* appellation's article on name similarity: + +| slug | appellation | was bound to | distance | +|---|---|---|---| +| `tarantino` | IGP Tarantino (Puglia) | `Trentino (vino)` — the Trentino DOC | ~900 km north | +| `rotae` | IGT Rotae (Molise) | `Roma (vino)` — the Roma DOC | different region (Lazio) | + +This was not only an SEO/identity problem. `rotae` had shipped a +**wiki-provenance terroir bullet about the wrong appellation** — "La DOC Roma +è stata approvata con DM 02.08.2011; la versione vigente del disciplinare +risale al DM 07.03.2014." — translated into all four locales. It passed the +≥ 0.6 fuzzy-coverage filter precisely *because* it is a faithful verbatim +quote; the filter checks that a bullet is grounded in its source, not that the +source is the right document. `tarantino` escaped content contamination (all 5 +of its bullets were `cahier`-provenance) but carried the wrong `sameAs`. + +Fixed: both pinned `{"missing": true}` in `raw/wikipedia/aoc_overrides.json` +under `it` with the reason, their cached article files replaced with a +`missing` marker recording the suppressed title, then IT 02d + 02e re-run for +the two slugs. Both are now 100 % `cahier`-grounded with `wiki_source_url: +null` (tarantino 6 facts, rotae 7). + +**The general lesson**: a `wiki`-provenance bullet is only as trustworthy as +the article-title match, and nothing downstream re-checks that match. A cheap +standing guard is to group the corpus by bound article title and look at any +title claimed by more than one appellation — which is exactly how these two +surfaced. Worth running after each `02b_fetch_aoc_lexicon` sweep. + +--- + +## Cross-country — grape pills show another country's spelling ✅ fixed 2026-09-06 + +Found 2026-09-06 from a GB spot-check: the English PDO's pill reads **"Optima +113"**, but the UK product specification says plain "Optima" (and the GB +record stores it correctly). "Optima 113" is Germany's official +Bundessortenamt name, from the Geilweilerhof breeding selection 33-13-113 — +a real name for the variety, just not the British one. + +Root cause is one guard in `emit_html` (`scripts/04_build_maps.py`, the +`grape_names` loop): + +```python +if s_slug and s_name and s_name.lower() != s_slug: + grape_names[s_slug] = s_name +``` + +The comment directly above it states the intent — *"drives the pill label so +the rendered name matches what the regulator actually published"* — but +`s_name.lower() != s_slug` drops the record's own spelling precisely when it +is already clean ("Optima".lower() == "optima"), presumably as a payload-size +saving on the assumption the client can re-derive it. The client does not +re-derive it from the slug: it falls back to the corpus-wide `GRAPES_INFO` +display name, which is the **most frequent spelling across all countries** — +frequently another language's. + +Blast radius (whole corpus, 50,757 (record, grape) pairs): **6,792 displaced +labels, of which 1,047 are substantive** — a different word or number, not +just casing. Worst offenders: + +| records | record's own name | label actually shown | +|---:|---|---| +| 97 | Müller Thurgau / müller-thurgau | **Rizlingszilváni** (Hungarian) | +| 76 | muscat à petits grains | muscat à petits grains blancs | +| 71 | Alicante Bouschet | alicante henri bouschet | +| 35 | albariño (ES) | **alvarinho** (Portuguese) | +| 30 | godello (ES) | **Gouveio** (Portuguese) | +| 18 | Blauer Portugieser (DE/AT) | **Kékoportó** (Hungarian) | +| 17 | plantet (FR) | seibel 5455 | +| 4 | Optima (GB) | Optima 113 (German) | + +By country: it=228, fr=220, de=149, es=94, ro=67, pt=67, hr=48, hu=31, ch=26, +sk=23. So German and Austrian pages label Müller-Thurgau with its Hungarian +name, and Spanish pages label Albariño and Godello with their Portuguese ones +— directly contradicting the stated rule in CLAUDE.md that the pill shows +"the cahier's spelling ... verbatim with the VIVC prime name in brackets when +distinct". + +**Fixed 2026-09-06**: the guard was removed so `emit_html` carries the +record's own spelling unconditionally, with a comment recording why it must +not be re-added as an optimisation. Verified in the rendered panel — the +English PDO's pills now read *Optima*, *Regent*, *Kerner*, *Albarino +(Alvarinho)* instead of *Optima 113*, *Regent N.*, *Kerner B.*, *Alvarinho*; +Mosel keeps its own *Optima 113* and *Müller Thurgau*, Valdeorras its +*godello*. Each country now shows its regulator's spelling with the VIVC +canonical in brackets when distinct, which is what CLAUDE.md specifies. + +Cost: `grape_names` rides the lazily-fetched panel payload, not the startup +bundle. Panel payloads total 42.4 MB over 11,656 files (mean 3.6 KB); the +startup bundle is unchanged at 3.51 MB. Per-record name counts rose as +expected (english-wine 30→81, mosel 64→126, sussex 10→28). + +--- + +## Cross-country — one Wikidata QID claimed by several appellations (pre-existing) + +Found 2026-09-06 while wiring GB's `sameAs`. **18 QIDs are currently claimed +by 50 appellation records**, and — counting only the 1,659 pages that actually +emit JSON-LD — **15 Wikipedia articles are claimed as `sameAs` by 44 of +them**. Either way it is an identity error in the JSON-LD: +schema.org `sameAs` asserts *this page is about that entity*, so two +appellations cannot legitimately share one QID. Group +`raw/wikidata/qids-by-slug.json` by `qid` to reproduce. + +Two distinct causes, needing different fixes: + +- **Umbrella article shared by a whole country's corpus** — clearly wrong. + `Q582745` "Maltese wine" is claimed by all three MT records (malta, gozo, + maltese-islands); `Q9198993` "Vinarska oblast Cechy" by cz cechy + ceske. + GB hit exactly this and is already suppressed (see the United Kingdom + section). +- **Parent article inherited by sub-denominations / sibling names** — + `Q1067753` Vinho Verde across 10 records and `Q191034` Porto across 4 (both + via the P9854 eAmbrosia join, which returns the parent's GI), plus + `Q1058259` chablis + petit-chablis, `Q21427011` the three Calvados records, + `Q551484` the three Anjou records, `Q3288502` the two Marc d'Alsace. + +The suppression mechanism already exists (`raw/wikidata/slug_overrides.json`, +`{suppress: true}`), so the QID half of the fix is curation, not code — but it +touches 6 countries, so it wants its own pass rather than riding a country +addition. + +**The Wikipedia half is not yet fixable by curation.** `_entity_same_as` in +`scripts/_lib/map_template.py` takes its Wikipedia URL from +`terroir_facts.wiki_source_url`, which no override reaches, so suppressing a +QID leaves the shared article link in place — GB's two PDOs still both point +at *Wine from the United Kingdom*. The worst instance is HR: 14 records all +claim `Vinogradarska područja Republike Hrvatske`. + +Suggested fix, self-maintaining and covering all 44 at once: drop a Wikipedia +`sameAs` whenever the same article URL is claimed by more than one *indexable* +record. An article claimed by two appellations cannot identify either, so the +invariant is sound without per-record curation. It changes output for ~44 +records across several countries, so it belongs in its own change with a +before/after diff. + +--- + +## Cross-country — audit_terroir_facts covers only FR / ES / GB (pre-existing) + +`scripts/audit_terroir_facts.py` re-derives fuzzy coverage from each +country's own source documents, so it needs a per-country dispatch entry +(extracted dir, wiki cache dir, lien field, Wikipedia heading map, hint +char-cap, and the right hint *builder* — FR-style vs ES-style). Only `fr` and +`es` were ever wired; every country added since (PT, IT, AT, SI, HR, HU, RO, +BG, GR, DE, SK, CH, CZ, LU, BE, NL, MT, CY) was silently skipped — a +`KeyError` swallowed by the loop's broad `except` and printed as +`err : ''`, indistinguishable from a corrupt cache. + +The 2026-09-06 GB pass wired `gb` in and made the skip explicit: unsupported +countries are now counted and reported instead of masquerading as errors. The +full-corpus run quantifies the gap — **1,030 fact-carrying records across 18 +countries are unaudited**: + +``` +[skipped] countries with no source dispatch entry: at=30, be=10, bg=54, +ch=3, cy=11, cz=13, de=39, gr=147, hr=18, hu=41, it=522, lu=1, mt=3, +nl=21, pt=44, ro=46, si=17, sk=10 +``` + +against 4,470 bullets actually audited (FR + ES + GB). Italy alone is 522 +records, so it is the highest-value single entry to wire. Wiring the remaining 18 is a bounded, mechanical +job — each needs the 5 dispatch entries above, and getting the hint builder +or char-cap wrong produces *false* drift / erosion flags (both were hit while +wiring GB, and both are documented inline there), so each country's entry +should be validated against a record with a known-good `wiki`-provenance +bullet. + +--- + +## Cross-country — two slugs for one VIVC variety (pre-existing; surfaced by GB) + +Not introduced by the UK pipeline — GB is simply the first country whose +*single* variety roster names both spellings, which puts the same grape on +one panel twice. + +**`blaufrankisch` and `lemberger` are both VIVC #1459 BLAUFRAENKISCH.** +The project's own cache agrees: `raw/vivc/by-slug/blaufrankisch.json` and +`raw/vivc/by-slug/lemberger.json` both read `vivc_id: 1459`, and each +lists the other as a synonym. The split is an alias-chain artefact — +`kekfrankos → blaufrankisch` and `limberger → lemberger` are both +one-hop, and nothing folds the two heads together. + +Usage across the corpus (229 records, 11 countries): + +| slug | records | countries | +|---|---:|---| +| `blaufrankisch` | 124 | hu 72, hr 24, sk 14, cz 10, gb 4 | +| `lemberger` | 105 | ro 43, si 21, de 15, at 9, hu 8, gb 4, be 2, pt 2, es 1 | + +Folding them is the established policy (commit `a7570f2` "fold +VIVC-collision slug dupes … unify facet by variety"), but it moves the +grape facet for 11 countries, so it belongs in its own change with its +own before/after diff — not in a country addition. + +**`csabagyongye` / `zalagyongye` are mis-named.** The slug +`csabagyongye` is bound to VIVC #13374, whose prime name is **ZALA +GYOENGYE** — so the *binding* is right for the surface "Zala gyöngye" +but the slug name says Csaba. Meanwhile `zalagyongye` exists as a +separate slug with no VIVC binding at all (7 HU records), and +`perle-von-zala → csabagyongye` is commented "alternate German name for +Csabagyöngye" when Perle von Zala is the German name of *Zala*gyöngye. +Needs a curator pass over the pair; renaming a slug ripples into the +VIVC cache, the wiki pages and the translations, so likewise its own +change. + ## Cross-country — eAmbrosia register attachment endpoint (spike ✅; CZ + SI live; Phase-2 retrofit planned) The EU GI register public API diff --git a/VERIFICATION.md b/VERIFICATION.md index dac523c..81389dc 100644 --- a/VERIFICATION.md +++ b/VERIFICATION.md @@ -816,3 +816,71 @@ grapes; 8 with terroir facts. .venv/bin/python scripts/audit_cy_coverage.py # 11 wines, 7 figshare + 4 district-union, 11/11 grapes # Independent check: eAmbrosia register UI, filter country=Cyprus + Wine. ``` + +## United Kingdom (GB) — 2026-09-06 + +**Independent cross-check** of the corpus count against the regulator's own +register, which for the UK is *not* eAmbrosia: since Brexit the authority is +the **GOV.UK "protected food and drink names" register** run by DEFRA. + +- Source of truth queried directly (not via the cached index): + `https://www.gov.uk/api/search.json?filter_format=protected_food_drink_name&filter_register=wines&filter_country_of_origin=united-kingdom` + → **7 UK wine entries**, of which **6 `status=registered`** and 1 + `status=applied-for` (*The Crouch Valley*, PDO, applied 2023-03-06 — still + in assessment, correctly excluded from the corpus). +- Corpus: **6 wines — 4 PDO** (`PDO-GB-A1585` English, `PDO-GB-A1587` Welsh, + `PDO-GB-02365` Sussex, `PDO-GB-N1636` Darnibole) **+ 2 PGI** + (`PGI-GB-A1589` English Regional, `PGI-GB-A1590` Welsh Regional). + Matches WineGB's public PDO/PGI page. +- File numbers cross-checked against `raw/eambrosia-register/gi-index.json` + (countryId=gb, qualityProductType=Wine): all 6 present there too — the + five pre-Brexit names retained under the Withdrawal Agreement, Sussex + added 2025-01-31 under the UK-EU agreement. The GOV.UK register itself + publishes no file number, hence the bridge in stage 00. +- **Specification coverage: 6/6.** Every registered wine ships a public + product specification from `assets.publishing.service.gov.uk` (5 PDF + + Sussex's .docx). No stub tier, no national-spec fallback layer, no WAF. +- Grapes: **6/6 with a resolved roster** — English/Welsh PDO 81 each, + the two Regional PGIs 85 each, Sussex 28, Darnibole 1 (100% Bacchus, + as its specification states). Spot-check: the Sussex sparkling roster + (Chardonnay, Pinot Noir, Pinot Meunier, Arbanne, Pinot Gris, Pinot + Blanc, Petit Meslier, Pinot Noir Précoce) matches the specification's + §7.2 verbatim. +- Geometry: **6/6 on the map**, all from ONS Open Geography (OGL v3.0) — + Bétard 2022 is an EU PDO layer and has no `PDO-GB-*` rows. + Areas cross-checked against published administrative figures: + + | record | resolved km² | published km² | + |---|---:|---:| + | English / English Regional | 130 522 | 130 279 (England) | + | Welsh / Welsh Regional | 20 790 | 20 779 (Wales) | + | Sussex | 3 788 | 3 782 (E. Sussex 1 709 + W. Sussex 1 991 + Brighton & Hove 82) | + | Darnibole | 0.061 | "whole 5 hectare area" (its specification) | + +- **Darnibole placement check** (its boundary is reconstructed — see + CLAUDE.md): the resolved centroid sits 502 m from the ONS centroid of + Camel Valley's postcode PL30 5LG and 918 m from Nanstallon village, and + falls inside the English PDO polygon. An EU-DEM 25 m transect confirms + the parcels sit at 47–80 m on a ~13 % south-facing slope with the valley + floor at 11–14 m immediately south — matching the specification's "steep + south facing slope" bounded by the River Camel's old bed. +- Terroir-fact bullets (02d, country=gb, English source → fr/es/nl): + **51 bullets across 6/6 wines** (5–10 each), Anthropic batch; + 49 of 51 grounded in the specification itself (`provenance=cahier`), + which is the expected shape for a cahier-primary country. + +**Re-run recipe**: + +``` +.venv/bin/python scripts/gb/00_fetch_data.py # 6 registered (+1 pending), ONS boundaries +.venv/bin/python scripts/gb/01_fetch_specs.py # 6 specs (5 pdf + 1 docx) +.venv/bin/python scripts/gb/02_extract_specs.py # 6/6 extracted, 0 stubs +.venv/bin/python scripts/02b_fetch_aoc_lexicon.py --lang en --source raw/gb/specs-extracted +.venv/bin/python scripts/gb/02d_extract_terroir_facts.py --batch --provider anthropic +.venv/bin/python scripts/gb/02e_translate_terroir_facts.py --batch --provider anthropic +.venv/bin/python scripts/gb/03_generate_wiki.py +.venv/bin/python scripts/04_build_maps.py +.venv/bin/python scripts/audit_gb_coverage.py # 6 wines, 0 problems +# Independent check: https://www.gov.uk/protected-food-drink-names +# → filter Register = "Wines", Country of origin = "United Kingdom". +``` diff --git a/raw/wikipedia/aoc_overrides.json b/raw/wikipedia/aoc_overrides.json index e2adef8..4ec1cc8 100644 --- a/raw/wikipedia/aoc_overrides.json +++ b/raw/wikipedia/aoc_overrides.json @@ -219,35 +219,35 @@ }, "cs": { "cechy": { - "wiki_title": "Vinařská oblast Čechy", "page_url": "https://cs.wikipedia.org/wiki/Vina%C5%99sk%C3%A1_oblast_%C4%8Cechy", - "verification_quote": "Vinařská oblast Čechy je vinařská oblast, která obsahuje schválená území pro pěstování révy vinné." + "verification_quote": "Vinařská oblast Čechy je vinařská oblast, která obsahuje schválená území pro pěstování révy vinné.", + "wiki_title": "Vinařská oblast Čechy" }, "ceske": { - "wiki_title": "Vinařská oblast Čechy", + "note": "pinned 2026-05-31; české zemské víno (PGI) is coextensive with the Čechy wine region — reuse the region article for salience.", "page_url": "https://cs.wikipedia.org/wiki/Vina%C5%99sk%C3%A1_oblast_%C4%8Cechy", "verification_quote": "Vinařská oblast Čechy je vinařská oblast, která obsahuje schválená území pro pěstování révy vinné.", - "note": "pinned 2026-05-31; české zemské víno (PGI) is coextensive with the Čechy wine region — reuse the region article for salience." + "wiki_title": "Vinařská oblast Čechy" }, "morava": { - "wiki_title": "Vinařská oblast Morava", "page_url": "https://cs.wikipedia.org/wiki/Vina%C5%99sk%C3%A1_oblast_Morava", - "verification_quote": "Vinařská oblast Morava je vinařská oblast, která zahrnuje schválená území pro pěstování révy vinné na území Moravy." + "verification_quote": "Vinařská oblast Morava je vinařská oblast, která zahrnuje schválená území pro pěstování révy vinné na území Moravy.", + "wiki_title": "Vinařská oblast Morava" }, "novosedelske-slamove-vino": { "missing": true, "note": "researched 2026-05; no dedicated article; cs.wiki has pages for various Novosedly settlements but none about this single-vineyard straw-wine PDO" }, - "znojmo": { - "wiki_title": "Znojemská vinařská podoblast", + "znojemska": { + "note": "pinned 2026-05-31; the bare-name cascade hit the festival article 'Znojemské vinobraní' — wrong topic.", "page_url": "https://cs.wikipedia.org/wiki/Znojemsk%C3%A1_vina%C5%99sk%C3%A1_podoblast", - "verification_quote": "Znojemská vinařská podoblast leží ve vinařské oblasti Morava. Pokrývá vinařské obce okresů Znojmo a Třebíč." + "verification_quote": "Znojemská vinařská podoblast leží ve vinařské oblasti Morava. Pokrývá vinařské obce okresů Znojmo a Třebíč.", + "wiki_title": "Znojemská vinařská podoblast" }, - "znojemska": { - "wiki_title": "Znojemská vinařská podoblast", + "znojmo": { "page_url": "https://cs.wikipedia.org/wiki/Znojemsk%C3%A1_vina%C5%99sk%C3%A1_podoblast", "verification_quote": "Znojemská vinařská podoblast leží ve vinařské oblasti Morava. Pokrývá vinařské obce okresů Znojmo a Třebíč.", - "note": "pinned 2026-05-31; the bare-name cascade hit the festival article 'Znojemské vinobraní' — wrong topic." + "wiki_title": "Znojemská vinařská podoblast" } }, "de": { @@ -440,14 +440,14 @@ "note": "researched 2026-05; no dedicated Rosalia DAC article" }, "saarlandischer-landwein": { - "wiki_title": "Saarländischer Landwein", "page_url": "https://de.wikipedia.org/wiki/Saarl%C3%A4ndischer_Landwein", - "verification_quote": "Saarländischer Landwein wird einerseits auf 124 ha an der saarländischen Obermosel, die zum Bereich Moseltor im bestimmten Anbaugebiet Mosel gehören, andererseits auf knapp 17 ha Anbaufläche für Landwein an Südhängen an der Saar, der Nied, der Blies und anderen Orten angebaut." + "verification_quote": "Saarländischer Landwein wird einerseits auf 124 ha an der saarländischen Obermosel, die zum Bereich Moseltor im bestimmten Anbaugebiet Mosel gehören, andererseits auf knapp 17 ha Anbaufläche für Landwein an Südhängen an der Saar, der Nied, der Blies und anderen Orten angebaut.", + "wiki_title": "Saarländischer Landwein" }, "sachsischer-landwein": { - "wiki_title": "Sächsischer Landwein", "page_url": "https://de.wikipedia.org/wiki/S%C3%A4chsischer_Landwein", - "verification_quote": "Sächsischer Landwein gehört zu den deutschen Landweinen, die eine gehobene Stufe des „Deutschen Weins\" ohne Herkunftsbezeichnung darstellen und sich durch „gebietstypischen Charakter und ländliche Namensgebung\" auszeichnen." + "verification_quote": "Sächsischer Landwein gehört zu den deutschen Landweinen, die eine gehobene Stufe des „Deutschen Weins\" ohne Herkunftsbezeichnung darstellen und sich durch „gebietstypischen Charakter und ländliche Namensgebung\" auszeichnen.", + "wiki_title": "Sächsischer Landwein" }, "salzburg": { "missing": true, @@ -458,9 +458,9 @@ "note": "researched 2026-05; no dedicated wine article on de.wikipedia.org" }, "schwabischer-landwein": { - "wiki_title": "Schwäbischer Landwein", "page_url": "https://de.wikipedia.org/wiki/Schw%C3%A4bischer_Landwein", - "verification_quote": "Schwäbischer Landwein gehört zu den Deutschen Landweinen, die eine gehobene Stufe des Weins (früher: „Tafelwein\") darstellen und sich durch „gebietstypischen Charakter und ländliche Namensgebung\" auszeichnen." + "verification_quote": "Schwäbischer Landwein gehört zu den Deutschen Landweinen, die eine gehobene Stufe des Weins (früher: „Tafelwein\") darstellen und sich durch „gebietstypischen Charakter und ländliche Namensgebung\" auszeichnen.", + "wiki_title": "Schwäbischer Landwein" }, "schwyz": { "missing": true, @@ -555,14 +555,14 @@ "note": "researched 2026-05; Weststeiermark article is the geographic region west of the Mur, wine is one aspect among many" }, "wien": { - "wiki_title": "Weinbau in Wien", "page_url": "https://de.wikipedia.org/wiki/Weinbau_in_Wien", - "verification_quote": "Der Weinbau in Wien wird auf einer Fläche von 588 Hektar des Wiener Stadtgebietes betrieben." + "verification_quote": "Der Weinbau in Wien wird auf einer Fläche von 588 Hektar des Wiener Stadtgebietes betrieben.", + "wiki_title": "Weinbau in Wien" }, "wiener-gemischter-satz": { - "wiki_title": "Gemischter Satz", "page_url": "https://de.wikipedia.org/wiki/Gemischter_Satz", - "verification_quote": "Der Gemischte Satz ist die Bezeichnung für den Anbau von Wein, der aus unterschiedlichen Rebsorten in einem Weingarten besteht, sowie dann des daraus hergestellten Weins." + "verification_quote": "Der Gemischte Satz ist die Bezeichnung für den Anbau von Wein, der aus unterschiedlichen Rebsorten in einem Weingarten besteht, sowie dann des daraus hergestellten Weins.", + "wiki_title": "Gemischter Satz" }, "zug": { "missing": true, @@ -872,9 +872,9 @@ "note": "researched 2026-05 (bulk); GR IGP grandfathered VdP names typically lack dedicated el.wiki articles — pattern confirmed by BG 52/52 NONE result + GR PDO 27/31 NONE rate. Curator todo: per-slug verification on a future pass." }, "monemvasia-malvasia": { - "wiki_title": "Μαλβαζία", "page_url": "https://el.wikipedia.org/wiki/%CE%9C%CE%B1%CE%BB%CE%B2%CE%B1%CE%B6%CE%AF%CE%B1", - "verification_quote": "Μαλβαζία (ή Μαλβασία) είναι ένα γλυκό, λιαστό κρασί που παράγεται στην περιοχή της Μονεμβασιάς, προστατευόμενης ονομασίας προέλευσης (Π.Ο.Π.)." + "verification_quote": "Μαλβαζία (ή Μαλβασία) είναι ένα γλυκό, λιαστό κρασί που παράγεται στην περιοχή της Μονεμβασιάς, προστατευόμενης ονομασίας προέλευσης (Π.Ο.Π.).", + "wiki_title": "Μαλβαζία" }, "moschato-kefallinias": { "missing": true, @@ -905,9 +905,9 @@ "note": "researched 2026-05 (bulk); GR IGP grandfathered VdP names typically lack dedicated el.wiki articles — pattern confirmed by BG 52/52 NONE result + GR PDO 27/31 NONE rate. Curator todo: per-slug verification on a future pass." }, "nemea": { - "wiki_title": "Κρασί Νεμέας", "page_url": "https://el.wikipedia.org/wiki/%CE%9A%CF%81%CE%B1%CF%83%CE%AF_%CE%9D%CE%B5%CE%BC%CE%AD%CE%B1%CF%82", - "verification_quote": "Το κρασί Νεμέας (Ονομασίας Προέλευσης Ανωτέρας Ποιότητας) παράγεται στην ευρύτερη περιοχή της Νεμέας, η οποία αποτελεί την μεγαλύτερη αμπελουργική ζώνη της Ελλάδας." + "verification_quote": "Το κρασί Νεμέας (Ονομασίας Προέλευσης Ανωτέρας Ποιότητας) παράγεται στην ευρύτερη περιοχή της Νεμέας, η οποία αποτελεί την μεγαλύτερη αμπελουργική ζώνη της Ελλάδας.", + "wiki_title": "Κρασί Νεμέας" }, "opountia-lokridas": { "missing": true, @@ -1070,18 +1070,18 @@ "note": "researched 2026-05 (bulk); GR IGP grandfathered VdP names typically lack dedicated el.wiki articles — pattern confirmed by BG 52/52 NONE result + GR PDO 27/31 NONE rate. Curator todo: per-slug verification on a future pass." }, "robola-kefallinias": { - "wiki_title": "Ρομπόλα Κεφαλονιάς", "page_url": "https://el.wikipedia.org/wiki/%CE%A1%CE%BF%CE%BC%CF%80%CF%8C%CE%BB%CE%B1_%CE%9A%CE%B5%CF%86%CE%B1%CE%BB%CE%BF%CE%BD%CE%B9%CE%AC%CF%82", - "verification_quote": "Η Ρομπόλα Κεφαλονιάς είναι λευκό κρασί Προστατευόμενης Ονομασίας Προέλευσης (ΠΟΠ) που παράγεται αποκλειστικά στο νησί της Κεφαλονιάς." + "verification_quote": "Η Ρομπόλα Κεφαλονιάς είναι λευκό κρασί Προστατευόμενης Ονομασίας Προέλευσης (ΠΟΠ) που παράγεται αποκλειστικά στο νησί της Κεφαλονιάς.", + "wiki_title": "Ρομπόλα Κεφαλονιάς" }, "rodos": { "missing": true, "note": "researched 2026-05; Ρόδος is the island article; no dedicated Rhodes wine PDO article" }, "santorini": { - "wiki_title": "Βινσάντο Σαντορίνης", "page_url": "https://el.wikipedia.org/wiki/%CE%92%CE%B9%CE%BD%CF%83%CE%AC%CE%BD%CF%84%CE%BF_%CE%A3%CE%B1%CE%BD%CF%84%CE%BF%CF%81%CE%AF%CE%BD%CE%B7%CF%82", - "verification_quote": "Το βινσάντο της Σαντορίνης είναι παραδοσιακό γλυκό κρασί με σκούρο μπρούτζινο χρώμα, το οποίο ανήκει στην κατηγορία των κρασιών Π.Ο.Π Σαντορίνης. [Note: el.wiki has only the Vinsanto sub-style article; the broader Santorini PDO covers Assyrtiko whites too]" + "verification_quote": "Το βινσάντο της Σαντορίνης είναι παραδοσιακό γλυκό κρασί με σκούρο μπρούτζινο χρώμα, το οποίο ανήκει στην κατηγορία των κρασιών Π.Ο.Π Σαντορίνης. [Note: el.wiki has only the Vinsanto sub-style article; the broader Santorini PDO covers Assyrtiko whites too]", + "wiki_title": "Βινσάντο Σαντορίνης" }, "serres": { "missing": true, @@ -1156,6 +1156,26 @@ "note": "researched 2026-05 (bulk); GR IGP grandfathered VdP names typically lack dedicated el.wiki articles — pattern confirmed by BG 52/52 NONE result + GR PDO 27/31 NONE rate. Curator todo: per-slug verification on a future pass." } }, + "en": { + "gozo": { + "note": "Malta has no per-PDO/PGI en.wikipedia article; the umbrella Maltese-wine article grounds terroir facts (the CH/LU pattern).", + "page_url": "https://en.wikipedia.org/wiki/Maltese_wine", + "verification_quote": "Maltese wine dates back over two thousand years to the time of the Phoenicians.", + "wiki_title": "Maltese wine" + }, + "malta": { + "note": "Malta has no per-PDO/PGI en.wikipedia article; the umbrella Maltese-wine article grounds terroir facts (the CH/LU pattern).", + "page_url": "https://en.wikipedia.org/wiki/Maltese_wine", + "verification_quote": "Maltese wine dates back over two thousand years to the time of the Phoenicians.", + "wiki_title": "Maltese wine" + }, + "maltese-islands": { + "note": "Malta has no per-PDO/PGI en.wikipedia article; the umbrella Maltese-wine article grounds terroir facts (the CH/LU pattern).", + "page_url": "https://en.wikipedia.org/wiki/Maltese_wine", + "verification_quote": "Maltese wine dates back over two thousand years to the time of the Phoenicians.", + "wiki_title": "Maltese wine" + } + }, "es": { "3-riberas": { "page_url": "https://es.wikipedia.org/wiki/Tres_Riberas", @@ -1287,11 +1307,6 @@ "missing": true, "note": "researched 2026-05; no dedicated wine article on fr.wikipedia.org" }, - "moselle-luxembourgeoise": { - "wiki_title": "Viticulture au Luxembourg", - "page_url": "https://fr.wikipedia.org/wiki/Viticulture_au_Luxembourg", - "verification_quote": "Sa production se concentre principalement sur la Moselle luxembourgeoise" - }, "alsace-grand-cru-altenberg-de-bergheim": { "page_url": "https://fr.wikipedia.org/wiki/Altenberg-de-bergheim", "verification_quote": "Un alsace grand cru Altenberg de Bergheim, ou plus simplement un altenberg-de-bergheim, est un vin blanc français produit sur le lieu-dit Altenberg et plusieurs lieux-dits l'avoisinant, situés sur la commune de Bergheim, dans le département du Haut-Rhin, dans la collectivité européenne d'Alsace, au sein de la région Grand Est.", @@ -1590,9 +1605,9 @@ "note": "Searched Cidre_du_Perche and 'cidre du perche' via Google site: — no dedicated fr.wikipedia.org page; only general Cidre and Perche (région naturelle) articles." }, "cite-de-carcassonne": { - "wiki_title": "Cité-de-carcassonne", "page_url": "https://fr.wikipedia.org/wiki/Cit%C3%A9-de-carcassonne", - "verification_quote": "Un cité-de-carcassonne, anciennement dénommé vin de pays des coteaux de la cité de Carcassonne de 1974 jusqu'en 2009, est un vin français qui fait partie depuis 2025 de l'indication géographique protégée." + "verification_quote": "Un cité-de-carcassonne, anciennement dénommé vin de pays des coteaux de la cité de Carcassonne de 1974 jusqu'en 2009, est un vin français qui fait partie depuis 2025 de l'indication géographique protégée.", + "wiki_title": "Cité-de-carcassonne" }, "clos-de-vougeot-ou-clos-vougeot": { "page_url": "https://fr.wikipedia.org/wiki/Clos-de-vougeot", @@ -1605,14 +1620,14 @@ "wiki_title": "Cognac (eau-de-vie)" }, "collines-rhodaniennes": { - "wiki_title": "Collines-rhodaniennes", "page_url": "https://fr.wikipedia.org/wiki/Collines-rhodaniennes", - "verification_quote": "Un collines-rhodaniennes, appelé vin de pays des collines rhodaniennes jusqu'en 2009, est un vin français d'indication géographique protégée." + "verification_quote": "Un collines-rhodaniennes, appelé vin de pays des collines rhodaniennes jusqu'en 2009, est un vin français d'indication géographique protégée.", + "wiki_title": "Collines-rhodaniennes" }, "comte-tolosan": { - "wiki_title": "Comté-tolosan", "page_url": "https://fr.wikipedia.org/wiki/Comt%C3%A9-tolosan", - "verification_quote": "Un comté-tolosan, anciennement appelé vin de pays du comté tolosan de 1982 à 2009, est un vin français d'indication géographique protégée." + "verification_quote": "Un comté-tolosan, anciennement appelé vin de pays du comté tolosan de 1982 à 2009, est un vin français d'indication géographique protégée.", + "wiki_title": "Comté-tolosan" }, "cote-de-nuits-villages-ou-vins-fins-de-la-cote-de-nuits": { "page_url": "https://fr.wikipedia.org/wiki/C%C3%B4te-de-nuits-villages", @@ -1620,9 +1635,9 @@ "wiki_title": "Côte-de-nuits-villages" }, "cote-vermeille": { - "wiki_title": "Côte-vermeille (IGP)", "page_url": "https://fr.wikipedia.org/wiki/C%C3%B4te-vermeille_(IGP)", - "verification_quote": "Le côte-vermeille est un vin français d'indication géographique protégée destiné à labelliser des vins ne pouvant prétendre à une appellation d'origine." + "verification_quote": "Le côte-vermeille est un vin français d'indication géographique protégée destiné à labelliser des vins ne pouvant prétendre à une appellation d'origine.", + "wiki_title": "Côte-vermeille (IGP)" }, "coteaux-bourguignons-ou-bourgogne-grand-ordinaire-ou-bourgogne-ordinaire": { "page_url": "https://fr.wikipedia.org/wiki/Coteaux-bourguignons", @@ -1695,14 +1710,14 @@ "wiki_title": "Côtes-de-bourg" }, "cotes-de-la-charite": { - "wiki_title": "Côtes-de-la-charité", "page_url": "https://fr.wikipedia.org/wiki/C%C3%B4tes-de-la-charit%C3%A9", - "verification_quote": "Un côtes-de-la-charité, anciennement vin de pays des coteaux charitois, est un vignoble français d'indication géographique protégée de zone, produit dans le département de la Nièvre autour de La Charité-sur-Loire." + "verification_quote": "Un côtes-de-la-charité, anciennement vin de pays des coteaux charitois, est un vignoble français d'indication géographique protégée de zone, produit dans le département de la Nièvre autour de La Charité-sur-Loire.", + "wiki_title": "Côtes-de-la-charité" }, "cotes-de-meuse": { - "wiki_title": "Côtes-de-meuse (IGP)", "page_url": "https://fr.wikipedia.org/wiki/C%C3%B4tes-de-meuse_(IGP)", - "verification_quote": "Un côtes-de-meuse, appelé vin de pays des côtes de Meuse jusqu'en 2009 lorsqu'il fut reconnu comme IGP, est un vin français d'indication géographique protégée." + "verification_quote": "Un côtes-de-meuse, appelé vin de pays des côtes de Meuse jusqu'en 2009 lorsqu'il fut reconnu comme IGP, est un vin français d'indication géographique protégée.", + "wiki_title": "Côtes-de-meuse (IGP)" }, "cotes-de-thau": { "page_url": "https://fr.wikipedia.org/wiki/C%C3%B4tes-de-thau", @@ -1779,9 +1794,9 @@ "note": "researched 2026-05; no dedicated wine article on fr.wikipedia.org" }, "haute-vallee-de-l-aude": { - "wiki_title": "Haute-vallée-de-l'aude (IGP)", "page_url": "https://fr.wikipedia.org/wiki/Haute-vall%C3%A9e-de-l%27aude_(IGP)", - "verification_quote": "Le haute-vallée-de-l'aude est un vin français d'indication géographique protégée de zone qui a vocation à labelliser les vins ne pouvant postuler une appellation d'origine." + "verification_quote": "Le haute-vallée-de-l'aude est un vin français d'indication géographique protégée de zone qui a vocation à labelliser les vins ne pouvant postuler une appellation d'origine.", + "wiki_title": "Haute-vallée-de-l'aude (IGP)" }, "haute-vallee-de-l-orb": { "page_url": "https://fr.wikipedia.org/wiki/Haute-vall%C3%A9e-de-l%27orb", @@ -1794,18 +1809,18 @@ "wiki_title": "Hermitage (AOC)" }, "ile-de-beaute": { - "wiki_title": "Île-de-beauté", "page_url": "https://fr.wikipedia.org/wiki/%C3%8Ele-de-beaut%C3%A9", - "verification_quote": "Un île-de-beauté est un vin français d'indication géographique protégée (le nouveau nom depuis 2009 des vins de pays) de zone du vignoble de Corse." + "verification_quote": "Un île-de-beauté est un vin français d'indication géographique protégée (le nouveau nom depuis 2009 des vins de pays) de zone du vignoble de Corse.", + "wiki_title": "Île-de-beauté" }, "jura": { "missing": true, "note": "researched 2026-05; no dedicated wine article on fr.wikipedia.org" }, "le-pays-cathare": { - "wiki_title": "Le-pays-cathare (IGP)", "page_url": "https://fr.wikipedia.org/wiki/Le-pays-cathare_(IGP)", - "verification_quote": "Le le-pays-cathare, anciennement vin de pays cathare, est un vin français d'indication géographique protégée." + "verification_quote": "Le le-pays-cathare, anciennement vin de pays cathare, est un vin français d'indication géographique protégée.", + "wiki_title": "Le-pays-cathare (IGP)" }, "luzern": { "missing": true, @@ -1816,6 +1831,11 @@ "verification_quote": "Un marc d'Alsace, marc de gewurztraminer, marc de gewurz ou marc d'Alsace gewurztraminer de par sa dénomination légale, est une eau-de-vie fabriquée à partir du marc de gewurztraminer.", "wiki_title": "Marc d'Alsace" }, + "moselle-luxembourgeoise": { + "page_url": "https://fr.wikipedia.org/wiki/Viticulture_au_Luxembourg", + "verification_quote": "Sa production se concentre principalement sur la Moselle luxembourgeoise", + "wiki_title": "Viticulture au Luxembourg" + }, "moulis-ou-moulis-en-medoc": { "page_url": "https://fr.wikipedia.org/wiki/Moulis_(AOC)", "verification_quote": "Un moulis, ou moulis-en-médoc (les deux formes sont autorisées par le cahier des charges), est un vin rouge français d'appellation d'origine contrôlée produit autour du village de Moulis-en-Médoc dans le Médoc.", @@ -1844,18 +1864,18 @@ "wiki_title": "Pays-d'hérault" }, "pays-d-oc": { - "wiki_title": "Pays-d'oc (IGP)", "page_url": "https://fr.wikipedia.org/wiki/Pays-d%27oc_(IGP)", - "verification_quote": "Un pays-d'oc, anciennement appelé vin de pays d'Oc jusqu'en 2009, est un vin français d'indication géographique protégée (IGP)." + "verification_quote": "Un pays-d'oc, anciennement appelé vin de pays d'Oc jusqu'en 2009, est un vin français d'indication géographique protégée (IGP).", + "wiki_title": "Pays-d'oc (IGP)" }, "pays-de-brive": { "not_aoc_topic": true, "note": "Inventoried as not_aoc_topic per CURATOR_TODO.md. fr.wikipedia.org has Brive-la-Gaillarde and the bassin de Brive but no dedicated article for the Pays de Brive IGP wine." }, "pays-des-bouches-du-rhone": { - "wiki_title": "Pays-des-bouches-du-rhône", "page_url": "https://fr.wikipedia.org/wiki/Pays-des-bouches-du-rh%C3%B4ne", - "verification_quote": "Un pays-des-bouches-du-rhône, appelé vin de pays des Bouches-du-Rhône jusqu'en 2009, est un vin français d'indication géographique protégée." + "verification_quote": "Un pays-des-bouches-du-rhône, appelé vin de pays des Bouches-du-Rhône jusqu'en 2009, est un vin français d'indication géographique protégée.", + "wiki_title": "Pays-des-bouches-du-rhône" }, "pommeau-de-normandie": { "not_aoc_topic": true, @@ -1893,9 +1913,9 @@ "note": "researched 2026-05; no dedicated wine article on fr.wikipedia.org" }, "terres-du-midi": { - "wiki_title": "Terres-du-midi", "page_url": "https://fr.wikipedia.org/wiki/Terres-du-midi", - "verification_quote": "Un terres-du-midi est un vin qui peut être produit dans tout le vignoble du Languedoc-Roussillon, bénéficiant d'une indication géographique protégée (IGP) régionale." + "verification_quote": "Un terres-du-midi est un vin qui peut être produit dans tout le vignoble du Languedoc-Roussillon, bénéficiant d'une indication géographique protégée (IGP) régionale.", + "wiki_title": "Terres-du-midi" }, "thunersee": { "missing": true, @@ -1914,9 +1934,9 @@ "note": "researched 2026-05; no dedicated wine article on fr.wikipedia.org" }, "val-de-loire": { - "wiki_title": "Val-de-loire (IGP)", "page_url": "https://fr.wikipedia.org/wiki/Val-de-loire_(IGP)", - "verification_quote": "Un val-de-loire, anciennement appelé vin de pays du jardin de la France de 1981 à 2007, puis renommé vin de pays du Val de Loire jusqu'en 2009, est un vin français d'indication géographique protégée régionale." + "verification_quote": "Un val-de-loire, anciennement appelé vin de pays du jardin de la France de 1981 à 2007, puis renommé vin de pays du Val de Loire jusqu'en 2009, est un vin français d'indication géographique protégée régionale.", + "wiki_title": "Val-de-loire (IGP)" }, "valais-wallis": { "page_url": "https://fr.wikipedia.org/wiki/Vignoble_du_Valais", @@ -1970,34 +1990,34 @@ }, "hr": { "ponikve": { - "wiki_title": "Vinogorje Ponikve", "page_url": "https://hr.wikipedia.org/wiki/Vinogorje_Ponikve", - "verification_quote": "Vinogorje Ponikve su vinogradarski položaj koji se nalazi u blizini naselja Boljenovići na poluotoku Pelješcu u sastavu općine Ston u vinogradarskoj podregiji Srednja i Južna Dalmacija." + "verification_quote": "Vinogorje Ponikve su vinogradarski položaj koji se nalazi u blizini naselja Boljenovići na poluotoku Pelješcu u sastavu općine Ston u vinogradarskoj podregiji Srednja i Južna Dalmacija.", + "wiki_title": "Vinogorje Ponikve" } }, "hu": { "balaton": { - "wiki_title": "Balaton borrégió", "page_url": "https://hu.wikipedia.org/wiki/Balaton_borr%C3%A9gi%C3%B3", - "verification_quote": "A Balaton borrégió Magyarország hat borrégiójának egyike; négy megyén át a Balaton körül helyezkedik el, mintegy 10 718 hektár szőlőterülettel." + "verification_quote": "A Balaton borrégió Magyarország hat borrégiójának egyike; négy megyén át a Balaton körül helyezkedik el, mintegy 10 718 hektár szőlőterülettel.", + "wiki_title": "Balaton borrégió" }, "balatonmelleki": { "missing": true, "note": "researched 2026-05; 404 at /Balatonmelléki_borvidék; the PGI Landwein term has no dedicated article (Balaton borrégió is the macro-PDO, already pinned)" }, "csopak": { - "wiki_title": "Balatonfüred–Csopaki borvidék", "page_url": "https://hu.wikipedia.org/wiki/Balatonf%C3%BCred%E2%80%93Csopaki_borvid%C3%A9k", - "verification_quote": "A Balatonfüred-Csopaki borvidék Magyarország Dunántúli részén található, a Balaton tó északi partjának keleti medencéjében, kb. 2150 hektáron. [parent borvidék — Csopak village within it]" + "verification_quote": "A Balatonfüred-Csopaki borvidék Magyarország Dunántúli részén található, a Balaton tó északi partjának keleti medencéjében, kb. 2150 hektáron. [parent borvidék — Csopak village within it]", + "wiki_title": "Balatonfüred–Csopaki borvidék" }, "debroi-harslevelu": { "missing": true, "note": "researched 2026-05; only mentioned within Egri borvidék article; no standalone Debrői Hárslevelű article" }, "duna": { - "wiki_title": "Duna borrégió", "page_url": "https://hu.wikipedia.org/wiki/Duna_borr%C3%A9gi%C3%B3", - "verification_quote": "A Duna borrégió (más néven Alföldi borrégió) Magyarország legnagyobb borrégiója a Duna és a Tisza közötti területen." + "verification_quote": "A Duna borrégió (más néven Alföldi borrégió) Magyarország legnagyobb borrégiója a Duna és a Tisza közötti területen.", + "wiki_title": "Duna borrégió" }, "duna-tisza-kozi": { "missing": true, @@ -2008,18 +2028,18 @@ "note": "researched 2026-05; \"Dunántúl\" is a general Transdanubia geography article; no dedicated Dunántúli borrégió article" }, "etyeki-pezsgo": { - "wiki_title": "Etyek–Budai borvidék", "page_url": "https://hu.wikipedia.org/wiki/Etyek%E2%80%93Budai_borvid%C3%A9k", - "verification_quote": "Az Etyek–Budai borvidék a Felső-Pannon borrégió legkeletibb, a Buda, Etyek és a Velencei-tó környéki szőlőket felölelő része. [parent borvidék — no Etyeki Pezsgő-specific article]" + "verification_quote": "Az Etyek–Budai borvidék a Felső-Pannon borrégió legkeletibb, a Buda, Etyek és a Velencei-tó környéki szőlőket felölelő része. [parent borvidék — no Etyeki Pezsgő-specific article]", + "wiki_title": "Etyek–Budai borvidék" }, "felso-magyarorszag": { "missing": true, "note": "researched 2026-05; misidentification of Felső-Pannon borrégió (different region — west Hungary not north Hungary); no dedicated article for the Felső-Magyarországi PGI" }, "fured": { - "wiki_title": "Balatonfüred–Csopaki borvidék", "page_url": "https://hu.wikipedia.org/wiki/Balatonf%C3%BCred%E2%80%93Csopaki_borvid%C3%A9k", - "verification_quote": "A Balatonfüred-Csopaki borvidék Magyarország Dunántúli részén található, a Balaton tó északi partjának keleti medencéjében, kb. 2150 hektáron." + "verification_quote": "A Balatonfüred-Csopaki borvidék Magyarország Dunántúli részén található, a Balaton tó északi partjának keleti medencéjében, kb. 2150 hektáron.", + "wiki_title": "Balatonfüred–Csopaki borvidék" }, "izsaki-arany-sarfeher": { "missing": true, @@ -2030,18 +2050,18 @@ "note": "researched 2026-05; \"Káli-medence\" is a geological/national-park article with no wine content" }, "koszeg": { - "wiki_title": "Soproni borvidék", "page_url": "https://hu.wikipedia.org/wiki/Soproni_borvid%C3%A9k", - "verification_quote": "A Soproni borvidék Magyarország egyik legrégibb borvidéke. [parent borvidék — Kőszegi körzet sub-district within it]" + "verification_quote": "A Soproni borvidék Magyarország egyik legrégibb borvidéke. [parent borvidék — Kőszegi körzet sub-district within it]", + "wiki_title": "Soproni borvidék" }, "monor": { "missing": true, "note": "researched 2026-05; \"Monor\" is a town article; Monori sub-region noted on Kunsági borvidék but no dedicated wine article" }, "pannon": { - "wiki_title": "Pannon borrégió", "page_url": "https://hu.wikipedia.org/wiki/Pannon_borr%C3%A9gi%C3%B3", - "verification_quote": "A Pannon borrégió Magyarország hét borrégiójának egyike." + "verification_quote": "A Pannon borrégió Magyarország hét borrégiójának egyike.", + "wiki_title": "Pannon borrégió" }, "soltvadkerti": { "missing": true, @@ -2090,9 +2110,9 @@ "note": "researched 2026-05; demoted from FOUND — agent pinned to \"Montello rosso\" but that article is for the Montello Rosso DOCG (slug montello-rosso); Asolo Montello DOCG (red wine in the Asolo-Montello area) has no separate it.wiki article" }, "asolo-prosecco": { - "wiki_title": "Colli Asolani - Prosecco", "page_url": "https://it.wikipedia.org/wiki/Colli_Asolani_-_Prosecco_(vino)", - "verification_quote": "Colli Asolani - Prosecco (o Asolo - Prosecco) è una DOCG riservata ad alcuni vini la cui produzione è consentita nella provincia di Treviso." + "verification_quote": "Colli Asolani - Prosecco (o Asolo - Prosecco) è una DOCG riservata ad alcuni vini la cui produzione è consentita nella provincia di Treviso.", + "wiki_title": "Colli Asolani - Prosecco" }, "avola": { "missing": true, @@ -2131,9 +2151,9 @@ "note": "researched 2026-05; Casauria base name is a disambig page (commune/abbey); DOC described only as sub-zone within Montepulciano d'Abruzzo rosso article" }, "castel-del-monte-nero-di-troia-riserva": { - "wiki_title": "Castel del Monte Nero di Troia riserva", "page_url": "https://it.wikipedia.org/wiki/Castel_del_Monte_Nero_di_Troia_riserva", - "verification_quote": "Castel del Monte Nero di Troia è la denominazione di origine controllata e garantita di un vino prodotto in provincia di Barletta-Andria-Trani e nella città metropolitana di Bari." + "verification_quote": "Castel del Monte Nero di Troia è la denominazione di origine controllata e garantita di un vino prodotto in provincia di Barletta-Andria-Trani e nella città metropolitana di Bari.", + "wiki_title": "Castel del Monte Nero di Troia riserva" }, "catalanesca-del-monte-somma": { "missing": true, @@ -2176,18 +2196,18 @@ "note": "researched 2026-05; no it.wiki article" }, "colline-teramane-montepulciano-dabruzzo": { - "wiki_title": "Montepulciano d'Abruzzo Colline Teramane", "page_url": "https://it.wikipedia.org/wiki/Montepulciano_d%27Abruzzo_Colline_Teramane", - "verification_quote": "Montepulciano d'Abruzzo Colline Teramane è la denominazione relativa al disciplinare di alcuni vini a DOCG prodotti nei comuni di…" + "verification_quote": "Montepulciano d'Abruzzo Colline Teramane è la denominazione relativa al disciplinare di alcuni vini a DOCG prodotti nei comuni di…", + "wiki_title": "Montepulciano d'Abruzzo Colline Teramane" }, "conselvano": { "missing": true, "note": "researched 2026-05; no dedicated article; only listed in Vini IGT directory + Vini del Veneto" }, "corti-benedettine-del-padovano": { - "wiki_title": "Corti benedettine del Padovano", "page_url": "https://it.wikipedia.org/wiki/Corti_benedettine_del_Padovano", - "verification_quote": "Il Corti benedettine del padovano è un vino a DOC prodotto nelle province di Padova e Venezia. [Note: it.wiki article is about the DOC; slug refers to the IGT of the same name + area — wine info is region-relevant]" + "verification_quote": "Il Corti benedettine del padovano è un vino a DOC prodotto nelle province di Padova e Venezia. [Note: it.wiki article is about the DOC; slug refers to the IGT of the same name + area — wine info is region-relevant]", + "wiki_title": "Corti benedettine del Padovano" }, "costa-etrusco-romana": { "missing": true, @@ -2210,9 +2230,9 @@ "note": "researched 2026-05; no standalone it.wiki article for the dell'Emilia / Emilia IGT; appears only inside category and list pages" }, "dolcetto-di-ovada-superiore": { - "wiki_title": "Dolcetto di Ovada superiore", "page_url": "https://it.wikipedia.org/wiki/Dolcetto_di_Ovada_superiore", - "verification_quote": "Dolcetto di Ovada superiore o Ovada è la denominazione di origine controllata e garantita di un vino prodotto in provincia di Alessandria." + "verification_quote": "Dolcetto di Ovada superiore o Ovada è la denominazione di origine controllata e garantita di un vino prodotto in provincia di Alessandria.", + "wiki_title": "Dolcetto di Ovada superiore" }, "epomeo": { "missing": true, @@ -2259,27 +2279,27 @@ "note": "researched 2026-05; \"Marca Trevigiana\" article is about the medieval territorial designation around Treviso; IGT not primary subject" }, "martina": { - "wiki_title": "Martina Franca (vino)", "page_url": "https://it.wikipedia.org/wiki/Martina_Franca_(vino)", - "verification_quote": "Martina Franca o Martina è la denominazione di origine controllata di un vino prodotto nelle province di Taranto, Bari e Brindisi, in Puglia." + "verification_quote": "Martina Franca o Martina è la denominazione di origine controllata di un vino prodotto nelle province di Taranto, Bari e Brindisi, in Puglia.", + "wiki_title": "Martina Franca (vino)" }, "mitterberg": { "missing": true, "note": "researched 2026-05; only listed inside Vini a indicazione geografica tipica (Bolzano province IGT entry); no standalone article" }, "montello-rosso": { - "wiki_title": "Montello rosso", "page_url": "https://it.wikipedia.org/wiki/Montello_rosso", - "verification_quote": "Il Montello rosso o Montello è una DOCG riservata a un vino la cui produzione è consentita nella provincia di Treviso, in due comprensori collinari limitrofi posti ai piedi delle Dolomiti." + "verification_quote": "Il Montello rosso o Montello è una DOCG riservata a un vino la cui produzione è consentita nella provincia di Treviso, in due comprensori collinari limitrofi posti ai piedi delle Dolomiti.", + "wiki_title": "Montello rosso" }, "montenetto-di-brescia": { "missing": true, "note": "researched 2026-05; no it.wiki standalone article; en.wiki has Monte Netto (geological hill), it.wiki has only IGT list entry" }, "moscato-di-sorso": { - "wiki_title": "Moscato di Sorso-Sennori", "page_url": "https://it.wikipedia.org/wiki/Moscato_di_Sorso-Sennori", - "verification_quote": "Il Moscato di Sorso-Sennori è un vino DOC la cui produzione è consentita nella città metropolitana di Sassari." + "verification_quote": "Il Moscato di Sorso-Sennori è un vino DOC la cui produzione è consentita nella città metropolitana di Sassari.", + "wiki_title": "Moscato di Sorso-Sennori" }, "murgia": { "missing": true, @@ -2306,18 +2326,18 @@ "note": "researched 2026-05; \"Paestum_(vino)\" 404s; Paestum disambig page does not include a wine entry; IGT only in IGT list-page" }, "pinot-nero-dell-oltrepo-pavese": { - "wiki_title": "Pinot Nero dell'Oltrepò Pavese", "page_url": "https://it.wikipedia.org/wiki/Pinot_Nero_dell%27Oltrep%C3%B2_Pavese", - "verification_quote": "Il Pinot Nero dell'Oltrepò Pavese è un vino DOC rosso fermo, la cui produzione è consentita nella provincia di Pavia." + "verification_quote": "Il Pinot Nero dell'Oltrepò Pavese è un vino DOC rosso fermo, la cui produzione è consentita nella provincia di Pavia.", + "wiki_title": "Pinot Nero dell'Oltrepò Pavese" }, "pompeiano": { "missing": true, "note": "researched 2026-05; \"Rosso pompeiano\" is the iron-oxide pigment, not the wine; Pompeiano IGT only in IGT list-page" }, "portofino": { - "wiki_title": "Golfo del Tigullio-Portofino", "page_url": "https://it.wikipedia.org/wiki/Golfo_del_Tigullio-Portofino", - "verification_quote": "Golfo del Tigullio-Portofino o Portofino è una DOC riservata ad alcuni vini la cui produzione è consentita nella città metropolitana di Genova." + "verification_quote": "Golfo del Tigullio-Portofino o Portofino è una DOC riservata ad alcuni vini la cui produzione è consentita nella città metropolitana di Genova.", + "wiki_title": "Golfo del Tigullio-Portofino" }, "roccamonfina": { "missing": true, @@ -2332,23 +2352,27 @@ "note": "researched 2026-05; no it.wiki standalone article; only third-party wine sources" }, "rosso-conero": { - "wiki_title": "Rosso Conero", "page_url": "https://it.wikipedia.org/wiki/Rosso_Conero", - "verification_quote": "Il Rosso Conero è un vino DOC la cui produzione è consentita nella zona di Monte Cònero." + "verification_quote": "Il Rosso Conero è un vino DOC la cui produzione è consentita nella zona di Monte Cònero.", + "wiki_title": "Rosso Conero" + }, + "rotae": { + "missing": true, + "note": "2026-09-06: the title cascade bound IGT Rotae (Molise) to 'Roma (vino)' — the Roma DOC in Lazio — on name similarity. It produced a wiki-provenance terroir bullet about the Roma DOC's decree dates on the Rotae page. No dedicated it.wikipedia article for the Rotae IGT." }, "s-anna-di-isola-capo-rizzuto": { - "wiki_title": "Sant'Anna di Isola Capo Rizzuto", "page_url": "https://it.wikipedia.org/wiki/Sant%27Anna_di_Isola_Capo_Rizzuto", - "verification_quote": "Il Sant'Anna di Isola Capo Rizzuto è una DOC riservata ad alcuni vini." + "verification_quote": "Il Sant'Anna di Isola Capo Rizzuto è una DOC riservata ad alcuni vini.", + "wiki_title": "Sant'Anna di Isola Capo Rizzuto" }, "salina": { "missing": true, "note": "researched 2026-05; Salina is a disambig; Isola di Salina is the island; no dedicated IGT wine article" }, "scanzo": { - "wiki_title": "Moscato di Scanzo", "page_url": "https://it.wikipedia.org/wiki/Moscato_di_Scanzo", - "verification_quote": "Scanzo o Moscato di Scanzo è la denominazione di origine controllata e garantita di un vino prodotto in provincia di Bergamo." + "verification_quote": "Scanzo o Moscato di Scanzo è la denominazione di origine controllata e garantita di un vino prodotto in provincia di Bergamo.", + "wiki_title": "Moscato di Scanzo" }, "schaffhausen": { "missing": true, @@ -2374,6 +2398,10 @@ "missing": true, "note": "researched 2026-05; no dedicated wine article on it.wikipedia.org" }, + "tarantino": { + "missing": true, + "note": "2026-09-06: the title cascade bound IGP Tarantino (Puglia) to 'Trentino (vino)' — the Trentino DOC ~900 km north — on name similarity. No dedicated it.wikipedia article for the Tarantino IGP." + }, "terrazze-dell-imperiese": { "missing": true, "note": "researched 2026-05; mentioned only in the IGT list page; no dedicated wine article" @@ -2460,18 +2488,18 @@ "note": "researched 2026-05; no it.wiki article exists at the direct URL" }, "valtellina-rosso": { - "wiki_title": "Valtellina Rosso", "page_url": "https://it.wikipedia.org/wiki/Valtellina_Rosso", - "verification_quote": "Il Valtellina è un vino a DOC prodotto in provincia di Sondrio." + "verification_quote": "Il Valtellina è un vino a DOC prodotto in provincia di Sondrio.", + "wiki_title": "Valtellina Rosso" }, "vaud": { "missing": true, "note": "researched 2026-05; no dedicated wine article on it.wikipedia.org" }, "verdicchio-di-matelica-riserva": { - "wiki_title": "Verdicchio di Matelica riserva", "page_url": "https://it.wikipedia.org/wiki/Verdicchio_di_Matelica_riserva", - "verification_quote": "Verdicchio di Matelica riserva è la denominazione di origine controllata e garantita di un vino bianco prodotto nelle province di Macerata e Ancona." + "verification_quote": "Verdicchio di Matelica riserva è la denominazione di origine controllata e garantita di un vino bianco prodotto nelle province di Macerata e Ancona.", + "wiki_title": "Verdicchio di Matelica riserva" }, "vigneti-delle-dolomiti": { "missing": true, @@ -2500,33 +2528,33 @@ "note": "researched 2026-05; redlink on Denominações de origem portuguesas; Alenquer pages are about the município" }, "alentejano": { - "wiki_title": "Vinho do Alentejo", "page_url": "https://pt.wikipedia.org/wiki/Vinho_do_Alentejo", - "verification_quote": "Todos os produtores de vinho da região que pretendem usar a Denominação de Origem Controlada (DOC Alentejo) ou a Indicação Geográfica (Regional Alentejano)…" + "verification_quote": "Todos os produtores de vinho da região que pretendem usar a Denominação de Origem Controlada (DOC Alentejo) ou a Indicação Geográfica (Regional Alentejano)…", + "wiki_title": "Vinho do Alentejo" }, "beira-interior": { - "wiki_title": "Vinho da Beira Interior", "page_url": "https://pt.wikipedia.org/wiki/Vinho_da_Beira_Interior", - "verification_quote": "O Vinho da Beira Interior é um vinho português produzido nas regiões de Castelo Rodrigo, Cova da Beira e Pinhel, constituindo uma Denominação de Origem Controlada cuja demarcação remonta a 2 de Novembro de 1999." + "verification_quote": "O Vinho da Beira Interior é um vinho português produzido nas regiões de Castelo Rodrigo, Cova da Beira e Pinhel, constituindo uma Denominação de Origem Controlada cuja demarcação remonta a 2 de Novembro de 1999.", + "wiki_title": "Vinho da Beira Interior" }, "bucelas": { "missing": true, "note": "researched 2026-05; Bucelas page is about the parish/town in Loures; brief wine note only" }, "carcavelos": { - "wiki_title": "Vinho de Carcavelos", "page_url": "https://pt.wikipedia.org/wiki/Vinho_de_Carcavelos", - "verification_quote": "Carcavelos DOC é a mais pequena região demarcada vinícola portuguesa e situa-se em torno da freguesia de Carcavelos, nos Municípios de Cascais e de Oeiras." + "verification_quote": "Carcavelos DOC é a mais pequena região demarcada vinícola portuguesa e situa-se em torno da freguesia de Carcavelos, nos Municípios de Cascais e de Oeiras.", + "wiki_title": "Vinho de Carcavelos" }, "colares": { - "wiki_title": "Colares DOC", "page_url": "https://pt.wikipedia.org/wiki/Colares_DOC", - "verification_quote": "Colares é uma pequena região vinícola portuguesa no concelho de Sintra, em redor da vila de Colares, entre a serra e o Atlântico." + "verification_quote": "Colares é uma pequena região vinícola portuguesa no concelho de Sintra, em redor da vila de Colares, entre a serra e o Atlântico.", + "wiki_title": "Colares DOC" }, "dao": { - "wiki_title": "Região Demarcada do Dão", "page_url": "https://pt.wikipedia.org/wiki/Regi%C3%A3o_Demarcada_do_D%C3%A3o", - "verification_quote": "A Região Demarcada do Dão foi instituída em data desconhecida de 1908, situada no centro de Portugal, na província da Beira Alta." + "verification_quote": "A Região Demarcada do Dão foi instituída em data desconhecida de 1908, situada no centro de Portugal, na província da Beira Alta.", + "wiki_title": "Região Demarcada do Dão" }, "do-tejo": { "missing": true, @@ -2553,9 +2581,9 @@ "note": "researched 2026-05; Lagos (Portugal) is the município, not the DOC" }, "madeira": { - "wiki_title": "Vinho da Madeira", "page_url": "https://pt.wikipedia.org/wiki/Vinho_da_Madeira", - "verification_quote": "O vinho da Madeira, ou simplesmente vinho Madeira, é um vinho fortificado, com elevado teor alcoólico, produzido nas encostas e adegas da Região Demarcada da Ilha da Madeira." + "verification_quote": "O vinho da Madeira, ou simplesmente vinho Madeira, é um vinho fortificado, com elevado teor alcoólico, produzido nas encostas e adegas da Região Demarcada da Ilha da Madeira.", + "wiki_title": "Vinho da Madeira" }, "madeirense": { "missing": true, @@ -2566,14 +2594,14 @@ "note": "researched 2026-05; no PT article for the IGP (Vinho Verde DOC article unrelated)" }, "pico": { - "wiki_title": "Vinho do Pico", "page_url": "https://pt.wikipedia.org/wiki/Vinho_do_Pico", - "verification_quote": "Vinho do Pico é a designação genérica dada aos vinhos produzidos na ilha do Pico, Açores." + "verification_quote": "Vinho do Pico é a designação genérica dada aos vinhos produzidos na ilha do Pico, Açores.", + "wiki_title": "Vinho do Pico" }, "setubal": { - "wiki_title": "Moscatel de Setúbal", "page_url": "https://pt.wikipedia.org/wiki/Moscatel_de_Set%C3%BAbal", - "verification_quote": "Moscatel de Setúbal é uma Denominação de Origem Controlada (DOC) portuguesa conhecida pelos vinhos generosos produzidos de castas moscatel." + "verification_quote": "Moscatel de Setúbal é uma Denominação de Origem Controlada (DOC) portuguesa conhecida pelos vinhos generosos produzidos de castas moscatel.", + "wiki_title": "Moscatel de Setúbal" }, "tavora-varosa": { "missing": true, @@ -2650,9 +2678,9 @@ "note": "researched 2026-05; only the Bujoru research station exists, not the wine region" }, "dealu-mare": { - "wiki_title": "Podgoria Dealu Mare", "page_url": "https://ro.wikipedia.org/wiki/Podgoria_Dealu_Mare", - "verification_quote": "Dealu Mare este o regiune viticolă cu mai multe podgorii, situată în județele Buzău și Prahova pe versantul sudic al dealurilor Istriței." + "verification_quote": "Dealu Mare este o regiune viticolă cu mai multe podgorii, situată în județele Buzău și Prahova pe versantul sudic al dealurilor Istriței.", + "wiki_title": "Podgoria Dealu Mare" }, "dealurile-crisanei": { "missing": true, @@ -2771,9 +2799,9 @@ "note": "researched 2026-05; Valea Mare-Podgoria, Argeș is a village; no dedicated Podgoria Ștefănești article" }, "tarnave": { - "wiki_title": "Podgoria Târnavelor", "page_url": "https://ro.wikipedia.org/wiki/Podgoria_T%C3%A2rnavelor", - "verification_quote": "Podgoria Tarnavelor este cea mai mare din Transilvania, concentrînd vigoarea plantațiilor cultivate între râurile Târnava Mare și Târnava Mică." + "verification_quote": "Podgoria Tarnavelor este cea mai mare din Transilvania, concentrînd vigoarea plantațiilor cultivate între râurile Târnava Mare și Târnava Mică.", + "wiki_title": "Podgoria Târnavelor" }, "terasele-dunarii": { "missing": true, @@ -2802,84 +2830,64 @@ "note": "researched 2026-05; sk.wiki has no all-Slovakia wine-PGI article (Slovenské_víno is 404)" }, "tokajske-vino-zo-slovenskej-oblasti": { - "wiki_title": "Tokajské víno", "page_url": "https://sk.wikipedia.org/wiki/Tokajsk%C3%A9_v%C3%ADno", - "verification_quote": "Tokajské víno je označenie vína z vinohradníckej oblasti Tokaj (scoped to Slovak Tokaj per zákon č. 313/2009 Z. z. o vinohradníctve a vinárstve)." + "verification_quote": "Tokajské víno je označenie vína z vinohradníckej oblasti Tokaj (scoped to Slovak Tokaj per zákon č. 313/2009 Z. z. o vinohradníctve a vinárstve).", + "wiki_title": "Tokajské víno" } }, "sl": { "bela-krajina": { - "wiki_title": "Vinorodni okoliš Bela krajina", "page_url": "https://sl.wikipedia.org/wiki/Vinorodni_okoli%C5%A1_Bela_krajina", - "verification_quote": "Vinorodni okoliš Bela krajina (1130 ha) je eden od treh vinorodnih okolišev 7700 hektarov obsegajoče slovenske vinorodne dežele Posavje." + "verification_quote": "Vinorodni okoliš Bela krajina (1130 ha) je eden od treh vinorodnih okolišev 7700 hektarov obsegajoče slovenske vinorodne dežele Posavje.", + "wiki_title": "Vinorodni okoliš Bela krajina" }, "bizeljcan": { "missing": true, "note": "researched 2026-05; sl.wiki has split articles \"Rdeči bizeljčan\" and \"Beli bizeljčan\" (red/white versions); no combined Bizeljčan appellation article" }, "bizeljsko-sremic": { - "wiki_title": "Vinorodni okoliš Bizeljsko - Sremič", "page_url": "https://sl.wikipedia.org/wiki/Vinorodni_okoli%C5%A1_Bizeljsko_-_Sremi%C4%8D", - "verification_quote": "Vinorodni okoliš Bizeljsko - Sremič (1700 ha) je eden od treh vinorodnih okolišev 7700 hektarov obsegajoče slovenske vinorodne dežele Posavje." + "verification_quote": "Vinorodni okoliš Bizeljsko - Sremič (1700 ha) je eden od treh vinorodnih okolišev 7700 hektarov obsegajoče slovenske vinorodne dežele Posavje.", + "wiki_title": "Vinorodni okoliš Bizeljsko - Sremič" }, "goriska-brda": { - "wiki_title": "Vinorodni okoliš Goriška brda", "page_url": "https://sl.wikipedia.org/wiki/Vinorodni_okoli%C5%A1_Gori%C5%A1ka_brda", - "verification_quote": "Vinorodni okoliš Goriška brda obsega 1800 ha površine in je del v 7055 ha obsegajoče slovenske vinorodne dežele Primorske." + "verification_quote": "Vinorodni okoliš Goriška brda obsega 1800 ha površine in je del v 7055 ha obsegajoče slovenske vinorodne dežele Primorske.", + "wiki_title": "Vinorodni okoliš Goriška brda" }, "kras": { - "wiki_title": "Vinorodni okoliš Kras", "page_url": "https://sl.wikipedia.org/wiki/Vinorodni_okoli%C5%A1_Kras", - "verification_quote": "Vinorodni okoliš Kras je s površino 575 hektarov najmanjši vinorodni okoliš v Sloveniji." + "verification_quote": "Vinorodni okoliš Kras je s površino 575 hektarov najmanjši vinorodni okoliš v Sloveniji.", + "wiki_title": "Vinorodni okoliš Kras" }, "podravje": { - "wiki_title": "Vinorodna dežela Podravje", "page_url": "https://sl.wikipedia.org/wiki/Vinorodna_de%C5%BEela_Podravje", - "verification_quote": "Vinorodna dežela Podravje je z okrog 6641 hektarji druga po velikosti izmed treh slovenskih vinorodnih dežel." + "verification_quote": "Vinorodna dežela Podravje je z okrog 6641 hektarji druga po velikosti izmed treh slovenskih vinorodnih dežel.", + "wiki_title": "Vinorodna dežela Podravje" }, "posavje": { - "wiki_title": "Vinorodna dežela Posavje", "page_url": "https://sl.wikipedia.org/wiki/Vinorodna_de%C5%BEela_Posavje", - "verification_quote": "Vinorodna dežela Posavje je ena izmed treh slovenskih vinorodnih dežel." + "verification_quote": "Vinorodna dežela Posavje je ena izmed treh slovenskih vinorodnih dežel.", + "wiki_title": "Vinorodna dežela Posavje" }, "prekmurje": { - "wiki_title": "Vinorodni okoliš Prekmurje", "page_url": "https://sl.wikipedia.org/wiki/Vinorodni_okoli%C5%A1_Prekmurje", - "verification_quote": "Vinorodni okoliš Prekmurje, tudi Prekmurske gorice (528 ha) je po površini manjši od dveh vinorodnih okolišev." + "verification_quote": "Vinorodni okoliš Prekmurje, tudi Prekmurske gorice (528 ha) je po površini manjši od dveh vinorodnih okolišev.", + "wiki_title": "Vinorodni okoliš Prekmurje" }, "primorska": { - "wiki_title": "Vinorodna dežela Primorska", "page_url": "https://sl.wikipedia.org/wiki/Vinorodna_de%C5%BEela_Primorska", - "verification_quote": "Vinorodna dežela Primorska je ena izmed treh vinorodnih dežel v Sloveniji." + "verification_quote": "Vinorodna dežela Primorska je ena izmed treh vinorodnih dežel v Sloveniji.", + "wiki_title": "Vinorodna dežela Primorska" }, "slovenska-istra": { - "wiki_title": "Vinorodni okoliš Koper", "page_url": "https://sl.wikipedia.org/wiki/Vinorodni_okoli%C5%A1_Koper", - "verification_quote": "Koprski vinorodni okoliš (2400 ha) spada v 7055 hektarov obsegajočo slovensko vinorodno deželo Primorsko. [Koprski okoliš = sl-language name for Slovenska Istra]" + "verification_quote": "Koprski vinorodni okoliš (2400 ha) spada v 7055 hektarov obsegajočo slovensko vinorodno deželo Primorsko. [Koprski okoliš = sl-language name for Slovenska Istra]", + "wiki_title": "Vinorodni okoliš Koper" }, "stajerska-slovenija": { "missing": true, "note": "researched 2026-05; no dedicated \"Vinorodni okoliš Štajerska Slovenija\" article; only mentioned as parent in sub-okoliš articles" } - }, - "en": { - "malta": { - "wiki_title": "Maltese wine", - "page_url": "https://en.wikipedia.org/wiki/Maltese_wine", - "verification_quote": "Maltese wine dates back over two thousand years to the time of the Phoenicians.", - "note": "Malta has no per-PDO/PGI en.wikipedia article; the umbrella Maltese-wine article grounds terroir facts (the CH/LU pattern)." - }, - "gozo": { - "wiki_title": "Maltese wine", - "page_url": "https://en.wikipedia.org/wiki/Maltese_wine", - "verification_quote": "Maltese wine dates back over two thousand years to the time of the Phoenicians.", - "note": "Malta has no per-PDO/PGI en.wikipedia article; the umbrella Maltese-wine article grounds terroir facts (the CH/LU pattern)." - }, - "maltese-islands": { - "wiki_title": "Maltese wine", - "page_url": "https://en.wikipedia.org/wiki/Maltese_wine", - "verification_quote": "Maltese wine dates back over two thousand years to the time of the Phoenicians.", - "note": "Malta has no per-PDO/PGI en.wikipedia article; the umbrella Maltese-wine article grounds terroir facts (the CH/LU pattern)." - } } -} \ No newline at end of file +} diff --git a/raw/wikipedia/grape_overrides.json b/raw/wikipedia/grape_overrides.json index 167089a..5a4c49c 100644 --- a/raw/wikipedia/grape_overrides.json +++ b/raw/wikipedia/grape_overrides.json @@ -196,6 +196,7 @@ "teoulier": "Téoulier", "terret-blanc-gris": "Terret (grape)", "touriga-nacional": "Touriga Nacional", + "triomphe-dalsace": "Triomphe d'Alsace", "ugni-blanc": "Trebbiano", "valdiguie": "Valdiguié", "verdejo-negro": "Verdejo", @@ -389,6 +390,7 @@ "teoulier": "Téoulier", "tintilla-de-rota": "Tintilla de Rota", "tourbat": "Torbato", + "triomphe-dalsace": "Triomphe d'Alsace", "trousseau-gris": "Trousseau", "ugni-blanc": "Trebbiano", "valdiguie": "Valdiguié", @@ -541,4 +543,4 @@ "kraljevina": "Kraljevina (sorta)", "zametovka": "Žametovka" } -} \ No newline at end of file +} diff --git a/scripts/04_build_maps.py b/scripts/04_build_maps.py index d5083db..926c2fb 100644 --- a/scripts/04_build_maps.py +++ b/scripts/04_build_maps.py @@ -98,6 +98,8 @@ from _lib.es.sigpac import SigpacIndex from _lib.es.zones import MAPA_ZONES_FILE, ESZoneIndex from _lib.fr_wine_region import derive_wine_region as derive_fr_wine_region +from _lib.gb.geometry import GBPolygonIndex +from _lib.gb.region import derive_region as derive_gb_region from _lib.geom_chain import ( _resolve_es_igp_fallback, _resolve_es_sigpac, @@ -203,6 +205,7 @@ EXTRACTED_BE = ROOT / "raw" / "be" / "dokumenten-extracted" EXTRACTED_NL = ROOT / "raw" / "nl" / "dokumenten-extracted" EXTRACTED_MT = ROOT / "raw" / "mt" / "dokumente-extracted" +EXTRACTED_GB = ROOT / "raw" / "gb" / "specs-extracted" NL_NUTS_GEOJSON = ROOT / "raw" / "nl" / "nuts" / "NUTS_RG_03M_2024_4326_LEVL_2.geojson" COMMUNES_GEOJSON = ROOT / "raw" / "ign" / "communes.geojson" WIKI = ROOT / "wiki" @@ -848,6 +851,16 @@ def main() -> int: if json_path.name == "_index.json": continue extracted_records.append(json.loads(json_path.read_text(encoding="utf-8"))) + # Multi-country: also iterate GB extracted records + # (raw/gb/specs-extracted/). 6 registered wine GIs (4 PDO + 2 PGI), + # every one of them fully extracted — the UK register publishes a + # product specification for all six, so there is no stub tier. + # Source language is en. + if EXTRACTED_GB.exists(): + for json_path in sorted(EXTRACTED_GB.glob("*.json")): + if json_path.name == "_index.json": + continue + extracted_records.append(json.loads(json_path.read_text(encoding="utf-8"))) # Augment ES records with national-pliego sidecar data — adds the # accessory varieties that the EU-OJ documento único omits. The # sidecar carries provenance (URL + sha256 + fetched_at) which @@ -1326,6 +1339,13 @@ def main() -> int: file=sys.stderr, ) mt_hits: Counter[str] = Counter() + gb_polygons = GBPolygonIndex() + print( + f"[load] GB polygons: {gb_polygons.n_countries} ONS countries, " + f"{gb_polygons.n_counties} ONS counties/UAs", + file=sys.stderr, + ) + gb_hits: Counter[str] = Counter() # Curator-reviewed geometry-outlier overrides — clips confirmed-spurious # parts (upstream-data errors) out of resolved polygons. See @@ -1358,6 +1378,7 @@ def main() -> int: _emit_be_features = False _emit_nl_features = False _emit_mt_features = False + _emit_gb_features = False # ES branch — Figshare PDO polygon → GISCO commune-union → parent # fallback. Stubs (`stub: True`) skip geometry; they appear in the @@ -2017,6 +2038,26 @@ def main() -> int: parent_geom_by_slug[record["slug"]] = geom parent_village_geom_by_slug[record["slug"]] = geom _emit_mt_features = True + elif country == "gb": + # GB branch — ONS administrative boundaries (Bétard is an EU + # PDO layer and carries no PDO-GB-* rows). England / Wales + # whole-country polygons for the four national GIs, the union + # of the three Sussex counties/UAs for Sussex, and for + # Darnibole an approximate boundary reconstructed from the + # parcel references on its own specification plan. + sib_v_geom = sib_name = sib_slug = None + cadastre_match = None + geom, geom_source, stats = gb_polygons.resolve( + record.get("file_number") or "" + ) + gb_hits[geom_source] += 1 + v_geom = geom + v_source = geom_source + v_stats = stats + if geom is not None and not geom.is_empty: + parent_geom_by_slug[record["slug"]] = geom + parent_village_geom_by_slug[record["slug"]] = geom + _emit_gb_features = True else: _emit_es_features = False _emit_pt_features = False @@ -2032,7 +2073,7 @@ def main() -> int: or _emit_sk_features or _emit_cz_features or _emit_ch_features or _emit_lu_features or _emit_be_features or _emit_nl_features - or _emit_mt_features): + or _emit_mt_features or _emit_gb_features): # Geometry already resolved above; skip the FR-specific chain. pass elif is_sub_denomination: @@ -2094,7 +2135,7 @@ def main() -> int: or _emit_sk_features or _emit_cz_features or _emit_ch_features or _emit_lu_features or _emit_be_features or _emit_nl_features - or _emit_mt_features): + or _emit_mt_features or _emit_gb_features): pass elif is_sub_denomination: # Prefer DGC's own parcellaire polygon as the village geometry — @@ -2231,7 +2272,7 @@ def main() -> int: # ES + PT + IT + AT + SI records have no `categorie` — every entry # is filtered to productType=WINE upstream in stage 00, so they're # all wines. - if record.get("country") in ("es", "pt", "it", "at", "de", "si", "hr", "hu", "ro", "bg", "gr", "cy", "sk", "cz", "ch", "lu", "be", "nl", "mt"): + if record.get("country") in ("es", "pt", "it", "at", "de", "si", "hr", "hu", "ro", "bg", "gr", "cy", "sk", "cz", "ch", "lu", "be", "nl", "mt", "gb"): is_wine = "1" else: is_wine = "1" if categorie.startswith("Vin") else "0" @@ -2485,6 +2526,10 @@ def main() -> int: # Islands" for the archipelago-wide PGI. Carried on the # record from stage 02. region_value = record.get("region") or "Maltese Islands" + elif record.get("country") == "gb": + # GB region = the home nation the specification demarcates the + # GI to (England / Wales). Carried on the record from stage 02. + region_value = derive_gb_region(record) or "United Kingdom" else: region_value = derive_fr_wine_region(record) common_props = { @@ -3131,6 +3176,23 @@ def _sources_for(record: dict) -> dict: "file_number": record.get("file_number") or "", "id_eambrosia": record.get("id_eambrosia") or "", } + if record.get("country") == "gb": + # United Kingdom: the DEFRA / GOV.UK product specification (PDF or + # .docx) served from the UK GI register. `boagri_url` and the + # EUR-Lex fields stay empty — nothing here is an EU-OJ document. + return { + "country": "gb", + "source_lang": "en", + "gov_uk_register_url": src.get("register_url") or "", + "gov_uk_spec_url": src.get("source_url") or "", + "spec_format": src.get("format") or "", + "spec_sha256": src.get("sha256") or "", + "filename": src.get("filename") or "", + "fetched_at": src.get("fetched_at") or "", + "file_number": record.get("file_number") or "", + "parser_template": record.get("parser_template") or "", + "geom_approximate": bool(record.get("geom_approximate")), + } if record.get("country") == "es": # The AOC-blob phase re-reads the on-disk extracted JSON (which # doesn't carry the augmentation), so fall back to the slug-keyed @@ -3507,6 +3569,7 @@ def emit_html( "be": EXTRACTED_BE, "nl": EXTRACTED_NL, "mt": EXTRACTED_MT, + "gb": EXTRACTED_GB, }.get(country, EXTRACTED) ext_path = ext_dir / f"{slug}.json" summary = "" @@ -3564,10 +3627,22 @@ def emit_html( # actually published (PT Douro shows "Aragonez", ES Rioja # shows "Tempranillo", FR Bandol shows "mourvèdre"), with # the VIVC canonical name added in brackets when distinct. + # Carry the record's spelling UNCONDITIONALLY. A previous + # `s_name.lower() != s_slug` guard dropped it whenever the name + # already matched its slug ("Optima" -> optima), on the + # assumption the client could re-derive it. It cannot: the + # client falls back to GRAPES_INFO[slug].name, which is the + # most frequent spelling ACROSS THE WHOLE CORPUS and is often + # another language's — so English "Optima" rendered as Germany's + # "Optima 113", German "Müller Thurgau" as Hungarian + # "Rizlingszilváni", Spanish "godello" as Portuguese "Gouveio" + # (1,047 substantive cases across 10 countries). `grape_names` + # rides the lazily-fetched panel payload, not the startup + # bundle, so carrying every name costs ~70 bytes per panel file. for d in (rec.get("grapes") or {}).get("details") or []: s_slug = d.get("slug") s_name = (d.get("name") or "").strip() - if s_slug and s_name and s_name.lower() != s_slug: + if s_slug and s_name: grape_names[s_slug] = s_name latin = _latin_form_or_empty(s_name) if latin: @@ -3817,9 +3892,10 @@ def gate_classify(rec: dict, has_children: bool = False) -> tuple[str, str | Non src_lang = rec.get("source_lang") or "fr" elif rec_country == "nl": src_lang = "nl" - elif rec_country == "mt": - # MT's country code is "mt" but its source language is "en" - # (Malta's EU single documents are published in English). + elif rec_country in ("mt", "gb"): + # MT's country code is "mt" and GB's is "gb", but both are + # English-source: Malta's EU single documents and the UK's + # DEFRA product specifications are written in English. src_lang = "en" else: # LU's country code is "lu" but its source language is "fr" — fall through to the "fr" default. @@ -3864,7 +3940,7 @@ def _src_lang_for(slug: str) -> str: return rec.get("source_lang") or "fr" if c == "nl": return "nl" - if c == "mt": + if c in ("mt", "gb"): return "en" # LU (country "lu") uses source_lang "fr" — falls through to the "fr" default. return c if c in ("es", "pt", "it", "at", "de", "si", "hr", "hu", "ro", "bg", "gr", "cy", "sk", "cz") else "fr" diff --git a/scripts/_lib/appellation_urls.json b/scripts/_lib/appellation_urls.json index 8e6fd0e..ce722ea 100644 --- a/scripts/_lib/appellation_urls.json +++ b/scripts/_lib/appellation_urls.json @@ -4704,6 +4704,14 @@ "Primorska": { "url": "https://vinskadruzba.si/", "label": "Vinska družba Slovenije d.o.o." + }, + "England": { + "url": "https://winegb.co.uk/", + "label": "WineGB" + }, + "Wales": { + "url": "https://winegb.co.uk/", + "label": "WineGB" } } } diff --git a/scripts/_lib/assets/app.js b/scripts/_lib/assets/app.js index b6ba506..070dd4b 100644 --- a/scripts/_lib/assets/app.js +++ b/scripts/_lib/assets/app.js @@ -1774,6 +1774,14 @@ const lab = sources.cantonal_reglement_label ? ' — ' + escapeHtml(sources.cantonal_reglement_label) : ''; links.push(`
  • ${LABELS.src_cantonal_reglement}${lab}
  • `); } + if (sources.gov_uk_spec_url) { + const fmtTail = sources.spec_format ? ' — ' + escapeHtml(String(sources.spec_format).toUpperCase()) : ''; + links.push(`
  • ${LABELS.src_gov_uk_spec}${fmtTail}
  • `); + } + if (sources.gov_uk_register_url) { + const fileNum = sources.file_number ? ' — ' + escapeHtml(sources.file_number) : ''; + links.push(`
  • ${LABELS.src_gov_uk_register}${fileNum}
  • `); + } if (sources.ofag_repertoire_url) { links.push(`
  • ${LABELS.src_ofag_repertoire}
  • `); } @@ -2020,6 +2028,7 @@ if (s.eur_lex_url && hasEambrosia) return [eambrosiaReg, s.source_lang === 'nl' ? 'enig document' : 'document unique', false, extra]; return null; } + if (country === 'gb') return s.gov_uk_spec_url ? ['DEFRA', 'product specification', false, extra] : null; if (country === 'at' || country === 'nl' || country === 'mt') return euFallback(); return null; } @@ -2133,6 +2142,8 @@ approxLine = `
    ${escapeHtml(LABELS.geom_approx_parent)}
    `; } else if (r.geom_source === 'aires-csv-dgc') { approxLine = `
    ${escapeHtml(LABELS.geom_approx_aires)}
    `; + } else if (r.geom_source === 'pdo-plan-parcel-hull-approx') { + approxLine = `
    ${escapeHtml(LABELS.geom_approx_pdo_plan)}
    `; } else if (r.geom_source === 'cadastre-lieu-dit-dgc' && r.cadastre_lieu_dit) { const src = `${escapeHtml(LABELS.geom_approx_cadastre_source_label)}`; approxLine = `
    ${fmt(LABELS.geom_approx_cadastre, { lieu_dit: escapeHtml(r.cadastre_lieu_dit), commune: escapeHtml(r.cadastre_commune || ''), source: src })}
    `; diff --git a/scripts/_lib/content_block.py b/scripts/_lib/content_block.py index 118feba..27e2220 100644 --- a/scripts/_lib/content_block.py +++ b/scripts/_lib/content_block.py @@ -55,6 +55,7 @@ "be": "enig document / document unique", "nl": "enig document", "mt": "single document", + "gb": "product specification", "ch": "règlement cantonal", } @@ -470,6 +471,7 @@ def render_sources(sources: dict | None, ctx: RenderCtx) -> str: "hu": "egységes dokumentum", "ro": "document unic", "bg": "единен документ", "gr": "ενιαίο έγγραφο", "cy": "ενιαίο έγγραφο", "sk": "jednotný dokument", "cz": "jednotný dokument", "nl": "enig document", "mt": "single document", + "gb": "product specification", } # national-spec source-org token → human regulator name (the literal token @@ -584,6 +586,10 @@ def eu_fallback() -> tuple[str, str, bool, str] | None: doc = "enig document" if s.get("source_lang") == "nl" else "document unique" return (eambrosia_reg, doc, False, extra) return None + if country == "gb": + if s.get("gov_uk_spec_url"): + return ("DEFRA", "product specification", False, extra) + return None if country in ("at", "nl", "mt"): return eu_fallback() return None diff --git a/scripts/_lib/gb/__init__.py b/scripts/_lib/gb/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/scripts/_lib/gb/darnibole.py b/scripts/_lib/gb/darnibole.py new file mode 100644 index 0000000..4392272 --- /dev/null +++ b/scripts/_lib/gb/darnibole.py @@ -0,0 +1,103 @@ +"""Darnibole PDO — an approximate boundary derived from the regulator's plan. + +`PDO-GB-N1636` Darnibole is a single-vineyard PDO at Camel Valley, +Nanstallon (Cornwall). Its product specification defines the demarcated +area **narratively** — "bordered to the West by a marked soil change to +alluvial sand (the old River Camel river bed) … The disused railway (now +the Camel Trail) demarcates the Southern boundary" — with no coordinates, +and no public polygon of it exists anywhere. + +What the specification *does* carry is a **"Plan of demarcated area"** +(page 5 of `protected-food-name-darnibole-wine.pdf`): an OS-based Rural +Payments Agency land-parcel map with the PDO boundary drawn in red over +numbered field parcels, each labelled "Vines planted" / "Not yet planted". + +Those parcel numbers are not arbitrary ids. Under the OS / RPA +convention a field parcel is identified by the **four-figure National +Grid reference of its centroid within its 1 km grid square** — two +digits of easting then two of northing, each in units of 10 m. So +parcel `7985` sits at 790 m E, 850 m N inside its square. + +That makes the plan georeferenceable, and this module reconstructs an +approximate boundary from it: the convex hull of the centroids of the +seven parcels the red line encloses. + +**Anchoring the square** (SX 03 67 → BNG easting base 203000, northing +base 67000) was verified three ways: + + 1. *Internal consistency.* Decoded north→south and west→east order + reproduces the plan's layout exactly, including the 1 km grid line + visible on the plan between parcels 5107/8101 (N 68xxx) and 5095 + (N 67950). + 2. *Terrain.* An elevation transect north from the block (EU-DEM 25 m) + puts the valley floor at 11–14 m over N 67300–67400 and climbs + steadily to 96 m by N 68100. The seven parcels sit at 47–80 m on a + ~13 % south-facing slope immediately above the old river bed — + precisely the specification's "steep south facing slope", bounded + south by the River Camel's old bed and north by land "above the + optimum thermal band". + 3. *Address.* The ONS centroid of Camel Valley's postcode (PL30 5LG) + is E 203164 N 67751 — on the same slope, ~330 m west of the block. + +**Accuracy.** The hull spans E 203490–203890 / N 67680–67950 and +measures **6.1 ha** against the specification's declared "whole 5 +hectare area" — so it is the right size in the right place, but it is a +reconstruction, not an official boundary: it interpolates between parcel +centroids rather than tracing the red line itself, so its edges fall +inside the true boundary by up to roughly half a field. Records built +from it carry `geom_approximate: True` and `geom_source: +"pdo-plan-parcel-hull-approx"` so the map panel can disclose it. + +Source: Product specification for Darnibole (PDO), DEFRA / GOV.UK, +Open Government Licence v3.0. +""" + +from __future__ import annotations + +from shapely.geometry import MultiPoint +from shapely.geometry.base import BaseGeometry + +# BNG easting/northing base of OS 1 km square SX 03 67. +_SQUARE_E0 = 203000 +_SQUARE_N0 = 67000 + +# The parcels enclosed by the red demarcation line on the plan, keyed by +# the RPA parcel number printed on it → (easting, northing) within the +# square in units of 10 m, plus the plan's own planting label. +DEMARCATED_PARCELS: dict[str, tuple[int, int, str]] = { + "5095": (50, 95, "not yet planted"), + "7985": (79, 85, "vines planted"), + "4976": (49, 76, "vines planted"), + "7873": (78, 73, "vines planted"), + "6372": (63, 72, "vines planted"), + "7270": (72, 70, "vines planted"), + "8968": (89, 68, "vines planted"), +} + +# Parcels drawn on the plan but OUTSIDE the red line, kept for +# provenance: 5317 / 5107 / 6921 / 8101 lie north of it, 6661 and 8748 +# (the older, hatched Camel Valley rows) south of it. +EXCLUDED_PARCELS: tuple[str, ...] = ("5317", "5107", "6921", "8101", "6661", "8748") + +GEOM_SOURCE = "pdo-plan-parcel-hull-approx" +SOURCE_LABEL = ( + "Approximate — reconstructed from the parcel references on the " + "PDO specification's plan of the demarcated area" +) +SOURCE_URL = ( + "https://assets.publishing.service.gov.uk/media/5fd36a7ad3bf7f3061e108aa/" + "protected-food-name-darnibole-wine.pdf" +) + + +def parcel_points_bng() -> list[tuple[int, int]]: + """The seven demarcated parcel centroids in EPSG:27700.""" + return [ + (_SQUARE_E0 + e * 10, _SQUARE_N0 + n * 10) + for e, n, _label in DEMARCATED_PARCELS.values() + ] + + +def boundary_bng() -> BaseGeometry: + """Approximate Darnibole boundary in EPSG:27700 (British National Grid).""" + return MultiPoint(parcel_points_bng()).convex_hull diff --git a/scripts/_lib/gb/geometry.py b/scripts/_lib/gb/geometry.py new file mode 100644 index 0000000..8fbb89c --- /dev/null +++ b/scripts/_lib/gb/geometry.py @@ -0,0 +1,179 @@ +"""GB-side geometry resolution — ONS administrative boundaries. + +The UK left the EU before Bétard 2022's snapshot logic applies to it, and +the dataset the other countries lean on (`raw/es/figshare/EU_PDO.gpkg`) is +an **EU** PDO layer — it carries no `PDO-GB-*` rows. GB therefore resolves +entirely against ONS Open Geography boundaries (Open Government Licence +v3.0), which is a better fit anyway: every UK wine GI is demarcated to +whole administrative units named in its product specification. + + - `raw/gb/ons/countries.geojson` — Countries (December 2025) UK BGC. + England and Wales, the `DEMARCATION` of four of the six GIs. + - `raw/gb/ons/counties.geojson` — Counties and Unitary Authorities + (December 2025) UK BGC. Sussex's specification demarcates "the + administrative boundaries of the counties of East and West Sussex"; + Brighton and Hove is a unitary authority carved out of East Sussex in + 1997 and sits between the two, so the union of the three CTYUAs + reconstructs the ceremonial Sussex the specification's own map shows. + +Stage 04 resolves each GB record by: + + 1. **ons-country** — England / Wales whole-country polygon, for the + English + Welsh PDOs and their Regional PGI counterparts. + 2. **ons-county-union** — Sussex: East Sussex ∪ West Sussex ∪ Brighton + and Hove. + 3. **pdo-plan-parcel-hull-approx** — Darnibole, reconstructed from the + parcel references on its specification's plan. Approximate by + construction; see `_lib/gb/darnibole.py` for the derivation and the + three-way anchor check. Carries `approximate=True` so the panel can + disclose it. + 4. **stub-no-geometry** — last resort; not hit in v1, all 6 resolve. +""" + +from __future__ import annotations + +import json +from pathlib import Path + +from shapely.geometry import shape +from shapely.geometry.base import BaseGeometry +from shapely.ops import transform, unary_union + +from _lib.gb.darnibole import GEOM_SOURCE as DARNIBOLE_GEOM_SOURCE +from _lib.gb.darnibole import boundary_bng as darnibole_boundary_bng + +ROOT = Path(__file__).resolve().parents[3] +COUNTRIES_GEOJSON = ROOT / "raw" / "gb" / "ons" / "countries.geojson" +COUNTIES_GEOJSON = ROOT / "raw" / "gb" / "ons" / "counties.geojson" + +# file_number → the ONS country whose polygon is the GI's territory. +GB_COUNTRY_TERRITORY: dict[str, str] = { + "PDO-GB-A1585": "England", # English PDO + "PGI-GB-A1589": "England", # English Regional PGI + "PDO-GB-A1587": "Wales", # Welsh PDO + "PGI-GB-A1590": "Wales", # Welsh Regional PGI +} + +# file_number → the ONS counties/unitary authorities whose union is the +# GI's territory. +GB_COUNTY_TERRITORY: dict[str, tuple[str, ...]] = { + "PDO-GB-02365": ("East Sussex", "West Sussex", "Brighton and Hove"), +} + +# file_number → GIs whose boundary is reconstructed rather than published. +GB_APPROXIMATE: tuple[str, ...] = ("PDO-GB-N1636",) # Darnibole + + +def _load_by_name(path: Path, name_field: str) -> dict[str, BaseGeometry]: + if not path.exists(): + return {} + fc = json.loads(path.read_text(encoding="utf-8")) + out: dict[str, BaseGeometry] = {} + for feat in fc.get("features") or []: + props = feat.get("properties") or {} + name = props.get(name_field) + geom = feat.get("geometry") + if not name or not geom: + continue + g = shape(geom) + if g is not None and not g.is_empty: + out[str(name)] = g + return out + + +class GBPolygonIndex: + """In-memory polygon index for GB records, backed by ONS boundaries.""" + + def __init__( + self, + countries_geojson: Path | None = None, + counties_geojson: Path | None = None, + ) -> None: + cpath = countries_geojson or COUNTRIES_GEOJSON + ypath = counties_geojson or COUNTIES_GEOJSON + # ONS publishes these layers with a year-suffixed name field + # (CTRY25NM / CTYUA25NM); accept whichever suffix is cached so a + # boundary refresh to a later vintage doesn't need a code change. + self._countries = self._load_any(cpath, "CTRY", "NM") + self._counties = self._load_any(ypath, "CTYUA", "NM") + self._darnibole_4326: BaseGeometry | None = None + + @staticmethod + def _load_any(path: Path, prefix: str, suffix: str) -> dict[str, BaseGeometry]: + if not path.exists(): + return {} + fc = json.loads(path.read_text(encoding="utf-8")) + feats = fc.get("features") or [] + field = "" + for feat in feats: + for key in (feat.get("properties") or {}): + if key.startswith(prefix) and key.endswith(suffix) and not key.endswith("NMW"): + field = key + break + if field: + break + return _load_by_name(path, field) if field else {} + + @property + def n_countries(self) -> int: + return len(self._countries) + + @property + def n_counties(self) -> int: + return len(self._counties) + + def darnibole_polygon(self) -> BaseGeometry | None: + """Approximate Darnibole boundary, reprojected BNG → WGS84.""" + if self._darnibole_4326 is None: + try: + from pyproj import Transformer + except ImportError: # pragma: no cover - pyproj is a hard dep of stage 04 + return None + tr = Transformer.from_crs("EPSG:27700", "EPSG:4326", always_xy=True) + self._darnibole_4326 = transform( + lambda x, y, z=None: tr.transform(x, y), darnibole_boundary_bng() + ) + return self._darnibole_4326 + + def resolve(self, file_number: str) -> tuple[BaseGeometry | None, str, dict]: + """Resolve geometry for one GB record by file_number. + Returns (geometry, geom_source, stats).""" + fn = (file_number or "").strip() + + country = GB_COUNTRY_TERRITORY.get(fn) + if country: + geom = self._countries.get(country) + if geom is not None and not geom.is_empty: + return geom, "ons-country", {"matched": -1, "unmatched": 0, "unit": country} + return None, "stub-no-geometry", {"matched": 0, "unmatched": 1, "unit": country} + + counties = GB_COUNTY_TERRITORY.get(fn) + if counties: + polys = [ + self._counties[name] for name in counties + if name in self._counties and not self._counties[name].is_empty + ] + stats = { + "matched": -1 if polys else 0, + "unmatched": len(counties) - len(polys), + "members": len(counties), + "resolved": len(polys), + } + if polys: + return unary_union(polys), "ons-county-union", stats + return None, "stub-no-geometry", stats + + if fn in GB_APPROXIMATE: + geom = self.darnibole_polygon() + if geom is not None and not geom.is_empty: + return geom, DARNIBOLE_GEOM_SOURCE, { + "matched": -1, "unmatched": 0, "approximate": True, + } + return None, "stub-no-geometry", {"matched": 0, "unmatched": 1} + + return None, "stub-no-geometry", {"matched": 0, "unmatched": 0} + + +def is_approximate(file_number: str) -> bool: + """True when the GI's boundary is a reconstruction, not a published one.""" + return (file_number or "").strip() in GB_APPROXIMATE diff --git a/scripts/_lib/gb/region.py b/scripts/_lib/gb/region.py new file mode 100644 index 0000000..b60da2e --- /dev/null +++ b/scripts/_lib/gb/region.py @@ -0,0 +1,59 @@ +"""GB region facet — the home nation each UK wine GI is demarcated to. + +The UK wines register is small (6 registered GIs) and every product +specification states its territory in a single `DEMARCATION:` / +`Demarcation:` field whose value is a home nation: + + - English PDO (PDO-GB-A1585) → England + - English Regional PGI (PGI-GB-A1589) → England + - Welsh PDO (PDO-GB-A1587) → Wales + - Welsh Regional PGI (PGI-GB-A1590) → Wales + - Sussex PDO (PDO-GB-02365) → England (East + West Sussex) + - Darnibole PDO (PDO-GB-N1636) → England (a 5 ha single + vineyard in Cornwall) + +Sussex and Darnibole sit geographically inside the English PDO's +territory but are **not** sub-denominations of it: each is a +first-class PDO in its own right on the UK register (Darnibole's +specification says so explicitly — "It qualified under the 'English' +PDO last year, but is considered unique and sufficiently different so +as to merit its own PDO"). v1 therefore models the corpus flat, the +way the CZ podoblasti are modelled as siblings of Čechy / Morava. + +Region labels follow the AT/IT/ES/SI/HR/HU/RO/BG/GR/DE/SK/CZ/NL/MT +convention — shown in their native form, not gettext-translated. +""" + +from __future__ import annotations + +_REGION_BY_FILE_NUMBER: dict[str, str] = { + "PDO-GB-A1585": "England", + "PGI-GB-A1589": "England", + "PDO-GB-A1587": "Wales", + "PGI-GB-A1590": "Wales", + "PDO-GB-02365": "England", + "PDO-GB-N1636": "England", +} + +# Fallback for a GI added to the register after this table was written: +# the specification's own DEMARCATION field, normalised. +_DEMARCATION_TO_REGION: dict[str, str] = { + "england": "England", + "wales": "Wales", + "scotland": "Scotland", + "northern ireland": "Northern Ireland", + "east and west sussex": "England", +} + + +def derive_region(record: dict) -> str: + fn = (record.get("file_number") or "").strip() + if fn in _REGION_BY_FILE_NUMBER: + return _REGION_BY_FILE_NUMBER[fn] + demarcation = (record.get("demarcation") or "").strip().lower() + if demarcation in _DEMARCATION_TO_REGION: + return _DEMARCATION_TO_REGION[demarcation] + for key, region in _DEMARCATION_TO_REGION.items(): + if key in demarcation: + return region + return "United Kingdom" diff --git a/scripts/_lib/gb/spec.py b/scripts/_lib/gb/spec.py new file mode 100644 index 0000000..24a730c --- /dev/null +++ b/scripts/_lib/gb/spec.py @@ -0,0 +1,456 @@ +"""Parsers for the three UK wine product-specification layouts. + +Unlike the EUR-Lex countries, the UK register does not publish one +template — the six specifications come in three shapes, because they +were written under three different regimes: + +1. **`defra-pfn-2011`** — the four December-2011 DEFRA specifications + (English PDO, Welsh PDO, English Regional PGI, Welsh Regional PGI). + A header block:: + + PROTECTED NAME: ENGLISH + DEMARCATION: ENGLAND + + then one `PART n: ` block per grapevine category + (`STILL WINE`, `QUALITY SPARKLING WINE`, and a closing `GENERAL + PROVISIONS` part that carries no wine). Each wine part opens with the + link narrative — latitude, growing season, diurnal range, acidity — + followed by `SPECIFICATION` and a run of upper-case subsections, of + which `VINE VARIETIES` and `MAXIMUM YIELDS` are the ones we read. + The still-wine variety roster is semicolon-separated; the sparkling + one is a short line-per-variety list. + +2. **`defra-pfn-application`** — Darnibole, the 2017 single-vineyard + PDO, filed on the EU application form: a numbered outline (`1. Details + of protection` … `7. Demarcated area`) with lettered sub-items. It has + **no link/terroir section** — the file stops after the demarcated area + and its plan — so the terroir narrative is taken from `7 b) Definition + of the demarcated area`, which is where the slate subsoil, the slope + and the aspect are actually described. + +3. **`uk-gi-single-document`** — Sussex, the only registration made under + the post-Brexit UK scheme (2022). A numbered EU-style single document + (`1. Applicant(s)` … `13. Inspection and certification`) with a proper + `9. Link` section carrying `9.1` natural + human factors, `9.2` + characteristics and `9.3` the causal link. Varieties live in + `7.2 Viticulture practices`, listed separately for sparkling and still. + +All three are Open Government Licence v3.0, © Crown copyright. + +Every UK specification lists varieties as a flat roster with no +principal/accessory split (the same shape as PT/IT/HR/BG/SK), so stage 02 +resolves every match as `principal`. +""" + +from __future__ import annotations + +import re + +# ---------------------------------------------------------------- shared + +# `PART 2: QUALITY SPARKLING WINE` → the shared style-taxonomy slug. +# Order matters: the most specific pattern must win, so "QUALITY +# SPARKLING WINE" is tested before the bare "SPARKLING WINE". +PART_STYLE_MARKERS: tuple[tuple[re.Pattern, str], ...] = ( + (re.compile(r"quality\s+sparkling\s+wine", re.I), "sparkling-quality"), + (re.compile(r"semi[\s-]?sparkling|pearl\s+wine", re.I), "semi-sparkling"), + (re.compile(r"sparkling\s+wine", re.I), "sparkling"), + (re.compile(r"liqueur\s+wine", re.I), "vin-de-liqueur"), + (re.compile(r"\bstill\s+wine", re.I), ""), # colour comes from the grapes +) + +COLOUR_BY_KEYWORD: dict[str, str] = { + "red wine": "red", "red wines": "red", + "white wine": "white", "white wines": "white", + "rosé wine": "rose", "rose wine": "rose", "rosé wines": "rose", + "rosé": "rose", +} + +# Section titles that are never a variety roster, guarding the loose +# keyword match used by the numbered templates. +_GRAPE_TITLE_BLOCKLIST = ("proof of origin", "labelling", "inspection") + +# Boilerplate lines inside a variety block. +_GRAPE_LINE_DROP = ( + "shall be made from the following", + "the following grape varieties", + "permitted grape vine varieties", + "vine varieties", + "grape varieties", + "grape variety", + "the vineyard owner must keep", + "are permitted within", +) + +# Real typos in the source documents. The corpus rule is to fold source +# typos rather than hand-edit the regulator's text (see the INAO cahier +# precedent) — but these two are *structural*: a stray comma and a line +# break split one variety name into two, which no grape-alias entry can +# repair because the fragments are separate candidates by then. +# - Sussex §7.2 still-wine roster reads "… Pinot Noir, Pinot Noir, +# Précoce, Regent …" for "Pinot Noir, Pinot Noir Précoce, Regent". +# - The same roster hyphenates across a line break: "Müller- Thurgau". +_SOURCE_TYPO_REPAIRS: tuple[tuple[re.Pattern, str], ...] = ( + (re.compile(r"Pinot\s+Noir\s*,\s*Précoce", re.I), "Pinot Noir Précoce"), + (re.compile(r"Müller-\s+Thurgau", re.I), "Müller-Thurgau"), + # - The DEFRA rosters run alphabetically (… Gamaret; Gamay; + # Garanoir; Gewurztraminer …) but drop the semicolon between + # Gamay and Garanoir, merging two Swiss-crossing entries into one. + (re.compile(r"\bGamay\s+Garanoir\b", re.I), "Gamay; Garanoir"), +) + + +def repair_source_typos(text: str) -> str: + for pattern, replacement in _SOURCE_TYPO_REPAIRS: + text = pattern.sub(replacement, text) + return text + + +def normalise_text(text: str) -> str: + """Fold form feeds to newlines and squeeze intra-line whitespace. + + `pdftotext -layout` emits a form feed at every page break; a section + that starts immediately after one would otherwise not match a + line-anchored header regex. + """ + text = text.replace("\x0c", "\n").replace("’", "'").replace("“", '"') + text = text.replace("”", '"').replace("„", '"').replace("‚", "'") + lines = [re.sub(r"[ \t\r\v]+", " ", ln).rstrip() for ln in text.splitlines()] + return "\n".join(lines) + + +def _clean_block(text: str) -> str: + lines = [ln.strip() for ln in (text or "").splitlines()] + out = [ln for ln in lines if ln] + return "\n".join(out).strip() + + +# ------------------------------------------------- template 1: DEFRA 2011 + +_HEADER_FIELD_RE = re.compile( + r"^\s*(PROTECTED NAME|DEMARCATION)\s*:\s*(.+?)\s*$", re.M) +_PART_RE = re.compile(r"^\s*PART\s+(\d+)\s*:\s*(.+?)\s*$", re.M) + +# An upper-case subsection header inside a PART. Allows the trailing +# "…: 80 hl/ha" value form of MAXIMUM YIELDS, and the two-line wrap that +# `MINIMUM NATURAL, ACTUAL AND TOTAL ALCOHOLIC STRENGTHS AND\nENRICHMENT` +# produces, by matching each physical line independently. +_UPPER_HEADER_RE = re.compile( + r"^(?P[A-Z][A-Z0-9 ,\-/&\.\(\)']{3,}?)\s*(?::\s*(?P<value>.*))?$") + + +def _is_upper_header(line: str) -> bool: + s = line.strip() + if len(s) < 4 or s.startswith(("-", "(", "\u2022", "\uf0b7")): + return False + head = s.split(":", 1)[0] + letters = [c for c in head if c.isalpha()] + if len(letters) < 3: + return False + if any(c.islower() for c in letters): + return False + # A wrapped sentence in caps is still a header here; a bullet is not. + return bool(_UPPER_HEADER_RE.match(s)) + + +def parse_defra_pfn_2011(text: str) -> dict: + """Parse a December-2011 DEFRA product specification.""" + text = normalise_text(text) + fields = {m.group(1).lower().replace(" ", "_"): m.group(2).strip() + for m in _HEADER_FIELD_RE.finditer(text)} + + parts: list[dict] = [] + matches = list(_PART_RE.finditer(text)) + for i, m in enumerate(matches): + end = matches[i + 1].start() if i + 1 < len(matches) else len(text) + body = text[m.end():end] + parts.append({"num": m.group(1), "title": m.group(2).strip(), "body": body}) + + varieties: list[str] = [] + yields: list[str] = [] + narratives: list[str] = [] + part_titles: list[str] = [] + + for part in parts: + title = part["title"] + if "GENERAL PROVISIONS" in title.upper(): + continue + part_titles.append(title) + body = part["body"] + # The link narrative is everything before the SPECIFICATION marker. + spec_split = re.split(r"^\s*SPECIFICATION\s*$", body, maxsplit=1, flags=re.M) + narrative = _clean_block(spec_split[0]) + if narrative: + narratives.append(f"{title.title()}\n{narrative}") + subsections = _split_upper_sections(spec_split[1] if len(spec_split) > 1 else "") + for stitle, sbody in subsections: + up = stitle.upper() + if up.startswith("VINE VARIETIES"): + varieties.append(sbody) + elif up.startswith("MAXIMUM YIELD"): + value = sbody.strip() or stitle.split(":", 1)[-1].strip() + yields.append(f"{title.title()}: {value}" if value else "") + + return { + "template": "defra-pfn-2011", + "protected_name": fields.get("protected_name", ""), + "demarcation": fields.get("demarcation", ""), + "part_titles": part_titles, + "roles": { + "geo_area": fields.get("demarcation", ""), + # Blank-line separated: each PART's roster is its own run, so + # the last name of one ("… Zweigeltrebe") is never joined to + # the first of the next ("Acolon …") as a line wrap. + "grape_varieties": "\n\n".join(v for v in varieties if v), + "link_to_terroir": "\n\n".join(narratives), + "description": narratives[0] if narratives else "", + "yields": "\n".join(y for y in yields if y), + }, + } + + +def _split_upper_sections(text: str) -> list[tuple[str, str]]: + """Split a SPECIFICATION body on its upper-case subsection headers.""" + lines = text.splitlines() + marks: list[tuple[int, str]] = [ + (i, ln.strip()) for i, ln in enumerate(lines) if _is_upper_header(ln) + ] + out: list[tuple[str, str]] = [] + for j, (idx, title) in enumerate(marks): + end = marks[j + 1][0] if j + 1 < len(marks) else len(lines) + inline = title.split(":", 1)[1].strip() if ":" in title else "" + body = "\n".join(lines[idx + 1:end]) + out.append((title, _clean_block(f"{inline}\n{body}" if inline else body))) + return out + + +# --------------------------------------- templates 2+3: numbered outlines + +# `3. Product details`, `7 a). NUTS Area`, `9.1 Details of the …` +_NUM_HEADER_RE = re.compile( + r"^\s*(?P<num>\d+(?:\s*[a-z]\))?(?:\.\d+)*)\s*[\.\)]?\s+(?P<title>[A-Za-z][^\n]*?)\s*:?\s*$" +) +# `a) Category`, `b) Description`, `c) Analytic characteristics:` +_ALPHA_HEADER_RE = re.compile(r"^\s*(?P<num>[a-z])\)\s*(?P<title>[A-Za-z][^\n]*?)\s*:?\s*$") + + +def _split_numbered(text: str) -> list[tuple[str, str, str]]: + """Return [(number, title, body)] for a numbered/lettered outline.""" + lines = text.splitlines() + marks: list[tuple[int, str, str]] = [] + for i, ln in enumerate(lines): + m = _NUM_HEADER_RE.match(ln) or _ALPHA_HEADER_RE.match(ln) + if not m: + continue + title = m.group("title").strip() + # Analytical rows ("1. Actual and Total Alcoholic Strengths: …") + # and prose sentences are not headers. + if len(title) > 90 or title.endswith((".", ",")): + continue + marks.append((i, re.sub(r"\s+", "", m.group("num")), title)) + out: list[tuple[str, str, str]] = [] + for j, (idx, num, title) in enumerate(marks): + end = marks[j + 1][0] if j + 1 < len(marks) else len(lines) + out.append((num, title, _clean_block("\n".join(lines[idx + 1:end])))) + return out + + +def _pick(sections: list[tuple[str, str, str]], keywords: tuple[str, ...], + blocklist: tuple[str, ...] = ()) -> str: + for kw in keywords: + for _num, title, body in sections: + tlow = title.lower() + if kw not in tlow or any(b in tlow for b in blocklist): + continue + if body.strip(): + return body + return "" + + +def _pick_with_children(sections: list[tuple[str, str, str]], + keywords: tuple[str, ...]) -> str: + """Section body plus every `n.x` child — Sussex's `9. Link` is an + empty header whose content is entirely in 9.1 / 9.2 / 9.3.""" + for kw in keywords: + for num, title, body in sections: + if kw not in title.lower(): + continue + chunks = [body] if body.strip() else [] + for cnum, ctitle, cbody in sections: + if cnum.startswith(f"{num}.") and cbody.strip(): + chunks.append(f"{ctitle}\n{cbody}") + if chunks: + return "\n\n".join(chunks) + return "" + + +def parse_numbered_spec(text: str, template: str) -> dict: + """Parse Darnibole's application form or Sussex's single document.""" + text = normalise_text(text) + sections = _split_numbered(text) + + demarcation = "" + m = re.search(r"^\s*Demarcation\s*:\s*(.+?)\s*$", text, re.M | re.I) + if m: + demarcation = m.group(1).strip() + + geo_area = _pick(sections, ( + "definition of the demarcated area", "demarcated area", + "geographical area", "delimited area", + )) + link = _pick_with_children(sections, ( + "link between the characteristics", "link", "details of the geographical area", + )) + # Darnibole has no link section at all: its terroir narrative — ancient + # slate subsoil, the steep south-facing slope, the thermal band — is in + # the demarcated-area definition. + if not link: + link = geo_area + grapes = _pick(sections, ( + "viticulture practices", "wine grape variety", "grape variety", + "grape varieties", "vine variet", + ), _GRAPE_TITLE_BLOCKLIST) + description = _pick_with_children(sections, ( + "description of the wine", "description", + )) + yields = _pick(sections, ("maximum yield", "harvest yield")) + categories = _pick(sections, ("category of the grapevine products", "category")) + + return { + "template": template, + "protected_name": _pick(sections, ("name of product to be registered", + "name(s) to be registered")), + # Parenthesised deliberately: without it Python binds the + # conditional to the whole `or` expression, so an explicitly + # stated "Demarcation:" line is discarded whenever the area + # section happens to be empty. + "demarcation": demarcation or (geo_area.splitlines()[0] if geo_area else ""), + "part_titles": [ln.strip() for ln in categories.splitlines() if ln.strip()], + "roles": { + "geo_area": geo_area, + "grape_varieties": grapes, + "link_to_terroir": link, + "description": description, + "yields": yields, + }, + } + + +# ------------------------------------------------------------- dispatcher + +def detect_template(text: str) -> str: + head = normalise_text(text)[:4000] + if _HEADER_FIELD_RE.search(head) and _PART_RE.search(normalise_text(text)): + return "defra-pfn-2011" + if re.search(r"^\s*1\.\s*Details of protection", head, re.M | re.I): + return "defra-pfn-application" + return "uk-gi-single-document" + + +def parse_spec(text: str) -> dict: + template = detect_template(text) + if template == "defra-pfn-2011": + return parse_defra_pfn_2011(text) + return parse_numbered_spec(text, template) + + +# ------------------------------------------------------------ grape lines + +_ITEM_SPLIT_RE = re.compile(r"[;\n]|,(?![^()]*\))|\s+and\s+(?=[A-Z])") + +# "Permitted Grape Vine Varieties: Chardonnay, Pinot Noir, …" — the roster +# starts after the label, on the same line. +_ROSTER_LABEL_RE = re.compile( + r"^.*?(?:permitted grape vine varieties|shall be made from the following" + r"|are permitted within[^:]*|following grape varieties)\s*:?\s*", re.I) +# "100% Bacchus." — Darnibole states its single variety as a proportion. +_PROPORTION_RE = re.compile(r"^\s*\d+(?:[.,]\d+)?\s*%\s*") + + +def _strip_bullet(line: str) -> str: + """Drop a leading list bullet. `pdftotext` renders the Wingdings + bullet the DEFRA specifications use as U+F0B7 (private-use area).""" + return re.sub(r"^[\s\u2022\u00b7\uf0b7\-\u2013]+", "", line).strip() + + +def _looks_like_roster(line: str) -> bool: + """True when a line is a list of variety names rather than prose. + + Every UK roster is a run of short names; the surrounding prose ("The + vineyard owner must keep detailed records of yield and size of each + parcel…") is long-winded even where it contains commas. Judging by + the *shape* of the split items keeps both apart without needing a + per-document rule. + """ + stripped = _ROSTER_LABEL_RE.sub("", _strip_bullet(line), count=1) + items = [i.strip(" .;,:") for i in _ITEM_SPLIT_RE.split(stripped)] + items = [i for i in items if i] + if not items: + return False + short = [i for i in items if len(i) <= 40 and len(i.split()) <= 4] + if len(items) == 1: + # A one-per-line sparkling roster ("Chardonnay") or "100% Bacchus". + only = _PROPORTION_RE.sub("", items[0]) + return bool(short) and bool(re.match(r"^[A-Z\u00c0-\u00dc]", only)) + return len(short) / len(items) >= 0.75 + + +def _roster_runs(text: str) -> list[str]: + """Group contiguous roster lines into runs and rejoin each correctly. + + A roster wrapped across lines by the PDF layout ("… Black Hamburg; + Blau\nPortugueser; Blauburger …") must be joined with a *space*, or + the wrap splits "Blau Portugueser" into two varieties. A roster + written one-name-per-line (the bulleted sparkling lists) has no + separators at all, so its lines must be joined with a *separator* + instead. Which of the two a run is gets decided by whether its lines + actually carry `;` / `,`. + """ + runs: list[list[str]] = [] + current: list[str] = [] + for line in text.splitlines(): + if _looks_like_roster(line): + current.append(_ROSTER_LABEL_RE.sub("", _strip_bullet(line), count=1)) + elif current: + runs.append(current) + current = [] + if current: + runs.append(current) + + out: list[str] = [] + for run in runs: + separated = sum(1 for ln in run if ";" in ln or "," in ln) + joiner = " " if separated * 2 >= len(run) else "; " + out.append(joiner.join(run)) + return out + + +def grape_candidates(section_text: str) -> list[str]: + """Split a UK variety roster into individual candidate names. + + The still-wine rosters are semicolon-separated and run to ~85 names; + the sparkling ones are bulleted one-per-line; Sussex's are + comma-separated with a trailing "X and Y". Parenthesised synonym + tails — "Fruhburgunder (Pinot Noir Precoce)", "Rulander (Synonyms: + Pinot Gris, Pinot Grigio)" — are stripped, since the head name is the + one the regulator authorises. Prose lines are skipped entirely, so + the record-keeping and yield-dispensation paragraphs around Sussex's + two rosters never reach the matcher. + """ + text = repair_source_typos(section_text or "") + out: list[str] = [] + for run in _roster_runs(text): + for raw in _ITEM_SPLIT_RE.split(run): + item = raw.strip().strip(".;,: ") + if not item: + continue + if any(d in item.lower() for d in _GRAPE_LINE_DROP): + continue + item = _PROPORTION_RE.sub("", item) + item = re.sub(r"\s*\([^)]*\)\s*", " ", item).strip() + if not item or len(item) > 40 or len(item.split()) > 4: + continue + if not re.search(r"[A-Za-z]", item): + continue + out.append(item) + return out diff --git a/scripts/_lib/geometry_outlier_overrides.json b/scripts/_lib/geometry_outlier_overrides.json index 96fd19e..cca678e 100644 --- a/scripts/_lib/geometry_outlier_overrides.json +++ b/scripts/_lib/geometry_outlier_overrides.json @@ -28,14 +28,16 @@ ], "_match_tol_km": 1.5, "_area_tol_pct": 20.0, - "clip": { "garda": { "file_number": "PDO-IT-A1320", "geom_source": "figshare-pdo", "drop": [ { - "centroid_latlon": [45.0667, 7.5236], + "centroid_latlon": [ + 45.0667, + 7.5236 + ], "area_km2": 29.517, "reason": "Fragment 225 km W of Lake Garda, near Avigliana (Piedmont). Lies 100% inside the Piemonte regional DOC (PDO-IT-A1224). The Garda disciplinare restricts the DOC to the provinces of Brescia, Mantova and Verona — no Piedmont component exists. Bétard 2022 EU_PDO.gpkg cross-attribution error.", "verified": "2026-05-22" @@ -47,7 +49,10 @@ "geom_source": "figshare-pdo", "drop": [ { - "centroid_latlon": [45.0029, 8.2519], + "centroid_latlon": [ + 45.0029, + 8.2519 + ], "area_km2": 17.305, "reason": "Fragment 205 km SW of the Adige valley, in the Monferrato (Piedmont). Lies 100% inside Grignolino d'Asti / Barbera del Monferrato / Monferrato DOCs. Valdadige DOC is the Adige valley (Trento / Bolzano / Verona). Bétard 2022 cross-attribution error — mirror of the Garda case.", "verified": "2026-05-22" @@ -59,13 +64,19 @@ "geom_source": "figshare-pdo", "drop": [ { - "centroid_latlon": [45.9239, 11.1067], + "centroid_latlon": [ + 45.9239, + 11.1067 + ], "area_km2": 5.286, "reason": "Fragment ~215 km NE of the Asti / Monferrato main body, near Trento. Lies 100% inside delle Venezie / Valdadige / Casteller (Trentino). Bétard 2022 cross-attribution error.", "verified": "2026-05-22" }, { - "centroid_latlon": [45.9642, 11.1270], + "centroid_latlon": [ + 45.9642, + 11.127 + ], "area_km2": 4.914, "reason": "Second Trentino fragment, ~218 km NE near Trento, shared verbatim with the other three Monferrato-family DOCs. Bétard 2022 cross-attribution error.", "verified": "2026-05-22" @@ -77,13 +88,19 @@ "geom_source": "figshare-pdo", "drop": [ { - "centroid_latlon": [45.9239, 11.1067], + "centroid_latlon": [ + 45.9239, + 11.1067 + ], "area_km2": 5.286, "reason": "Trentino fragment near Trento (~204 km from the Monferrato main body), shared verbatim across the Monferrato-family DOCs. Bétard 2022 cross-attribution error.", "verified": "2026-05-22" }, { - "centroid_latlon": [45.9642, 11.1270], + "centroid_latlon": [ + 45.9642, + 11.127 + ], "area_km2": 4.914, "reason": "Second Trentino fragment near Trento (~207 km away), shared verbatim across the Monferrato-family DOCs. Bétard 2022 cross-attribution error.", "verified": "2026-05-22" @@ -95,13 +112,19 @@ "geom_source": "figshare-pdo", "drop": [ { - "centroid_latlon": [45.9239, 11.1067], + "centroid_latlon": [ + 45.9239, + 11.1067 + ], "area_km2": 5.286, "reason": "Trentino fragment near Trento (~204 km from the Monferrato main body), shared verbatim across the Monferrato-family DOCs. Bétard 2022 cross-attribution error.", "verified": "2026-05-22" }, { - "centroid_latlon": [45.9642, 11.1270], + "centroid_latlon": [ + 45.9642, + 11.127 + ], "area_km2": 4.914, "reason": "Second Trentino fragment near Trento (~207 km away), shared verbatim across the Monferrato-family DOCs. Bétard 2022 cross-attribution error.", "verified": "2026-05-22" @@ -113,13 +136,19 @@ "geom_source": "figshare-pdo", "drop": [ { - "centroid_latlon": [45.9239, 11.1067], + "centroid_latlon": [ + 45.9239, + 11.1067 + ], "area_km2": 5.286, "reason": "Trentino fragment near Trento (~204 km from the Monferrato main body), shared verbatim across the Monferrato-family DOCs. Bétard 2022 cross-attribution error.", "verified": "2026-05-22" }, { - "centroid_latlon": [45.9642, 11.1270], + "centroid_latlon": [ + 45.9642, + 11.127 + ], "area_km2": 4.914, "reason": "Second Trentino fragment near Trento (~207 km away), shared verbatim across the Monferrato-family DOCs. Bétard 2022 cross-attribution error.", "verified": "2026-05-22" @@ -127,13 +156,16 @@ ] } }, - "whitelist": { "madeira": "Atlantic archipelago — the DOP spans Madeira, Porto Santo and the Desertas; disjoint island parts are expected.", "madeirense": "Atlantic archipelago — same island footprint as Madeira DOP.", "islas-canarias": "Atlantic archipelago — the DO covers all seven Canary Islands; disjoint island parts are expected.", "sicilia": "Sicily plus its minor-island groups (Pantelleria, Aeolian, Egadi, Ustica) — disjoint island parts are expected.", "cava": "Spanish sparkling-wine DO produced in legally listed municipalities across Catalonia, La Rioja, Aragon, Navarra, the Basque Country, Extremadura and Valencia — genuinely multi-region and disjoint.", - "saale-unstrut": "German Anbaugebiet legally extended to Werder (Havel) in Brandenburg — the Werderaner Wachtelberg, Germany's northernmost vineyard. BLE Produktspezifikation §11.3 names the Brandenburg-side Gemarkungen Werder (Havel), Phöben, Plessow and Neu Töplitz; the ~117 km² polygon ~150 km NE of the main Sachsen-Anhalt body (~52.38°N, 12.88°E) is therefore legitimate, not a Bétard cross-attribution." + "saale-unstrut": "German Anbaugebiet legally extended to Werder (Havel) in Brandenburg — the Werderaner Wachtelberg, Germany's northernmost vineyard. BLE Produktspezifikation §11.3 names the Brandenburg-side Gemarkungen Werder (Havel), Phöben, Plessow and Neu Töplitz; the ~117 km² polygon ~150 km NE of the main Sachsen-Anhalt body (~52.38°N, 12.88°E) is therefore legitimate, not a Bétard cross-attribution.", + "english-wine": "Offshore England — the detached parts are the Isles of Scilly (~40-54 km off Land's End) plus other coastal islands of the ONS England country polygon. The PDO is demarcated to ENGLAND, so they are legitimately in the area; Scilly has commercial vineyard plantings.", + "english-regional-wine": "Offshore England — same ONS England footprint as the English PDO.", + "welsh-wine": "Offshore Wales — the detached part is off Anglesey (Holy Island area) in the ONS Wales country polygon. The PDO is demarcated to WALES, so the island parts are legitimately in the area.", + "welsh-regional-wine": "Offshore Wales — same ONS Wales footprint as the Welsh PDO." } } diff --git a/scripts/_lib/grape_corpus.py b/scripts/_lib/grape_corpus.py index 56de2c6..b32ee2f 100644 --- a/scripts/_lib/grape_corpus.py +++ b/scripts/_lib/grape_corpus.py @@ -50,6 +50,7 @@ ("ch", ROOT / "raw" / "ch" / "dokumente-extracted"), # Malta — source language is English (EU single documents are EN). ("en", ROOT / "raw" / "mt" / "dokumente-extracted"), + ("en", ROOT / "raw" / "gb" / "specs-extracted"), ("hu", ROOT / "raw" / "hu" / "dokumentumok-extracted"), ("nl", ROOT / "raw" / "nl" / "dokumenten-extracted"), # Belgium — per-record source_lang (nl Flemish / fr Walloon); nl covers diff --git a/scripts/_lib/grape_lexicon.py b/scripts/_lib/grape_lexicon.py index 2f11e46..5490c58 100644 --- a/scripts/_lib/grape_lexicon.py +++ b/scripts/_lib/grape_lexicon.py @@ -1758,6 +1758,34 @@ def _ends_with_colour_word(name: str) -> bool: # Mariafeld clone of Pinot noir (VIVC #9279 carries the synonym). # A clone, not a variety — fold; the record keeps its spelling. "mario-feld": "pinot-noir", + # --- GB (DEFRA / GOV.UK product specifications) -------------------- + # The English + Welsh rosters are the corpus's longest single lists + # (~85 names) and spell several varieties in forms no other country + # uses. VIVC ids verified against the catalogue 2026-09. + "blau-portugueser": "blauer-portugieser", # VIVC #9620 PORTUGIESER BLAU — DEFRA drops the -er + "elbling-white": "elbling", # VIVC #3865 ELBLING WEISS — English gloss of Weißer Elbling + "cascade": "cascade", # VIVC #2139 CASCADE (Seibel 13053) — French-American hybrid + "madeleine-angevine": "madeleine-angevine", # VIVC #7062 MADELEINE ANGEVINE + "madeleine-sylvaner": "madeleine-sylvaner", # VIVC #7070 MADELEINE SYLVANER + # VIVC #12931 VELTLINER ROT — a distinct cultivar from Frühroter + # Veltliner (`velteliner-rouge-precoce`), which the fuzzy matcher + # scores at 88; keep them apart. + "roter-veltliner": "roter-veltliner", + "veltliner-rot": "roter-veltliner", + "senator": "fruehgipfler", # VIVC #4269 FRUEHGIPFLER — SENATOR is its synonym + "fruehgipfler": "fruehgipfler", + "fruhgipfler": "fruehgipfler", + # VIVC lists TRIOMPHE against both DODRELYABI (#3616) and TRIOMPHE + # D'ALSACE (#12650); in a UK roster it is the Alsace hybrid, which is + # widely planted in England and Wales. + "triomphe": "triomphe-dalsace", + "triomphe-dalsace": "triomphe-dalsace", + "triomphe-d-alsace": "triomphe-dalsace", + # No VIVC accession. A black Russian/Caucasus cultivar grown outdoors + # in the UK, named after the cosmonaut; described as a black grape by + # the RHS plant register (rhs.org.uk/plants/184101) and by English + # vineyard sources. + "gagarin-blue": "gagarin-blue", } # Default colour for each well-known variety. When the parser extracts a @@ -1772,6 +1800,14 @@ def _ends_with_colour_word(name: str) -> bool: # canonical, and the distinct mutations (`-blanc`, `-gris`) stay separate # because their suffix ≠ the default. DEFAULT_COLOUR: dict[str, str] = { + # --- GB (DEFRA / GOV.UK product specifications) -------------------- + "cascade": "noir", # VIVC #2139 + "madeleine-angevine": "blanc", # VIVC #7062 + "madeleine-sylvaner": "blanc", # VIVC #7070 + "roter-veltliner": "rose", # VIVC #12931 VELTLINER ROT + "fruehgipfler": "blanc", # VIVC #4269 + "triomphe-dalsace": "noir", # VIVC #12650 + "gagarin-blue": "noir", # no VIVC accession; RHS plant register "chardonnay": "blanc", "chenin": "blanc", "muscadelle": "blanc", diff --git a/scripts/_lib/map_template.py b/scripts/_lib/map_template.py index d157a92..44de0b5 100644 --- a/scripts/_lib/map_template.py +++ b/scripts/_lib/map_template.py @@ -142,6 +142,11 @@ def build_labels(_: Callable[[str], str]) -> dict[str, str]: "(commune de {commune}, {source})." ), "geom_approx_cadastre_source_label": _("cadastre.data.gouv.fr"), + "geom_approx_pdo_plan": _( + "Aire approchée — reconstituée à partir des références " + "parcellaires du plan de l'aire délimitée annexé au cahier des " + "charges ; ce n'est pas une limite officielle." + ), "stack_header": _("{n} appellations à ce point"), "stack_cycle_hint": _("Cliquer à nouveau pour parcourir les autres"), "src_cahier": _("Cahier des charges (BO Agri, PDF)"), @@ -160,6 +165,8 @@ def build_labels(_: Callable[[str], str]) -> dict[str, str]: "src_eambrosia_id": _("Numéro de dossier"), "src_cantonal_reglement": _("Règlement cantonal sur la vigne et le vin"), "src_ofag_repertoire": _("Répertoire suisse des AOC (OFAG/BLW)"), + "src_gov_uk_spec": _("Cahier des charges (GOV.UK, PDF)"), + "src_gov_uk_register": _("Registre des IG du Royaume-Uni"), "legend_h": _("Légende couleurs"), "legend_bassin_h": _("Bassin viticole"), "legend_area_hint": _("Plus l'aire est petite, plus la teinte est dense."), @@ -327,6 +334,7 @@ def build_region_labels(_: Callable[[str], str]) -> dict[str, str]: "nl": "\U0001F1F3\U0001F1F1", # 🇳🇱 "mt": "\U0001F1F2\U0001F1F9", # 🇲🇹 "cy": "\U0001F1E8\U0001F1FE", # 🇨🇾 + "gb": "\U0001F1EC\U0001F1E7", # 🇬🇧 } @@ -357,6 +365,7 @@ def build_country_labels(_: Callable[[str], str]) -> dict[str, str]: "nl": _("Pays-Bas"), "mt": _("Malte"), "cy": _("Chypre"), + "gb": _("Royaume-Uni"), } diff --git a/scripts/_lib/summaries.py b/scripts/_lib/summaries.py index 06f0199..b60a736 100644 --- a/scripts/_lib/summaries.py +++ b/scripts/_lib/summaries.py @@ -32,10 +32,13 @@ def derive_summary(record: dict) -> str: keep the same SHA — the cache only re-translates AOCs whose summary was previously clipped mid-clause. """ - if record.get("country") in ("es", "pt", "lu"): + if record.get("country") in ("es", "pt", "lu", "gb"): # ES/PT records carry a pre-computed summary; LU records # likewise come pre-summarised from stage 02 (the cahier's - # white-wine description paragraph from section b). + # white-wine description paragraph from section b). GB the same: + # the DEFRA product specifications are not a numbered template at + # all, so there is no section "1"/"I" to fall back on — stage 02 + # derives the blurb from the routed description / link role. return record.get("summary", "") or "" sections = record.get("sections", {}) roles = record.get("section_roles") or {} diff --git a/scripts/audit_gb_coverage.py b/scripts/audit_gb_coverage.py new file mode 100644 index 0000000..b6ef63e --- /dev/null +++ b/scripts/audit_gb_coverage.py @@ -0,0 +1,134 @@ +"""Coverage audit for the United Kingdom corpus. + +Reports, per registered UK wine GI: whether its product specification was +fetched, which parser template read it, how many varieties and styles came +out, whether it carries terroir-source text and extracted terroir facts, +and how its geometry resolves. + +Unlike the eAmbrosia countries there is no stub tier to chase — every +registered UK wine ships a public specification — so the curator queue +here is short by construction: it lists pending applications (names still +in assessment, which stage 00 filters out of the corpus) and any +registered wine whose specification failed to fetch or parse. + + .venv/bin/python scripts/audit_gb_coverage.py + .venv/bin/python scripts/audit_gb_coverage.py --strict # non-zero on gaps +""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "scripts")) + +INDEX_IN = ROOT / "raw" / "gb" / "gov-uk" / "index.json" +GOVUK_MANIFEST = ROOT / "raw" / "gb" / "gov-uk" / "manifest.json" +SPECS_MANIFEST = ROOT / "raw" / "gb" / "specs" / "manifest.json" +EXTRACTED = ROOT / "raw" / "gb" / "specs-extracted" +TERROIR = ROOT / "raw" / "terroir-facts" + + +def _load(path: Path, default): + if not path.exists(): + return default + try: + return json.loads(path.read_text(encoding="utf-8")) + except (ValueError, OSError): + return default + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--strict", action="store_true", + help="exit non-zero when a registered wine is missing " + "a specification, varieties or terroir text") + args = ap.parse_args() + + if not INDEX_IN.exists(): + print(f"error: {INDEX_IN} missing — run scripts/gb/00_fetch_data.py first", + file=sys.stderr) + return 1 + + wines = _load(INDEX_IN, {"wines": []})["wines"] + govuk = _load(GOVUK_MANIFEST, {}) + specs = _load(SPECS_MANIFEST, {}).get("by_slug", {}) + + try: + from _lib.gb.geometry import GBPolygonIndex + gb_polygons = GBPolygonIndex() + except Exception as exc: # noqa: BLE001 - the audit must run without geopandas + print(f"[warn] geometry index unavailable ({exc}); skipping geom column", + file=sys.stderr) + gb_polygons = None + + print(f"United Kingdom — {len(wines)} registered wine GIs\n") + header = (f"{'slug':24s} {'kind':4s} {'region':8s} {'spec':6s} " + f"{'template':24s} {'grapes':>6s} {'styles':>6s} {'link':>6s} " + f"{'facts':>5s} geom") + print(header) + print("-" * len(header)) + + problems: list[str] = [] + n_facts_total = 0 + geom_counts: dict[str, int] = {} + + for w in wines: + slug = w["slug"] + meta = specs.get(slug, {}) + spec_ok = meta.get("status") == "ok" + rec = _load(EXTRACTED / f"{slug}.json", {}) + grapes = len(((rec.get("grapes") or {}).get("details")) or []) + styles = len(rec.get("styles") or []) + link = len(rec.get("link_to_terroir") or "") + facts_doc = _load(TERROIR / f"{slug}.json", {}) + n_facts = len(facts_doc.get("facts") or []) if facts_doc.get("country") == "gb" else 0 + n_facts_total += n_facts + + geom = "-" + if gb_polygons is not None: + _g, geom, _stats = gb_polygons.resolve(w.get("file_number") or "") + geom_counts[geom] = geom_counts.get(geom, 0) + 1 + + print(f"{slug:24s} {w['kind']:4s} {rec.get('region', '?'):8s} " + f"{'ok' if spec_ok else 'MISS':6s} " + f"{rec.get('parser_template', '-'):24s} {grapes:6d} {styles:6d} " + f"{link:6d} {n_facts:5d} {geom}") + + if not spec_ok: + problems.append(f"{slug}: no cached product specification") + elif not rec: + problems.append(f"{slug}: specification cached but not extracted") + else: + if grapes == 0: + problems.append(f"{slug}: no varieties resolved") + if link == 0: + problems.append(f"{slug}: no terroir source text") + + print("\nGeometry: " + ", ".join(f"{k}={v}" for k, v in sorted(geom_counts.items()))) + n_with_facts = sum( + 1 for w in wines + if (_load(TERROIR / f"{w['slug']}.json", {}) or {}).get("facts") + ) + print(f"Terroir facts: {n_facts_total} bullets across " + f"{n_with_facts}/{len(wines)} wines") + + pending = govuk.get("pending_applications") or [] + print(f"\nCurator queue — {len(pending)} pending application(s), " + f"{len(problems)} problem(s)") + for p in pending: + print(f" [pending] {p['name']} ({p['kind']}, applied {p['date_application']}) " + f"— {p['register_url']}") + for p in problems: + print(f" [problem] {p}") + if not pending and not problems: + print(" (nothing queued)") + + return 1 if (args.strict and problems) else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/audit_terroir_facts.py b/scripts/audit_terroir_facts.py index 135a5f1..da1096c 100644 --- a/scripts/audit_terroir_facts.py +++ b/scripts/audit_terroir_facts.py @@ -45,19 +45,33 @@ EXTRACTED_BY_COUNTRY = { "fr": ROOT / "raw" / "inao" / "cahier-extracted", "es": ROOT / "raw" / "es" / "pliegos-extracted", + "gb": ROOT / "raw" / "gb" / "specs-extracted", } WIKI_BY_COUNTRY = { "fr": ROOT / "raw" / "wikipedia" / "aocs" / "fr", "es": ROOT / "raw" / "wikipedia" / "aocs" / "es", + # GB's source language is English (the DEFRA product specifications). + "gb": ROOT / "raw" / "wikipedia" / "aocs" / "en", } LIEN_FIELD_BY_COUNTRY = { "fr": "lien_au_terroir", "es": "link_to_terroir", + "gb": "link_to_terroir", } FUZZY_THRESHOLD = 0.6 BULLET_SOFT_CAP = 140 WIKI_HINT_CHAR_CAP = 1500 +# Each country's stage 02d caps the Wikipedia hint at its own length; the +# audit must reproduce that cap or a perfectly-grounded `wiki` bullet whose +# quote sits past the audit's shorter cap reports as spuriously "eroded". +WIKI_HINT_CAP_BY_COUNTRY = { + "gb": 2800, # scripts/gb/02d_extract_terroir_facts.py +} + + +def wiki_hint_cap(country: str) -> int: + return WIKI_HINT_CAP_BY_COUNTRY.get(country, WIKI_HINT_CHAR_CAP) TOP_RE = re.compile(r"\b([1-9])°\s*[-–]\s*([A-ZÀ-Ý][^\n]{5,80})") SUB_RE = re.compile(r"\b([a-c])\)\s*[-–]?\s*([A-ZÀ-Ý][^\n]{5,80})") @@ -98,9 +112,25 @@ } WIKI_TO_SUBSECTION_ES["interactions"] = WIKI_TO_SUBSECTION_ES["facteurs_naturels"] +# English Wikipedia headings — mirror scripts/gb/02d_extract_terroir_facts.py. +WIKI_TO_SUBSECTION_EN: dict[str, list[str]] = { + "facteurs_naturels": [ + "Geography", "Geology", "Climate", "Soil", "Soils", "Terroir", + "Wine regions", "Regions", "Viticulture", "Vineyards", + ], + "facteurs_humains": [ + "History", "Grapes", "Grape varieties", "Varieties", "Production", + "Winemaking", "Wineries", + ], + "produit": ["Wines", "Styles", "Wine styles", "Types of wine", "Production"], + "interactions": [], +} +WIKI_TO_SUBSECTION_EN["interactions"] = WIKI_TO_SUBSECTION_EN["facteurs_naturels"] + WIKI_TO_SUBSECTION_BY_COUNTRY = { "fr": WIKI_TO_SUBSECTION_FR, "es": WIKI_TO_SUBSECTION_ES, + "gb": WIKI_TO_SUBSECTION_EN, } @@ -149,7 +179,8 @@ def _index_wiki_sections(full: str, headings: list[str]) -> dict[str, str]: def _build_subsection_hint_fr( - sub_key: str, wanted: list[str], section_text: dict[str, str] + sub_key: str, wanted: list[str], section_text: dict[str, str], + char_cap: int = WIKI_HINT_CHAR_CAP, ) -> str: """FR wiki-hint format used by `scripts/02d_extract_terroir_facts.py`.""" chunks: list[str] = [] @@ -160,13 +191,13 @@ def _build_subsection_hint_fr( if body: chunks.append(f"« {h} » : {body}") joined = "\n\n".join(chunks) - if len(joined) > WIKI_HINT_CHAR_CAP: - joined = joined[:WIKI_HINT_CHAR_CAP].rsplit(" ", 1)[0] + " […]" + if len(joined) > char_cap: + joined = joined[:char_cap].rsplit(" ", 1)[0] + " […]" return joined def _build_subsection_hint_es( - wiki_record: dict, headings: list[str] + wiki_record: dict, headings: list[str], char_cap: int = WIKI_HINT_CHAR_CAP, ) -> str: """ES wiki-hint format — mirrors `_wiki_hint_for_subsection` in `scripts/es/02d_extract_terroir_facts.py`. Different from the FR @@ -176,7 +207,7 @@ def _build_subsection_hint_es( `wiki`-only bullets.""" full = wiki_record.get("full_text") or "" if not full: - return (wiki_record.get("lead_extract") or "")[:WIKI_HINT_CHAR_CAP] + return (wiki_record.get("lead_extract") or "")[:char_cap] section_text = _index_wiki_sections(full, headings) pieces = [section_text["__intro__"]] if section_text.get("__intro__") else [] for h in headings: @@ -184,8 +215,8 @@ def _build_subsection_hint_es( pieces.append(f"# {h}\n{section_text[h]}") blob = "\n\n".join(pieces).strip() if blob: - return blob[:WIKI_HINT_CHAR_CAP] - return (wiki_record.get("lead_extract") or "")[:WIKI_HINT_CHAR_CAP] + return blob[:char_cap] + return (wiki_record.get("lead_extract") or "")[:char_cap] def load_wiki_hints(slug: str, country: str) -> tuple[dict[str, str], dict | None]: @@ -198,15 +229,21 @@ def load_wiki_hints(slug: str, country: str) -> tuple[dict[str, str], dict | Non data = json.loads(cache.read_text(encoding="utf-8")) if data.get("missing") or data.get("error"): return empty, data - if country == "es": + # GB's stage 02d builds its hint the ES way — full intro, then + # "# {heading}" blocks, raw char cap — not the FR way. Routing it + # through the FR builder matches no heading at all and leaves only a + # 400-char intro, which reports well-grounded `wiki` bullets as eroded. + if country in ("es", "gb"): + cap = wiki_hint_cap(country) out = { - sub_key: _build_subsection_hint_es(data, headings) + sub_key: _build_subsection_hint_es(data, headings, cap) for sub_key, headings in headings_map.items() } return out, data section_text = _index_wiki_sections(data.get("full_text", ""), data.get("sections", [])) + cap = wiki_hint_cap(country) out = { - sub_key: _build_subsection_hint_fr(sub_key, wanted, section_text) + sub_key: _build_subsection_hint_fr(sub_key, wanted, section_text, cap) for sub_key, wanted in headings_map.items() } return out, data @@ -226,10 +263,38 @@ def load_current_cahier(slug: str, country: str) -> tuple[str, str] | None: rec = json.loads(p.read_text(encoding="utf-8")) except Exception: # noqa: BLE001 return None - lien = (rec.get(field) or "").strip() + if country == "gb": + # GB's stage 02d grounds on a composite context, not the bare lien + # field — region + demarcated area + variety roster + the link + # section — and hashes that. Rebuild it identically here or every + # GB record reports a spurious `cahier-drift`. + # Mirrors `_cahier_context` in scripts/gb/02d_extract_terroir_facts.py. + lien = _cahier_context_gb(rec) + else: + lien = (rec.get(field) or "").strip() return lien, cahier_sha(lien) +def _cahier_context_gb(record: dict) -> str: + pieces = [f"Region: {record.get('region') or ''} ({record.get('kind') or ''})"] + geo = record.get("geo_area_brief") or "" + if geo: + pieces.append(f"Demarcated area: {geo}") + grapes = (record.get("grapes") or {}).get("details") or [] + if grapes: + names = ", ".join(g.get("name") or g.get("slug") for g in grapes[:40]) + pieces.append(f"Authorised varieties: {names}") + link = record.get("link_to_terroir") or "" + if link: + pieces.append(f"Link with the geographical area:\n{link}") + return "\n".join(pieces) + + +class UnsupportedCountry(Exception): + """Raised for a terroir-facts cache whose country has no source + dispatch entry — see EXTRACTED_BY_COUNTRY.""" + + def audit_one(cache_path: Path) -> dict: """Audit one terroir-facts cache file. Returns a dict with the per-AOC findings (drift flags + per-bullet coverage).""" @@ -237,6 +302,12 @@ def audit_one(cache_path: Path) -> dict: slug = data.get("slug") or cache_path.stem country = data.get("country") or "fr" facts = data.get("facts") or [] + if country not in EXTRACTED_BY_COUNTRY: + # This audit re-derives coverage from the country's own source + # documents, so it only covers countries wired into the dispatch + # tables above. Everything else is skipped explicitly rather than + # dying on a KeyError that the caller reports as a corrupt cache. + raise UnsupportedCountry(country) cur_cahier = load_current_cahier(slug, country) cur_lien = cur_cahier[0] if cur_cahier else "" @@ -382,9 +453,13 @@ def main() -> int: print(f"[audit] {len(files)} AOC caches", file=sys.stderr) audits: list[dict] = [] + unsupported: dict[str, int] = {} for p in files: try: a = audit_one(p) + except UnsupportedCountry as e: + unsupported[str(e)] = unsupported.get(str(e), 0) + 1 + continue except Exception as e: # noqa: BLE001 print(f" err {p.stem}: {e}", file=sys.stderr) continue @@ -392,6 +467,10 @@ def main() -> int: if not args.quiet: print_per_aoc(a, verbose=args.verbose) + if unsupported: + listed = ", ".join(f"{c}={n}" for c, n in sorted(unsupported.items())) + print(f" [skipped] countries with no source dispatch entry: {listed}", + file=sys.stderr) summary = summarize(audits) print("\n[audit] summary:", file=sys.stderr) print(json.dumps(summary, ensure_ascii=False, indent=2, default=str), file=sys.stderr) diff --git a/scripts/gb/00_fetch_data.py b/scripts/gb/00_fetch_data.py new file mode 100644 index 0000000..eba2433 --- /dev/null +++ b/scripts/gb/00_fetch_data.py @@ -0,0 +1,364 @@ +"""Fetch the public reference datasets the United Kingdom pipeline depends on. + +Pipeline stage 00 (gb). + +The UK is the corpus's first post-Brexit register: it is **not** in +eAmbrosia's live scheme (its pre-2021 GIs linger there as legacy rows, +but Sussex — registered 2022 — never appears), and Bétard 2022 is an EU +PDO layer carrying no `PDO-GB-*` rows. Both of the usual spines are +therefore unavailable, and both are replaced by UK-domestic ones: + +1. **GOV.UK "protected food and drink names" register** (the spine). + DEFRA publishes the UK GI schemes as a structured GOV.UK *finder* + whose documents are queryable through the site's own search API: + + https://www.gov.uk/api/search.json + ?filter_format=protected_food_drink_name + &filter_register=wines + &filter_country_of_origin=united-kingdom + + Each hit resolves to a per-GI content-API document carrying typed + metadata (protection type, register, status, application + UK/EU + registration dates, reason for protection) and — for every registered + wine — an attachment: the **product specification**. Stage 01 fetches + those; stage 02 parses them. + + v1 corpus: **6 registered wine GIs** — 4 PDO (English, Welsh, Sussex, + Darnibole) + 2 PGI (English Regional, Welsh Regional). Every one of + the 6 ships a public specification, so unlike ES/IT/GR/SI/HR/BG/SK/CZ + the UK needs no national-spec fallback tier and has **no stub tier at + all**. A 7th name, "The Crouch Valley" (PDO, applied for 2023-03-06), + is still in assessment; it is filtered out by `status=registered` the + way the eAmbrosia countries filter theirs, and reported by + `scripts/audit_gb_coverage.py` so the queue stays visible. + +2. **ONS Open Geography boundaries** (the geometry source), Open + Government Licence v3.0, "Contains OS data © Crown copyright and + database right", Source: Office for National Statistics: + + - Countries (December 2025) UK BGC — England + Wales, the + `DEMARCATION` of four of the six GIs. + - Counties and Unitary Authorities (December 2025) UK BGC — East + Sussex + West Sussex + Brighton and Hove, whose union is the + Sussex PDO's "administrative boundaries of the counties of East + and West Sussex". + + BGC ("generalised, clipped to the coastline") is the right + generalisation for a web map: full-resolution BFC is several times + larger with no visible gain at the zooms this corpus renders. + + Darnibole needs no boundary download — its approximate polygon is + reconstructed from the parcel references printed on its own + specification plan (see `scripts/_lib/gb/darnibole.py`). + +Outputs: +- raw/gb/gov-uk/index.json — the 6 registered wine GIs + spec URLs +- raw/gb/gov-uk/manifest.json — fetch metadata + the pending-application queue +- raw/gb/ons/countries.geojson — England + Wales +- raw/gb/ons/counties.geojson — the three Sussex CTYUAs +- raw/gb/ons/manifest.json — per-layer sha256 / feature counts / licence +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import re +import sys +import unicodedata +from datetime import datetime, timezone +from pathlib import Path + +import requests + +ROOT = Path(__file__).resolve().parents[2] +GOVUK_DIR = ROOT / "raw" / "gb" / "gov-uk" +INDEX_PATH = GOVUK_DIR / "index.json" +GOVUK_MANIFEST = GOVUK_DIR / "manifest.json" +ONS_DIR = ROOT / "raw" / "gb" / "ons" +ONS_MANIFEST = ONS_DIR / "manifest.json" + +SEARCH_URL = "https://www.gov.uk/api/search.json" +CONTENT_URL = "https://www.gov.uk/api/content" +REGISTER_BASE = "https://www.gov.uk/protected-food-drink-names" + +UA = ( + "open-wine-map/0.0.1 (https://github.com/devloed-com/open-wine-map; " + "mailto:winemap@devloed.com) python-requests" +) +GOVUK_LICENSE = "Open Government Licence v3.0 — © Crown copyright, DEFRA / GOV.UK" +ONS_LICENSE = ( + "Open Government Licence v3.0 — Source: Office for National Statistics " + "licensed under the Open Government Licence. Contains OS data " + "© Crown copyright and database right 2025." +) + +ONS_FS = ( + "https://services1.arcgis.com/ESMARspQHYMw9BZ9/arcgis/rest/services" +) +ONS_LAYERS = { + "countries.geojson": { + "service": "Countries_December_2025_Boundaries_UK_BGC", + "where": "CTRY25NM IN ('England','Wales')", + "out_fields": "CTRY25CD,CTRY25NM", + "expected": 2, + "label": "Countries (December 2025) Boundaries UK BGC", + }, + "counties.geojson": { + "service": "Counties_and_Unitary_Authorities_December_2025_Boundaries_UK_BGC", + "where": "CTYUA25NM IN ('East Sussex','West Sussex','Brighton and Hove')", + "out_fields": "CTYUA25CD,CTYUA25NM", + "expected": 3, + "label": "Counties and Unitary Authorities (December 2025) Boundaries UK BGC", + }, +} + +# The GOV.UK register does not publish a GI file number — the per-entry +# metadata carries protection type and dates but no identifier (only +# Sussex's specification states one in its own text, "PDO GB number: +# W0006"). The rest of the corpus keys on the `PDO-xx-*` / `PGI-xx-*` +# file number, and all six UK wines are also listed in the EU register +# (the four 2011 names and Darnibole as pre-Brexit registrations kept +# protected under the Withdrawal Agreement; Sussex added 2025-01-31 +# under the UK-EU agreement), so we bridge to those identifiers here. +# Verified against `raw/eambrosia-register/gi-index.json` +# (countryId=gb, qualityProductType=Wine). +_FILE_NUMBER_BY_SLUG: dict[str, str] = { + "english-wine": "PDO-GB-A1585", + "welsh-wine": "PDO-GB-A1587", + "english-regional-wine": "PGI-GB-A1589", + "welsh-regional-wine": "PGI-GB-A1590", + "darnibole": "PDO-GB-N1636", + "sussex": "PDO-GB-02365", +} + +# The GOV.UK register writes the shorthand names the labels use ("English", +# "Welsh Regional"); the wine is "English wine" etc. Keep the register's own +# `registered_name` as the appellation name — it is what the label carries — +# but slugify to a stable, unambiguous slug. +_SLUG_OVERRIDES = { + "english": "english-wine", + "welsh": "welsh-wine", + "english-regional": "english-regional-wine", + "welsh-regional": "welsh-regional-wine", +} + + +def slugify(s: str) -> str: + s = unicodedata.normalize("NFKD", s).encode("ascii", "ignore").decode() + return re.sub(r"[^A-Za-z0-9]+", "-", s).strip("-").lower() + + +def _sha256(data: bytes) -> str: + return hashlib.sha256(data).hexdigest() + + +def normalise_kind(protection_type: str) -> str: + """The GOV.UK register spells these out in English + ("protected-designation-of-origin-pdo"). The corpus convention — and, + concretely, the map's polygon-colour expression, which keys on + `kind == 'IGP'` — is the DOP/IGP pair every other country uses.""" + p = (protection_type or "").lower() + if "designation-of-origin" in p: + return "DOP" + if "geographical-indication" in p: + return "IGP" + return "" + + +def search_wines(session: requests.Session) -> list[dict]: + params = { + "filter_format": "protected_food_drink_name", + "filter_register": "wines", + "filter_country_of_origin": "united-kingdom", + "count": 100, + "fields": ",".join([ + "title", "link", "registered_name", "register", "status", + "protection_type", "country_of_origin", "class_category", + "date_application", "date_registration", "date_registration_eu", + "reason_for_protection", + ]), + } + r = session.get(SEARCH_URL, params=params, timeout=60) + r.raise_for_status() + return r.json().get("results") or [] + + +def fetch_detail(session: requests.Session, link: str) -> dict: + path = link.lstrip("/") + r = session.get(f"{CONTENT_URL}/{path}", timeout=60) + r.raise_for_status() + return r.json() + + +def _spec_attachments(detail: dict) -> list[dict]: + """Product-specification attachments, most useful first. + + A register entry links its product specification and (for the + post-2021 applications) a decision notice; Welsh entries additionally + carry a Welsh-language translation of the specification. Rank the + English specification first and keep the rest as provenance. + """ + out: list[dict] = [] + for att in (detail.get("details") or {}).get("attachments") or []: + url = att.get("url") or "" + if not url: + continue + title = (att.get("title") or "").strip() + low = f"{title} {url}".lower() + if "decision" in low and "notice" in low: + role = "decision-notice" + elif "welsh" in low and "translation" in low: + role = "specification-welsh-translation" + else: + role = "product-specification" + out.append({ + "role": role, + "title": title, + "url": url, + "content_type": att.get("content_type") or "", + }) + order = {"product-specification": 0, "specification-welsh-translation": 1, + "decision-notice": 2} + out.sort(key=lambda a: order.get(a["role"], 9)) + return out + + +def project(result: dict, detail: dict) -> dict: + meta = (detail.get("details") or {}).get("metadata") or {} + name = (meta.get("registered_name") or result.get("registered_name") or "").strip() + base = slugify(name) + slug = _SLUG_OVERRIDES.get(base, base) + atts = _spec_attachments(detail) + spec = next((a for a in atts if a["role"] == "product-specification"), None) + return { + "name": name, + "slug": slug, + "file_number": _FILE_NUMBER_BY_SLUG.get(slug, ""), + "kind": normalise_kind(meta.get("protection_type") or ""), + "protection_type": meta.get("protection_type") or "", + "status": meta.get("status") or "", + "register": meta.get("register") or "", + "class_category": (meta.get("class_category") or [""])[0], + "reason_for_protection": meta.get("reason_for_protection") or "", + "date_application": meta.get("date_application") or "", + "date_registration": meta.get("date_registration") or "", + "date_registration_eu": meta.get("date_registration_eu") or "", + "register_url": f"{REGISTER_BASE}/{(result.get('link') or '').rsplit('/', 1)[-1]}", + "spec_url": (spec or {}).get("url") or "", + "spec_title": (spec or {}).get("title") or "", + "attachments": atts, + } + + +def fetch_ons_layer(session: requests.Session, spec: dict) -> dict: + url = f"{ONS_FS}/{spec['service']}/FeatureServer/0/query" + params = { + "where": spec["where"], "outFields": spec["out_fields"], + "outSR": 4326, "f": "geojson", + } + print(f"[fetch] ONS {spec['service']}", file=sys.stderr) + r = session.get(url, params=params, timeout=180) + r.raise_for_status() + return r.json() + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--refresh", action="store_true", + help="re-download the ONS boundary layers even when cached") + args = ap.parse_args() + + session = requests.Session() + session.headers["User-Agent"] = UA + rc = 0 + + # ---- 1. the register spine ------------------------------------- + GOVUK_DIR.mkdir(parents=True, exist_ok=True) + results = search_wines(session) + print(f"[fetch] GOV.UK wines register: {len(results)} UK entries", file=sys.stderr) + + wines: list[dict] = [] + pending: list[dict] = [] + for res in results: + detail = fetch_detail(session, res.get("link") or "") + rec = project(res, detail) + if rec["status"] != "registered": + pending.append({k: rec[k] for k in + ("name", "slug", "kind", "status", "date_application", + "register_url")}) + continue + if not rec["file_number"]: + print(f"[warn] {rec['name']} ({rec['slug']}): no file number bridged — " + "add it to _FILE_NUMBER_BY_SLUG (region + geometry key on it)", + file=sys.stderr) + rc = 2 + if not rec["spec_url"]: + print(f"[warn] {rec['name']}: registered but no product specification " + "attachment on the register page", file=sys.stderr) + rc = 2 + wines.append(rec) + + wines.sort(key=lambda w: w["slug"]) + INDEX_PATH.write_text( + json.dumps({"wines": wines}, ensure_ascii=False, indent=2), encoding="utf-8") + + now = datetime.now(timezone.utc).isoformat(timespec="seconds") + GOVUK_MANIFEST.write_text(json.dumps({ + "generated_at": now, + "source_url": SEARCH_URL, + "register": "wines", + "country_of_origin": "united-kingdom", + "license": GOVUK_LICENSE, + "n_results": len(results), + "n_registered": len(wines), + "n_with_spec": sum(1 for w in wines if w["spec_url"]), + "pending_applications": pending, + }, ensure_ascii=False, indent=2, sort_keys=True), encoding="utf-8") + print(f"[done] register: {len(wines)} registered wine GIs " + f"({sum(1 for w in wines if w['spec_url'])} with a product specification), " + f"{len(pending)} pending → {INDEX_PATH.relative_to(ROOT)}", file=sys.stderr) + + # ---- 2. ONS boundaries ----------------------------------------- + ONS_DIR.mkdir(parents=True, exist_ok=True) + layers: dict[str, dict] = {} + for fname, spec in ONS_LAYERS.items(): + out_path = ONS_DIR / fname + if out_path.exists() and not args.refresh: + fc = json.loads(out_path.read_text(encoding="utf-8")) + else: + fc = fetch_ons_layer(session, spec) + out_path.write_text(json.dumps(fc, ensure_ascii=False), encoding="utf-8") + feats = fc.get("features") or [] + if fc.get("exceededTransferLimit"): + print(f"[error] {fname}: exceededTransferLimit — result truncated", + file=sys.stderr) + rc = 2 + if len(feats) != spec["expected"]: + print(f"[warn] {fname}: got {len(feats)} features, " + f"expected {spec['expected']}", file=sys.stderr) + rc = rc or 2 + data = out_path.read_bytes() + layers[fname] = { + "service": spec["service"], + "label": spec["label"], + "where": spec["where"], + "sha256": _sha256(data), + "bytes": len(data), + "n_features": len(feats), + "expected": spec["expected"], + } + + ONS_MANIFEST.write_text(json.dumps({ + "generated_at": now, + "source_url": ONS_FS, + "license": ONS_LICENSE, + "layers": layers, + }, ensure_ascii=False, indent=2, sort_keys=True), encoding="utf-8") + print(f"[done] ONS boundaries → {ONS_DIR.relative_to(ROOT)}", file=sys.stderr) + return rc + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/gb/01_fetch_specs.py b/scripts/gb/01_fetch_specs.py new file mode 100644 index 0000000..977d7a3 --- /dev/null +++ b/scripts/gb/01_fetch_specs.py @@ -0,0 +1,168 @@ +"""Fetch each UK wine GI's product specification from GOV.UK. + +Pipeline stage 01 (gb). + +The register entries written by stage 00 each carry a `spec_url` on +`assets.publishing.service.gov.uk`. There is no WAF, no cookie gate and +no JavaScript challenge — unlike the EUR-Lex countries, which is why the +UK pipeline ships no `01b_solve_waf.py` sibling. + +Two document formats are served, keyed by `Content-Type`: + + - **PDF** (5 of 6) — the four 2011 DEFRA specifications plus Darnibole. + - **.docx** (1 of 6) — Sussex, the only post-2021 UK-scheme + registration. Handled downstream by the stdlib zip → + `word/document.xml` route the HR pipeline already uses. + +Cached documents are reused unless `--refresh` is passed; the manifest +records sha256 + byte count + fetched_at per slug so stage 02 can attach +provenance and so a rerun is a no-op. + +Reads: raw/gb/gov-uk/index.json +Writes: raw/gb/specs/<slug>.{pdf,docx} + manifest.json +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import sys +import time +from datetime import datetime, timezone +from pathlib import Path + +import requests + +ROOT = Path(__file__).resolve().parents[2] +INDEX_IN = ROOT / "raw" / "gb" / "gov-uk" / "index.json" +OUT_DIR = ROOT / "raw" / "gb" / "specs" +MANIFEST_PATH = OUT_DIR / "manifest.json" + +UA = ( + "open-wine-map/0.0.1 (https://github.com/devloed-com/open-wine-map; " + "mailto:winemap@devloed.com) python-requests" +) +LICENSE = "Open Government Licence v3.0 — © Crown copyright, DEFRA / GOV.UK" + +_EXT_BY_CONTENT_TYPE = { + "application/pdf": ".pdf", + "application/vnd.openxmlformats-officedocument.wordprocessingml.document": ".docx", + "application/msword": ".doc", +} + + +def _sha256(data: bytes) -> str: + return hashlib.sha256(data).hexdigest() + + +def _ext_for(content_type: str, url: str) -> str: + ct = (content_type or "").split(";")[0].strip().lower() + if ct in _EXT_BY_CONTENT_TYPE: + return _EXT_BY_CONTENT_TYPE[ct] + for ext in (".pdf", ".docx", ".doc"): + if url.lower().endswith(ext): + return ext + return ".bin" + + +def _cached(slug: str) -> Path | None: + for ext in (".pdf", ".docx", ".doc", ".bin"): + p = OUT_DIR / f"{slug}{ext}" + if p.exists(): + return p + return None + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--only", action="append", default=[], + help="restrict to slugs containing this substring (repeatable)") + ap.add_argument("--refresh", action="store_true", + help="re-download even when a cached document exists") + ap.add_argument("--delay", type=float, default=1.0, + help="seconds to pause between downloads (default 1.0)") + args = ap.parse_args() + + if not INDEX_IN.exists(): + print(f"error: {INDEX_IN} missing — run scripts/gb/00_fetch_data.py first", + file=sys.stderr) + return 1 + + wines = json.loads(INDEX_IN.read_text(encoding="utf-8"))["wines"] + if args.only: + needles = [s.lower() for s in args.only] + wines = [w for w in wines if any(n in w["slug"].lower() for n in needles)] + + OUT_DIR.mkdir(parents=True, exist_ok=True) + manifest: dict = {} + if MANIFEST_PATH.exists(): + try: + manifest = json.loads(MANIFEST_PATH.read_text(encoding="utf-8")) + except (ValueError, OSError): + manifest = {} + by_slug: dict[str, dict] = manifest.get("by_slug", {}) + + session = requests.Session() + session.headers["User-Agent"] = UA + ok = cached = failed = 0 + + for w in wines: + slug, url = w["slug"], w.get("spec_url") or "" + if not url: + by_slug[slug] = {"status": "no-spec-url", "source_url": ""} + failed += 1 + print(f"[miss] {slug}: no product-specification URL", file=sys.stderr) + continue + existing = _cached(slug) + if existing and not args.refresh: + cached += 1 + print(f"[skip] {slug}: cached ({existing.name})", file=sys.stderr) + continue + try: + r = session.get(url, timeout=120) + r.raise_for_status() + except requests.RequestException as exc: + by_slug[slug] = {"status": "fetch-error", "source_url": url, + "error": str(exc)[:300]} + failed += 1 + print(f"[fail] {slug}: {exc}", file=sys.stderr) + continue + ext = _ext_for(r.headers.get("Content-Type", ""), r.url) + # A format change (docx → pdf on a re-registration) must not leave + # the previous document behind for stage 02 to pick up. + for stale in (OUT_DIR.glob(f"{slug}.*")): + if stale.suffix != ext and stale.name != "manifest.json": + stale.unlink() + out_path = OUT_DIR / f"{slug}{ext}" + out_path.write_bytes(r.content) + by_slug[slug] = { + "status": "ok", + "filename": out_path.name, + "format": ext.lstrip("."), + "source_url": url, + "final_url": r.url, + "content_type": r.headers.get("Content-Type", ""), + "bytes": len(r.content), + "sha256": _sha256(r.content), + "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "register_url": w.get("register_url", ""), + "spec_title": w.get("spec_title", ""), + } + ok += 1 + print(f"[ok] {slug}: {len(r.content)} bytes → {out_path.name}", file=sys.stderr) + if args.delay: + time.sleep(args.delay) + + MANIFEST_PATH.write_text(json.dumps({ + "generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "license": LICENSE, + "by_slug": dict(sorted(by_slug.items())), + }, ensure_ascii=False, indent=2), encoding="utf-8") + print(f"[done] fetched={ok} cached={cached} failed={failed} " + f"→ {OUT_DIR.relative_to(ROOT)}", file=sys.stderr) + return 1 if failed else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/gb/02_extract_specs.py b/scripts/gb/02_extract_specs.py new file mode 100644 index 0000000..ac08154 --- /dev/null +++ b/scripts/gb/02_extract_specs.py @@ -0,0 +1,296 @@ +"""Build one record per UK wine GI from its cached product specification. + +Pipeline stage 02 (gb). + +Every registered UK wine GI ships a public specification (stage 01), so +this stage has **no stub path**: all six records come out fully +extracted. That is unique in the corpus — ES/IT/GR/SI/HR/BG/SK/CZ all +need a national-spec fallback tier for their grandfathered names. + +The three document layouts are handled by `scripts/_lib/gb/spec.py`; +this stage is the glue: text extraction (`pdftotext -layout` for the five +PDFs, stdlib zip → `word/document.xml` for Sussex's .docx), grape +matching, style derivation and record assembly. + +Grape roles: no UK specification splits principal from accessory — each +lists a flat roster of authorised varieties — so every match resolves as +`principal`, the same convention as PT/IT/HR/BG/SK. + +v1 models the 6 wine GIs as a **flat corpus**. Sussex and Darnibole sit +geographically inside the English PDO's territory but are first-class +PDOs on the register rather than sub-denominations of it (Darnibole's +own specification makes the point explicitly), so they are siblings — +the way the CZ podoblasti are siblings of Čechy / Morava. + +Reads: raw/gb/gov-uk/index.json + raw/gb/specs/*.{pdf,docx} + manifest.json +Writes: raw/gb/specs-extracted/*.json + _index.json + raw/gb/extraction-unknowns.json (unmatched variety candidates) +""" + +from __future__ import annotations + +import argparse +import html as html_lib +import json +import re +import subprocess +import sys +import zipfile +from pathlib import Path + +from tqdm import tqdm + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts")) +from _lib.gb.geometry import is_approximate # noqa: E402 +from _lib.gb.region import derive_region # noqa: E402 +from _lib.gb.spec import ( # noqa: E402 + COLOUR_BY_KEYWORD, + PART_STYLE_MARKERS, + grape_candidates, + parse_spec, +) +from _lib.grape_entity import ( # noqa: E402 + flush_unknowns_queue, + match_variety, + set_pliego_context, +) + +INDEX_IN = ROOT / "raw" / "gb" / "gov-uk" / "index.json" +SPECS_DIR = ROOT / "raw" / "gb" / "specs" +SPECS_MANIFEST = SPECS_DIR / "manifest.json" +OUT_DIR = ROOT / "raw" / "gb" / "specs-extracted" +INDEX_OUT = OUT_DIR / "_index.json" +UNKNOWNS_OUT = ROOT / "raw" / "gb" / "extraction-unknowns.json" + +_GRAPE_COLOUR_TO_STYLE = {"blanc": "white", "noir": "red", "gris": "white", "rose": "rose"} + + +def pdf_text(path: Path) -> str: + proc = subprocess.run( + ["pdftotext", "-layout", str(path), "-"], + capture_output=True, text=True, check=False, + ) + return proc.stdout or "" + + +def docx_text(path: Path) -> str: + """Plain text from a .docx, via the stdlib (the HR pipeline's route). + + `<w:pPr>` paragraph-property blocks are dropped first so numbering + and style markup can't leak into the text, and `</w:p>` becomes a + newline so paragraphs survive as lines. + """ + with zipfile.ZipFile(path) as zf: + xml = zf.read("word/document.xml").decode("utf-8", errors="replace") + xml = re.sub(r"<w:pPr>.*?</w:pPr>", "", xml, flags=re.S) + xml = re.sub(r"</w:p\s*>", "\n", xml) + xml = re.sub(r"<w:tab\s*/>", " ", xml) + xml = re.sub(r"<[^>]+>", "", xml) + return html_lib.unescape(xml) + + +def spec_text(path: Path) -> str: + if path.suffix.lower() == ".docx": + return docx_text(path) + return pdf_text(path) + + +def parse_grapes(section_text: str) -> dict: + """Resolve the variety roster to lexicon slugs; all `principal`.""" + out: dict[str, list] = { + "principal": [], "accessory": [], "observation": [], "details": [], + } + seen: set[str] = set() + for cand in grape_candidates(section_text): + match = match_variety(cand) + if match is None or match.slug in seen: + continue + seen.add(match.slug) + out["principal"].append(match.slug) + out["details"].append({ + "slug": match.slug, + "name": cand, + "role": "principal", + "colour": match.colour, + }) + return out + + +def parse_styles(parsed: dict, grape_details: list[dict], wine_name: str) -> list[str]: + """Derive styles from the specification's own wine-category parts. + + The DEFRA 2011 template names its categories as `PART n: STILL WINE` + / `PART n: QUALITY SPARKLING WINE`; Sussex names them in section 5 + ("Traditional method quality sparkling wine", "Quality still wine"). + Colour comes from the varieties, since none of the UK specifications + describes its wines by colour in a keyword form. + """ + found: set[str] = set() + blob = " ".join(parsed.get("part_titles") or []) + roles = parsed.get("roles") or {} + blob_full = f"{blob} {roles.get('description', '')} {wine_name}" + for pattern, slug in PART_STYLE_MARKERS: + if slug and pattern.search(blob): + found.add(slug) + for kw, colour in COLOUR_BY_KEYWORD.items(): + if re.search(rf"\b{re.escape(kw)}\b", blob_full, re.I): + found.add(colour) + for g in grape_details: + base = _GRAPE_COLOUR_TO_STYLE.get(g.get("colour") or "", "") + if base: + found.add(base) + if base == "red": + found.add("rose") + return sorted(found) + + +def derive_summary(parsed: dict, wine: dict, max_chars: int = 600) -> str: + roles = parsed.get("roles") or {} + text = re.sub(r"\s+", " ", roles.get("description") or roles.get("link_to_terroir") or "") + # The DEFRA parts prefix their narrative with the category name. + text = re.sub(r"^(Still Wine|Quality Sparkling Wine)\s+", "", text).strip() + if not text: + demarcation = (parsed.get("demarcation") or "").strip() + kind = "PDO" if wine.get("kind") == "DOP" else "PGI" + return (f"{wine['name']} is a United Kingdom wine {kind}" + + (f" demarcated to {demarcation.title()}." if demarcation else ".")) + if len(text) <= max_chars: + return text + cut = text[:max_chars].rsplit(". ", 1)[0] + return cut + ("." if not cut.endswith(".") else "") + + +def build_record(wine: dict, parsed: dict, meta: dict) -> dict: + roles = dict(parsed.get("roles") or {}) + grapes = parse_grapes(roles.get("grape_varieties", "")) + styles = parse_styles(parsed, grapes["details"], wine["name"]) + demarcation = (parsed.get("demarcation") or "").strip() + region = derive_region({"file_number": wine["file_number"], + "demarcation": demarcation}) + return { + "country": "gb", + "source_lang": "en", + "file_number": wine["file_number"], + "id_eambrosia": "", + "slug": wine["slug"], + "name": wine["name"], + "kind": wine["kind"], + "is_sub_denomination": False, + "parent_slug": "", + "region": region, + "demarcation": demarcation, + "categories": [wine["kind"]] if wine.get("kind") else [], + "summary": derive_summary(parsed, wine), + "sections": {}, + "section_titles": {}, + "section_roles": roles, + "grapes": grapes, + "styles": styles, + "geo_area_brief": roles.get("geo_area", ""), + "link_to_terroir": roles.get("link_to_terroir", ""), + "max_yields": roles.get("yields", ""), + "parser_template": parsed.get("template", ""), + # Darnibole's boundary is reconstructed from its specification's + # own plan rather than published as data; the panel discloses it. + "geom_approximate": is_approximate(wine["file_number"]), + "date_registration": wine.get("date_registration", ""), + "date_registration_eu": wine.get("date_registration_eu", ""), + "reason_for_protection": wine.get("reason_for_protection", ""), + "publications": [], + "producer_group": {"name": "", "url": ""}, + "source": { + "kind": "gov-uk-product-specification", + "filename": meta.get("filename", ""), + "format": meta.get("format", ""), + "source_url": meta.get("source_url", ""), + "final_url": meta.get("final_url", ""), + "register_url": wine.get("register_url", ""), + "sha256": meta.get("sha256", ""), + "bytes": meta.get("bytes", 0), + "fetched_at": meta.get("fetched_at", ""), + "license": "Open Government Licence v3.0 — © Crown copyright, DEFRA / GOV.UK", + }, + "stub": False, + } + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--only", action="append", default=[]) + args = ap.parse_args() + + if not INDEX_IN.exists(): + print(f"error: {INDEX_IN} missing — run scripts/gb/00_fetch_data.py first", + file=sys.stderr) + return 1 + + wines = json.loads(INDEX_IN.read_text(encoding="utf-8"))["wines"] + if args.only: + needles = [s.lower() for s in args.only] + wines = [w for w in wines if any(n in w["slug"].lower() for n in needles)] + + manifest: dict = {} + if SPECS_MANIFEST.exists(): + manifest = json.loads(SPECS_MANIFEST.read_text(encoding="utf-8")).get("by_slug", {}) + + OUT_DIR.mkdir(parents=True, exist_ok=True) + index: dict[str, dict] = {} + extracted = failed = 0 + + for w in tqdm(wines, desc="extract-gb-specs", leave=False): + slug = w["slug"] + set_pliego_context(slug) + meta = manifest.get(slug, {}) + fname = meta.get("filename") or "" + path = SPECS_DIR / fname if fname else None + if path is None or not path.exists(): + print(f"[fail] {slug}: no cached specification " + "(run scripts/gb/01_fetch_specs.py)", file=sys.stderr) + failed += 1 + continue + text = spec_text(path) + if not text.strip(): + print(f"[fail] {slug}: empty text from {path.name}", file=sys.stderr) + failed += 1 + continue + parsed = parse_spec(text) + record = build_record(w, parsed, meta) + if not record["grapes"]["principal"]: + print(f"[warn] {slug}: no varieties resolved from " + f"{record['parser_template']}", file=sys.stderr) + (OUT_DIR / f"{slug}.json").write_text( + json.dumps(record, ensure_ascii=False, indent=2), encoding="utf-8") + extracted += 1 + index[slug] = { + "country": "gb", + "source_lang": "en", + "file_number": w["file_number"], + "slug": slug, + "name": w["name"], + "kind": w["kind"], + "region": record["region"], + "filename": f"{slug}.json", + "is_sub_denomination": False, + "parent_slug": "", + "stub": False, + "parser_template": record["parser_template"], + "n_grapes": len(record["grapes"]["details"]), + "n_styles": len(record["styles"]), + "link_chars": len(record["link_to_terroir"]), + } + + set_pliego_context(None) + INDEX_OUT.write_text( + json.dumps(index, ensure_ascii=False, indent=2, sort_keys=True), encoding="utf-8") + n_unknowns = flush_unknowns_queue(UNKNOWNS_OUT) + if n_unknowns: + print(f"[entity] {n_unknowns} unknown variety candidates → " + f"{UNKNOWNS_OUT.relative_to(ROOT)}", file=sys.stderr) + print(f"[done] extracted={extracted} failed={failed} → {OUT_DIR.relative_to(ROOT)}", + file=sys.stderr) + return 1 if failed else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/gb/02d_extract_terroir_facts.py b/scripts/gb/02d_extract_terroir_facts.py new file mode 100644 index 0000000..45ae45e --- /dev/null +++ b/scripts/gb/02d_extract_terroir_facts.py @@ -0,0 +1,617 @@ +"""Extract noteworthy terroir facts for each GB wine using the bounded, +dual-source LLM layer. + +Unlike Malta — the other English-source corpus, whose amendment +communications carry no link section — every UK product specification +does carry a real terroir narrative, so GB uses the **standard +cahier-primary** model rather than the Wikipedia-primary CH/MT one: + + - The four DEFRA 2011 specifications open each wine part with the link + text proper (northerly latitude, long growing season, high diurnal + range, the resulting acidity), 2.4-2.5 KB per record. + - Sussex has a full `9. Link` section — soils (South Downs chalk, the + greensands), climate, human factors — 5.5 KB. + - Darnibole has no link section, so its terroir narrative is its + `7 b) Definition of the demarcated area`: the ancient slate subsoil, + the steep south-facing slope, the thermal band. Stage 02 already + routes it to `link_to_terroir`. + +en.wikipedia.org stays the secondary salience hint, exactly as elsewhere. +Records are **not** skipped when Wikipedia is missing — the regulator +text alone is sufficient grounding, which is the whole point of the +cahier-primary model. + +`source_lang` is always "en". EN is the canonical rendered surface, so +the bullets need no translation for `/`; stage 02e only translates them +into fr/es/nl. + +The 4-subsection structure + JSON schema match AT/MT exactly so stage 04 +renders GB facts through the same code path. + +Providers: anthropic / mistral / ollama / manual. Batch via +`--batch --provider anthropic`. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import shutil +import sys +import tempfile +import time +from concurrent.futures import ThreadPoolExecutor, as_completed +from datetime import datetime, timezone +from difflib import SequenceMatcher +from pathlib import Path + +from tqdm import tqdm + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 + +EXTRACTED = ROOT / "raw" / "gb" / "specs-extracted" +WIKI_AOCS_ROOT = ROOT / "raw" / "wikipedia" / "aocs" +CACHE_DIR = ROOT / "raw" / "terroir-facts" +MANIFEST = CACHE_DIR / "manifest-gb.json" + +SOURCE_LANG = "en" +MIN_WIKI_CHARS = 400 +FUZZY_THRESHOLD = 0.6 +WIKI_HINT_CHAR_CAP = 2800 + + +SUBSECTIONS = [ + {"key": "facteurs_naturels", "max_bullets": 5}, + {"key": "facteurs_humains", "max_bullets": 2}, + {"key": "produit", "max_bullets": 2}, + {"key": "interactions", "max_bullets": 1}, +] + + +SUBSECTION_LABELS = { + "facteurs_naturels": "Natural factors (geology, soils, climate, relief)", + "facteurs_humains": "Historical and human factors (history, practices)", + "produit": "Product characteristics (sensory profile)", + "interactions": "Causal interactions (terroir / wine link)", +} + + +SUBSECTION_TOPICS = { + "facteurs_naturels": "geology and soils (South Downs and Chiltern chalk, the greensands, Kimmeridgian and Portland limestone, greensand, clay, slate), cool maritime climate, northerly latitude (above 49.9°N) and the long growing season it gives, high diurnal range, sunshine hours and growing degree days, aspect and shelter, rainfall", + "facteurs_humains": "history of viticulture in England and Wales (Roman and monastic roots, the post-1950s revival, the sparkling-wine expansion), wine education and research (Plumpton College), traditional-method sparkling production, hand harvesting, choice of varieties for a cool climate", + "produit": "wine colours, aromas, structure, natural acidity, autolytic character of the traditional-method sparkling wines, the character of the still wines", + "interactions": "explicit link between the British terroir and the character of the wine — how the chalk, the cool maritime climate, the long growing season and the high acidity express themselves in the wine", +} + + +WIKI_TO_SUBSECTION = { + "facteurs_naturels": [ + "Geography", "Geology", "Climate", "Soil", "Soils", "Terroir", + "Wine regions", "Regions", "Viticulture", "Vineyards", + ], + "facteurs_humains": [ + "History", "Indigenous grapes", "Grapes", "Grape varieties", + "Varieties", "Production", "Winemaking", "Wineries", + ], + "produit": [ + "Wines", "Styles", "Wine styles", "Types of wine", "Production", + ], + "interactions": [], +} +WIKI_TO_SUBSECTION["interactions"] = WIKI_TO_SUBSECTION["facteurs_naturels"] + + +EXTRACT_SYSTEM = """You extract noteworthy facts about a United Kingdom wine appellation from its product specification (primary source — the "Link with the geographical area" text, the demarcated area and the authorised varieties) and, where present, the English Wikipedia article (secondary source). + +Target reader: an informed wine lover or sommelier. Noteworthy: named geological formations (the South Downs chalk, the greensands, slate subsoil), specific soil types, the cool maritime climate and northerly latitude, growing-season length and diurnal range, distinctive viticultural and winemaking practices, precise sensory profile, dated historical anchors. + +═══ SOURCE 1: Product specification (user message below) ═══ + +═══ SOURCE 2: Wikipedia extract (relevant sections, may be empty) ═══ +{wiki_hint} + +Sub-section handled: {label} +Preferred categories: {topics} + +Strict rules: +- Each bullet MUST be backed by AT LEAST ONE quote: `cahier_quote` (verbatim from the product specification) OR `wiki_quote` (verbatim from the Wikipedia extract). Prefer `cahier_quote` — the specification is the regulator source. +- If only Wikipedia mentions a noteworthy fact, keep only `wiki_quote` — it will be attributed to Wikipedia at render time. +- Quotes are VERBATIM (copy-paste) from their source. NEVER attribute to a source text that does not appear in it. +- No value judgements ("exceptional", "prestigious"…). +- No figures absent from both sources. +- At most {max_bullets} bullets, each ≤ 140 characters. +- If neither the specification nor Wikipedia contains a concrete noteworthy fact for this sub-section, return an empty list. + +Reply ONLY in JSON, no preamble: +{{"facts": [{{"bullet": "…", "cahier_quote": "…", "wiki_quote": "…"}}, ...]}} +Use an empty string "" for the missing quote.""" + + +USER_LEAD = "Sub-section: {label}\n\nRegulator context:\n\n{ctx}" + + +# ─────────────────────────────────────────────────────────────── helpers ── + + +def normalize(s: str) -> str: + return " ".join((s or "").split()).lower() + + +def wiki_sha(text: str) -> str: + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def fuzzy_coverage(quote: str, source: str) -> float: + q = normalize(quote) + s = normalize(source) + if not q: + return 0.0 + match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( + 0, len(q), 0, len(s) + ) + return match.size / len(q) + + +def _find_heading(full: str, heading: str) -> int: + idx = full.find(f"\n\n{heading}\n\n") + if idx == -1: + idx = full.find(f"\n{heading}\n") + return idx + + +def _index_wiki_sections(full: str, headings: list[str]) -> dict[str, str]: + positions = sorted( + (idx, h) for h in headings if (idx := _find_heading(full, h)) != -1 + ) + section_text: dict[str, str] = {} + intro_end = positions[0][0] if positions else len(full) + section_text["__intro__"] = full[:intro_end].strip() + for i, (start, h) in enumerate(positions): + end = positions[i + 1][0] if i + 1 < len(positions) else len(full) + section_text[h] = full[start:end].strip() + return section_text + + +def _wiki_hint_for_subsection( + wiki_record: dict, sub_key: str, char_cap: int = WIKI_HINT_CHAR_CAP, +) -> str: + if not wiki_record or wiki_record.get("missing") or wiki_record.get("error"): + return "" + full = wiki_record.get("full_text") or "" + if not full: + return (wiki_record.get("lead_extract") or wiki_record.get("extract", ""))[:char_cap] + headings = WIKI_TO_SUBSECTION.get(sub_key, []) + sections = _index_wiki_sections(full, headings) + pieces = [sections["__intro__"]] if sections.get("__intro__") else [] + for h in headings: + if h in sections: + pieces.append(f"# {h}\n{sections[h]}") + blob = "\n\n".join(pieces).strip() + return (blob[:char_cap] + if blob + else (wiki_record.get("lead_extract") or wiki_record.get("extract", ""))[:char_cap]) + + +def _ground_facts( + raw_facts: list[dict], cahier_ctx: str, wiki_hint: str, +) -> tuple[list[dict], int]: + kept: list[dict] = [] + dropped = 0 + for f in raw_facts: + bullet = (f.get("bullet") or "").strip() + cq = (f.get("cahier_quote") or "").strip() + wq = (f.get("wiki_quote") or "").strip() + if not bullet: + dropped += 1 + continue + cc = fuzzy_coverage(cq, cahier_ctx) if cq else 0.0 + wc = fuzzy_coverage(wq, wiki_hint) if wq else 0.0 + c_ok = cc >= FUZZY_THRESHOLD + w_ok = wc >= FUZZY_THRESHOLD + if not (c_ok or w_ok): + dropped += 1 + continue + if c_ok and w_ok: + provenance = "both" + elif c_ok: + provenance = "cahier" + else: + provenance = "wiki" + kept.append({ + **f, + "cahier_coverage": round(cc, 3), + "wiki_coverage": round(wc, 3), + "provenance": provenance, + }) + return kept, dropped + + +# ─────────────────────────────────────────────── source-text + targets ── + + +def _cahier_context(record: dict) -> str: + """The product specification's own terroir narrative, plus the + demarcated area and variety roster as supporting context. Bullets + grounded in this block carry `provenance=cahier`.""" + pieces: list[str] = [] + pieces.append(f"Region: {record.get('region') or ''} ({record.get('kind') or ''})") + geo = record.get("geo_area_brief") or "" + if geo: + pieces.append(f"Demarcated area: {geo}") + grapes = (record.get("grapes") or {}).get("details") or [] + if grapes: + names = ", ".join(g.get("name") or g.get("slug") for g in grapes[:40]) + pieces.append(f"Authorised varieties: {names}") + link = record.get("link_to_terroir") or "" + if link: + pieces.append(f"Link with the geographical area:\n{link}") + return "\n".join(pieces) + + +def _wiki_record_for(slug: str) -> dict: + path = WIKI_AOCS_ROOT / SOURCE_LANG / f"{slug}.json" + if not path.exists(): + return {} + try: + return json.loads(path.read_text(encoding="utf-8")) + except (ValueError, OSError): + return {} + + +def _has_usable_wiki(rec: dict) -> bool: + if not rec or rec.get("missing") or rec.get("error"): + return False + full = rec.get("full_text") or "" + if len(full) >= MIN_WIKI_CHARS: + return True + extract = rec.get("lead_extract") or rec.get("extract", "") + return len(extract) >= MIN_WIKI_CHARS + + +def collect_targets() -> list[dict]: + out: list[dict] = [] + for jp in sorted(EXTRACTED.glob("*.json")): + if jp.name.startswith("_"): + continue + rec = json.loads(jp.read_text(encoding="utf-8")) + if rec.get("is_sub_denomination"): + continue + # Unlike MT/CH, a missing Wikipedia article is not disqualifying: + # the specification's own link section is the primary source. + wiki = _wiki_record_for(rec["slug"]) + if not _has_usable_wiki(wiki): + wiki = {} + rec["_wiki_record"] = wiki + rec["_cahier_ctx"] = _cahier_context(rec) + out.append(rec) + return out + + +# ─────────────────────────────────────────────────────────────── core loop ── + + +def _process_subsection(provider, model_id: str, record: dict, sub: dict): + label = SUBSECTION_LABELS[sub["key"]] + topic = SUBSECTION_TOPICS[sub["key"]] + wiki = record.get("_wiki_record") or {} + cahier_ctx = record.get("_cahier_ctx") or "" + wiki_hint = _wiki_hint_for_subsection(wiki, sub["key"]) + system = EXTRACT_SYSTEM.format( + wiki_hint=wiki_hint or "(no Wikipedia extract available)", + label=label, topics=topic, max_bullets=sub["max_bullets"], + ) + user = USER_LEAD.format(label=label, ctx=cahier_ctx) + try: + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + except Exception as e: # noqa: BLE001 + return [], 0, str(e) + payload, perr = llm_json.parse_facts(raw) + if payload is None: + return [], 0, perr or "no JSON in response" + kept, dropped = _ground_facts(payload.get("facts") or [], cahier_ctx, wiki_hint) + return kept, dropped, "" + + +def _process_record(provider, model_id: str, record: dict) -> dict: + slug = record["slug"] + wiki = record.get("_wiki_record") or {} + cahier_ctx = record.get("_cahier_ctx") or "" + + all_facts: list[dict] = [] + n_dropped_total = 0 + sub_errors: list[tuple[str, str]] = [] + for sub in SUBSECTIONS: + kept, dropped, err = _process_subsection(provider, model_id, record, sub) + n_dropped_total += dropped + if err: + sub_errors.append((sub["key"], err)) + continue + for f in kept: + f["subsection"] = sub["key"] + all_facts.append(f) + + payload = { + "country": "gb", + "source_lang": SOURCE_LANG, + "slug": slug, + "name": record.get("name") or slug, + "facts": all_facts, + "n_dropped": n_dropped_total, + "model": model_id, + "model_kind": provider.kind, + "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "cahier_source_sha": wiki_sha(cahier_ctx), + "cahier_source_pdf_url": (record.get("source") or {}).get("source_url", ""), + "cahier_source_kind": "gov-uk-product-specification", + "wiki_source_revision": wiki.get("revision"), + "wiki_source_url": wiki.get("page_url"), + "subsection_errors": sub_errors, + } + cache.write_json(CACHE_DIR / f"{slug}.json", payload) + return payload + + +def _is_cache_valid(record: dict) -> bool: + slug = record["slug"] + p = CACHE_DIR / f"{slug}.json" + if not p.exists(): + return False + try: + existing = json.loads(p.read_text(encoding="utf-8")) + except (ValueError, OSError): + return False + if existing.get("country") != "gb": + return False + if existing.get("cahier_source_sha") != wiki_sha(record.get("_cahier_ctx") or ""): + return False + wiki = record.get("_wiki_record") or {} + if existing.get("wiki_source_revision") != wiki.get("revision"): + return False + return True + + +# ─────────────────────────────────────────────────────── round-trip flow ── + + +def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: + items: list[dict] = [] + n_records_emitted = 0 + for rec in collect_targets(): + if skip_cached and _is_cache_valid(rec): + continue + if limit and n_records_emitted >= limit: + break + n_records_emitted += 1 + for sub in SUBSECTIONS: + wiki_hint = _wiki_hint_for_subsection(rec.get("_wiki_record") or {}, sub["key"]) + label = SUBSECTION_LABELS[sub["key"]] + items.append({ + "slug": rec["slug"], + "lang": SOURCE_LANG, + "subsection": sub["key"], + "subsection_label": label, + "max_bullets": sub["max_bullets"], + "system_prompt": EXTRACT_SYSTEM.format( + wiki_hint=wiki_hint or "(no Wikipedia extract)", + label=label, topics=SUBSECTION_TOPICS[sub["key"]], + max_bullets=sub["max_bullets"], + ), + "cahier_ctx": rec.get("_cahier_ctx") or "", + "wiki_hint": wiki_hint, + "cahier_source_sha": wiki_sha(rec.get("_cahier_ctx") or ""), + "wiki_source_revision": (rec.get("_wiki_record") or {}).get("revision"), + "facts": [], + }) + cache.write_json(out_path, { + "exported_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "n_items": len(items), + "items": items, + }) + print(f"[02d/gb] wrote {out_path} ({len(items)} items across " + f"{len({i['slug'] for i in items})} wines)", file=sys.stderr) + return 0 + + +def import_todo(in_path: Path, *, translator_id: str, translator_kind: str) -> int: + if not in_path.exists(): + print(f"error: {in_path} does not exist.", file=sys.stderr) + return 1 + payload = json.loads(in_path.read_text(encoding="utf-8")) + by_slug: dict[str, list[dict]] = {} + for it in payload.get("items") or []: + by_slug.setdefault(it["slug"], []).append(it) + record_index = {r["slug"]: r for r in collect_targets()} + counts = {"wrote": 0, "sha_mismatch": 0, "unknown_slug": 0} + for slug, slug_items in by_slug.items(): + rec = record_index.get(slug) + if rec is None: + counts["unknown_slug"] += 1 + continue + cahier_ctx = rec.get("_cahier_ctx") or "" + first = slug_items[0] + if first.get("cahier_source_sha") and first["cahier_source_sha"] != wiki_sha(cahier_ctx): + counts["sha_mismatch"] += 1 + continue + facts: list[dict] = [] + for it in slug_items: + kept, _ = _ground_facts(it.get("facts") or [], cahier_ctx, it.get("wiki_hint") or "") + for f in kept: + f["subsection"] = it.get("subsection") or "facteurs_naturels" + facts.append(f) + wiki = rec.get("_wiki_record") or {} + cache.write_json(CACHE_DIR / f"{slug}.json", { + "country": "gb", "source_lang": SOURCE_LANG, "slug": slug, + "name": rec.get("name") or slug, "facts": facts, + "model": translator_id, "model_kind": translator_kind, + "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "cahier_source_sha": wiki_sha(cahier_ctx), + "cahier_source_pdf_url": (rec.get("source") or {}).get("source_url", ""), + "cahier_source_kind": "gov-uk-product-specification", + "wiki_source_revision": wiki.get("revision"), + "wiki_source_url": wiki.get("page_url"), + }) + counts["wrote"] += 1 + print(f"[02d/gb] wrote {counts['wrote']} cache files; " + f"skipped sha_mismatch={counts['sha_mismatch']}, " + f"unknown_slug={counts['unknown_slug']}", file=sys.stderr) + return 0 + + +# ───────────────────────────────────────────────────────────────── main ── + + +def _build_argparser() -> argparse.ArgumentParser: + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--provider", default="ollama", + choices=("anthropic", "mistral", "ollama", "manual")) + ap.add_argument("--model", default=None) + ap.add_argument("--ollama-url", default=providers.DEFAULT_OLLAMA_URL) + ap.add_argument("--mistral-url", default=providers.DEFAULT_MISTRAL_URL) + ap.add_argument("--workers", type=int, default=1) + ap.add_argument("--limit", type=int, default=0) + ap.add_argument("--only", action="append", default=[]) + ap.add_argument("--refresh", action="store_true") + ap.add_argument("--batch", action="store_true", + help="submit all wines to the provider Batch API") + roundtrip.add_arguments(ap) + return ap + + +def _dispatch_emit_or_import(args) -> int | None: + rc = roundtrip.validate_emit_import(args) + if rc is not None: + return rc + if args.emit_todo: + return emit_todo(Path(args.emit_todo), skip_cached=not args.all, limit=args.limit) + if args.import_path: + return import_todo(Path(args.import_path), + translator_id=args.translator_id, + translator_kind=args.translator_kind) + return None + + +def _run_batch(args) -> int: + if not batch.supports(args.provider): + print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) + return 1 + model_id = args.model or batch.default_model(args.provider) + targets = collect_targets() + if args.only: + needles = [s.lower() for s in args.only] + targets = [r for r in targets if any(n in r["slug"].lower() for n in needles)] + if args.limit: + targets = targets[: args.limit] + targets = [r for r in targets if args.refresh or not _is_cache_valid(r)] + if not targets: + print("[02d/gb] batch: nothing to do.", file=sys.stderr) + return 0 + print(f"[02d/gb] batch: {len(targets)} wines (provider={args.provider}, " + f"model={model_id})", file=sys.stderr) + + def run_loop(prov): + global CACHE_DIR + if getattr(prov, "kind", "") == "collecting": + keep = CACHE_DIR + CACHE_DIR = Path(tempfile.mkdtemp(prefix="batch-02d-gb-")) + try: + for rec in targets: + _process_record(prov, model_id, rec) + finally: + shutil.rmtree(CACHE_DIR, ignore_errors=True) + CACHE_DIR = keep + else: + CACHE_DIR.mkdir(parents=True, exist_ok=True) + for rec in targets: + _process_record(prov, model_id, rec) + + batch.run_two_pass( + provider=args.provider, model=model_id, + sidecar=ROOT / "raw" / ".batch" / "02d-gb.json", + run_loop=run_loop, + ) + return 0 + + +def main() -> int: + args = _build_argparser().parse_args() + if not EXTRACTED.exists(): + print(f"error: {EXTRACTED} missing — run scripts/gb/02_extract_specs.py first", + file=sys.stderr) + return 1 + + sub_rc = _dispatch_emit_or_import(args) + if sub_rc is not None: + return sub_rc + + terroir_verbatim.emit_for_country( + country="gb", extracted_dir=EXTRACTED, cache_dir=CACHE_DIR, + default_source_lang=SOURCE_LANG, cahier_source_kind="gov-uk-product-specification", + only=args.only, log_prefix="[02d/gb]", + ) + + if args.batch: + return _run_batch(args) + + targets = collect_targets() + if args.only: + needles = [s.lower() for s in args.only] + targets = [r for r in targets if any(n in r["slug"].lower() for n in needles)] + if args.limit: + targets = targets[: args.limit] + if not args.refresh: + targets = [r for r in targets if not _is_cache_valid(r)] + if not targets: + print("[02d/gb] nothing to do — all caches valid.", file=sys.stderr) + return 0 + + provider, model_id = providers.make_provider( + args.provider, model=args.model, ollama_url=args.ollama_url, + mistral_url=args.mistral_url, + ) + if provider is None: + print(f"[02d/gb] manual provider: {len(targets)} wines need extraction.", + file=sys.stderr) + return 1 + + CACHE_DIR.mkdir(parents=True, exist_ok=True) + print(f"[02d/gb] {len(targets)} MT wines to extract " + f"(provider={args.provider}, model={model_id}, workers={args.workers})", + file=sys.stderr) + + workers = max(1, args.workers) + n_done = n_facts = 0 + t0 = time.time() + if workers <= 1: + for rec in tqdm(targets, desc="terroir-gb", leave=False): + res = _process_record(provider, model_id, rec) + n_done += 1 + n_facts += len(res.get("facts", [])) + else: + with ThreadPoolExecutor(max_workers=workers) as ex: + futures = {ex.submit(_process_record, provider, model_id, r): r for r in targets} + for fut in tqdm(as_completed(futures), total=len(targets), desc="terroir-gb", leave=False): + try: + res = fut.result() + n_done += 1 + n_facts += len(res.get("facts", [])) + except Exception as e: # noqa: BLE001 + print(f" worker exception: {e}", file=sys.stderr) + + elapsed = time.time() - t0 + MANIFEST.write_text(json.dumps({ + "generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "provider": args.provider, "model": model_id, + "n_wines_processed": n_done, "n_facts_total": n_facts, + "elapsed_seconds": int(elapsed), + }, ensure_ascii=False, indent=2, sort_keys=True), encoding="utf-8") + print(f"[02d/gb] done: {n_done} wines, {n_facts} facts, {elapsed/60:.1f} min", + file=sys.stderr) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/gb/02e_translate_terroir_facts.py b/scripts/gb/02e_translate_terroir_facts.py new file mode 100644 index 0000000..5673a60 --- /dev/null +++ b/scripts/gb/02e_translate_terroir_facts.py @@ -0,0 +1,363 @@ +"""Translate GB per-wine terroir-fact bullets into the missing locales. + +GB analog of `scripts/mt/02e_translate_terroir_facts.py`. The UK's source +language is always English (the DEFRA product specifications are written +in English, and EN is the canonical rendered surface), so the bullets are +already in EN — this stage only produces fr / es / nl +(`{en, fr, es, nl} - {en}`). + +Reads: raw/terroir-facts/<slug>.json (where country == 'gb') +Writes: raw/translations/terroir-facts/<lang>/<slug>.json +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import sys +import time +from concurrent.futures import ThreadPoolExecutor, as_completed +from datetime import datetime, timezone +from pathlib import Path + +from tqdm import tqdm + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 + +TERROIR_FACTS = ROOT / "raw" / "terroir-facts" +CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" + +TARGET_LOCALES = ("en", "fr", "es", "nl") +LOCALE_NAME = {"en": "English", "fr": "French", "es": "Spanish", "nl": "Dutch"} +SOURCE_LANG_NAME = {"en": "English"} + + +SYSTEM_PROMPT_TEMPLATE = """You translate short {source_lang_name} bullets describing a United Kingdom wine appellation's terroir, history, and wine character into {lang_name}. The bullets come from the appellation's DEFRA product specification + English Wikipedia context, aimed at sommelier students and wine enthusiasts. + +Rules: +- Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. Array length MUST equal the input list length. +- Preserve British proper nouns verbatim: appellation and place names (English, Welsh, English Regional, Welsh Regional, Sussex, East Sussex, West Sussex, Darnibole, Cornwall, the South Downs, the Weald, Camel Valley, Plumpton College), geological terms (Kimmeridgian, greensand, chalk, slate), grape names (Bacchus, Seyval Blanc, Madeleine Angevine, Chardonnay, Pinot Noir, Pinot Meunier, …), and the UK scheme terms PDO / PGI. +- Translate descriptive vocabulary naturally for a wine-literate reader. +- Match each source bullet's length and register; do not add commentary, footnotes, or explanations. +- Output ONLY the JSON array, no preface, no markdown fences.""" + + +def facts_sha(facts: list[dict]) -> str: + blob = "\n".join((f.get("bullet") or "") for f in facts) + return hashlib.sha256(blob.encode("utf-8")).hexdigest() + + +def parse_array(raw: str, expected_len: int) -> list[str] | None: + arr, _err = llm_json.parse_str_array(raw, expected_len) + return arr + + +def cache_path(lang: str, slug: str) -> Path: + return CACHE_ROOT / lang / f"{slug}.json" + + +def load_existing(lang: str, slug: str) -> dict | None: + return cache.read_json_or_none(cache_path(lang, slug)) + + +def write_cache( + *, lang: str, slug: str, source_lang: str, src_facts: list[dict], + translated_bullets: list[str], src_data: dict, + translator: str, translator_kind: str, +) -> None: + facts_out = [ + { + "bullet": translated_bullets[i], + "subsection": src_facts[i].get("subsection", "facteurs_naturels"), + "provenance": src_facts[i].get("provenance", "wiki"), + } + for i in range(len(src_facts)) + ] + payload = { + "country": "gb", + "source_lang": source_lang, + "slug": slug, + "lang": lang, + "facts": facts_out, + "source_facts_sha": facts_sha(src_facts), + "wiki_source_url": src_data.get("wiki_source_url") or "", + "cahier_source_pdf_url": src_data.get("cahier_source_pdf_url") or "", + "translator": translator, + "translator_kind": translator_kind, + "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + } + cache.write_json(cache_path(lang, slug), payload) + + +def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: + return bool( + existing + and existing.get("source_facts_sha") == sha + and len(existing.get("facts") or []) == expected_len + ) + + +def _load_gb_source(path: Path) -> dict | None: + d = cache.read_json_or_none(path) + if not d or not d.get("facts"): + return None + if d.get("country") != "gb": + return None + return d + + +def enumerate_jobs(languages: tuple[str, ...], *, skip_cached: bool = True) -> list[dict]: + jobs: list[dict] = [] + for f in sorted(TERROIR_FACTS.glob("*.json")): + if f.name.startswith("_") or f.name.startswith("manifest"): + continue + src = _load_gb_source(f) + if src is None: + continue + source_lang = src.get("source_lang") or "en" + src_facts = src["facts"] + sha = facts_sha(src_facts) + for lang in languages: + if lang == source_lang: + continue + if skip_cached and _is_fresh_cache( + load_existing(lang, src["slug"]), sha, len(src_facts), + ): + continue + jobs.append({ + "slug": src["slug"], "lang": lang, "source_lang": source_lang, + "src_facts": src_facts, "src_data": src, "sha": sha, + }) + return jobs + + +def build_user_prompt(src_facts: list[dict]) -> str: + lines = [f"{i + 1}. {f.get('bullet', '')}" for i, f in enumerate(src_facts)] + return "Translate the following bullets:\n\n" + "\n".join(lines) + + +def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: + system = SYSTEM_PROMPT_TEMPLATE.format( + source_lang_name=SOURCE_LANG_NAME.get(job["source_lang"], job["source_lang"]), + lang_name=LOCALE_NAME[job["lang"]], + ) + user = build_user_prompt(job["src_facts"]) + try: + raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + except Exception as e: # noqa: BLE001 + return None, f"call: {e}" + parsed = parse_array(raw, len(job["src_facts"])) + if parsed is None: + return None, f"parse_or_length_mismatch: {raw[:200]!r}" + return parsed, None + + +def emit_todo(out_path: Path, *, languages: tuple[str, ...], skip_cached: bool) -> int: + jobs = enumerate_jobs(languages, skip_cached=skip_cached) + items = [ + { + "slug": j["slug"], "lang": j["lang"], "source_lang": j["source_lang"], + "source_facts_sha": j["sha"], + "source_bullets": [f.get("bullet", "") for f in j["src_facts"]], + "translated_bullets": [], + } + for j in jobs + ] + cache.write_json(out_path, { + "exported_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "languages": list(languages), "n_items": len(items), "items": items, + }) + counts = {loc: sum(1 for j in jobs if j["lang"] == loc) for loc in languages} + pretty = ", ".join(f"{loc}={n}" for loc, n in counts.items()) + print(f"[02e/gb] wrote {out_path} ({len(items)} items: {pretty})", file=sys.stderr) + return 0 + + +def _import_one(it: dict, src_index: dict, *, translator_id: str, translator_kind: str) -> str: + slug = it.get("slug") or "" + lang = it.get("lang") or "" + if lang not in TARGET_LOCALES: + return "skipped_unknown_lang" + src = src_index.get(slug) + if src is None: + return "skipped_unknown_slug" + src_facts = src.get("facts") or [] + if not src_facts: + return "skipped_unknown_slug" + if it.get("source_facts_sha") and it["source_facts_sha"] != facts_sha(src_facts): + return "skipped_sha" + translated = it.get("translated_bullets") or [] + if len(translated) != len(src_facts) or any(not (s or "").strip() for s in translated): + return "skipped_empty" + write_cache( + lang=lang, slug=slug, + source_lang=it.get("source_lang") or src.get("source_lang") or "en", + src_facts=src_facts, translated_bullets=[s.strip() for s in translated], + src_data=src, translator=translator_id, translator_kind=translator_kind, + ) + return "wrote" + + +def import_todo(in_path: Path, *, translator_id: str, translator_kind: str) -> int: + if not in_path.exists(): + print(f"error: {in_path} does not exist.", file=sys.stderr) + return 1 + payload = json.loads(in_path.read_text(encoding="utf-8")) + src_index: dict[str, dict] = {} + for f in TERROIR_FACTS.glob("*.json"): + if f.name.startswith("_") or f.name.startswith("manifest"): + continue + try: + d = json.loads(f.read_text(encoding="utf-8")) + except Exception: # noqa: BLE001 + continue + if d.get("country") == "gb": + src_index[d["slug"]] = d + counts = {"wrote": 0, "skipped_sha": 0, "skipped_empty": 0, + "skipped_unknown_slug": 0, "skipped_unknown_lang": 0} + for it in payload.get("items") or []: + counts[_import_one(it, src_index, translator_id=translator_id, + translator_kind=translator_kind)] += 1 + pretty = ", ".join(f"{k}={v}" for k, v in counts.items() if v) + print(f"[02e/gb] {pretty}", file=sys.stderr) + return 0 + + +def _build_argparser() -> argparse.ArgumentParser: + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--provider", default="ollama", + choices=("anthropic", "mistral", "ollama", "manual")) + ap.add_argument("--model", default=None) + ap.add_argument("--ollama-url", default=providers.DEFAULT_OLLAMA_URL) + ap.add_argument("--mistral-url", default=providers.DEFAULT_MISTRAL_URL) + ap.add_argument("--lang", action="append", choices=TARGET_LOCALES, default=None) + ap.add_argument("--limit", type=int, default=0) + ap.add_argument("--workers", type=int, default=1) + ap.add_argument("--refresh", action="store_true") + ap.add_argument("--batch", action="store_true") + roundtrip.add_arguments(ap) + return ap + + +def _process_one_job(provider, model_id: str, job: dict) -> tuple[bool, str | None]: + translated, err = translate_one(provider, job) + if err or translated is None: + return False, err or "unknown" + write_cache( + lang=job["lang"], slug=job["slug"], source_lang=job["source_lang"], + src_facts=job["src_facts"], translated_bullets=translated, + src_data=job["src_data"], translator=model_id, translator_kind=provider.kind, + ) + return True, None + + +def _run_translation_loop(provider, model_id: str, jobs: list[dict], workers: int = 1): + done = 0 + skipped: list[tuple[str, str, str]] = [] + if workers <= 1: + for job in tqdm(jobs, desc="translate-gb", leave=False): + ok, err = _process_one_job(provider, model_id, job) + if ok: + done += 1 + else: + skipped.append((job["lang"], job["slug"], err or "unknown")) + time.sleep(0.05) + return done, skipped + with ThreadPoolExecutor(max_workers=workers) as ex: + futures = {ex.submit(_process_one_job, provider, model_id, j): j for j in jobs} + for fut in tqdm(as_completed(futures), total=len(jobs), desc="translate-gb", leave=False): + job = futures[fut] + try: + ok, err = fut.result() + except Exception as e: # noqa: BLE001 + skipped.append((job["lang"], job["slug"], f"worker exception: {e}")) + continue + if ok: + done += 1 + else: + skipped.append((job["lang"], job["slug"], err or "unknown")) + return done, skipped + + +def _run_batch(args, languages: tuple[str, ...]) -> int: + if not batch.supports(args.provider): + print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) + return 1 + model_id = args.model or batch.default_model(args.provider) + jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.limit: + jobs = jobs[: args.limit] + if not jobs: + print("[02e/gb] batch: nothing to do.", file=sys.stderr) + return 0 + print(f"[02e/gb] batch: {len(jobs)} translations (provider={args.provider}, " + f"model={model_id}, locales={','.join(languages)})", file=sys.stderr) + batch.run_two_pass( + provider=args.provider, model=model_id, + sidecar=ROOT / "raw" / ".batch" / "02e-gb.json", + run_loop=lambda prov: _run_translation_loop(prov, model_id, jobs, workers=1), + ) + return 0 + + +def _dispatch_emit_or_import(args, languages: tuple[str, ...]) -> int | None: + rc = roundtrip.validate_emit_import(args) + if rc is not None: + return rc + if args.emit_todo: + return emit_todo(Path(args.emit_todo), languages=languages, skip_cached=not args.all) + if args.import_path: + return import_todo(Path(args.import_path), translator_id=args.translator_id, + translator_kind=args.translator_kind) + return None + + +def main() -> int: + args = _build_argparser().parse_args() + if not TERROIR_FACTS.exists(): + print("error: raw/terroir-facts is missing — run " + "scripts/gb/02d_extract_terroir_facts.py first", file=sys.stderr) + return 1 + languages = tuple(args.lang) if args.lang else TARGET_LOCALES + + sub_rc = _dispatch_emit_or_import(args, languages) + if sub_rc is not None: + return sub_rc + + if args.batch: + return _run_batch(args, languages) + + jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.limit: + jobs = jobs[: args.limit] + if not jobs: + print("[02e/gb] nothing to do — all caches up to date.", file=sys.stderr) + return 0 + + provider, model_id = providers.make_provider( + args.provider, model=args.model, ollama_url=args.ollama_url, + mistral_url=args.mistral_url, + ) + if provider is None: + print(f"[02e/gb] manual provider: {len(jobs)} entries need translation.", + file=sys.stderr) + return 1 + + workers = max(1, args.workers) + print(f"[02e/gb] {len(jobs)} translations to fetch " + f"(provider={args.provider}, model={model_id}, " + f"locales={','.join(languages)}, workers={workers})", file=sys.stderr) + done, skipped = _run_translation_loop(provider, model_id, jobs, workers=workers) + print(f"[02e/gb] translated: {done}, skipped: {len(skipped)}", file=sys.stderr) + if skipped: + for lang, slug, err in skipped[:20]: + print(f" skip {lang}/{slug}: {err[:160]}", file=sys.stderr) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/gb/03_generate_wiki.py b/scripts/gb/03_generate_wiki.py new file mode 100644 index 0000000..afd73da --- /dev/null +++ b/scripts/gb/03_generate_wiki.py @@ -0,0 +1,265 @@ +"""Generate one wiki/<slug>.md per GB wine record + extend +wiki/_index.json with GB entries. + +Pipeline stage 03 (gb). Mirrors `scripts/mt/03_generate_wiki.py` — the +other English-source corpus — for the United Kingdom. Reads +`raw/gb/specs-extracted/*.json`, emits per-record markdown pages with +English section headings, and merges GB entries into `wiki/_index.json` +(preserving entries from the other countries). + +Sources cited per page are the UK ones: the DEFRA product specification +and the GOV.UK register entry. No EUR-Lex or eAmbrosia link is emitted — +the UK register is the authority for these names. +""" + +from __future__ import annotations + +import argparse +import json +import re +import sys +from datetime import datetime, timezone +from pathlib import Path + +from tqdm import tqdm + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT / "scripts")) +from _lib.gb.region import derive_region # noqa: E402 + +EXTRACTED = ROOT / "raw" / "gb" / "specs-extracted" +WIKI = ROOT / "wiki" +WIKI_INDEX = WIKI / "_index.json" +TERROIR_FACTS = ROOT / "raw" / "terroir-facts" +APPELLATION_NOTES = ROOT / "scripts" / "_lib" / "appellation_notes.json" + +_FACTS_SLUGS: frozenset[str] | None = None +_NOTES: dict[str, dict] | None = None + + +def _terroir_facts_slugs() -> frozenset[str]: + global _FACTS_SLUGS + if _FACTS_SLUGS is None: + slugs: set[str] = set() + if TERROIR_FACTS.exists(): + for p in TERROIR_FACTS.glob("*.json"): + if p.stem.startswith("manifest"): + continue + try: + if json.loads(p.read_text(encoding="utf-8")).get("facts"): + slugs.add(p.stem) + except (ValueError, OSError): + continue + _FACTS_SLUGS = frozenset(slugs) + return _FACTS_SLUGS + + +def _appellation_notes() -> dict[str, dict]: + global _NOTES + if _NOTES is None: + _NOTES = {} + if APPELLATION_NOTES.exists(): + try: + raw = json.loads(APPELLATION_NOTES.read_text(encoding="utf-8")) + _NOTES = {k: v for k, v in raw.items() if not k.startswith("__")} + except (ValueError, OSError): + _NOTES = {} + return _NOTES + + +SECTION_LABELS = { + "summary": "Summary", + "geo": "Demarcated area", + "grapes": "Grape varieties", + "link": "Link with the geographical area", + "note": "Note", + "sources": "Sources", +} + + +def _resolve_region(record: dict) -> str: + return record.get("region") or derive_region({ + "file_number": record.get("file_number") or "", + "demarcation": record.get("demarcation") or "", + }) + + +def _truncate(text: str, max_chars: int) -> str: + text = re.sub(r"\s+", " ", text).strip() + if len(text) <= max_chars: + return text + cut = text[:max_chars].rsplit(". ", 1)[0] + return cut + ("." if not cut.endswith(".") else "") + + +def _render_note_section(slug: str) -> list[str]: + note = _appellation_notes().get(slug) + if not note: + return [] + text = ((note.get("note") or {}).get("en") or "").strip() + if not text: + return [] + body = [f"## {SECTION_LABELS['note']}", "", text, ""] + sources = note.get("sources") or [] + if sources: + for s in sources: + label = (s.get("label") or "").strip() + url = (s.get("url") or "").strip() + if label and url: + body.append(f"- [{label}]({url})") + body.append("") + return body + + +def render_record(record: dict) -> str: + name = record["name"] + slug = record["slug"] + kind = record.get("kind", "DOP") + region = _resolve_region(record) + + src = record.get("source") or {} + summary = (record.get("summary") or "").strip() + geo = (record.get("geo_area_brief") or "").strip() + link = (record.get("link_to_terroir") or "").strip() + grape_details = (record.get("grapes") or {}).get("details") or [] + + fm = [ + "---", + f"title: {name}", + f"type: {kind.lower()}", + f"slug: {slug}", + "country: gb", + f"region: {region}", + f"kind: {kind}", + f"file_number: {record.get('file_number') or ''}", + f"parser_template: {record.get('parser_template') or ''}", + ] + if record.get("stub"): + fm += [ + "stub: true", + f"stub_reason: {record.get('stub_reason') or ''}", + ] + fm += [ + "sources:", + f" product_specification: {src.get('source_url') or ''}", + f" register_entry: {src.get('register_url') or ''}", + f" specification_filename: {src.get('filename') or ''}", + "---", + "", + f"# {name}", + "", + ] + + body: list[str] = [] + + _facts = _terroir_facts_slugs() + if summary and slug not in _facts: + body += [ + f"## {SECTION_LABELS['summary']}", + "", + _truncate(summary, max_chars=1200), + "", + ] + + if geo: + body += [ + f"## {SECTION_LABELS['geo']}", + "", + _truncate(geo, max_chars=2000), + "", + ] + + if grape_details: + body += [ + f"## {SECTION_LABELS['grapes']}", + "", + ", ".join(d.get("name") or d.get("slug") for d in grape_details), + "", + ] + + if link: + body += [ + f"## {SECTION_LABELS['link']}", + "", + _truncate(link, max_chars=2000), + "", + ] + + body += _render_note_section(slug) + + body += [ + f"## {SECTION_LABELS['sources']}", + "", + f"- Product specification (DEFRA): <{src.get('source_url') or ''}>", + f"- UK GI register entry: <{src.get('register_url') or ''}>", + f"- File number: `{record.get('file_number') or ''}`", + "", + "_Product-specification text: © Crown copyright, DEFRA / GOV.UK. " + "Licensed under the Open Government Licence v3.0._", + "", + ] + return "\n".join(fm + body) + + +def index_entry(record: dict) -> dict: + return { + "country": "gb", + "file_number": record.get("file_number") or "", + "name": record["name"], + "kind": record.get("kind", "DOP"), + "region": _resolve_region(record), + "is_sub_denomination": False, + "parent_slug": "", + "parent_name": "", + "categories": [record.get("kind", "DOP")], + "stub": False, + "page": f"{record['slug']}.md", + } + + +def main() -> int: + if not EXTRACTED.exists(): + print(f"error: {EXTRACTED} missing — run scripts/gb/02_extract_specs.py first", + file=sys.stderr) + return 1 + + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--only", action="append", default=[]) + args = ap.parse_args() + + files = sorted(p for p in EXTRACTED.glob("*.json") if not p.name.startswith("_")) + if args.only: + needles = [s.lower() for s in args.only] + files = [p for p in files if any(n in p.stem.lower() for n in needles)] + + WIKI.mkdir(parents=True, exist_ok=True) + written = 0 + gb_index: dict[str, dict] = {} + for f in tqdm(files, desc="gb-wiki", leave=False): + rec = json.loads(f.read_text(encoding="utf-8")) + out_path = WIKI / f"{rec['slug']}.md" + out_path.write_text(render_record(rec), encoding="utf-8") + gb_index[rec["slug"]] = index_entry(rec) + written += 1 + + existing: dict[str, dict] = {} + if WIKI_INDEX.exists(): + try: + existing = json.loads(WIKI_INDEX.read_text(encoding="utf-8")) + except (ValueError, OSError): + existing = {} + other_kept = {k: v for k, v in existing.items() if v.get("country") != "gb"} + merged = {**other_kept, **gb_index} + WIKI_INDEX.write_text(json.dumps(merged, ensure_ascii=False, indent=2, sort_keys=True), + encoding="utf-8") + print( + f"[gb/03] wrote {written} GB wiki pages, merged index " + f"({len(other_kept)} non-GB + {len(gb_index)} GB = {len(merged)} entries) " + f"@ {datetime.now(timezone.utc).isoformat(timespec='seconds')}", + file=sys.stderr, + ) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/fixtures/gb_defra_application_darnibole.txt b/tests/fixtures/gb_defra_application_darnibole.txt new file mode 100644 index 0000000..7e70180 --- /dev/null +++ b/tests/fixtures/gb_defra_application_darnibole.txt @@ -0,0 +1,31 @@ +Darnibole Bacchus wine – Protected Designation of +Origin (PDO) +1. Details of protection +a) Name(s) to be registered: DARNIBOLE b) Equivalent term(s): Darnibole Bacchus c) + +Geographical indication type: PDO + +3. Product details +a) Category +Wine + +b) Description +Darnibole Bacchus is a dry still Cornish wine produced at Camel Valley winery from +grapes grown in a unique and delineated area known as Darnibole. + +4. Wine grape variety: +100% Bacchus. + +6. Maximum yield +8 tonnes of grape per hectare + +7. Demarcated area +a) NUTS Area +UKK3, UKK30 [Cornwall and Isles of Scilly] + +b) Definition of the demarcated area +The whole 5 hectare area, demarcated as Darnibole, is based on an area of ancient slate +subsoil bordered to the West by a marked soil change to alluvial sand (the old River Camel +river bed) and to the East the land levels out providing for not enough of a steep slope with +the aspect also starting to face slightly East of South. The disused railway (now the Camel +Trail) demarcates the Southern boundary. diff --git a/tests/fixtures/gb_defra_pfn_2011_english.txt b/tests/fixtures/gb_defra_pfn_2011_english.txt new file mode 100644 index 0000000..0ea0550 --- /dev/null +++ b/tests/fixtures/gb_defra_pfn_2011_english.txt @@ -0,0 +1,73 @@ +Department for Environment Food and Rural Affairs + +December 2011 + +ENGLISH WINE - PROTECTED DESIGNATION OF ORIGIN (PDO) + + +PROTECTED NAME: ENGLISH + +DEMARCATION: ENGLAND + +(Wine produced from vines growing below a height of 220 metres above sea level are +eligible for the Scheme). + + PART 1: STILL WINE + +English Wine is made from grapes grown close to the limit for viticulture. All vineyards +are positioned at above 49.9 degrees north leading to long daylight hours in the +growing season. The climate is temperate with few summer days above 30°C. The +diurnal temperature range is high. + +The northerly latitude of the vineyards in this PDO creates the long growing season +and long daylight hours that are key to the development of strong aromatic flavours. + + +SPECIFICATION + +ACIDIFICATION, DE-ACIDIFICATION AND SWEETENING + +De-acidification of wines is permitted only up to a limit of 1g/l expressed as +tartaric acid. + +VINE VARIETIES + +Acolon; Bacchus; Black Hamburg; Blau +Portugueser; Cabernet Blanc; Gamaret; Gamay +Garanoir; Huxelrebe; Kernling; Madeleine +Angevine; Madeleine Sylvaner; Marechal Foch; Ortega; Petit Meslier; +Roter Veltliner; Rulander (Synonyms: Pinot Gris, Pinot Grigio); Senator; Seyval Blanc; +Solaris; Triomphe; Zala Gyongye; Zweigeltrebe + +WINE-MAKING METHODS + +During the process of harvesting, wine-making and storage, wine-makers must ensure +that potential individual English Wines are distinguishable from other Wines. + +MAXIMUM YIELDS: 80 hl/ha + + + PART 2: QUALITY SPARKLING WINE + +English Quality Sparkling Wines are made from grapes grown close to the limit for +viticulture. The moderate temperatures lead to the high acidity and low pH which is +the backbone of fine sparkling wines. + +SPECIFICATION + +VINE VARIETIES + +English Quality Sparkling Wines shall be made from the following vine varieties: + Chardonnay + Pinot Noir + Pinot Noir Précoce + Pinot Meunier + Pinot Blanc + Pinot Gris + +MAXIMUM YIELDS: 80 hl/ha + + + PART 3: GENERAL PROVISIONS + +The control body is Wine Standards of the Food Standards Agency. diff --git a/tests/fixtures/gb_uk_single_document_sussex.txt b/tests/fixtures/gb_uk_single_document_sussex.txt new file mode 100644 index 0000000..a534845 --- /dev/null +++ b/tests/fixtures/gb_uk_single_document_sussex.txt @@ -0,0 +1,36 @@ +Product specification for Sussex +Demarcation: East and West Sussex +PDO GB number: W0006 +1. Applicant(s) +Name: Sussex Wine Producers (SWP) +3. Details of protection +3.1 Name of product to be registered +'Sussex' +5. Category of the grapevine products: +'Sussex' sparkling wine +Traditional method quality sparkling wine +'Sussex' still wine +Quality still wine +6. Description of the wine(s) +'Sussex' sparkling wine is a traditional method quality sparkling white, rosé and red wine made from classic sparkling wine grape varieties grown within the administrative boundaries of the counties of East and West Sussex. +7.2 Viticulture practices +'Sussex' sparkling wine +Permitted Grape Vine Varieties: Sussex sparkling wine shall be made from the following grape varieties: +Chardonnay, Pinot Noir, Pinot Meunier, Arbanne, Pinot Gris, Pinot Blanc, Petit Meslier and Pinot Noir Précoce. +The vineyard owner must keep detailed records of yield and size of each parcel of vineyard land and make those available for inspection by Wine Standards and the Scheme Manager to confirm PDO status. +Harvest yields: Under normal conditions the maximum harvest yield shall be 12 tonnes per hectare. +(Note*: This level will only be authorized in exceptional circumstances, when it can be proved to the SWP that sugar levels, acidity and flavour are not being jeopardised, or in fact where conditions are such that reducing yield could be detrimental to the quality of the crop.) +'Sussex' still wine +Permitted Grape Vine Varieties: The following grape varieties are permitted within the Sussex PDO for still wine production: +Acolon, Albarino, Auxerrois, Bacchus, Chardonnay, Dornfelder, Gamay, Huxelrebe, Müller- Thurgau, Orion, Ortega, Pinot Blanc, Pinot Gris, Pinot Meunier, Pinot Noir, Pinot Noir, Précoce, Regent, Regner, Reichensteiner, Riesling, Rondo, Roter Veltliner, Sauvignon Blanc, Schonburger, Siegerrebe, Solaris. +8. Definition of the demarcated area +East and West Sussex +The current administrative boundary of West Sussex rises up from a point east of the town of Emsworth. +9. Link +9.1 Details of the geographical area (natural and human factors) +Soils +The majority of Sussex vineyards are based either on the chalk of the South Downs or on the mineral rich greensands to the north. +9.3 Link between the characteristics of the geographical area and the quality of the wine. +The high latitude of the Sussex vineyards gives the opportunity to have a longer growing season. +10. Proof of origin +During the process of harvest and winemaking, the producer must ensure that records are kept. diff --git a/tests/test_gb_parser.py b/tests/test_gb_parser.py new file mode 100644 index 0000000..27deaa2 --- /dev/null +++ b/tests/test_gb_parser.py @@ -0,0 +1,241 @@ +"""Fixture-based regression tests for the United Kingdom (GB) parsers. + +One parser module, `scripts/_lib/gb/spec.py`, covering three document +layouts — the UK register is the only corpus where a single country ships +three unrelated templates: + + - **defra-pfn-2011** — the December-2011 DEFRA specifications + (English/Welsh PDO + their Regional PGIs). A header block, then one + `PART n: <CATEGORY>` per grapevine category, each with an upper-case + `SPECIFICATION` subsection run. Two seams matter here: + * The still roster is semicolon-separated and **wraps across PDF + lines**, so a run of roster lines must be rejoined with a SPACE + ("… Black Hamburg; Blau\\nPortugueser …" is one variety, not two). + * The sparkling roster is bulleted one-per-line with NO separators + (pdftotext renders the Wingdings bullet as U+F0B7), so that run + must instead be rejoined with a SEPARATOR — the opposite rule. + Getting this backwards merges six varieties into one string. + Plus the two PART rosters must not bleed into each other + ("Zweigeltrebe" + "Acolon" -> "Zweigeltrebe Acolon"). + + - **uk-gi-single-document** — Sussex, the only post-Brexit UK-scheme + registration. Numbered EU-style outline; `9. Link` is an empty header + whose content lives entirely in 9.1 / 9.3, so routing must gather + children. Its variety rosters sit in a section that also carries + record-keeping and yield-dispensation PROSE, which must not reach the + grape matcher. + + - **defra-pfn-application** — Darnibole, on the 2017 EU application + form. It has NO link section, so `link_to_terroir` must fall back to + `7 b) Definition of the demarcated area` (where the slate subsoil and + the slope actually are), and its single variety is stated as a + proportion ("100% Bacchus"). + +Two documented source typos are repaired structurally rather than through +`GRAPE_ALIAS`, because they split one name into two before any alias +could apply: Sussex's "Pinot Noir, Pinot Noir, Précoce" and the DEFRA +rosters' missing semicolon between Gamay and Garanoir. + +Also pins `scripts/_lib/gb/darnibole.py` — the boundary reconstructed +from the OS/RPA parcel references on the specification's own plan. + +Real cached documents live under raw/gb/specs/ (gitignored); the fixtures +here are short, redacted excerpts. +""" +from __future__ import annotations + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.gb import darnibole # noqa: E402 +from _lib.gb.region import derive_region # noqa: E402 +from _lib.gb.spec import detect_template, grape_candidates, parse_spec # noqa: E402 + +# ─────────────────────────────────────────────── defra-pfn-2011 (English) ── + + +def test_defra_2011_template_and_header(fixture_text): + parsed = parse_spec(fixture_text("gb_defra_pfn_2011_english.txt")) + assert parsed["template"] == "defra-pfn-2011" + assert parsed["protected_name"] == "ENGLISH" + assert parsed["demarcation"] == "ENGLAND" + # GENERAL PROVISIONS carries no wine and must be dropped. + assert parsed["part_titles"] == ["STILL WINE", "QUALITY SPARKLING WINE"] + + +def test_defra_2011_roles(fixture_text): + roles = parse_spec(fixture_text("gb_defra_pfn_2011_english.txt"))["roles"] + assert roles["geo_area"] == "ENGLAND" + # The link narrative is the pre-SPECIFICATION prose of each wine part. + assert "49.9 degrees north" in roles["link_to_terroir"] + assert "backbone of fine sparkling wines" in roles["link_to_terroir"] + # …and must NOT swallow the regulatory subsections that follow it. + assert "tartaric acid" not in roles["link_to_terroir"] + assert "Still Wine: 80 hl/ha" in roles["yields"] + + +def test_defra_2011_line_wrapped_names_survive(fixture_text): + """A roster wrapped by the PDF layout must rejoin with a space.""" + names = grape_candidates( + parse_spec(fixture_text("gb_defra_pfn_2011_english.txt"))["roles"]["grape_varieties"] + ) + assert "Blau Portugueser" in names + assert "Madeleine Angevine" in names + assert "Marechal Foch" in names + assert "Blau" not in names and "Portugueser" not in names + assert "Angevine" not in names + + +def test_defra_2011_bulleted_roster_splits_per_line(fixture_text): + """The sparkling roster has no separators — one variety per bullet.""" + names = grape_candidates( + parse_spec(fixture_text("gb_defra_pfn_2011_english.txt"))["roles"]["grape_varieties"] + ) + for expected in ("Chardonnay", "Pinot Noir", "Pinot Noir Précoce", + "Pinot Meunier", "Pinot Blanc", "Pinot Gris"): + assert expected in names, expected + # The bullet glyph itself must never survive into a candidate. + assert not any("\uf0b7" in n for n in names) + + +def test_defra_2011_parts_do_not_bleed_together(fixture_text): + """The last name of PART 1's roster must not glue to PART 2's first.""" + names = grape_candidates( + parse_spec(fixture_text("gb_defra_pfn_2011_english.txt"))["roles"]["grape_varieties"] + ) + assert "Zweigeltrebe" in names + assert not any(n.startswith("Zweigeltrebe ") for n in names) + + +def test_defra_2011_source_typo_and_synonym_tail(fixture_text): + names = grape_candidates( + parse_spec(fixture_text("gb_defra_pfn_2011_english.txt"))["roles"]["grape_varieties"] + ) + # Missing semicolon in the source: alphabetical order (Gamaret, Gamay, + # Garanoir, Gewurztraminer) proves these are two entries. + assert "Gamay" in names and "Garanoir" in names + assert "Gamay Garanoir" not in names + # Parenthesised synonym tails are stripped; the head name is the one + # the regulator authorises. + assert "Rulander" in names + assert not any(n.startswith("Rulander (") for n in names) + + +# ────────────────────────────────────────── uk-gi-single-document (Sussex) ── + + +def test_sussex_template_and_link_gathers_children(fixture_text): + parsed = parse_spec(fixture_text("gb_uk_single_document_sussex.txt")) + assert parsed["template"] == "uk-gi-single-document" + assert parsed["demarcation"] == "East and West Sussex" + link = parsed["roles"]["link_to_terroir"] + # `9. Link` is an empty header — 9.1 and 9.3 are its content. + assert "chalk of the South Downs" in link + assert "longer growing season" in link + + +def test_sussex_prose_never_reaches_the_matcher(fixture_text): + names = grape_candidates( + parse_spec(fixture_text("gb_uk_single_document_sussex.txt"))["roles"]["grape_varieties"] + ) + joined = " | ".join(names) + for prose in ("vineyard owner", "exceptional circumstances", "tonnes per hectare", + "jeopardised", "Scheme Manager"): + assert prose not in joined, prose + + +def test_sussex_source_typos_repaired(fixture_text): + names = grape_candidates( + parse_spec(fixture_text("gb_uk_single_document_sussex.txt"))["roles"]["grape_varieties"] + ) + # "… Pinot Noir, Pinot Noir, Précoce, Regent …" — a stray comma splits + # one name in two; no alias could repair it after the split. + assert "Pinot Noir Précoce" in names + assert "Précoce" not in names + # Hyphenation across a line break. + assert "Müller-Thurgau" in names + assert "Müller- Thurgau" not in names + assert "Regent" in names + + +# ──────────────────────────────────── defra-pfn-application (Darnibole) ── + + +def test_darnibole_template_and_terroir_fallback(fixture_text): + parsed = parse_spec(fixture_text("gb_defra_application_darnibole.txt")) + assert parsed["template"] == "defra-pfn-application" + # No link section exists, so the demarcated-area definition stands in. + link = parsed["roles"]["link_to_terroir"] + assert "ancient slate" in link + assert link == parsed["roles"]["geo_area"] + + +def test_darnibole_proportion_prefix_stripped(fixture_text): + names = grape_candidates( + parse_spec(fixture_text("gb_defra_application_darnibole.txt"))["roles"]["grape_varieties"] + ) + assert names == ["Bacchus"] + + +# ───────────────────────────────────────────────── darnibole geometry ── + + +def test_darnibole_hull_matches_the_declared_area(): + """The hull of the seven demarcated parcel centroids should land close + to the specification's declared "whole 5 hectare area".""" + hull = darnibole.boundary_bng() + assert len(darnibole.DEMARCATED_PARCELS) == 7 + hectares = hull.area / 1e4 + assert 5.0 <= hectares <= 8.0, hectares + minx, miny, maxx, maxy = hull.bounds + # OS 1 km square SX 03 67, on the south-facing slope above the Camel. + assert 203000 <= minx and maxx <= 204000 + assert 67000 <= miny and maxy <= 68000 + + +def test_darnibole_parcels_are_not_the_excluded_ones(): + """Parcels drawn on the plan but outside the red line stay out.""" + assert not set(darnibole.DEMARCATED_PARCELS) & set(darnibole.EXCLUDED_PARCELS) + + +# ──────────────────────────────────────────────────────── region facet ── + + +def test_region_by_file_number_and_demarcation_fallback(): + assert derive_region({"file_number": "PDO-GB-A1585"}) == "England" + assert derive_region({"file_number": "PGI-GB-A1590"}) == "Wales" + assert derive_region({"file_number": "PDO-GB-02365"}) == "England" # Sussex + assert derive_region({"file_number": "PDO-GB-N1636"}) == "England" # Darnibole + # A GI registered after this table was written falls back to the + # specification's own DEMARCATION field. + assert derive_region({"file_number": "PDO-GB-9999", + "demarcation": "WALES"}) == "Wales" + assert derive_region({"file_number": "PDO-GB-9999"}) == "United Kingdom" + + +def test_demarcation_survives_an_empty_area_section(): + """An explicit "Demarcation:" line must win even when no area section + parses — the region facet falls back to it, so losing it would drop a + future GI to "United Kingdom".""" + from _lib.gb.spec import parse_numbered_spec + + parsed = parse_numbered_spec( + "Product specification for Testshire\n" + "Demarcation: Testshire\n" + "3. Details of protection\n" + "3.1 Name of product to be registered\n" + "Testshire\n", + "uk-gi-single-document", + ) + assert parsed["roles"]["geo_area"] == "" + assert parsed["demarcation"] == "Testshire" + + +def test_detect_template_is_stable_across_all_three(fixture_text): + assert detect_template(fixture_text("gb_defra_pfn_2011_english.txt")) == "defra-pfn-2011" + assert detect_template( + fixture_text("gb_defra_application_darnibole.txt")) == "defra-pfn-application" + assert detect_template( + fixture_text("gb_uk_single_document_sussex.txt")) == "uk-gi-single-document"