From 8a26508222d307f88c9e4b9108a620640672403b Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Wed, 3 Jun 2026 21:42:47 +0200 Subject: [PATCH 01/41] SEO/GEO phase 1: externalize map-data bundle + deploy MIME/redirect guards MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Externalize the inline JS data out of each locale homepage so the HTML shell is small and the data caches across pages — the prerequisite for the per-appellation pre-rendered pages in later phases. - map_template.render(): the ~12.3 MB AOCS + ~0.6 MB grape-tooltip blobs are no longer inlined; they ship in a hashed, render-blocking /data/aocs...js that assigns window.__OWM_DATA before the app runs (read synchronously, no boot re-thread). render() now returns (html, data_filename, data_bytes). Homepage HTML 13.3 MB -> 449 KB (-96.6%). - grapes_info is serialized with sort_keys so the cache-busting hash is deterministic across runs (its dict order is set-derived); aocs keeps its build order, which drives the sidebar appellation list. Reproducible-build / no-op-rerun contract holds. - 04_build_maps.py: write the per-locale bundle to wiki/data/ and prune stale hashed files. - deploy.py: set Content-Type by file extension on upload (html/json/js/xml/ webmanifest/...; pmtiles and others keep octet-stream), and a non-fatal post-deploy check that warns if the apex host isn't 301-redirecting to www (guards the Search Console "Duplicate without user-selected canonical" issue; the redirect itself is a Bunny edge rule, not in this repo). Co-Authored-By: Claude Opus 4.8 (1M context) --- scripts/04_build_maps.py | 13 ++++++-- scripts/_lib/map_template.py | 50 ++++++++++++++++++++++++---- scripts/deploy.py | 63 +++++++++++++++++++++++++++++++++++- 3 files changed, 116 insertions(+), 10 deletions(-) diff --git a/scripts/04_build_maps.py b/scripts/04_build_maps.py index a232624..fc2e3c4 100644 --- a/scripts/04_build_maps.py +++ b/scripts/04_build_maps.py @@ -5656,12 +5656,21 @@ def _src_lang_for(slug: str) -> str: aocs_for_lang = overlay_translated_facts(aocs_for_lang, facts_translations) out = (WIKI / "index.html") if lang == "en" else (WIKI / lang / "index.html") out.parent.mkdir(parents=True, exist_ok=True) - # Pass a swapped facets dict so the per-locale `aocs` is what gets serialised. + # Pass a swapped facets dict so the per-locale `aocs` is the data bundle. per_locale_facets = {**facets, "aocs": aocs_for_lang} - html_out = render_map_html( + html_out, data_filename, data_bytes = render_map_html( **per_locale_facets, locale=lang, grapes_info=lex, styles_info=styles_lex, ) out.write_text(html_out, encoding="utf-8") + # External per-locale data bundle (AOCS + grape tooltips) referenced by + # the page's render-blocking " ) - return _TEMPLATE.format( + # The two large per-locale reference blobs (AOCS ~12 MB, grape tooltips + # ~0.6 MB) ship in an external hashed /data/ script rather than inline, so + # the HTML shell is small and the data caches once per locale across every + # page that locale will serve (homepages now; per-appellation pages later). + # Serialise after the grapes_info mutation above so the bianchello note is + # included. A render-blocking + + ", "kind": "AOC", "country": "fr"}) + assert "" not in out + assert "<script>" in out + + +def test_sources_block_branches() -> None: + out = _render({"name": "F", "kind": "AOC", "country": "fr", + "sources": {"boagri": "http://b/p.pdf", "homologation_date": "2024-01-01", + "show_texte": "http://inao/t", "id_eambrosia": "EUGI/1", + "file_number": "PDO-FR-0001"}}) + assert "

Sources

" in out + assert "Cahier des charges" in out and "homologated 2024-01-01" in out + assert "eAmbrosia" in out and "PDO-FR-0001" in out + + +def test_deterministic_output_and_sig() -> None: + rec = {"name": "D", "kind": "AOC", "country": "fr", "region": "BORDEAUX", + "grapes_principal": ["merlot", "aragonez"], "geom_source": "parcellaire"} + a, b = _render(rec), _render(rec) + assert a == b # byte-identical incl. data-ssr-sig + + +def test_fr_source_marker_only_off_locale() -> None: + rec = {"name": "M", "kind": "AOC", "country": "fr", "summary": "Texte."} + assert "(French)" not in _render(rec, locale="fr") # native locale: no marker + assert "(French)" in _render(rec, locale="en") # off-locale: marker shown From 5ffa8afce31ea3be64449cdaaab58fedca3c987a Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Wed, 3 Jun 2026 22:29:45 +0200 Subject: [PATCH 03/41] SEO/GEO phase 3: gated per-appellation entity pages (pilot) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Pre-render a crawlable HTML page per appellation for a gated pilot set (23 well-known parents x 4 locales = 92 pages), reusing each locale's shell + the shared Phase-1 data bundle so the pages cost almost nothing on the CDN. map_template.render(): - Parameterised the (title / description / hreflang / og / json-ld) and added a {ssr_content} body slot; split the format call into a shared shell_kwargs (constant across a locale's pages) + per-page slots. The homepage output is unchanged except the new SSR-removal JS. - entity_slugs path emits {slug: html} per pilot slug, each with a self-referential canonical, a per-slug hreflang cluster (x-default -> EN), a templated (NON-verbatim) title + meta description, Place/AdministrativeArea + BreadcrumbList JSON-LD (GeoShape box derived from bbox, lat/lng order), and the Phase-2 content_block injected as
(carrying data-ssr-sig). EN entities live at /en/; fr/es/nl at //. - The app JS removes #ssr-content on boot — the interactive panel supersedes the no-JS / crawler fallback article. 04_build_maps.py: - gate_classify folds sub-denominations + stubs + no-geometry records; an indexable record needs resolved geometry plus its own grapes/summary/terroir. A hand-picked pilot allowlist (intersected with the gate) + a dry-run counter; writes wiki///index.html; sitemap expanded to per-slug 4-locale hreflang clusters (96 URLs: 4 home + 92 entity). Determinism: the inline VIVC-derived blobs (VIVC_SIBLINGS / SLUG_TO_CANONICAL / GRAPE_SYNONYMS) and GRAPE_SEARCH_INDEX were non-deterministic across builds (set-iteration order) — fixed by sorting vivc_groups member lists, sort_keys on the three dicts, and a slug tiebreaker on the index sort. The homepage is now byte-identical across builds (no-op rerun / reproducibility holds). Verified: 92 entity pages; self-canonical + reciprocal hreflang; valid Place/BreadcrumbList JSON-LD; SSR article present pre-JS; node --check on every generated script; FR/ES/NL locale chrome correct; full test suite + dup-keys green. Staging gates (NOT verifiable locally, must confirm before relying on indexation): (1) Bunny must serve the origin ///index.html for the no-trailing-slash URL (origin-file-wins over the SPA rewrite) — the canonical uses the no-slash form matching the shipped client deep-link; (2) facet "open" buttons are still buttons, so pilot pages are sitemap-discoverable but not yet internally linked (Phase 3.5 follow-up); (3) the communes_matched == -1 display quirk is shared with the JS panel (a > 0 guard is a deferred shared fix). Co-Authored-By: Claude Opus 4.8 (1M context) --- scripts/04_build_maps.py | 65 +++++++++- scripts/_lib/map_template.py | 226 +++++++++++++++++++++++++++++++---- 2 files changed, 264 insertions(+), 27 deletions(-) diff --git a/scripts/04_build_maps.py b/scripts/04_build_maps.py index fc2e3c4..b013406 100644 --- a/scripts/04_build_maps.py +++ b/scripts/04_build_maps.py @@ -5572,6 +5572,42 @@ def sort_facet(d: dict[str, int]) -> list[tuple[str, int]]: vivc_by_slug=_load_vivc_by_slug(), ) fr_styles_lex = load_style_lexicon("fr") + + # --- Phase 3 pilot: gated per-appellation entity pages ---------------- + # gate_classify decides which records earn an indexable, crawlable page. + # v1 FOLDS every sub-denomination (its narrative/grapes are parent- + # inherited at render time — CLAUDE.md) and any stub / no-geometry record; + # an indexable record needs resolved geometry plus its own grapes / summary + # / terroir. The pilot emits only a small hand-picked allowlist (intersected + # with the gate) so indexation + demand can be validated before a corpus- + # wide rollout. Slugs absent from the corpus are silently dropped. + def gate_classify(rec: dict) -> tuple[str, str | None]: + if rec.get("is_sub_denomination"): + return ("fold", rec.get("parent_slug")) + if rec.get("is_stub") or rec.get("geom_source") == "stub-no-geometry" or not rec.get("bbox"): + return ("fold", None) + has_own = bool( + rec.get("terroir_facts") or rec.get("summary") or (rec.get("grapes_principal") or []) + ) + return ("index", None) if has_own else ("fold", None) + + pilot_candidates = [ + "priorat", "montsant", "rioja", "ribera-del-duero", "rias-baixas", + "barolo", "barbaresco", "chianti", "brunello-di-montalcino", "soave", + "chablis", "sancerre", "chateauneuf-du-pape", "bandol", "pauillac", + "margaux", "douro", "alentejo", "rheingau", "mosel", "tokaj", + "santorini", "nemea", + ] + pilot_entity_slugs = [ + s for s in pilot_candidates if aocs.get(s) and gate_classify(aocs[s])[0] == "index" + ] + _pilot_dropped = [s for s in pilot_candidates if s not in pilot_entity_slugs] + print( + f"[entity-pilot] {len(pilot_entity_slugs)}/{len(pilot_candidates)} pilot slugs indexable; " + f"dropped (missing or gated): {_pilot_dropped}", + file=sys.stderr, + ) + for lang in LOCALES: lex = build_grapes_info(lang) if lang == "fr": @@ -5658,8 +5694,9 @@ def _src_lang_for(slug: str) -> str: out.parent.mkdir(parents=True, exist_ok=True) # Pass a swapped facets dict so the per-locale `aocs` is the data bundle. per_locale_facets = {**facets, "aocs": aocs_for_lang} - html_out, data_filename, data_bytes = render_map_html( + html_out, data_filename, data_bytes, entity_pages = render_map_html( **per_locale_facets, locale=lang, grapes_info=lex, styles_info=styles_lex, + entity_slugs=pilot_entity_slugs, ) out.write_text(html_out, encoding="utf-8") # External per-locale data bundle (AOCS + grape tooltips) referenced by @@ -5671,6 +5708,13 @@ def _src_lang_for(slug: str) -> str: for _old in data_dir.glob(f"aocs.{lang}.*.js"): _old.unlink() (data_dir / data_filename).write_bytes(data_bytes) + # Pre-rendered per-appellation pages (the gated pilot set; empty unless + # entity_slugs was passed to render). EN entities live under /en/ + # (the CDN serves the EN shell for /en/*), fr/es/nl under //. + for ent_slug, ent_html in entity_pages.items(): + ent_dir = WIKI / ("en" if lang == "en" else lang) / ent_slug + ent_dir.mkdir(parents=True, exist_ok=True) + (ent_dir / "index.html").write_text(ent_html, encoding="utf-8") # EN's home stays at / (canonical), but its appellation deep-links live # under /en/ so a single CDN rewrite ( // → # //index.html ) covers all four locales. Emit the same page at @@ -5698,7 +5742,7 @@ def _src_lang_for(slug: str) -> str: file=sys.stderr, ) - write_seo_files() + write_seo_files(pilot_entity_slugs) def _hreflang_alternates(paths_by_lang: dict[str, str]) -> str: @@ -5773,7 +5817,7 @@ def copy_brand_assets() -> None: print(f"[assets] mirrored {ASSETS_SRC.relative_to(ROOT)} → {ASSETS_OUT.relative_to(ROOT)} ({copied} updated)", file=sys.stderr) -def write_seo_files() -> None: +def write_seo_files(entity_slugs: list[str] | None = None) -> None: """Emit wiki/robots.txt and wiki/sitemap.xml. The map IS the homepage: `/` (EN canonical), `/fr/`, `/es/`, `/nl/`. The @@ -5791,6 +5835,19 @@ def write_seo_files() -> None: for lang in LOCALES ] + # Per-appellation entity pages (gated pilot): one per locale per slug + # (en under /en/), each carrying the slug's hreflang cluster + # (x-default -> EN). + n_entity = 0 + for slug in entity_slugs or []: + ent_paths = {lang: f"/{lang}/{slug}" for lang in LOCALES} + ent_alts = _hreflang_alternates(ent_paths) + for lang in LOCALES: + url_blocks.append( + _sitemap_url_block(f"{SITE_BASE_URL}{ent_paths[lang]}", today, ent_alts) + ) + n_entity += 1 + sitemap = ( '\n' ' None: (WIKI / "robots.txt").write_text(robots, encoding="utf-8") print( f"[seo] wrote {WIKI.relative_to(ROOT)}/robots.txt and sitemap.xml " - f"({len(LOCALES)} URLs)", + f"({len(url_blocks)} URLs: {len(LOCALES)} home + {n_entity} entity)", file=sys.stderr, ) diff --git a/scripts/_lib/map_template.py b/scripts/_lib/map_template.py index da6a8f5..eb166e7 100644 --- a/scripts/_lib/map_template.py +++ b/scripts/_lib/map_template.py @@ -17,6 +17,7 @@ import json from collections.abc import Callable +from _lib.content_block import RenderCtx, esc, render_content_block from _lib.i18n import load_translations @@ -751,10 +752,137 @@ def _build_grape_search_index( "count_principal": counts["principal"].get(canon, 0), "count_accessory": counts["accessory"].get(canon, 0), }) - index.sort(key=lambda e: (-e["count"], e["label"].casefold())) + index.sort(key=lambda e: (-e["count"], e["label"].casefold(), e["slug"])) return index +_HOMEPAGE_HREFLANG = ( + '\n' + '\n' + '\n' + '\n' + '' +) + + +def _entity_path(locale: str, slug: str) -> str: + """Per-appellation URL path. EN entities live under /en/ (the CDN + serves the EN shell for /en/*; bare / 404s); fr/es/nl under + //. Matches the shipped client deep-link form (no trailing + slash).""" + return f"/en/{slug}" if locale == "en" else f"/{locale}/{slug}" + + +def _entity_hreflang_block(slug: str) -> str: + rows = [ + f'' + for lg in ("en", "fr", "es", "nl") + ] + rows.append(f'') + return "\n".join(rows) + + +def _clamp(text: str, n: int = 160) -> str: + text = " ".join((text or "").split()) + if len(text) <= n: + return text + return text[:n].rsplit(" ", 1)[0].rstrip(" ,.;:") + "…" + + +def _entity_grape_names(rec: dict, grapes_info: dict, limit: int) -> list[str]: + out = [] + for g in (rec.get("grapes_principal") or [])[:limit]: + nm = ( + (rec.get("grape_names") or {}).get(g) + or (grapes_info.get(g) or {}).get("name") + or g.replace("-", " ") + ) + out.append(nm) + return out + + +def _build_entity_jsonld(slug, rec, canonical_url, locale, country_labels, region) -> str: + """schema.org Place (AdministrativeArea) + BreadcrumbList for one + appellation. Pre-serialised to an opaque string (its braces are data, not + str.format slots).""" + name = rec.get("name") or slug + country = country_labels.get(rec.get("country") or "", "") + contained: list[dict] = [] + if rec.get("is_sub_denomination") and rec.get("parent_name"): + contained.append({"@type": "AdministrativeArea", "name": rec["parent_name"]}) + if region: + contained.append({"@type": "AdministrativeArea", "name": region}) + if country: + contained.append({"@type": "Country", "name": country}) + for alias in rec.get("country_aliases") or []: + an = country_labels.get(alias, "") + if an: + contained.append({"@type": "Country", "name": an}) + place = { + "@context": "https://schema.org", + "@type": "AdministrativeArea", + "name": name, + "url": canonical_url, + } + bbox = rec.get("bbox") + if bbox and len(bbox) == 4: + # stored [minLng, minLat, maxLng, maxLat]; GeoShape box wants + # "minLat minLng maxLat maxLng". + place["geo"] = {"@type": "GeoShape", "box": f"{bbox[1]} {bbox[0]} {bbox[3]} {bbox[2]}"} + if contained: + place["containedInPlace"] = contained if len(contained) > 1 else contained[0] + home = f"{_SITE_BASE_URL}/" if locale == "en" else f"{_SITE_BASE_URL}/{locale}/" + crumbs = [{"@type": "ListItem", "position": 1, "name": "Open Wine Map", "item": home}] + pos = 2 + if country: + crumbs.append({"@type": "ListItem", "position": pos, "name": country}) + pos += 1 + if rec.get("is_sub_denomination") and rec.get("parent_name") and rec.get("parent_slug"): + crumbs.append({ + "@type": "ListItem", "position": pos, "name": rec["parent_name"], + "item": f"{_SITE_BASE_URL}{_entity_path(locale, rec['parent_slug'])}", + }) + pos += 1 + crumbs.append({"@type": "ListItem", "position": pos, "name": name, "item": canonical_url}) + breadcrumb = {"@context": "https://schema.org", "@type": "BreadcrumbList", + "itemListElement": crumbs} + return ( + '" + ) + + +def _build_entity_meta(slug, rec, locale, labels, region_labels, country_labels, grapes_info) -> dict: + """Per-appellation values: self-canonical, per-slug hreflang cluster, + templated (non-verbatim) title + meta description, and Place/BreadcrumbList + JSON-LD.""" + name = rec.get("name") or slug + kind = rec.get("kind") or "" + region = region_labels.get(rec.get("region") or "", rec.get("region") or "") + country = country_labels.get(rec.get("country") or "", "") + canonical_url = f"{_SITE_BASE_URL}{_entity_path(locale, slug)}" + geo = ", ".join(x for x in (region, country) if x) + head_bits = " · ".join(x for x in (kind, geo) if x) + title = f"{name} — {head_bits} · Open Wine Map" if head_bits else f"{name} · Open Wine Map" + gnames = _entity_grape_names(rec, grapes_info, 4) + desc = f"{name}, {geo}" if geo else name + if kind: + desc = f"{desc} · {kind}" + if gnames: + desc = f"{desc}. {labels.get('facet_principal_h', '')}: {', '.join(gnames)}" + desc = _clamp(desc, 160) + return { + "canonical_url": canonical_url, + "page_title": esc(title), + "meta_description": esc(desc), + "og_title": esc(title), + "og_description": esc(desc), + "hreflang_block": _entity_hreflang_block(slug), + "jsonld_html": _build_entity_jsonld(slug, rec, canonical_url, locale, country_labels, region), + } + + def render( *, layer_url: str, @@ -770,10 +898,14 @@ def render( styles_info: dict | None = None, vivc_by_slug: dict | None = None, area_quartiles: tuple[float, float] = (0.0, 1.0), -) -> tuple[str, str, bytes]: + entity_slugs: list[str] | None = None, +) -> tuple[str, str, bytes, dict[str, str]]: """Render the full map page (index.html) for one locale. - Returns `(html, data_filename, data_bytes)`. The large per-locale `aocs` + Returns `(html, data_filename, data_bytes, entity_pages)`, where + `entity_pages` is `{slug: html}` for each `entity_slugs` member — the + pre-rendered per-appellation pages reusing this locale's shell + data bundle + (empty unless `entity_slugs` is given). The large per-locale `aocs` and `grapes_info` blobs are NOT inlined; they are serialised into `data_bytes` (a `window.__OWM_DATA=…;` script) which the caller writes to `wiki/data/`, referenced by a render-blocking `'/'' inside their + bodies). A lossless-reconstruction assert guards against future drift.""" + style_open, style_close = "" + data_tag = '' + app_open, app_close = "" + pre, _r = t.split(style_open, 1) + css, post = _r.split(style_close, 1) + mid, _a = post.split(data_tag, 1) + gap, _a2 = _a.split(app_open, 1) + app, foot = _a2.split(app_close, 1) + reconstructed = ( + pre + style_open + css + style_close + + mid + data_tag + gap + app_open + app + app_close + foot + ) + if reconstructed != t: + raise AssertionError("map_template: _split_template is not lossless") + page_template = ( + pre + + '' + + mid + + data_tag + + gap + + '' + + foot + ) + return page_template, css, app + + +_PAGE_TEMPLATE, _STYLE_CSS, _APP_JS = _split_template(_TEMPLATE) From 9f687f41d78e28241096dbbce11c8248bf85356c Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Fri, 5 Jun 2026 12:50:38 +0200 Subject: [PATCH 07/41] SEO/GEO: hide server-rendered card from JS visitors (kill boot flash) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The externalized app..js (~407KB cached fetch) takes ~190ms to boot, during which the server-rendered #ssr-content card was visible before the map replaced it — a white-card flash. Gate it: the pre-paint inline script (which already sets view-mode/theme classes before paint) now also adds `js` to , and `html.js #ssr-content{display:none}` hides the card from first paint for JS visitors (the app removes it on boot anyway). Non-JS visitors and non-rendering crawlers still receive the card in the HTML, so SEO + accessibility are unchanged. Verified locally: with JS the card is never visible during boot (no flash); the #ssr-content article remains in the raw HTML source. Co-Authored-By: Claude Opus 4.8 (1M context) --- scripts/_lib/map_template.py | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/scripts/_lib/map_template.py b/scripts/_lib/map_template.py index d29b9f5..4a75432 100644 --- a/scripts/_lib/map_template.py +++ b/scripts/_lib/map_template.py @@ -1305,6 +1305,11 @@ def _emit(slug: str, meta: dict, ssr: str) -> None: var mode = 'simple'; try {{ if (localStorage.getItem('view_mode') === 'advanced') mode = 'advanced'; }} catch (e) {{}} document.documentElement.classList.add('mode-' + mode); + // Mark JS available before paint so the server-rendered #ssr-content card + // can be hidden via CSS (rule below). The app swaps it for the live panel, + // so JS users skip the brief boot flash; non-JS visitors and non-rendering + // crawlers still receive the card in the HTML. + document.documentElement.classList.add('js'); }})(); (function () {{ // Resolve the effective theme before first paint so dark mode never @@ -1320,6 +1325,10 @@ def _emit(slug: str, meta: dict, ssr: str) -> None: ' inside their - bodies). A lossless-reconstruction assert guards against future drift.""" + The ' + inside its body); a lossless-reconstruction assert guards against drift.""" style_open, style_close = "" - data_tag = '' - app_open, app_close = "" pre, _r = t.split(style_open, 1) css, post = _r.split(style_close, 1) - mid, _a = post.split(data_tag, 1) - gap, _a2 = _a.split(app_open, 1) - app, foot = _a2.split(app_close, 1) - reconstructed = ( - pre + style_open + css + style_close - + mid + data_tag + gap + app_open + app + app_close + foot - ) - if reconstructed != t: + if pre + style_open + css + style_close + post != t: raise AssertionError("map_template: _split_template is not lossless") - page_template = ( - pre - + '' - + mid - + data_tag - + gap - + '' - + foot - ) - return page_template, css, app + page_template = pre + '' + post + return page_template, css + + +_PAGE_TEMPLATE, _STYLE_CSS = _split_template(_TEMPLATE) + +# The map application JS lives in assets/app.js — a real .js file (lint-able, +# eslint-able) rather than an escaped Python string. Build-time per-locale +# values are injected via `__OWM___` tokens that sit exactly where the +# old `{slot}` format fields were (literal braces are already real in the .js), +# so `_render_app_js` is byte-for-byte equivalent to the old +# `_APP_JS.format(**kwargs)` (verified by the golden comparator). app.js is now +# the sole source — edit it directly; do not move the JS back into _TEMPLATE. +_APP_JS_SOURCE = (Path(__file__).resolve().parent / "assets" / "app.js").read_text( + encoding="utf-8" +) +_OWM_TOKEN_RE = re.compile(r"__OWM_(\w+?)__") + +def _render_app_js(kwargs: dict[str, str]) -> str: + """Fill the per-locale build tokens in assets/app.js (single pass, so a + value can never be re-scanned for another token).""" + missing: list[str] = [] -_PAGE_TEMPLATE, _STYLE_CSS, _APP_JS = _split_template(_TEMPLATE) + def _sub(m: "re.Match[str]") -> str: + key = m.group(1) + if key not in kwargs: + missing.append(key) + return m.group(0) + return str(kwargs[key]) + + out = _OWM_TOKEN_RE.sub(_sub, _APP_JS_SOURCE) + if missing: + raise KeyError(f"app.js: unfilled build tokens {sorted(set(missing))}") + return out From 26471bafaabed22b2daa2636519ce21522d2c7ec Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Wed, 10 Jun 2026 23:41:56 +0200 Subject: [PATCH 28/41] refactor: extract ES/SI/HR augmenters + shared caches to _lib/augment/ (no-op) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Phase 6 foundation + first slice. The 13 slug-keyed provenance caches that the national-spec augmenters populate and _sources_for()/the panel-blob phase read back move to _lib/augment/_shared.py, so both writer and readers reference the SAME dict objects across the module split (the augmenters mutate in place — .clear()/[slug]=, never reassign — so the imported names stay bound). The ES/SI/HR augmenters + their sidecar-dir constants move to _lib/augment/{es,si, hr}.py verbatim. 10 augmenters (+ IT extras, CZ helper, the 02f lexicon/geom blocks) remain in 04 for follow-up. Verified move-only: golden comparator vs the post-Phase-3/4 baseline reports 'identical' after a full stage-04 rebuild; 63 tests + ruff clean. Co-Authored-By: Claude Opus 4.8 (1M context) --- scripts/04_build_maps.py | 379 ++----------------------------- scripts/_lib/augment/__init__.py | 0 scripts/_lib/augment/_shared.py | 120 ++++++++++ scripts/_lib/augment/es.py | 87 +++++++ scripts/_lib/augment/hr.py | 103 +++++++++ scripts/_lib/augment/si.py | 108 +++++++++ 6 files changed, 436 insertions(+), 361 deletions(-) create mode 100644 scripts/_lib/augment/__init__.py create mode 100644 scripts/_lib/augment/_shared.py create mode 100644 scripts/_lib/augment/es.py create mode 100644 scripts/_lib/augment/hr.py create mode 100644 scripts/_lib/augment/si.py diff --git a/scripts/04_build_maps.py b/scripts/04_build_maps.py index 1ca304c..9caf50f 100644 --- a/scripts/04_build_maps.py +++ b/scripts/04_build_maps.py @@ -41,6 +41,24 @@ from _lib.at.gemeinde import ATCommuneIndex from _lib.at.geometry import ATPolygonIndex from _lib.at.region import derive_bundesland as derive_at_bundesland +from _lib.augment._shared import ( + _BG_NATIONAL_SPEC_BY_SLUG, + _CY_NATIONAL_SPEC_BY_SLUG, + _CZ_CHZO_BY_SLUG, + _CZ_NATIONAL_SPEC_BY_SLUG, + _DE_PRODUKTSPEZIFIKATION_BY_SLUG, + _ES_NATIONAL_PLIEGO_BY_SLUG, + _GR_NATIONAL_SPEC_BY_SLUG, + _HR_SPECIFIKACIJA_BY_SLUG, + _HU_NATIONAL_SPEC_BY_SLUG, + _IT_MASAF_BY_SLUG, + _RO_NATIONAL_SPEC_BY_SLUG, + _SI_SPECIFIKACIJA_BY_SLUG, + _SK_NATIONAL_SPEC_BY_SLUG, +) +from _lib.augment.es import augment_es_records_with_national_pliegos +from _lib.augment.hr import augment_hr_records_with_specifikacija +from _lib.augment.si import augment_si_records_with_specifikacija from _lib.be.geometry import BEPolygonIndex from _lib.be.region import derive_region as derive_be_region from _lib.bg.geometry import BGPolygonIndex @@ -131,7 +149,6 @@ ROOT = Path(__file__).resolve().parent.parent EXTRACTED = ROOT / "raw" / "inao" / "cahier-extracted" EXTRACTED_ES = ROOT / "raw" / "es" / "pliegos-extracted" -NATIONAL_PLIEGOS_ES = ROOT / "raw" / "es" / "national-pliegos-extracted" ES_FIGSHARE_GPKG = ROOT / "raw" / "es" / "figshare" / "EU_PDO.gpkg" ES_GISCO_LAU_ZIP = ROOT / "raw" / "es" / "gisco" / "LAU_RG_01M_2024_3035.shp.zip" ES_SIGPAC_DIR = ROOT / "raw" / "es" / "sigpac" @@ -149,9 +166,7 @@ EXTRACTED_DE = ROOT / "raw" / "de" / "dokumente-extracted" PRODUKTSPEZIFIKATION_DE = ROOT / "raw" / "de" / "produktspezifikationen-extracted" EXTRACTED_SI = ROOT / "raw" / "si" / "dokumenti-extracted" -SPECIFIKACIJE_SI = ROOT / "raw" / "si" / "specifikacije-extracted" EXTRACTED_HR = ROOT / "raw" / "hr" / "dokumenti-extracted" -SPECIFIKACIJE_HR = ROOT / "raw" / "hr" / "specifikacije-extracted" EXTRACTED_HU = ROOT / "raw" / "hu" / "dokumentumok-extracted" EXTRACTED_RO = ROOT / "raw" / "ro" / "dokumente-extracted" EXTRACTED_BG = ROOT / "raw" / "bg" / "dokumenti-extracted" @@ -199,99 +214,6 @@ _DISAMBIG_SUFFIX = re.compile(r"\s*\([^)]*\)\s*$") -# Slug-keyed cache of national-pliego provenance, populated by -# augment_es_records_with_national_pliegos() and read by _sources_for() -# later in the build. Needed because the AOC-blob phase re-reads each -# extracted JSON from disk (where the augmentation isn't persisted) — -# this lookup gives that phase access to the same provenance the -# in-memory record carries. -_ES_NATIONAL_PLIEGO_BY_SLUG: dict[str, dict] = {} - - -# Slug-keyed cache of MASAF disciplinare provenance + augmented payload, -# populated by augment_de_records_with_produktspezifikation() and read -# by _sources_for() / panel rendering — same shape as the IT/ES caches. -_DE_PRODUKTSPEZIFIKATION_BY_SLUG: dict[str, dict] = {} - -# populated by augment_it_records_with_masaf() and read by _sources_for() -# / the AOC-blob phase (which re-reads each on-disk extracted JSON, -# bypassing in-memory augmentation). -_IT_MASAF_BY_SLUG: dict[str, dict] = {} - -# Slug-keyed cache of CZ national-spec provenance, populated by -# augment_cz_records_with_national_specs(). Mirrors the ES/IT/DE caches. -# Czech wine law publishes one national variety roster (Vyhláška 88/2017 -# Sb. Příloha č. 2) that applies to every jakostní víno regardless of -# podoblast, so every augmented CZ wine carries the same provenance -# block — but per-record so _sources_for() can surface it uniformly. -_CZ_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} - -# Slug-keyed cache of CZ CHZO-spec provenance (the SZPI „moravské“ / -# „české“ product specifications). Every CZ wine sits in one of the two -# regions (Morava / Čechy), so all 13 carry the region spec's provenance -# — the terroir bullets (02d) ground on its section-1 region description. -# Populated by augment_cz_records_with_national_specs(). -_CZ_CHZO_BY_SLUG: dict[str, dict] = {} - -# Slug-keyed cache of SI specifikacija provenance + augmented payload, -# populated by augment_si_records_with_specifikacija(). Two source -# patterns feed it (MKGP per-wine .doc, Uradni list RS pravilnik HTML); -# the sidecar's `parser_template` distinguishes them for attribution. -_SI_SPECIFIKACIJA_BY_SLUG: dict[str, dict] = {} - -# Slug-keyed cache of HR specifikacija provenance, populated by -# augment_hr_records_with_specifikacija(). The 16 grandfathered HR wines -# whose EU-OJ JEDINSTVENI DOKUMENT was never published are augmented from -# the Ministarstvo poljoprivrede per-wine SPECIFIKACIJA PROIZVODA (stage -# 02f). `parser_template` distinguishes the lettered .doc/.pdf path -# (mps-specifikacija-v1) from the docx fallback (mps-specifikacija-docx). -_HR_SPECIFIKACIJA_BY_SLUG: dict[str, dict] = {} - -# Slug-keyed cache of GR national-spec provenance, populated by -# augment_gr_records_with_national_specs(). 132 of the 138 grandfathered -# GR wines are augmented from the ΥΠΑΑΤ national προδιαγραφή / τεχνικός -# φάκελος (stage 02f) — 87 structured-PDF ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ, 43 `.doc`, -# 2 `.docx`. `parser_template` (gr-national-{pdf,doc,docx}) distinguishes. -_GR_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} -# Slug-keyed cache of CY national-spec provenance, populated by -# augment_cy_records_with_national_specs(). All 11 CY wines are -# grandfathered names augmented from the moa.gov.cy τεχνικός φάκελος -# (stage 02f) — a Greek ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ PDF, OCR'd when image-only. -_CY_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} - -# Slug-keyed cache of RO national-spec provenance, populated by -# augment_ro_records_with_national_specs(). The 14 grandfathered RO wines -# (only an Ares(...) reference in eAmbrosia, no EU-OJ DOCUMENT UNIC) are -# augmented from the ONVPV caiet de sarcini (stage 02f, onvpv-caiet-de- -# sarcini-v1 parser). Unlike GR/HR, the merge also carries `geo_communes` -# so the 2 grandfathered IGPs resolve via the GISCO commune-union chain. -_RO_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} - -# Slug-keyed cache of HU national-spec provenance, populated by -# augment_hu_records_with_national_specs(). The 15 grandfathered HU wines -# (only an Ares(...) reference in eAmbrosia, no EU-OJ EGYSÉGES DOKUMENTUM) -# are augmented from the Agrárminisztérium termékleírás PDF (stage 02f, -# hu-termekleiras-v1 parser). The merge carries grapes + terroir text + -# geo_communes so the panel + 02d ground on the national spec. -_HU_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} - -# Slug-keyed cache of BG national-spec provenance, populated by -# augment_bg_records_with_national_specs(). The 51 grandfathered BG wines -# (only an Ares(...) reference in eAmbrosia, no EU-OJ ЕДИНЕН ДОКУМЕНТ) are -# augmented from the ИАЛВ / IAVV per-wine продуктова спецификация (stage -# 02f, iavv-specifikacija-v1 parser — eavw.com PDF, numbered 1–8 template: -# 5 сортове / 6 Връзка с географския район / 3 район). -_BG_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} - -# Slug-keyed cache of SK national-spec provenance, populated by -# augment_sk_records_with_national_specs(). The 5 grandfathered SK wines -# (only an Ares(...) reference in eAmbrosia, no EU-OJ JEDNOTNÝ DOKUMENT) are -# augmented from the ÚPV SR per-wine špecifikácia výrobku (stage 02f, -# upv-sr-specifikacia-v1 parser — indprop.gov.sk PDF, lettered a–i template: -# f) označenie odrôd / g) údaje potvrdzujúce spojitosť / d) zemepisná oblasť). -_SK_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} - - # Cross-border PDOs that physically extend across more than one country. # Keyed by eAmbrosia file_number → list of secondary country codes (the # primary country sits in `record.country` from stage 02). The map shows @@ -306,271 +228,6 @@ } -def augment_es_records_with_national_pliegos(records: list[dict]) -> int: - """In-place merge of national-pliego sidecar varieties into each ES - record's `grapes` field. Returns the number of records augmented. - - Mutations per record: - - new variety slugs (those NOT already in principal ∪ accessory) are - appended to `grapes.accessory` - - matching entries are appended to `grapes.details` with - `role="accessory"` and `source="national-pliego"` so the UI can - distinguish doc-único-canonical varieties from pliego-augmented ones - - a top-level `national_pliego` block carries provenance for - `_sources_for()` to surface in the panel - """ - _ES_NATIONAL_PLIEGO_BY_SLUG.clear() - if not NATIONAL_PLIEGOS_ES.exists(): - return 0 - augmented = 0 - for record in records: - if record.get("country") != "es": - continue - slug = record.get("slug") - if not slug: - continue - sidecar_path = NATIONAL_PLIEGOS_ES / f"{slug}.json" - if not sidecar_path.exists(): - continue - sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) - new_slugs = list(sidecar.get("delta_vs_oj", {}).get("new_slugs") or []) - if not new_slugs: - # Still stamp provenance — the pliego was parsed even if it - # added nothing new. Skip the merge but keep attribution - # consistent for the audit. - nat_provenance = { - "url": sidecar.get("source", {}).get("url", ""), - "sha256": sidecar.get("source", {}).get("sha256", ""), - "fetched_at": sidecar.get("source", {}).get("fetched_at", ""), - "parser_template": sidecar.get("parser_template", ""), - "added_slugs": [], - } - record["national_pliego"] = nat_provenance - _ES_NATIONAL_PLIEGO_BY_SLUG[slug] = nat_provenance - continue - grapes = dict(record.get("grapes") or {}) - principal = list(grapes.get("principal") or []) - accessory = list(grapes.get("accessory") or []) - details = list(grapes.get("details") or []) - existing = set(principal) | set(accessory) - added: list[str] = [] - slug_to_detail = {d.get("slug"): d for d in sidecar.get("varieties", [])} - for s in new_slugs: - if s in existing: - continue - accessory.append(s) - existing.add(s) - added.append(s) - detail = dict(slug_to_detail.get(s) or {"slug": s, "name": s, "colour": ""}) - detail["role"] = "accessory" - detail["source"] = "national-pliego" - details.append(detail) - grapes["accessory"] = accessory - grapes["details"] = details - record["grapes"] = grapes - nat_provenance = { - "url": sidecar.get("source", {}).get("url", ""), - "sha256": sidecar.get("source", {}).get("sha256", ""), - "fetched_at": sidecar.get("source", {}).get("fetched_at", ""), - "parser_template": sidecar.get("parser_template", ""), - "added_slugs": added, - } - record["national_pliego"] = nat_provenance - _ES_NATIONAL_PLIEGO_BY_SLUG[slug] = nat_provenance - if added: - augmented += 1 - return augmented - - -def augment_si_records_with_specifikacija(records: list[dict]) -> int: - """In-place merge of SI national-spec sidecar data into stub records. - - Sibling of `augment_it_records_with_masaf`. The 16 grandfathered SI - wines (every wine except Cviček) ship as content-stubs because their - eAmbrosia entry has no fetchable EU-OJ ENOTNI DOKUMENT URL. Stage - 02f (`scripts/si/02f_extract_specifikacije.py`) extracts the - canonical Slovenian regulator source — either an MKGP per-wine - `.doc` specifikacija proizvoda or an Uradni list RS pravilnik HTML — - into a sidecar at `raw/si/specifikacije-extracted/.json`. - - For each SI stub with a matching sidecar: - - summary ← MKGP §2 (opis vin) or pravilnik §2 - - grapes ← MKGP §6 (sorte) or pravilnik Article-5 + - priloga 2 (priporočene → principal, - dovoljene → accessory) - - geo_area_brief ← MKGP §4 (opredelitev geografskega območja) - - link_to_terroir ← MKGP §7 (povezava z geografskim območjem) - (pravilnik-derived records have empty - link_to_terroir — pravilniki are regulatory - lists, not narrative documents) - - styles ← MKGP-derived colour/sparkling/predikat tags - - section_roles ← unified role dict so 02d can read terroir - text uniformly - - stub_reason ← prefixed `specifikacija:` so the audit can - tell EU-OJ-extracted from spec-augmented - - specifikacija ← provenance block (url, sha256, fetched_at, - parser_template, source_org, license) - - `record["stub"]` stays True — the record is still NOT an EU-OJ - extraction, just augmented with the canonical Slovenian regulator - source. Stage 03 / 04 callers use the `specifikacija` block to - distinguish and to render attribution. - - Returns the number of records augmented. - """ - _SI_SPECIFIKACIJA_BY_SLUG.clear() - if not SPECIFIKACIJE_SI.exists(): - return 0 - augmented = 0 - for record in records: - if record.get("country") != "si": - continue - if not record.get("stub"): - continue - slug = record.get("slug") - if not slug: - continue - sidecar_path = SPECIFIKACIJE_SI / f"{slug}.json" - if not sidecar_path.exists(): - continue - try: - sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) - except (ValueError, OSError): - continue - - src = sidecar.get("source") or {} - provenance = { - "url": src.get("url") or "", - "final_url": src.get("final_url") or "", - "sha256": src.get("sha256") or "", - "bytes": src.get("bytes") or 0, - "fetched_at": src.get("fetched_at") or "", - "format": src.get("format") or "", - "source_org": src.get("source_org") or "", - "license": src.get("license") or "", - "parser_template": sidecar.get("parser_template") or "", - "matched_okoliši": sidecar.get("matched_okoliši") or [], - } - - if sidecar.get("summary"): - record["summary"] = sidecar["summary"] - if sidecar.get("grapes"): - record["grapes"] = sidecar["grapes"] - if sidecar.get("geo_area_brief"): - record["geo_area_brief"] = sidecar["geo_area_brief"] - if sidecar.get("link_to_terroir"): - record["link_to_terroir"] = sidecar["link_to_terroir"] - if sidecar.get("styles"): - existing_styles = set(record.get("styles") or []) - record["styles"] = sorted(existing_styles | set(sidecar["styles"])) - - section_roles = dict(record.get("section_roles") or {}) - for role in ("description", "geo_area", "grape_varieties", "link_to_terroir"): - sidecar_roles = sidecar.get("section_roles") or {} - if sidecar_roles.get(role): - section_roles[role] = sidecar_roles[role] - record["section_roles"] = section_roles - - if record.get("stub_reason") and not record["stub_reason"].startswith("specifikacija:"): - record["stub_reason"] = f"specifikacija:{record['stub_reason']}" - record["specifikacija"] = provenance - _SI_SPECIFIKACIJA_BY_SLUG[slug] = provenance - augmented += 1 - return augmented - - -def augment_hr_records_with_specifikacija(records: list[dict]) -> int: - """In-place merge of HR national-spec sidecar data into stub records. - - Sibling of `augment_si_records_with_specifikacija`. The 16 - grandfathered HR wines (everything except Muškat momjanski + Ponikve) - ship as content-stubs because their eAmbrosia entry has no fetchable - EU-OJ JEDINSTVENI DOKUMENT URL. Stage 02f - (`scripts/hr/02f_extract_specifikacije.py`) extracts the canonical - Ministarstvo poljoprivrede per-wine SPECIFIKACIJA PROIZVODA (14 - `.doc`, 1 `.docx`, 1 PDF) into a sidecar at - `raw/hr/specifikacije-extracted/.json`. - - For each HR stub with a matching sidecar: - - summary ← lettered section b) (opis svojstava vina) - - grapes ← section f) (sorte vinove loze; colour-grouped, - all principal — the MPS spec has no - principal/accessory split, same as PT/IT) - - geo_area_brief ← section d) (granice područja) - - link_to_terroir ← section g) (…povezane sa zemljopisnim - uvjetima); empty for the Primorska docx - - styles ← colour/sparkling/dessert/liqueur tags - - section_roles ← unified role dict so 02d reads terroir text - - stub_reason ← prefixed `specifikacija:` so the audit can - tell EU-OJ-extracted from spec-augmented - - specifikacija ← provenance block (url, sha256, fetched_at, - parser_template, source_org, license) - - `record["stub"]` stays True — the record is still NOT an EU-OJ - extraction, just augmented with the canonical Croatian regulator - source. Returns the number of records augmented. - """ - _HR_SPECIFIKACIJA_BY_SLUG.clear() - if not SPECIFIKACIJE_HR.exists(): - return 0 - augmented = 0 - for record in records: - if record.get("country") != "hr": - continue - if not record.get("stub"): - continue - slug = record.get("slug") - if not slug: - continue - sidecar_path = SPECIFIKACIJE_HR / f"{slug}.json" - if not sidecar_path.exists(): - continue - try: - sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) - except (ValueError, OSError): - continue - - src = sidecar.get("source") or {} - provenance = { - "url": src.get("url") or "", - "final_url": src.get("final_url") or "", - "sha256": src.get("sha256") or "", - "bytes": src.get("bytes") or 0, - "fetched_at": src.get("fetched_at") or "", - "format": src.get("format") or "", - "source_org": src.get("source_org") or "", - "license": src.get("license") or "", - "parser_template": sidecar.get("parser_template") or "", - } - - if sidecar.get("summary"): - record["summary"] = sidecar["summary"] - if sidecar.get("grapes") and (sidecar["grapes"].get("principal") - or sidecar["grapes"].get("accessory")): - record["grapes"] = sidecar["grapes"] - if sidecar.get("geo_area_brief"): - record["geo_area_brief"] = sidecar["geo_area_brief"] - if sidecar.get("link_to_terroir"): - record["link_to_terroir"] = sidecar["link_to_terroir"] - if sidecar.get("styles"): - existing_styles = set(record.get("styles") or []) - record["styles"] = sorted(existing_styles | set(sidecar["styles"])) - - section_roles = dict(record.get("section_roles") or {}) - for role in ("description", "geo_area", "grape_varieties", "link_to_terroir"): - sidecar_roles = sidecar.get("section_roles") or {} - if sidecar_roles.get(role): - section_roles[role] = sidecar_roles[role] - record["section_roles"] = section_roles - - if record.get("stub_reason") and not record["stub_reason"].startswith("specifikacija:"): - record["stub_reason"] = f"specifikacija:{record['stub_reason']}" - record["specifikacija"] = provenance - _HR_SPECIFIKACIJA_BY_SLUG[slug] = provenance - augmented += 1 - return augmented - - def augment_gr_records_with_national_specs(records: list[dict]) -> int: """In-place merge of GR national-spec sidecar data into stub records. diff --git a/scripts/_lib/augment/__init__.py b/scripts/_lib/augment/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/scripts/_lib/augment/_shared.py b/scripts/_lib/augment/_shared.py new file mode 100644 index 0000000..0aa70db --- /dev/null +++ b/scripts/_lib/augment/_shared.py @@ -0,0 +1,120 @@ +"""Shared state for the stage-04 national-spec augmenters. + +The per-country augmenters (``_lib/augment/.py``) populate slug-keyed +provenance caches that ``_sources_for()`` and the panel-blob phase in +``04_build_maps.py`` read back later (the AOC-blob phase re-reads each +extracted JSON from disk, where the in-memory augmentation isn't persisted, +so these caches give it access to the same provenance the in-memory record +carries). Both the writer (the augmenter) and the readers (stage 04) MUST +reference the *same* dict object, so the caches live here and are imported by +both sides. The per-country sidecar directories live here too for the same +reason. Moved verbatim out of 04_build_maps.py — no behaviour change. +""" +from __future__ import annotations + +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[3] + +# Per-country national-spec sidecar directories read by the augmenters. +NATIONAL_PLIEGOS_ES = ROOT / "raw" / "es" / "national-pliegos-extracted" +MASAF_DISCIPLINARI_IT = ROOT / "raw" / "it" / "masaf-disciplinari-extracted" +IT_REGIONAL_REGISTERS = ROOT / "raw" / "it" / "regional-variety-registers" +PRODUKTSPEZIFIKATION_DE = ROOT / "raw" / "de" / "produktspezifikationen-extracted" +SPECIFIKACIJE_SI = ROOT / "raw" / "si" / "specifikacije-extracted" +SPECIFIKACIJE_HR = ROOT / "raw" / "hr" / "specifikacije-extracted" +NATIONAL_SPECS_BG = ROOT / "raw" / "bg" / "national-specs-extracted" +NATIONAL_SPECS_GR = ROOT / "raw" / "gr" / "national-specs-extracted" +NATIONAL_SPECS_CY = ROOT / "raw" / "cy" / "national-specs-extracted" +NATIONAL_SPECS_RO = ROOT / "raw" / "ro" / "national-specs-extracted" +NATIONAL_SPECS_HU = ROOT / "raw" / "hu" / "national-specs-extracted" +NATIONAL_SPECS_SK = ROOT / "raw" / "sk" / "national-specs-extracted" +NATIONAL_SPECS_CZ = ROOT / "raw" / "cz" / "national-specs" + +# Slug-keyed provenance caches. Populated by the per-country augmenter named +# in each comment; read by _sources_for() / the panel-blob phase in stage 04. +_ES_NATIONAL_PLIEGO_BY_SLUG: dict[str, dict] = {} + + +# Slug-keyed cache of MASAF disciplinare provenance + augmented payload, +# populated by augment_de_records_with_produktspezifikation() and read +# by _sources_for() / panel rendering — same shape as the IT/ES caches. +_DE_PRODUKTSPEZIFIKATION_BY_SLUG: dict[str, dict] = {} + +# populated by augment_it_records_with_masaf() and read by _sources_for() +# / the AOC-blob phase (which re-reads each on-disk extracted JSON, +# bypassing in-memory augmentation). +_IT_MASAF_BY_SLUG: dict[str, dict] = {} + +# Slug-keyed cache of CZ national-spec provenance, populated by +# augment_cz_records_with_national_specs(). Mirrors the ES/IT/DE caches. +# Czech wine law publishes one national variety roster (Vyhláška 88/2017 +# Sb. Příloha č. 2) that applies to every jakostní víno regardless of +# podoblast, so every augmented CZ wine carries the same provenance +# block — but per-record so _sources_for() can surface it uniformly. +_CZ_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} + +# Slug-keyed cache of CZ CHZO-spec provenance (the SZPI „moravské“ / +# „české“ product specifications). Every CZ wine sits in one of the two +# regions (Morava / Čechy), so all 13 carry the region spec's provenance +# — the terroir bullets (02d) ground on its section-1 region description. +# Populated by augment_cz_records_with_national_specs(). +_CZ_CHZO_BY_SLUG: dict[str, dict] = {} + +# Slug-keyed cache of SI specifikacija provenance + augmented payload, +# populated by augment_si_records_with_specifikacija(). Two source +# patterns feed it (MKGP per-wine .doc, Uradni list RS pravilnik HTML); +# the sidecar's `parser_template` distinguishes them for attribution. +_SI_SPECIFIKACIJA_BY_SLUG: dict[str, dict] = {} + +# Slug-keyed cache of HR specifikacija provenance, populated by +# augment_hr_records_with_specifikacija(). The 16 grandfathered HR wines +# whose EU-OJ JEDINSTVENI DOKUMENT was never published are augmented from +# the Ministarstvo poljoprivrede per-wine SPECIFIKACIJA PROIZVODA (stage +# 02f). `parser_template` distinguishes the lettered .doc/.pdf path +# (mps-specifikacija-v1) from the docx fallback (mps-specifikacija-docx). +_HR_SPECIFIKACIJA_BY_SLUG: dict[str, dict] = {} + +# Slug-keyed cache of GR national-spec provenance, populated by +# augment_gr_records_with_national_specs(). 132 of the 138 grandfathered +# GR wines are augmented from the ΥΠΑΑΤ national προδιαγραφή / τεχνικός +# φάκελος (stage 02f) — 87 structured-PDF ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ, 43 `.doc`, +# 2 `.docx`. `parser_template` (gr-national-{pdf,doc,docx}) distinguishes. +_GR_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} +# Slug-keyed cache of CY national-spec provenance, populated by +# augment_cy_records_with_national_specs(). All 11 CY wines are +# grandfathered names augmented from the moa.gov.cy τεχνικός φάκελος +# (stage 02f) — a Greek ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ PDF, OCR'd when image-only. +_CY_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} + +# Slug-keyed cache of RO national-spec provenance, populated by +# augment_ro_records_with_national_specs(). The 14 grandfathered RO wines +# (only an Ares(...) reference in eAmbrosia, no EU-OJ DOCUMENT UNIC) are +# augmented from the ONVPV caiet de sarcini (stage 02f, onvpv-caiet-de- +# sarcini-v1 parser). Unlike GR/HR, the merge also carries `geo_communes` +# so the 2 grandfathered IGPs resolve via the GISCO commune-union chain. +_RO_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} + +# Slug-keyed cache of HU national-spec provenance, populated by +# augment_hu_records_with_national_specs(). The 15 grandfathered HU wines +# (only an Ares(...) reference in eAmbrosia, no EU-OJ EGYSÉGES DOKUMENTUM) +# are augmented from the Agrárminisztérium termékleírás PDF (stage 02f, +# hu-termekleiras-v1 parser). The merge carries grapes + terroir text + +# geo_communes so the panel + 02d ground on the national spec. +_HU_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} + +# Slug-keyed cache of BG national-spec provenance, populated by +# augment_bg_records_with_national_specs(). The 51 grandfathered BG wines +# (only an Ares(...) reference in eAmbrosia, no EU-OJ ЕДИНЕН ДОКУМЕНТ) are +# augmented from the ИАЛВ / IAVV per-wine продуктова спецификация (stage +# 02f, iavv-specifikacija-v1 parser — eavw.com PDF, numbered 1–8 template: +# 5 сортове / 6 Връзка с географския район / 3 район). +_BG_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} + +# Slug-keyed cache of SK national-spec provenance, populated by +# augment_sk_records_with_national_specs(). The 5 grandfathered SK wines +# (only an Ares(...) reference in eAmbrosia, no EU-OJ JEDNOTNÝ DOKUMENT) are +# augmented from the ÚPV SR per-wine špecifikácia výrobku (stage 02f, +# upv-sr-specifikacia-v1 parser — indprop.gov.sk PDF, lettered a–i template: +# f) označenie odrôd / g) údaje potvrdzujúce spojitosť / d) zemepisná oblasť). +_SK_NATIONAL_SPEC_BY_SLUG: dict[str, dict] = {} diff --git a/scripts/_lib/augment/es.py b/scripts/_lib/augment/es.py new file mode 100644 index 0000000..870e304 --- /dev/null +++ b/scripts/_lib/augment/es.py @@ -0,0 +1,87 @@ +"""ES national-pliego variety augmentation (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. The shared +provenance cache + sidecar dir live in `_shared` so the writer here and the +`_sources_for()` reader in stage 04 reference the same objects. +""" +from __future__ import annotations + +import json + +from ._shared import _ES_NATIONAL_PLIEGO_BY_SLUG, NATIONAL_PLIEGOS_ES + + +def augment_es_records_with_national_pliegos(records: list[dict]) -> int: + """In-place merge of national-pliego sidecar varieties into each ES + record's `grapes` field. Returns the number of records augmented. + + Mutations per record: + - new variety slugs (those NOT already in principal ∪ accessory) are + appended to `grapes.accessory` + - matching entries are appended to `grapes.details` with + `role="accessory"` and `source="national-pliego"` so the UI can + distinguish doc-único-canonical varieties from pliego-augmented ones + - a top-level `national_pliego` block carries provenance for + `_sources_for()` to surface in the panel + """ + _ES_NATIONAL_PLIEGO_BY_SLUG.clear() + if not NATIONAL_PLIEGOS_ES.exists(): + return 0 + augmented = 0 + for record in records: + if record.get("country") != "es": + continue + slug = record.get("slug") + if not slug: + continue + sidecar_path = NATIONAL_PLIEGOS_ES / f"{slug}.json" + if not sidecar_path.exists(): + continue + sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) + new_slugs = list(sidecar.get("delta_vs_oj", {}).get("new_slugs") or []) + if not new_slugs: + # Still stamp provenance — the pliego was parsed even if it + # added nothing new. Skip the merge but keep attribution + # consistent for the audit. + nat_provenance = { + "url": sidecar.get("source", {}).get("url", ""), + "sha256": sidecar.get("source", {}).get("sha256", ""), + "fetched_at": sidecar.get("source", {}).get("fetched_at", ""), + "parser_template": sidecar.get("parser_template", ""), + "added_slugs": [], + } + record["national_pliego"] = nat_provenance + _ES_NATIONAL_PLIEGO_BY_SLUG[slug] = nat_provenance + continue + grapes = dict(record.get("grapes") or {}) + principal = list(grapes.get("principal") or []) + accessory = list(grapes.get("accessory") or []) + details = list(grapes.get("details") or []) + existing = set(principal) | set(accessory) + added: list[str] = [] + slug_to_detail = {d.get("slug"): d for d in sidecar.get("varieties", [])} + for s in new_slugs: + if s in existing: + continue + accessory.append(s) + existing.add(s) + added.append(s) + detail = dict(slug_to_detail.get(s) or {"slug": s, "name": s, "colour": ""}) + detail["role"] = "accessory" + detail["source"] = "national-pliego" + details.append(detail) + grapes["accessory"] = accessory + grapes["details"] = details + record["grapes"] = grapes + nat_provenance = { + "url": sidecar.get("source", {}).get("url", ""), + "sha256": sidecar.get("source", {}).get("sha256", ""), + "fetched_at": sidecar.get("source", {}).get("fetched_at", ""), + "parser_template": sidecar.get("parser_template", ""), + "added_slugs": added, + } + record["national_pliego"] = nat_provenance + _ES_NATIONAL_PLIEGO_BY_SLUG[slug] = nat_provenance + if added: + augmented += 1 + return augmented diff --git a/scripts/_lib/augment/hr.py b/scripts/_lib/augment/hr.py new file mode 100644 index 0000000..02979ea --- /dev/null +++ b/scripts/_lib/augment/hr.py @@ -0,0 +1,103 @@ +"""HR national-spec (specifikacija) augmentation (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. The shared +provenance cache + sidecar dir live in `_shared` so the writer here and the +`_sources_for()` reader in stage 04 reference the same objects. +""" +from __future__ import annotations + +import json + +from ._shared import _HR_SPECIFIKACIJA_BY_SLUG, SPECIFIKACIJE_HR + + +def augment_hr_records_with_specifikacija(records: list[dict]) -> int: + """In-place merge of HR national-spec sidecar data into stub records. + + Sibling of `augment_si_records_with_specifikacija`. The 16 + grandfathered HR wines (everything except Muškat momjanski + Ponikve) + ship as content-stubs because their eAmbrosia entry has no fetchable + EU-OJ JEDINSTVENI DOKUMENT URL. Stage 02f + (`scripts/hr/02f_extract_specifikacije.py`) extracts the canonical + Ministarstvo poljoprivrede per-wine SPECIFIKACIJA PROIZVODA (14 + `.doc`, 1 `.docx`, 1 PDF) into a sidecar at + `raw/hr/specifikacije-extracted/.json`. + + For each HR stub with a matching sidecar: + - summary ← lettered section b) (opis svojstava vina) + - grapes ← section f) (sorte vinove loze; colour-grouped, + all principal — the MPS spec has no + principal/accessory split, same as PT/IT) + - geo_area_brief ← section d) (granice područja) + - link_to_terroir ← section g) (…povezane sa zemljopisnim + uvjetima); empty for the Primorska docx + - styles ← colour/sparkling/dessert/liqueur tags + - section_roles ← unified role dict so 02d reads terroir text + - stub_reason ← prefixed `specifikacija:` so the audit can + tell EU-OJ-extracted from spec-augmented + - specifikacija ← provenance block (url, sha256, fetched_at, + parser_template, source_org, license) + + `record["stub"]` stays True — the record is still NOT an EU-OJ + extraction, just augmented with the canonical Croatian regulator + source. Returns the number of records augmented. + """ + _HR_SPECIFIKACIJA_BY_SLUG.clear() + if not SPECIFIKACIJE_HR.exists(): + return 0 + augmented = 0 + for record in records: + if record.get("country") != "hr": + continue + if not record.get("stub"): + continue + slug = record.get("slug") + if not slug: + continue + sidecar_path = SPECIFIKACIJE_HR / f"{slug}.json" + if not sidecar_path.exists(): + continue + try: + sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + + src = sidecar.get("source") or {} + provenance = { + "url": src.get("url") or "", + "final_url": src.get("final_url") or "", + "sha256": src.get("sha256") or "", + "bytes": src.get("bytes") or 0, + "fetched_at": src.get("fetched_at") or "", + "format": src.get("format") or "", + "source_org": src.get("source_org") or "", + "license": src.get("license") or "", + "parser_template": sidecar.get("parser_template") or "", + } + + if sidecar.get("summary"): + record["summary"] = sidecar["summary"] + if sidecar.get("grapes") and (sidecar["grapes"].get("principal") + or sidecar["grapes"].get("accessory")): + record["grapes"] = sidecar["grapes"] + if sidecar.get("geo_area_brief"): + record["geo_area_brief"] = sidecar["geo_area_brief"] + if sidecar.get("link_to_terroir"): + record["link_to_terroir"] = sidecar["link_to_terroir"] + if sidecar.get("styles"): + existing_styles = set(record.get("styles") or []) + record["styles"] = sorted(existing_styles | set(sidecar["styles"])) + + section_roles = dict(record.get("section_roles") or {}) + for role in ("description", "geo_area", "grape_varieties", "link_to_terroir"): + sidecar_roles = sidecar.get("section_roles") or {} + if sidecar_roles.get(role): + section_roles[role] = sidecar_roles[role] + record["section_roles"] = section_roles + + if record.get("stub_reason") and not record["stub_reason"].startswith("specifikacija:"): + record["stub_reason"] = f"specifikacija:{record['stub_reason']}" + record["specifikacija"] = provenance + _HR_SPECIFIKACIJA_BY_SLUG[slug] = provenance + augmented += 1 + return augmented diff --git a/scripts/_lib/augment/si.py b/scripts/_lib/augment/si.py new file mode 100644 index 0000000..2cd698e --- /dev/null +++ b/scripts/_lib/augment/si.py @@ -0,0 +1,108 @@ +"""SI national-spec (specifikacija) augmentation (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. The shared +provenance cache + sidecar dir live in `_shared` so the writer here and the +`_sources_for()` reader in stage 04 reference the same objects. +""" +from __future__ import annotations + +import json + +from ._shared import _SI_SPECIFIKACIJA_BY_SLUG, SPECIFIKACIJE_SI + + +def augment_si_records_with_specifikacija(records: list[dict]) -> int: + """In-place merge of SI national-spec sidecar data into stub records. + + Sibling of `augment_it_records_with_masaf`. The 16 grandfathered SI + wines (every wine except Cviček) ship as content-stubs because their + eAmbrosia entry has no fetchable EU-OJ ENOTNI DOKUMENT URL. Stage + 02f (`scripts/si/02f_extract_specifikacije.py`) extracts the + canonical Slovenian regulator source — either an MKGP per-wine + `.doc` specifikacija proizvoda or an Uradni list RS pravilnik HTML — + into a sidecar at `raw/si/specifikacije-extracted/.json`. + + For each SI stub with a matching sidecar: + - summary ← MKGP §2 (opis vin) or pravilnik §2 + - grapes ← MKGP §6 (sorte) or pravilnik Article-5 + + priloga 2 (priporočene → principal, + dovoljene → accessory) + - geo_area_brief ← MKGP §4 (opredelitev geografskega območja) + - link_to_terroir ← MKGP §7 (povezava z geografskim območjem) + (pravilnik-derived records have empty + link_to_terroir — pravilniki are regulatory + lists, not narrative documents) + - styles ← MKGP-derived colour/sparkling/predikat tags + - section_roles ← unified role dict so 02d can read terroir + text uniformly + - stub_reason ← prefixed `specifikacija:` so the audit can + tell EU-OJ-extracted from spec-augmented + - specifikacija ← provenance block (url, sha256, fetched_at, + parser_template, source_org, license) + + `record["stub"]` stays True — the record is still NOT an EU-OJ + extraction, just augmented with the canonical Slovenian regulator + source. Stage 03 / 04 callers use the `specifikacija` block to + distinguish and to render attribution. + + Returns the number of records augmented. + """ + _SI_SPECIFIKACIJA_BY_SLUG.clear() + if not SPECIFIKACIJE_SI.exists(): + return 0 + augmented = 0 + for record in records: + if record.get("country") != "si": + continue + if not record.get("stub"): + continue + slug = record.get("slug") + if not slug: + continue + sidecar_path = SPECIFIKACIJE_SI / f"{slug}.json" + if not sidecar_path.exists(): + continue + try: + sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + + src = sidecar.get("source") or {} + provenance = { + "url": src.get("url") or "", + "final_url": src.get("final_url") or "", + "sha256": src.get("sha256") or "", + "bytes": src.get("bytes") or 0, + "fetched_at": src.get("fetched_at") or "", + "format": src.get("format") or "", + "source_org": src.get("source_org") or "", + "license": src.get("license") or "", + "parser_template": sidecar.get("parser_template") or "", + "matched_okoliši": sidecar.get("matched_okoliši") or [], + } + + if sidecar.get("summary"): + record["summary"] = sidecar["summary"] + if sidecar.get("grapes"): + record["grapes"] = sidecar["grapes"] + if sidecar.get("geo_area_brief"): + record["geo_area_brief"] = sidecar["geo_area_brief"] + if sidecar.get("link_to_terroir"): + record["link_to_terroir"] = sidecar["link_to_terroir"] + if sidecar.get("styles"): + existing_styles = set(record.get("styles") or []) + record["styles"] = sorted(existing_styles | set(sidecar["styles"])) + + section_roles = dict(record.get("section_roles") or {}) + for role in ("description", "geo_area", "grape_varieties", "link_to_terroir"): + sidecar_roles = sidecar.get("section_roles") or {} + if sidecar_roles.get(role): + section_roles[role] = sidecar_roles[role] + record["section_roles"] = section_roles + + if record.get("stub_reason") and not record["stub_reason"].startswith("specifikacija:"): + record["stub_reason"] = f"specifikacija:{record['stub_reason']}" + record["specifikacija"] = provenance + _SI_SPECIFIKACIJA_BY_SLUG[slug] = provenance + augmented += 1 + return augmented From e3542239d76b80159c526c19914644d7515db3fb Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Thu, 11 Jun 2026 07:32:34 +0200 Subject: [PATCH 29/41] refactor: extract remaining 9 augmenters to _lib/augment/ (no-op) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Completes Phase 6.1: bg/cy/gr/sk/ro/hu/de + it (4 fns: masaf, _backfill, regional-registers, synthesize-sottozone) + cz (with its colour-block map + in-function grape-lexicon import) move verbatim to _lib/augment/.py. _IT_REGISTER_BY_SLUG joins the shared caches; _IT_MENZIONI_BY_SLUG stays in 04 (only 04's menzioni-harvest touches it). HU/CZ sidecar dirs stay importable from _shared (04 reads them outside the augmenter). synthesize_it_sottozone's call order in main() is preserved. 04_build_maps.py is now ~1400 lines lighter. Verified move-only: golden comparator vs the (yesterday) baseline differs ONLY in sitemap.xml, and ONLY in (the build date rolled 06-10→06-11; grep -vc lastmod = 0). Every HTML page, app.js, aocs.js, data and robots are byte-identical; write_seo_files was untouched. 63 tests + ruff clean. Co-Authored-By: Claude Opus 4.8 (1M context) --- scripts/04_build_maps.py | 1123 +------------------------------ scripts/_lib/augment/_shared.py | 4 + scripts/_lib/augment/bg.py | 88 +++ scripts/_lib/augment/cy.py | 78 +++ scripts/_lib/augment/cz.py | 200 ++++++ scripts/_lib/augment/de.py | 187 +++++ scripts/_lib/augment/gr.py | 90 +++ scripts/_lib/augment/hu.py | 101 +++ scripts/_lib/augment/it.py | 283 ++++++++ scripts/_lib/augment/ro.py | 92 +++ scripts/_lib/augment/sk.py | 88 +++ 11 files changed, 1227 insertions(+), 1107 deletions(-) create mode 100644 scripts/_lib/augment/bg.py create mode 100644 scripts/_lib/augment/cy.py create mode 100644 scripts/_lib/augment/cz.py create mode 100644 scripts/_lib/augment/de.py create mode 100644 scripts/_lib/augment/gr.py create mode 100644 scripts/_lib/augment/hu.py create mode 100644 scripts/_lib/augment/it.py create mode 100644 scripts/_lib/augment/ro.py create mode 100644 scripts/_lib/augment/sk.py diff --git a/scripts/04_build_maps.py b/scripts/04_build_maps.py index 9caf50f..26331f1 100644 --- a/scripts/04_build_maps.py +++ b/scripts/04_build_maps.py @@ -52,13 +52,29 @@ _HR_SPECIFIKACIJA_BY_SLUG, _HU_NATIONAL_SPEC_BY_SLUG, _IT_MASAF_BY_SLUG, + _IT_REGISTER_BY_SLUG, _RO_NATIONAL_SPEC_BY_SLUG, _SI_SPECIFIKACIJA_BY_SLUG, _SK_NATIONAL_SPEC_BY_SLUG, + NATIONAL_SPECS_CZ, + NATIONAL_SPECS_HU, ) +from _lib.augment.bg import augment_bg_records_with_national_specs +from _lib.augment.cy import augment_cy_records_with_national_specs +from _lib.augment.cz import augment_cz_records_with_national_specs +from _lib.augment.de import augment_de_records_with_produktspezifikation from _lib.augment.es import augment_es_records_with_national_pliegos +from _lib.augment.gr import augment_gr_records_with_national_specs from _lib.augment.hr import augment_hr_records_with_specifikacija +from _lib.augment.hu import augment_hu_records_with_national_specs +from _lib.augment.it import ( + augment_it_records_with_masaf, + augment_it_records_with_regional_registers, + synthesize_it_sottozone_records, +) +from _lib.augment.ro import augment_ro_records_with_national_specs from _lib.augment.si import augment_si_records_with_specifikacija +from _lib.augment.sk import augment_sk_records_with_national_specs from _lib.be.geometry import BEPolygonIndex from _lib.be.region import derive_region as derive_be_region from _lib.bg.geometry import BGPolygonIndex @@ -104,7 +120,6 @@ from _lib.it.comune import ITCommuneIndex from _lib.it.geometry import ITPolygonIndex from _lib.it.region import derive_regione as derive_it_regione -from _lib.it.sottozona import extract_sottozone as extract_it_sottozone from _lib.it.zones import ITZoneIndex from _lib.lieu_dit import LieuDitIndex, derive_climat_name from _lib.lu.geometry import LUPolygonIndex @@ -159,28 +174,18 @@ EXTRACTED_PT = ROOT / "raw" / "pt" / "cadernos-extracted" PT_CAOP_DIR = ROOT / "raw" / "pt" / "caop" EXTRACTED_IT = ROOT / "raw" / "it" / "disciplinari-extracted" -MASAF_DISCIPLINARI_IT = ROOT / "raw" / "it" / "masaf-disciplinari-extracted" -IT_REGIONAL_REGISTERS = ROOT / "raw" / "it" / "regional-variety-registers" EXTRACTED_AT = ROOT / "raw" / "at" / "dokumente-extracted" AT_STATISTIK_DIR = ROOT / "raw" / "at" / "statistik" EXTRACTED_DE = ROOT / "raw" / "de" / "dokumente-extracted" -PRODUKTSPEZIFIKATION_DE = ROOT / "raw" / "de" / "produktspezifikationen-extracted" EXTRACTED_SI = ROOT / "raw" / "si" / "dokumenti-extracted" EXTRACTED_HR = ROOT / "raw" / "hr" / "dokumenti-extracted" EXTRACTED_HU = ROOT / "raw" / "hu" / "dokumentumok-extracted" EXTRACTED_RO = ROOT / "raw" / "ro" / "dokumente-extracted" EXTRACTED_BG = ROOT / "raw" / "bg" / "dokumenti-extracted" -NATIONAL_SPECS_BG = ROOT / "raw" / "bg" / "national-specs-extracted" EXTRACTED_GR = ROOT / "raw" / "gr" / "dokumenti-extracted" -NATIONAL_SPECS_GR = ROOT / "raw" / "gr" / "national-specs-extracted" EXTRACTED_CY = ROOT / "raw" / "cy" / "dokumenti-extracted" -NATIONAL_SPECS_CY = ROOT / "raw" / "cy" / "national-specs-extracted" -NATIONAL_SPECS_RO = ROOT / "raw" / "ro" / "national-specs-extracted" -NATIONAL_SPECS_HU = ROOT / "raw" / "hu" / "national-specs-extracted" EXTRACTED_SK = ROOT / "raw" / "sk" / "dokumenty-extracted" -NATIONAL_SPECS_SK = ROOT / "raw" / "sk" / "national-specs-extracted" EXTRACTED_CZ = ROOT / "raw" / "cz" / "dokumenty-extracted" -NATIONAL_SPECS_CZ = ROOT / "raw" / "cz" / "national-specs" EXTRACTED_CH = ROOT / "raw" / "ch" / "dokumente-extracted" CH_SWISSTOPO_GPKG = ROOT / "raw" / "ch" / "swisstopo" / "swissboundaries3d_2026-01_2056_5728.gpkg" CH_SITG_GEOJSON = ROOT / "raw" / "ch" / "geoportals" / "sitg-vit-vignoble-ao.geojson" @@ -228,1108 +233,12 @@ } -def augment_gr_records_with_national_specs(records: list[dict]) -> int: - """In-place merge of GR national-spec sidecar data into stub records. - - Sibling of `augment_si_records_with_specifikacija`. 138 of 147 GR - wines ship as content-stubs (no fetchable EU-OJ ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ). - Stage 02f (`scripts/gr/02f_extract_national_specs.py`) parses the - ΥΠΑΑΤ national προδιαγραφή / τεχνικός φάκελος fetched by stage 01c - into `raw/gr/national-specs-extracted/.json` (132 of 138; the - other 6 are unresolved — see CURATOR_TODO.md). - - For each GR stub with a matching sidecar: - - grapes ← §6 ΟΙΝΟΠΟΙΗΣΙΜΕΣ ΠΟΙΚΙΛΙΕΣ (PDF list) or the - grape section's capitalised-token scan (.doc prose) - - link_to_terroir ← §7 ΔΕΣΜΟΣ ΜΕ ΤΗΝ ΓΕΩΓΡΑΦΙΚΗ ΠΕΡΙΟΧΗ - - geo_area_brief / summary / styles ← matching sections - - section_roles ← unified role dict so 02d reads terroir uniformly - - stub_reason ← prefixed `national-spec:` so the audit can tell - EU-OJ-extracted from spec-augmented wines - - national_spec ← provenance block (url, sha256, format, …) - - `record["stub"]` stays True — still NOT an EU-OJ extraction, just - augmented with the canonical ΥΠΑΑΤ source. Returns count augmented. - """ - _GR_NATIONAL_SPEC_BY_SLUG.clear() - if not NATIONAL_SPECS_GR.exists(): - return 0 - augmented = 0 - for record in records: - if record.get("country") != "gr" or not record.get("stub"): - continue - slug = record.get("slug") - if not slug: - continue - sidecar_path = NATIONAL_SPECS_GR / f"{slug}.json" - if not sidecar_path.exists(): - continue - try: - sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) - except (ValueError, OSError): - continue - - src = sidecar.get("source") or {} - provenance = { - "url": src.get("source_url") or "", - "sha256": src.get("sha256") or "", - "fetched_at": src.get("fetched_at") or "", - "format": src.get("format") or "", - "source_org": src.get("source_org") or "ypaat", - "filename": src.get("filename") or "", - "parser_template": sidecar.get("parser_template") or "", - } - - if sidecar.get("summary"): - record["summary"] = sidecar["summary"] - if sidecar.get("grapes") and (sidecar["grapes"].get("principal") - or sidecar["grapes"].get("accessory")): - record["grapes"] = sidecar["grapes"] - if sidecar.get("geo_area_brief"): - record["geo_area_brief"] = sidecar["geo_area_brief"] - if sidecar.get("link_to_terroir"): - record["link_to_terroir"] = sidecar["link_to_terroir"] - if sidecar.get("styles"): - record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) - - section_roles = dict(record.get("section_roles") or {}) - for role in ("description", "geo_area", "grape_varieties", "link_to_terroir"): - sidecar_roles = sidecar.get("section_roles") or {} - if sidecar_roles.get(role): - section_roles[role] = sidecar_roles[role] - record["section_roles"] = section_roles - - if record.get("stub_reason") and not record["stub_reason"].startswith("national-spec:"): - record["stub_reason"] = f"national-spec:{record['stub_reason']}" - record["national_spec"] = provenance - _GR_NATIONAL_SPEC_BY_SLUG[slug] = provenance - augmented += 1 - return augmented - - -def augment_cy_records_with_national_specs(records: list[dict]) -> int: - """In-place merge of CY national-spec sidecar data into stub records. - - Sibling of `augment_gr_records_with_national_specs`. All 11 CY wines - ship as content-stubs (no fetchable EU-OJ ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ). Stage 02f - (`scripts/cy/02f_extract_national_specs.py`) parses the moa.gov.cy - Department-of-Agriculture τεχνικός φάκελος (Greek single-document - PDF, OCR'd when image-only) into `raw/cy/national-specs-extracted/ - .json`; this merges grapes / terroir text / styles / geo-area - into the in-memory stub. `record["stub"]` stays True. Returns the - count augmented.""" - _CY_NATIONAL_SPEC_BY_SLUG.clear() - if not NATIONAL_SPECS_CY.exists(): - return 0 - augmented = 0 - for record in records: - if record.get("country") != "cy" or not record.get("stub"): - continue - slug = record.get("slug") - if not slug: - continue - sidecar_path = NATIONAL_SPECS_CY / f"{slug}.json" - if not sidecar_path.exists(): - continue - try: - sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) - except (ValueError, OSError): - continue - - src = sidecar.get("source") or {} - provenance = { - "url": src.get("source_url") or "", - "sha256": src.get("sha256") or "", - "fetched_at": src.get("fetched_at") or "", - "format": src.get("format") or "", - "source_org": src.get("source_org") or "moa-cy", - "filename": src.get("filename") or "", - "parser_template": sidecar.get("parser_template") or "", - } - - if sidecar.get("summary"): - record["summary"] = sidecar["summary"] - if sidecar.get("grapes") and (sidecar["grapes"].get("principal") - or sidecar["grapes"].get("accessory")): - record["grapes"] = sidecar["grapes"] - if sidecar.get("geo_area_brief"): - record["geo_area_brief"] = sidecar["geo_area_brief"] - if sidecar.get("link_to_terroir"): - record["link_to_terroir"] = sidecar["link_to_terroir"] - if sidecar.get("styles"): - record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) - - section_roles = dict(record.get("section_roles") or {}) - for role in ("description", "geo_area", "grape_varieties", "link_to_terroir"): - sidecar_roles = sidecar.get("section_roles") or {} - if sidecar_roles.get(role): - section_roles[role] = sidecar_roles[role] - record["section_roles"] = section_roles - - if record.get("stub_reason") and not record["stub_reason"].startswith("national-spec:"): - record["stub_reason"] = f"national-spec:{record['stub_reason']}" - record["national_spec"] = provenance - _CY_NATIONAL_SPEC_BY_SLUG[slug] = provenance - augmented += 1 - return augmented - - -def augment_bg_records_with_national_specs(records: list[dict]) -> int: - """In-place merge of BG national-spec sidecar data into stub records. - - Sibling of `augment_gr_records_with_national_specs`. 51 of 54 BG wines - ship as content-stubs (no fetchable EU-OJ ЕДИНЕН ДОКУМЕНТ). Stage 02f - (`scripts/bg/02f_extract_national_specs.py`) parses the ИАЛВ / IAVV - per-wine продуктова спецификация PDF fetched by stage 01c into - `raw/bg/national-specs-extracted/.json` (51 of 51). - - For each BG stub with a matching sidecar: - - grapes ← section 5 (Винени сортове грозде, colour-split) - - link_to_terroir ← section 6 (Връзка с географския район) - - geo_area_brief / summary / styles ← matching sections - - section_roles ← unified role dict so 02d reads terroir uniformly - - stub_reason ← prefixed `national-spec:` so the audit can tell - EU-OJ-extracted from spec-augmented wines - - national_spec ← provenance block (url, sha256, format, …) - - `record["stub"]` stays True — still NOT an EU-OJ extraction, just - augmented with the canonical ИАЛВ source. Returns count augmented. - """ - _BG_NATIONAL_SPEC_BY_SLUG.clear() - if not NATIONAL_SPECS_BG.exists(): - return 0 - augmented = 0 - for record in records: - if record.get("country") != "bg" or not record.get("stub"): - continue - slug = record.get("slug") - if not slug: - continue - sidecar_path = NATIONAL_SPECS_BG / f"{slug}.json" - if not sidecar_path.exists(): - continue - try: - sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) - except (ValueError, OSError): - continue - - src = sidecar.get("source") or {} - provenance = { - "url": src.get("url") or "", - "sha256": src.get("sha256") or "", - "fetched_at": src.get("fetched_at") or "", - "format": src.get("format") or "", - "source_org": src.get("source_org") or "iavv", - "filename": src.get("filename") or "", - "parser_template": sidecar.get("parser_template") or "", - } - - if sidecar.get("summary"): - record["summary"] = sidecar["summary"] - if sidecar.get("grapes") and (sidecar["grapes"].get("principal") - or sidecar["grapes"].get("accessory")): - record["grapes"] = sidecar["grapes"] - if sidecar.get("geo_area_brief"): - record["geo_area_brief"] = sidecar["geo_area_brief"] - if sidecar.get("link_to_terroir"): - record["link_to_terroir"] = sidecar["link_to_terroir"] - if sidecar.get("styles"): - record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) - - section_roles = dict(record.get("section_roles") or {}) - for role in ("description", "geo_area", "grape_varieties", "link_to_terroir"): - sidecar_roles = sidecar.get("section_roles") or {} - if sidecar_roles.get(role): - section_roles[role] = sidecar_roles[role] - record["section_roles"] = section_roles - - if record.get("stub_reason") and not record["stub_reason"].startswith("national-spec:"): - record["stub_reason"] = f"national-spec:{record['stub_reason']}" - record["national_spec"] = provenance - _BG_NATIONAL_SPEC_BY_SLUG[slug] = provenance - augmented += 1 - return augmented - - -def augment_sk_records_with_national_specs(records: list[dict]) -> int: - """In-place merge of SK national-spec sidecar data into stub records. - - Sibling of `augment_bg_records_with_national_specs`. 5 of the SK - content-stubs (no fetchable EU-OJ JEDNOTNÝ DOKUMENT) are augmented from - the ÚPV SR (indprop.gov.sk) per-wine špecifikácia výrobku. Stage 02f - (`scripts/sk/02f_extract_national_specs.py`) parses each text-layer PDF - fetched by stage 01c into `raw/sk/national-specs-extracted/.json`. - - For each SK stub with a matching sidecar: - - grapes ← section f) označenie odrody alebo odrôd - - link_to_terroir ← section g) údaje potvrdzujúce spojitosť - - geo_area_brief / summary / styles ← matching sections - - section_roles ← unified role dict so 02d reads terroir uniformly - - stub_reason ← prefixed `national-spec:` so the audit can tell - EU-OJ-extracted from spec-augmented wines - - national_spec ← provenance block (url, sha256, format, …) - - `record["stub"]` stays True — still NOT an EU-OJ extraction, just - augmented with the canonical ÚPV SR source. Returns count augmented. - """ - _SK_NATIONAL_SPEC_BY_SLUG.clear() - if not NATIONAL_SPECS_SK.exists(): - return 0 - augmented = 0 - for record in records: - if record.get("country") != "sk" or not record.get("stub"): - continue - slug = record.get("slug") - if not slug: - continue - sidecar_path = NATIONAL_SPECS_SK / f"{slug}.json" - if not sidecar_path.exists(): - continue - try: - sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) - except (ValueError, OSError): - continue - - src = sidecar.get("source") or {} - provenance = { - "url": src.get("url") or "", - "sha256": src.get("sha256") or "", - "fetched_at": src.get("fetched_at") or "", - "format": src.get("format") or "", - "source_org": src.get("source_org") or "upv-sr", - "filename": src.get("filename") or "", - "parser_template": sidecar.get("parser_template") or "", - } - - if sidecar.get("summary"): - record["summary"] = sidecar["summary"] - if sidecar.get("grapes") and (sidecar["grapes"].get("principal") - or sidecar["grapes"].get("accessory")): - record["grapes"] = sidecar["grapes"] - if sidecar.get("geo_area_brief"): - record["geo_area_brief"] = sidecar["geo_area_brief"] - if sidecar.get("link_to_terroir"): - record["link_to_terroir"] = sidecar["link_to_terroir"] - if sidecar.get("styles"): - record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) - - section_roles = dict(record.get("section_roles") or {}) - for role in ("description", "geo_area", "grape_varieties", "link_to_terroir"): - sidecar_roles = sidecar.get("section_roles") or {} - if sidecar_roles.get(role): - section_roles[role] = sidecar_roles[role] - record["section_roles"] = section_roles - - if record.get("stub_reason") and not record["stub_reason"].startswith("national-spec:"): - record["stub_reason"] = f"national-spec:{record['stub_reason']}" - record["national_spec"] = provenance - _SK_NATIONAL_SPEC_BY_SLUG[slug] = provenance - augmented += 1 - return augmented - - -def augment_ro_records_with_national_specs(records: list[dict]) -> int: - """In-place merge of RO national-spec sidecar data into stub records. - - Sibling of `augment_gr_records_with_national_specs`. The 14 - grandfathered RO wines (eAmbrosia carries only a non-fetchable - `Ares(...)` reference — no EU-OJ DOCUMENT UNIC) ship as content-stubs. - Stage 02f (`scripts/ro/02f_extract_national_specs.py`) parses the - ONVPV caiet de sarcini fetched by stage 01c into - `raw/ro/national-specs-extracted/.json`. - - For each RO stub with a matching sidecar: - - grapes ← §IV Soiurile de struguri (colour-grouped) - - link_to_terroir ← §II Legătura cu aria geografică - - geo_communes ← §III Delimitarea geografică (drives the GISCO - commune-union geometry for the 2 grandfathered - IGPs — the RO-specific delta vs. GR/HR) - - geo_area_brief / summary / styles ← matching sections - - section_roles ← unified role dict so 02d reads terroir uniformly - - stub_reason ← prefixed `national-spec:` - - national_spec ← provenance block (url, sha256, format, …) - - `record["stub"]` stays True. Returns count augmented. - """ - _RO_NATIONAL_SPEC_BY_SLUG.clear() - if not NATIONAL_SPECS_RO.exists(): - return 0 - augmented = 0 - for record in records: - if record.get("country") != "ro" or not record.get("stub"): - continue - slug = record.get("slug") - if not slug: - continue - sidecar_path = NATIONAL_SPECS_RO / f"{slug}.json" - if not sidecar_path.exists(): - continue - try: - sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) - except (ValueError, OSError): - continue - - src = sidecar.get("source") or {} - provenance = { - "url": src.get("source_url") or "", - "sha256": src.get("sha256") or "", - "fetched_at": src.get("fetched_at") or "", - "format": src.get("format") or "", - "source_org": src.get("source_org") or "onvpv", - "filename": src.get("filename") or "", - "parser_template": sidecar.get("parser_template") or "", - } - - if sidecar.get("summary"): - record["summary"] = sidecar["summary"] - if sidecar.get("grapes") and (sidecar["grapes"].get("principal") - or sidecar["grapes"].get("accessory")): - record["grapes"] = sidecar["grapes"] - if sidecar.get("geo_area_brief"): - record["geo_area_brief"] = sidecar["geo_area_brief"] - if sidecar.get("geo_communes"): - record["geo_communes"] = sidecar["geo_communes"] - if sidecar.get("link_to_terroir"): - record["link_to_terroir"] = sidecar["link_to_terroir"] - if sidecar.get("styles"): - record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) - - section_roles = dict(record.get("section_roles") or {}) - for role in ("geo_area", "grape_varieties", "link_to_terroir"): - sidecar_roles = sidecar.get("section_roles") or {} - if sidecar_roles.get(role): - section_roles[role] = sidecar_roles[role] - record["section_roles"] = section_roles - - if record.get("stub_reason") and not record["stub_reason"].startswith("national-spec:"): - record["stub_reason"] = f"national-spec:{record['stub_reason']}" - record["national_spec"] = provenance - _RO_NATIONAL_SPEC_BY_SLUG[slug] = provenance - augmented += 1 - return augmented - - -def augment_hu_records_with_national_specs(records: list[dict]) -> int: - """In-place merge of HU national-spec sidecar data into stub records. - - Sibling of `augment_ro_records_with_national_specs`. The 15 - grandfathered HU wines (eAmbrosia carries only a non-fetchable - `Ares(...)` reference — no EU-OJ EGYSÉGES DOKUMENTUM) ship as - content-stubs. Stage 02f (`scripts/hu/02f_extract_national_specs.py`) - parses the Agrárminisztérium termékleírás PDF fetched by stage 01c - into `raw/hu/national-specs-extracted/.json`. - - For each HU stub with a matching sidecar: - - grapes ← VI. ENGEDÉLYEZETT SZŐLŐFAJTÁK - - link_to_terroir ← VII. KAPCSOLAT A FÖLDRAJZI TERÜLETTEL - - geo_communes ← IV. KÖRÜLHATÁROLT TERÜLET (commune-precision; - geometry still prefers the Bétard polygon - these wines already have, so this is a record) - - geo_area_brief / summary / styles ← matching sections - - section_roles ← unified role dict so 02d reads terroir uniformly - - stub_reason ← prefixed `national-spec:` - - national_spec ← provenance block (url, sha256, format, …) - - `record["stub"]` stays True. Returns count augmented. - """ - _HU_NATIONAL_SPEC_BY_SLUG.clear() - if not NATIONAL_SPECS_HU.exists(): - return 0 - augmented = 0 - for record in records: - if record.get("country") != "hu": - continue - slug = record.get("slug") - if not slug: - continue - sidecar_path = NATIONAL_SPECS_HU / f"{slug}.json" - if not sidecar_path.exists(): - continue - try: - sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) - except (ValueError, OSError): - continue - - src = sidecar.get("source") or {} - provenance = { - "url": src.get("source_url") or "", - "sha256": src.get("sha256") or "", - "fetched_at": src.get("fetched_at") or "", - "format": src.get("format") or "", - "source_org": src.get("source_org") or "agrarminiszterium", - "filename": src.get("filename") or "", - "parser_template": sidecar.get("parser_template") or "", - } - - # Fill-if-empty: a stub is fully empty so this fills everything; - # a non-stub with a thin EU extraction (e.g. Badacsony, whose - # awkward doc structure left the grape section unrouted) gets only - # its EMPTY fields filled — good EUR-Lex data is never clobbered. - cur_grapes = record.get("grapes") or {} - if (sidecar.get("summary") and not record.get("summary")): - record["summary"] = sidecar["summary"] - if (sidecar.get("grapes") - and (sidecar["grapes"].get("principal") or sidecar["grapes"].get("accessory")) - and not (cur_grapes.get("principal") or cur_grapes.get("accessory"))): - record["grapes"] = sidecar["grapes"] - if sidecar.get("geo_area_brief") and not record.get("geo_area_brief"): - record["geo_area_brief"] = sidecar["geo_area_brief"] - if sidecar.get("geo_communes") and not record.get("geo_communes"): - record["geo_communes"] = sidecar["geo_communes"] - if sidecar.get("dulok") and not record.get("dulok"): - record["dulok"] = sidecar["dulok"] - if sidecar.get("link_to_terroir") and not record.get("link_to_terroir"): - record["link_to_terroir"] = sidecar["link_to_terroir"] - if sidecar.get("styles"): - record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) - - section_roles = dict(record.get("section_roles") or {}) - sidecar_roles = sidecar.get("section_roles") or {} - for role in ("geo_area", "grape_varieties", "link_to_terroir"): - if sidecar_roles.get(role) and not section_roles.get(role): - section_roles[role] = sidecar_roles[role] - record["section_roles"] = section_roles - - if record.get("stub") and record.get("stub_reason") \ - and not record["stub_reason"].startswith("national-spec:"): - record["stub_reason"] = f"national-spec:{record['stub_reason']}" - record["national_spec"] = provenance - _HU_NATIONAL_SPEC_BY_SLUG[slug] = provenance - augmented += 1 - return augmented - - -def _backfill_it_nonstub_from_masaf(record: dict, sidecar: dict) -> bool: - """Fill ONLY the empty fields of a non-stub IT record from its MASAF - sidecar — the documento unico is canonical, but some OJ docs omit the - geo area or variety list, and the national disciplinare carries them. - Never overwrites populated docunico data. Returns True if anything - was filled.""" - filled = False - g = record.get("grapes") or {} - if sidecar.get("grapes") and not (g.get("principal") or g.get("accessory")): - record["grapes"] = sidecar["grapes"] - filled = True - if sidecar.get("menzioni") and not record.get("menzioni"): - record["menzioni"] = sidecar["menzioni"] - filled = True - section_roles = dict(record.get("section_roles") or {}) - if sidecar.get("geo_area_brief") and not (record.get("geo_area_brief") or "").strip(): - record["geo_area_brief"] = sidecar["geo_area_brief"] - section_roles["geo_area"] = sidecar["geo_area_brief"] - filled = True - if sidecar.get("link_to_terroir") and not (record.get("link_to_terroir") or "").strip(): - record["link_to_terroir"] = sidecar["link_to_terroir"] - section_roles["link_to_terroir"] = sidecar["link_to_terroir"] - filled = True - if filled: - record["section_roles"] = section_roles - record["masaf_backfill"] = True - return filled - - -def augment_it_records_with_masaf(records: list[dict]) -> int: - """In-place merge of MASAF disciplinare sidecar data into IT stub - records. Only stubs are touched — wines whose documento unico was - extracted in stage 02 already carry canonical EUR-Lex data and - shouldn't be overwritten. - - For each IT stub with a matching sidecar at - raw/it/masaf-disciplinari-extracted/.json the following - fields are merged: - - summary ← Article 1 first paragraph - - regione ← derived from Article 3 / 9 text - - grapes ← parsed from Article 2 (principal-only) - - geo_area_brief ← Article 3 body - - link_to_terroir ← Article 9 body - - section_roles ← {grape_varieties, geo_area, link_to_terroir, ...} - - stub_reason ← prefixed "masaf:" so the audit can tell - doc-unico-extracted from masaf-augmented - - masaf ← provenance block (url, sha256, fetched_at, - parser_template, bundle_key, archive_path) - - `record["stub"]` stays True — the record is still NOT a documento - unico extraction, just augmented. Stage 03 / 04 callers use the - `masaf` block to distinguish. - - Returns the number of records augmented. - """ - _IT_MASAF_BY_SLUG.clear() - if not MASAF_DISCIPLINARI_IT.exists(): - return 0 - augmented = 0 - for record in records: - if record.get("country") != "it": - continue - slug = record.get("slug") - if not slug: - continue - sidecar_path = MASAF_DISCIPLINARI_IT / f"{slug}.json" - if not sidecar_path.exists(): - continue - try: - sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) - except (ValueError, OSError): - continue - - # Non-stub records carry canonical EUR-Lex documento-unico data — - # only BACKFILL fields the documento unico left empty (some OJ - # docs omit the area or variety list), never overwrite. Stubs get - # the full merge below. - if not record.get("stub"): - if _backfill_it_nonstub_from_masaf(record, sidecar): - augmented += 1 - continue - - # Build the provenance block (also cached for the AOC-blob phase). - src = sidecar.get("source") or {} - match_info = sidecar.get("match") or {} - provenance = { - "filename": src.get("filename") or "", - "sha256": src.get("sha256") or "", - "bytes": src.get("bytes") or 0, - "fetched_at": src.get("fetched_at") or "", - "parser_template": sidecar.get("parser_template") or "", - "bundle_key": src.get("bundle_key") or "", - "archive_path": src.get("archive_path") or "", - "match_how": match_info.get("how") or "", - "pdf_filename": match_info.get("pdf_filename") or "", - # When an override pinned the URL, surface it for the panel. - "override_url": src.get("url") or "", - "override_source_org": src.get("source_org") or "", - } - - # Merge augmented fields onto the record. Replace rather than - # union — the record was a stub so there's nothing to lose. - if sidecar.get("summary"): - record["summary"] = sidecar["summary"] - if sidecar.get("regione") and not record.get("regione"): - record["regione"] = sidecar["regione"] - if sidecar.get("grapes"): - record["grapes"] = sidecar["grapes"] - if sidecar.get("menzioni") and not record.get("menzioni"): - record["menzioni"] = sidecar["menzioni"] - if sidecar.get("geo_area_brief"): - record["geo_area_brief"] = sidecar["geo_area_brief"] - if sidecar.get("link_to_terroir"): - record["link_to_terroir"] = sidecar["link_to_terroir"] - # IT MASAF is the last national-spec layer to carry styles; merge them - # the same way every other augment does (union, never clobber). The - # disciplinare's tipologie + organoleptic articles supply the markers - # (spumante / passito / vin santo / dolce) the grape-colour floor can't - # infer; the floor still backfills any colour the scan missed. - if sidecar.get("styles"): - record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) - section_roles = dict(record.get("section_roles") or {}) - if sidecar.get("grapes"): - section_roles.setdefault("grape_varieties", "") - if sidecar.get("geo_area_brief"): - section_roles["geo_area"] = sidecar["geo_area_brief"] - if sidecar.get("link_to_terroir"): - section_roles["link_to_terroir"] = sidecar["link_to_terroir"] - if sidecar.get("summary"): - section_roles["description"] = sidecar["summary"] - record["section_roles"] = section_roles - - if record.get("stub_reason") and not record["stub_reason"].startswith("masaf:"): - record["stub_reason"] = f"masaf:{record['stub_reason']}" - record["masaf"] = provenance - _IT_MASAF_BY_SLUG[slug] = provenance - augmented += 1 - return augmented - - -_IT_REGISTER_BY_SLUG: dict[str, dict] = {} # Slug → menzioni (MGA/UGA cru name list) for the panel chip section. # Populated after the IT augments (menzioni live on the in-memory record, # not in the feature props or the on-disk stub) and read by the aocs blob. _IT_MENZIONI_BY_SLUG: dict[str, list] = {} -def augment_it_records_with_regional_registers(records: list[dict]) -> int: - """Fill the grape roster of regional-IGT records whose disciplinare - defers to the Region's authorised-variety register (the annex is - absent from the MASAF PDF). Each region sidecar at - raw/it/regional-variety-registers/.json lists the IGT slugs - (`igts`) that draw from it. Only applied when the record still has no - grapes, so a varietal IGT (e.g. catalanesca-del-monte-somma, excluded - from the `igts` lists) is never given a whole regional roster. - - Returns the number of records given a roster.""" - _IT_REGISTER_BY_SLUG.clear() - sources = IT_REGIONAL_REGISTERS / "sources.json" - if not sources.exists(): - return 0 - by_slug: dict[str, dict] = {} - for region in json.loads(sources.read_text(encoding="utf-8")): - if region.startswith("_"): - continue - sidecar_path = IT_REGIONAL_REGISTERS / f"{region}.json" - if not sidecar_path.exists(): - continue - try: - sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) - except (ValueError, OSError): - continue - for igt in sidecar.get("igts", []): - by_slug[igt] = sidecar - - augmented = 0 - for record in records: - if record.get("country") != "it": - continue - slug = record.get("slug") - sidecar = by_slug.get(slug) - if not sidecar: - continue - g = record.get("grapes") or {} - if g.get("principal") or g.get("accessory"): - continue - slugs = [v["slug"] for v in sidecar.get("varieties", [])] - if not slugs: - continue - record["grapes"] = { - "principal": slugs, - "accessory": [], - "observation": [], - "details": [ - {"slug": v["slug"], "name": v["name"], "role": "principal", - "colour": v.get("colour", ""), - "source": "regional-variety-register"} - for v in sidecar["varieties"] - ], - } - src = sidecar.get("source") or {} - provenance = { - "region": sidecar.get("region", ""), - "url": src.get("url", ""), - "source_org": src.get("source_org", ""), - "note": src.get("note", ""), - "sha256": src.get("sha256", ""), - "n_varieties": len(slugs), - } - record["regional_register"] = provenance - _IT_REGISTER_BY_SLUG[slug] = provenance - augmented += 1 - return augmented - - -def synthesize_it_sottozone_records(records: list[dict]) -> int: - """Emit first-class sub-denomination records for Italian sottozone - detected in the MASAF disciplinare (Chianti's 7, Valtellina's 5, - Bardolino's 3, …). The EU documento unico rarely names them, so - stage 02 emits none — they live in the national disciplinare's - Article 1, which 02f cached in the sidecar's `article_bodies`. - - Each sottozona becomes a child record mirroring the ES subzona / - FR DGC model: `is_sub_denomination=True`, `parent_slug`, - `parent_name`, `parent_id_eambrosia`, inheriting the parent's - grapes / styles / terroir / regione. Geometry resolves via the - stage-04 `parent-appellation` inheritance step. Appended to - `records` (processed after every parent, so parent geometry is - available). Returns the number of sottozona records created.""" - if not MASAF_DISCIPLINARI_IT.exists(): - return 0 - existing = {r.get("slug") for r in records if r.get("country") == "it"} - new_records: list[dict] = [] - for record in list(records): - if record.get("country") != "it" or record.get("is_sub_denomination"): - continue - slug = record.get("slug") - sidecar_path = MASAF_DISCIPLINARI_IT / f"{slug}.json" - if not slug or not sidecar_path.exists(): - continue - try: - sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) - except (ValueError, OSError): - continue - bodies = sidecar.get("article_bodies") or {} - text = " ".join( - [sidecar.get("geo_area_brief") or "", bodies.get("1", ""), bodies.get("3", "")] - ) - parent_name = record.get("name") or slug - for sz in extract_it_sottozone(text, parent_name): - sz_slug = f"{slug}-{sz['slug']}" - if not sz["slug"] or sz_slug in existing: - continue - existing.add(sz_slug) - child = dict(record) - child.update({ - "slug": sz_slug, - "name": f"{parent_name} {sz['name']}", - "is_sub_denomination": True, - "parent_slug": slug, - "parent_name": parent_name, - "parent_id_eambrosia": record.get("id_eambrosia") or "", - "menzioni": [], - "sottozona_source": "masaf-disciplinare-article-1", - }) - new_records.append(child) - records.extend(new_records) - return len(new_records) - - -def augment_de_records_with_produktspezifikation(records: list[dict]) -> int: - """In-place merge of BLE-Produktspezifikation sidecar data into DE - parent-Anbaugebiet records. - - The EU Einziges Dokument for German wines doesn't carry a principal/ - accessory split — section 7 is a flat list. The BLE national - Produktspezifikation (Amtliches Werk §5 UrhG) names individual - varieties with their own Mindestmostgewicht threshold in §3.2 (Mosel - → Riesling/Elbling/Müller-Thurgau/Dornfelder). Stage 02f extracts - that split into raw/de/produktspezifikationen-extracted/.json; - this augment re-tags the in-memory record's grapes block accordingly. - - Two sidecar categories are merged (both written by stage 02f): - - the 13 Anbaugebiete (regional PDOs), with a principal/accessory - split from §3.2 Mindestmostgewicht; and - - the 15 Landwein g.g.A. that ship as stubs (no EU Einziges - Dokument). Their BLE spec has no role split, so they arrive as - `section-8-flat-no-split` (all-principal) and fold their full - roster + §-Zusammenhang terroir text into the stub record. - Einzellage sub-denominations are still skipped (they inherit the - parent Anbaugebiet's varieties at render time). - - For records with `role_split_method == "section-3.2-principal"`: - - re-tag existing record["grapes"]["details"] items as - principal/accessory based on the sidecar's slug sets - - rebuild record["grapes"]["principal"] / ["accessory"] lists - - fold any new sidecar slugs not already in the EU record - - For records with `role_split_method == "section-8-flat-no-split"` - (Anbaugebiete whose §3.2 doesn't enumerate per-variety thresholds — - Franken, Württemberg in v1): the existing all-principal default - stands; the sidecar just records the BLE source as provenance. - - Stage 04 reads the cached provenance later in the AOC-blob phase - via `_DE_PRODUKTSPEZIFIKATION_BY_SLUG`. - - Returns the number of records augmented. - """ - _DE_PRODUKTSPEZIFIKATION_BY_SLUG.clear() - if not PRODUKTSPEZIFIKATION_DE.exists(): - return 0 - augmented = 0 - for record in records: - if record.get("country") != "de": - continue - if record.get("is_sub_denomination"): - continue - slug = record.get("slug") or "" - sidecar_path = PRODUKTSPEZIFIKATION_DE / f"{slug}.json" - if not sidecar_path.exists(): - continue - try: - sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) - except (ValueError, OSError): - continue - - method = sidecar.get("role_split_method") or "" - src = sidecar.get("source") or {} - provenance = { - "url": src.get("url") or "", - "sha256": src.get("sha256") or "", - "bytes": src.get("bytes") or 0, - "fetched_at": src.get("fetched_at") or "", - "source_org": src.get("source_org") or "BLE", - "license": src.get("license") or "Amtliches Werk §5 UrhG", - "role_split_method": method, - "n_principal": sidecar.get("n_principal") or 0, - "n_accessory": sidecar.get("n_accessory") or 0, - } - - # BLE §8/§9 "Zusammenhang" terroir backfill — when the EU - # Einziges Dokument is sparse (stub or no link_to_terroir), use - # the BLE PDF's terroir block so 02d can extract facts from it. - # Affects Ahr, Baden, Hessische Bergstraße, Rheingau, Sachsen, - # Saale-Unstrut in v1. We also record the BLE PDF URL as the - # cahier_source for downstream provenance. - bz = (sidecar.get("zusammenhang_text") or "").strip() - eu_terroir = (record.get("link_to_terroir") or "").strip() - if bz and len(eu_terroir) < 400: - record["link_to_terroir"] = bz - section_roles = dict(record.get("section_roles") or {}) - section_roles["link_to_terroir"] = bz - record["section_roles"] = section_roles - # Surface the BLE PDF as the canonical terroir source so the - # panel + 02d attribute it correctly. - rec_src = dict(record.get("source") or {}) - rec_src["terroir_source_url"] = src.get("url") or "" - rec_src["terroir_source_org"] = "BLE" - record["source"] = rec_src - provenance["terroir_backfilled"] = True - - # For both role-split methods, fold the sidecar's variety roster - # into the in-memory record. The default fold tags non-sidecar - # EU slugs as `accessory` (the BLE PDF's §3.2 is the principal - # allowlist); the flat-no-split method instead tags everything - # as principal because the document doesn't enumerate a split. - if method in ("section-3.2-principal", "section-8-flat-no-split"): - # Build slug → role from the sidecar. - sidecar_role: dict[str, str] = {} - sidecar_details: list[dict] = [] - for g in sidecar.get("grapes") or []: - s = g.get("slug") - if s: - sidecar_role[s] = g.get("role") or "principal" - sidecar_details.append(g) - - # Re-tag the EU-record's existing details. Keep the EU - # record's display name (matches the wine's own - # Einziges-Dokument spelling). When the sidecar has an - # authoritative §3.2 split, slugs NOT named in the sidecar - # default to "accessory" — the BLE PDF's §3.2 is the - # principal-allowlist, so anything outside it is - # implicitly "alle übrigen Rebsorten". When the sidecar is - # flat-no-split, every slug is principal (the regulator - # didn't enumerate a split). - unmatched_default = ( - "principal" if method == "section-8-flat-no-split" else "accessory" - ) - grapes = dict(record.get("grapes") or {}) - details_in = grapes.get("details") or [] - new_details: list[dict] = [] - eu_slugs: set[str] = set() - for d in details_in: - new_d = dict(d) - s = new_d.get("slug") - if s: - new_d["role"] = sidecar_role.get(s, unmatched_default) - new_details.append(new_d) - if s: - eu_slugs.add(s) - - # Fold in any sidecar varieties that the EU record missed - # (the §8 list is more complete than the EU section 7 for - # some Anbaugebiete). - for g in sidecar_details: - s = g.get("slug") - if s and s not in eu_slugs: - new_details.append({ - "slug": s, - "name": g.get("name", s), - "role": g.get("role", "accessory"), - "colour": g.get("colour", ""), - "source": "ble-produktspezifikation", - }) - - # Rebuild principal / accessory lists from the re-tagged - # details (deterministic + dedup-by-first-seen). - principal: list[str] = [] - accessory: list[str] = [] - seen_p: set[str] = set() - seen_a: set[str] = set() - for d in new_details: - s = d.get("slug") - if not s: - continue - if d.get("role") == "accessory": - if s in seen_a: - continue - seen_a.add(s) - accessory.append(s) - else: - if s in seen_p: - continue - seen_p.add(s) - principal.append(s) - grapes["principal"] = principal - grapes["accessory"] = accessory - grapes["details"] = new_details - record["grapes"] = grapes - - record["produktspezifikation"] = provenance - _DE_PRODUKTSPEZIFIKATION_BY_SLUG[slug] = provenance - augmented += 1 - return augmented - - -def augment_cz_records_with_national_specs(records: list[dict]) -> int: - """In-place merge of the Czech national variety roster (Vyhláška - č. 88/2017 Sb. Příloha č. 2) into every CZ wine record. - - Czech wine law publishes one national variety list (35 white + 26 - red + 6 zemské-víno = 67 varieties) that applies to every jakostní - víno regardless of podoblast — no per-appellation restriction. So - every CZ wine that *should* carry a variety list (10 of 13, all - except 3 newer single-vineyard / single-varietal PDOs whose - Vyhláška-88 status is undocumented) gets the same fold: - - - white-wine PDOs/PGIs → all 35 white varieties as `principal` - - red-wine PDOs/PGIs → all 26 red varieties as `principal` - - mixed (most macros + podoblasti) → both lists folded; - zemské-víno-only varieties go under `accessory` (they apply - only to the lower zemské-víno PGI tier). - - Since the Vyhláška doesn't enumerate a per-appellation principal/ - accessory split (it's a flat national authorisation), we mark - everything `principal` and let the panel render "All Czech - jakostní vína authorise these 67 varieties — see Vyhláška č. - 88/2017 Sb." in the provenance. - - Stage 04 reads the cached provenance later via - `_CZ_NATIONAL_SPEC_BY_SLUG`. - - Returns the number of records augmented. - """ - _CZ_NATIONAL_SPEC_BY_SLUG.clear() - _CZ_CHZO_BY_SLUG.clear() - if not NATIONAL_SPECS_CZ.exists(): - return 0 - - # Load the two SZPI CHZO region specs (terroir source + style roster + - # provenance), keyed by region. Both PGIs are the spec's own subject; - # the macro CHOPs + podoblasti in that region share its section-1 - # terroir description (rendered by 02d) and cite the same SZPI PDF. - chzo_by_region: dict[str, dict] = {} - for key in ("chzo-moravske", "chzo-ceske"): - p = NATIONAL_SPECS_CZ / f"{key}.json" - if not p.exists(): - continue - try: - d = json.loads(p.read_text(encoding="utf-8")) - except (ValueError, OSError): - continue - if d.get("region"): - chzo_by_region[d["region"]] = d - # The 2 PGI slugs whose own product specification this is (they - # inherit the spec's per-style roster; the CHOPs/podoblasti keep - # grape-colour-inferred styles only). - chzo_pgi_slugs = {"moravske", "ceske"} - - varieties_path = NATIONAL_SPECS_CZ / "varieties.json" - manifest_path = NATIONAL_SPECS_CZ / "manifest.json" - if not varieties_path.exists(): - return 0 - try: - spec = json.loads(varieties_path.read_text(encoding="utf-8")) - except (ValueError, OSError): - return 0 - try: - manifest = json.loads(manifest_path.read_text(encoding="utf-8")) if manifest_path.exists() else {} - except (ValueError, OSError): - manifest = {} - src_meta = (manifest.get("sources") or {}).get("vyhlaska-88-2017") or {} - - # Build a slug → match details index for the lexicon match. - from _lib.grape_entity import match_variety, set_pliego_context # noqa: E402 - sidecar_details: list[dict] = [] - set_pliego_context("vyhlaska-88-2017") - for v in spec.get("varieties") or []: - m = match_variety(v.get("name") or "") - if m is None: - continue - sidecar_details.append({ - "slug": m.slug, - "name": v.get("name"), - "role": "principal", - "colour": m.colour or _COLOUR_FROM_CZ_BLOCK.get(v.get("colour", ""), ""), - "source": "vyhlaska-88-2017", - }) - set_pliego_context(None) - - provenance_base = { - "url": src_meta.get("canonical_url") or src_meta.get("fetch_url") or "", - "fetch_url": src_meta.get("fetch_url") or "", - "title": src_meta.get("title") or "", - "sbirka_castka": src_meta.get("sbirka_castka") or "", - "sha256": src_meta.get("sha256") or "", - "fetched_at": (src_meta.get("fetched_at") or "") - if isinstance(src_meta, dict) else "", - "source_org": "sbirka", - "license": "Czech law text per §3(d) of the Czech Copyright Act", - "n_varieties": spec.get("n_total") or 0, - "n_white": spec.get("n_white") or 0, - "n_red": spec.get("n_red") or 0, - "n_zemske": spec.get("n_zemske") or 0, - } - - augmented = 0 - for record in records: - if record.get("country") != "cz": - continue - slug = record.get("slug") or "" - # Build the per-record details list. Start from the sidecar - # (the national variety roster) since the EU-OJ extracted record - # has no grapes (every CZ wine is a stub in v1). Folding by - # role: everything `principal` because the Vyhláška doesn't - # split. - grapes = dict(record.get("grapes") or {}) - existing_slugs = {d.get("slug") for d in (grapes.get("details") or []) if d.get("slug")} - new_details = list(grapes.get("details") or []) - for d in sidecar_details: - if d["slug"] in existing_slugs: - continue - existing_slugs.add(d["slug"]) - new_details.append(d) - principal = [d["slug"] for d in new_details if d["slug"] and d.get("role") != "accessory"] - accessory = [d["slug"] for d in new_details if d["slug"] and d.get("role") == "accessory"] - # Dedup while preserving order. - principal_seen: set[str] = set() - principal_ordered = [s for s in principal if not (s in principal_seen or principal_seen.add(s))] - accessory_seen: set[str] = set() - accessory_ordered = [s for s in accessory if not (s in accessory_seen or accessory_seen.add(s))] - grapes["principal"] = principal_ordered - grapes["accessory"] = accessory_ordered - grapes["details"] = new_details - record["grapes"] = grapes - # Czech wine law publishes no per-appellation wine-description - # section, so styles can't be read from a spec the way HR/SI/BG - # do. Infer the base colour styles from the authorised variety - # roster instead (the BE colour-distribution fallback): a - # blanc/gris variety authorises white, a noir variety authorises - # red + rosé. Every CZ wine carries the national roster, so all - # carry white/red/rosé — honest (any CZ jakostní víno appellation - # may be made in any colour) and it makes CZ wines findable in the - # style facet instead of invisible. The single straw-wine PDO - # (Novosedelské Slámové víno) additionally carries vin-de-paille, - # evident from its own name. - colours = {d.get("colour") for d in new_details if d.get("colour")} - styles = set(record.get("styles") or []) - if colours & {"blanc", "gris"}: - styles.add("white") - if "noir" in colours: - styles.add("red") - styles.add("rose") - if slug == "novosedelske-slamove-vino": - styles.add("vin-de-paille") - # The 2 PGIs ("zemské víno") additionally carry the real style - # roster from their SZPI CHZO spec section 2 (sparkling / - # semi-sparkling / vin-de-liqueur on top of the colour bases). - chzo = chzo_by_region.get(record.get("region") or "") - if chzo and slug in chzo_pgi_slugs: - styles |= set(chzo.get("styles") or []) - record["styles"] = sorted(styles) - # All CZ wines cite the region's CHZO spec as their terroir - # source (02d grounds on its section-1 region description), so - # surface its provenance uniformly for the panel source block. - if chzo: - chzo_prov = { - "url": chzo.get("source_url") or "", - "title": chzo.get("source_title") or "", - "region": chzo.get("region") or "", - "source_org": chzo.get("source_org") or "szpi", - "sha256": chzo.get("source_sha256") or "", - "parser_template": chzo.get("parser_template") or "", - } - record["chzo_spec"] = chzo_prov - _CZ_CHZO_BY_SLUG[slug] = chzo_prov - record["national_spec"] = provenance_base - _CZ_NATIONAL_SPEC_BY_SLUG[slug] = provenance_base - augmented += 1 - return augmented - - -# CZ variety-block colour → grape-entity colour mapping. The Vyhláška -# 88/2017 block headers are "Bílé moštové odrůdy" (blanc) / "Modré -# moštové odrůdy" (noir) / "Odrůdy pro výrobu zemských vín" (mixed -# colours, kept as `zemske` here — falls back via match_variety()). -_COLOUR_FROM_CZ_BLOCK: dict[str, str] = { - "blanc": "blanc", - "noir": "noir", - "zemske": "", # mixed; let match_variety supply the per-variety colour -} # Simple-mode style buckets: collapses the fine-grained style tags into the diff --git a/scripts/_lib/augment/_shared.py b/scripts/_lib/augment/_shared.py index 0aa70db..5fff533 100644 --- a/scripts/_lib/augment/_shared.py +++ b/scripts/_lib/augment/_shared.py @@ -46,6 +46,10 @@ # bypassing in-memory augmentation). _IT_MASAF_BY_SLUG: dict[str, dict] = {} +# Slug-keyed cache of IT regional-variety-register provenance, populated by +# augment_it_records_with_regional_registers() and read by _sources_for(). +_IT_REGISTER_BY_SLUG: dict[str, dict] = {} + # Slug-keyed cache of CZ national-spec provenance, populated by # augment_cz_records_with_national_specs(). Mirrors the ES/IT/DE caches. # Czech wine law publishes one national variety roster (Vyhláška 88/2017 diff --git a/scripts/_lib/augment/bg.py b/scripts/_lib/augment/bg.py new file mode 100644 index 0000000..1e5348a --- /dev/null +++ b/scripts/_lib/augment/bg.py @@ -0,0 +1,88 @@ +"""BG national-spec (ИАЛВ продуктова спецификация) (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. The shared +provenance cache + sidecar dir live in `_shared` (same objects as the +`_sources_for()` reader in stage 04). +""" +from __future__ import annotations + +import json + +from ._shared import _BG_NATIONAL_SPEC_BY_SLUG, NATIONAL_SPECS_BG + + +def augment_bg_records_with_national_specs(records: list[dict]) -> int: + """In-place merge of BG national-spec sidecar data into stub records. + + Sibling of `augment_gr_records_with_national_specs`. 51 of 54 BG wines + ship as content-stubs (no fetchable EU-OJ ЕДИНЕН ДОКУМЕНТ). Stage 02f + (`scripts/bg/02f_extract_national_specs.py`) parses the ИАЛВ / IAVV + per-wine продуктова спецификация PDF fetched by stage 01c into + `raw/bg/national-specs-extracted/.json` (51 of 51). + + For each BG stub with a matching sidecar: + - grapes ← section 5 (Винени сортове грозде, colour-split) + - link_to_terroir ← section 6 (Връзка с географския район) + - geo_area_brief / summary / styles ← matching sections + - section_roles ← unified role dict so 02d reads terroir uniformly + - stub_reason ← prefixed `national-spec:` so the audit can tell + EU-OJ-extracted from spec-augmented wines + - national_spec ← provenance block (url, sha256, format, …) + + `record["stub"]` stays True — still NOT an EU-OJ extraction, just + augmented with the canonical ИАЛВ source. Returns count augmented. + """ + _BG_NATIONAL_SPEC_BY_SLUG.clear() + if not NATIONAL_SPECS_BG.exists(): + return 0 + augmented = 0 + for record in records: + if record.get("country") != "bg" or not record.get("stub"): + continue + slug = record.get("slug") + if not slug: + continue + sidecar_path = NATIONAL_SPECS_BG / f"{slug}.json" + if not sidecar_path.exists(): + continue + try: + sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + + src = sidecar.get("source") or {} + provenance = { + "url": src.get("url") or "", + "sha256": src.get("sha256") or "", + "fetched_at": src.get("fetched_at") or "", + "format": src.get("format") or "", + "source_org": src.get("source_org") or "iavv", + "filename": src.get("filename") or "", + "parser_template": sidecar.get("parser_template") or "", + } + + if sidecar.get("summary"): + record["summary"] = sidecar["summary"] + if sidecar.get("grapes") and (sidecar["grapes"].get("principal") + or sidecar["grapes"].get("accessory")): + record["grapes"] = sidecar["grapes"] + if sidecar.get("geo_area_brief"): + record["geo_area_brief"] = sidecar["geo_area_brief"] + if sidecar.get("link_to_terroir"): + record["link_to_terroir"] = sidecar["link_to_terroir"] + if sidecar.get("styles"): + record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) + + section_roles = dict(record.get("section_roles") or {}) + for role in ("description", "geo_area", "grape_varieties", "link_to_terroir"): + sidecar_roles = sidecar.get("section_roles") or {} + if sidecar_roles.get(role): + section_roles[role] = sidecar_roles[role] + record["section_roles"] = section_roles + + if record.get("stub_reason") and not record["stub_reason"].startswith("national-spec:"): + record["stub_reason"] = f"national-spec:{record['stub_reason']}" + record["national_spec"] = provenance + _BG_NATIONAL_SPEC_BY_SLUG[slug] = provenance + augmented += 1 + return augmented diff --git a/scripts/_lib/augment/cy.py b/scripts/_lib/augment/cy.py new file mode 100644 index 0000000..731c2c3 --- /dev/null +++ b/scripts/_lib/augment/cy.py @@ -0,0 +1,78 @@ +"""CY national-spec (moa.gov.cy τεχνικός φάκελος) (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. The shared +provenance cache + sidecar dir live in `_shared` (same objects as the +`_sources_for()` reader in stage 04). +""" +from __future__ import annotations + +import json + +from ._shared import _CY_NATIONAL_SPEC_BY_SLUG, NATIONAL_SPECS_CY + + +def augment_cy_records_with_national_specs(records: list[dict]) -> int: + """In-place merge of CY national-spec sidecar data into stub records. + + Sibling of `augment_gr_records_with_national_specs`. All 11 CY wines + ship as content-stubs (no fetchable EU-OJ ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ). Stage 02f + (`scripts/cy/02f_extract_national_specs.py`) parses the moa.gov.cy + Department-of-Agriculture τεχνικός φάκελος (Greek single-document + PDF, OCR'd when image-only) into `raw/cy/national-specs-extracted/ + .json`; this merges grapes / terroir text / styles / geo-area + into the in-memory stub. `record["stub"]` stays True. Returns the + count augmented.""" + _CY_NATIONAL_SPEC_BY_SLUG.clear() + if not NATIONAL_SPECS_CY.exists(): + return 0 + augmented = 0 + for record in records: + if record.get("country") != "cy" or not record.get("stub"): + continue + slug = record.get("slug") + if not slug: + continue + sidecar_path = NATIONAL_SPECS_CY / f"{slug}.json" + if not sidecar_path.exists(): + continue + try: + sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + + src = sidecar.get("source") or {} + provenance = { + "url": src.get("source_url") or "", + "sha256": src.get("sha256") or "", + "fetched_at": src.get("fetched_at") or "", + "format": src.get("format") or "", + "source_org": src.get("source_org") or "moa-cy", + "filename": src.get("filename") or "", + "parser_template": sidecar.get("parser_template") or "", + } + + if sidecar.get("summary"): + record["summary"] = sidecar["summary"] + if sidecar.get("grapes") and (sidecar["grapes"].get("principal") + or sidecar["grapes"].get("accessory")): + record["grapes"] = sidecar["grapes"] + if sidecar.get("geo_area_brief"): + record["geo_area_brief"] = sidecar["geo_area_brief"] + if sidecar.get("link_to_terroir"): + record["link_to_terroir"] = sidecar["link_to_terroir"] + if sidecar.get("styles"): + record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) + + section_roles = dict(record.get("section_roles") or {}) + for role in ("description", "geo_area", "grape_varieties", "link_to_terroir"): + sidecar_roles = sidecar.get("section_roles") or {} + if sidecar_roles.get(role): + section_roles[role] = sidecar_roles[role] + record["section_roles"] = section_roles + + if record.get("stub_reason") and not record["stub_reason"].startswith("national-spec:"): + record["stub_reason"] = f"national-spec:{record['stub_reason']}" + record["national_spec"] = provenance + _CY_NATIONAL_SPEC_BY_SLUG[slug] = provenance + augmented += 1 + return augmented diff --git a/scripts/_lib/augment/cz.py b/scripts/_lib/augment/cz.py new file mode 100644 index 0000000..3d5f9dd --- /dev/null +++ b/scripts/_lib/augment/cz.py @@ -0,0 +1,200 @@ +"""CZ national-spec + CHZO augmentation (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. The shared +provenance caches + sidecar dir live in `_shared` (same objects as the +`_sources_for()` reader in stage 04). The CZ colour-block map and the +in-function grape-lexicon import move with the function. +""" +from __future__ import annotations + +import json + +from ._shared import _CZ_CHZO_BY_SLUG, _CZ_NATIONAL_SPEC_BY_SLUG, NATIONAL_SPECS_CZ + +_COLOUR_FROM_CZ_BLOCK: dict[str, str] = { + "blanc": "blanc", + "noir": "noir", + "zemske": "", # mixed; let match_variety supply the per-variety colour +} + + +def augment_cz_records_with_national_specs(records: list[dict]) -> int: + """In-place merge of the Czech national variety roster (Vyhláška + č. 88/2017 Sb. Příloha č. 2) into every CZ wine record. + + Czech wine law publishes one national variety list (35 white + 26 + red + 6 zemské-víno = 67 varieties) that applies to every jakostní + víno regardless of podoblast — no per-appellation restriction. So + every CZ wine that *should* carry a variety list (10 of 13, all + except 3 newer single-vineyard / single-varietal PDOs whose + Vyhláška-88 status is undocumented) gets the same fold: + + - white-wine PDOs/PGIs → all 35 white varieties as `principal` + - red-wine PDOs/PGIs → all 26 red varieties as `principal` + - mixed (most macros + podoblasti) → both lists folded; + zemské-víno-only varieties go under `accessory` (they apply + only to the lower zemské-víno PGI tier). + + Since the Vyhláška doesn't enumerate a per-appellation principal/ + accessory split (it's a flat national authorisation), we mark + everything `principal` and let the panel render "All Czech + jakostní vína authorise these 67 varieties — see Vyhláška č. + 88/2017 Sb." in the provenance. + + Stage 04 reads the cached provenance later via + `_CZ_NATIONAL_SPEC_BY_SLUG`. + + Returns the number of records augmented. + """ + _CZ_NATIONAL_SPEC_BY_SLUG.clear() + _CZ_CHZO_BY_SLUG.clear() + if not NATIONAL_SPECS_CZ.exists(): + return 0 + + # Load the two SZPI CHZO region specs (terroir source + style roster + + # provenance), keyed by region. Both PGIs are the spec's own subject; + # the macro CHOPs + podoblasti in that region share its section-1 + # terroir description (rendered by 02d) and cite the same SZPI PDF. + chzo_by_region: dict[str, dict] = {} + for key in ("chzo-moravske", "chzo-ceske"): + p = NATIONAL_SPECS_CZ / f"{key}.json" + if not p.exists(): + continue + try: + d = json.loads(p.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + if d.get("region"): + chzo_by_region[d["region"]] = d + # The 2 PGI slugs whose own product specification this is (they + # inherit the spec's per-style roster; the CHOPs/podoblasti keep + # grape-colour-inferred styles only). + chzo_pgi_slugs = {"moravske", "ceske"} + + varieties_path = NATIONAL_SPECS_CZ / "varieties.json" + manifest_path = NATIONAL_SPECS_CZ / "manifest.json" + if not varieties_path.exists(): + return 0 + try: + spec = json.loads(varieties_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + return 0 + try: + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) if manifest_path.exists() else {} + except (ValueError, OSError): + manifest = {} + src_meta = (manifest.get("sources") or {}).get("vyhlaska-88-2017") or {} + + # Build a slug → match details index for the lexicon match. + from _lib.grape_entity import match_variety, set_pliego_context # noqa: E402 + sidecar_details: list[dict] = [] + set_pliego_context("vyhlaska-88-2017") + for v in spec.get("varieties") or []: + m = match_variety(v.get("name") or "") + if m is None: + continue + sidecar_details.append({ + "slug": m.slug, + "name": v.get("name"), + "role": "principal", + "colour": m.colour or _COLOUR_FROM_CZ_BLOCK.get(v.get("colour", ""), ""), + "source": "vyhlaska-88-2017", + }) + set_pliego_context(None) + + provenance_base = { + "url": src_meta.get("canonical_url") or src_meta.get("fetch_url") or "", + "fetch_url": src_meta.get("fetch_url") or "", + "title": src_meta.get("title") or "", + "sbirka_castka": src_meta.get("sbirka_castka") or "", + "sha256": src_meta.get("sha256") or "", + "fetched_at": (src_meta.get("fetched_at") or "") + if isinstance(src_meta, dict) else "", + "source_org": "sbirka", + "license": "Czech law text per §3(d) of the Czech Copyright Act", + "n_varieties": spec.get("n_total") or 0, + "n_white": spec.get("n_white") or 0, + "n_red": spec.get("n_red") or 0, + "n_zemske": spec.get("n_zemske") or 0, + } + + augmented = 0 + for record in records: + if record.get("country") != "cz": + continue + slug = record.get("slug") or "" + # Build the per-record details list. Start from the sidecar + # (the national variety roster) since the EU-OJ extracted record + # has no grapes (every CZ wine is a stub in v1). Folding by + # role: everything `principal` because the Vyhláška doesn't + # split. + grapes = dict(record.get("grapes") or {}) + existing_slugs = {d.get("slug") for d in (grapes.get("details") or []) if d.get("slug")} + new_details = list(grapes.get("details") or []) + for d in sidecar_details: + if d["slug"] in existing_slugs: + continue + existing_slugs.add(d["slug"]) + new_details.append(d) + principal = [d["slug"] for d in new_details if d["slug"] and d.get("role") != "accessory"] + accessory = [d["slug"] for d in new_details if d["slug"] and d.get("role") == "accessory"] + # Dedup while preserving order. + principal_seen: set[str] = set() + principal_ordered = [s for s in principal if not (s in principal_seen or principal_seen.add(s))] + accessory_seen: set[str] = set() + accessory_ordered = [s for s in accessory if not (s in accessory_seen or accessory_seen.add(s))] + grapes["principal"] = principal_ordered + grapes["accessory"] = accessory_ordered + grapes["details"] = new_details + record["grapes"] = grapes + # Czech wine law publishes no per-appellation wine-description + # section, so styles can't be read from a spec the way HR/SI/BG + # do. Infer the base colour styles from the authorised variety + # roster instead (the BE colour-distribution fallback): a + # blanc/gris variety authorises white, a noir variety authorises + # red + rosé. Every CZ wine carries the national roster, so all + # carry white/red/rosé — honest (any CZ jakostní víno appellation + # may be made in any colour) and it makes CZ wines findable in the + # style facet instead of invisible. The single straw-wine PDO + # (Novosedelské Slámové víno) additionally carries vin-de-paille, + # evident from its own name. + colours = {d.get("colour") for d in new_details if d.get("colour")} + styles = set(record.get("styles") or []) + if colours & {"blanc", "gris"}: + styles.add("white") + if "noir" in colours: + styles.add("red") + styles.add("rose") + if slug == "novosedelske-slamove-vino": + styles.add("vin-de-paille") + # The 2 PGIs ("zemské víno") additionally carry the real style + # roster from their SZPI CHZO spec section 2 (sparkling / + # semi-sparkling / vin-de-liqueur on top of the colour bases). + chzo = chzo_by_region.get(record.get("region") or "") + if chzo and slug in chzo_pgi_slugs: + styles |= set(chzo.get("styles") or []) + record["styles"] = sorted(styles) + # All CZ wines cite the region's CHZO spec as their terroir + # source (02d grounds on its section-1 region description), so + # surface its provenance uniformly for the panel source block. + if chzo: + chzo_prov = { + "url": chzo.get("source_url") or "", + "title": chzo.get("source_title") or "", + "region": chzo.get("region") or "", + "source_org": chzo.get("source_org") or "szpi", + "sha256": chzo.get("source_sha256") or "", + "parser_template": chzo.get("parser_template") or "", + } + record["chzo_spec"] = chzo_prov + _CZ_CHZO_BY_SLUG[slug] = chzo_prov + record["national_spec"] = provenance_base + _CZ_NATIONAL_SPEC_BY_SLUG[slug] = provenance_base + augmented += 1 + return augmented + + +# CZ variety-block colour → grape-entity colour mapping. The Vyhláška +# 88/2017 block headers are "Bílé moštové odrůdy" (blanc) / "Modré +# moštové odrůdy" (noir) / "Odrůdy pro výrobu zemských vín" (mixed +# colours, kept as `zemske` here — falls back via match_variety()). diff --git a/scripts/_lib/augment/de.py b/scripts/_lib/augment/de.py new file mode 100644 index 0000000..61ef544 --- /dev/null +++ b/scripts/_lib/augment/de.py @@ -0,0 +1,187 @@ +"""DE BLE Produktspezifikation variety role split + terroir (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. The shared +provenance cache + sidecar dir live in `_shared` (same objects as the +`_sources_for()` reader in stage 04). +""" +from __future__ import annotations + +import json + +from ._shared import _DE_PRODUKTSPEZIFIKATION_BY_SLUG, PRODUKTSPEZIFIKATION_DE + + +def augment_de_records_with_produktspezifikation(records: list[dict]) -> int: + """In-place merge of BLE-Produktspezifikation sidecar data into DE + parent-Anbaugebiet records. + + The EU Einziges Dokument for German wines doesn't carry a principal/ + accessory split — section 7 is a flat list. The BLE national + Produktspezifikation (Amtliches Werk §5 UrhG) names individual + varieties with their own Mindestmostgewicht threshold in §3.2 (Mosel + → Riesling/Elbling/Müller-Thurgau/Dornfelder). Stage 02f extracts + that split into raw/de/produktspezifikationen-extracted/.json; + this augment re-tags the in-memory record's grapes block accordingly. + + Two sidecar categories are merged (both written by stage 02f): + - the 13 Anbaugebiete (regional PDOs), with a principal/accessory + split from §3.2 Mindestmostgewicht; and + - the 15 Landwein g.g.A. that ship as stubs (no EU Einziges + Dokument). Their BLE spec has no role split, so they arrive as + `section-8-flat-no-split` (all-principal) and fold their full + roster + §-Zusammenhang terroir text into the stub record. + Einzellage sub-denominations are still skipped (they inherit the + parent Anbaugebiet's varieties at render time). + + For records with `role_split_method == "section-3.2-principal"`: + - re-tag existing record["grapes"]["details"] items as + principal/accessory based on the sidecar's slug sets + - rebuild record["grapes"]["principal"] / ["accessory"] lists + - fold any new sidecar slugs not already in the EU record + + For records with `role_split_method == "section-8-flat-no-split"` + (Anbaugebiete whose §3.2 doesn't enumerate per-variety thresholds — + Franken, Württemberg in v1): the existing all-principal default + stands; the sidecar just records the BLE source as provenance. + + Stage 04 reads the cached provenance later in the AOC-blob phase + via `_DE_PRODUKTSPEZIFIKATION_BY_SLUG`. + + Returns the number of records augmented. + """ + _DE_PRODUKTSPEZIFIKATION_BY_SLUG.clear() + if not PRODUKTSPEZIFIKATION_DE.exists(): + return 0 + augmented = 0 + for record in records: + if record.get("country") != "de": + continue + if record.get("is_sub_denomination"): + continue + slug = record.get("slug") or "" + sidecar_path = PRODUKTSPEZIFIKATION_DE / f"{slug}.json" + if not sidecar_path.exists(): + continue + try: + sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + + method = sidecar.get("role_split_method") or "" + src = sidecar.get("source") or {} + provenance = { + "url": src.get("url") or "", + "sha256": src.get("sha256") or "", + "bytes": src.get("bytes") or 0, + "fetched_at": src.get("fetched_at") or "", + "source_org": src.get("source_org") or "BLE", + "license": src.get("license") or "Amtliches Werk §5 UrhG", + "role_split_method": method, + "n_principal": sidecar.get("n_principal") or 0, + "n_accessory": sidecar.get("n_accessory") or 0, + } + + # BLE §8/§9 "Zusammenhang" terroir backfill — when the EU + # Einziges Dokument is sparse (stub or no link_to_terroir), use + # the BLE PDF's terroir block so 02d can extract facts from it. + # Affects Ahr, Baden, Hessische Bergstraße, Rheingau, Sachsen, + # Saale-Unstrut in v1. We also record the BLE PDF URL as the + # cahier_source for downstream provenance. + bz = (sidecar.get("zusammenhang_text") or "").strip() + eu_terroir = (record.get("link_to_terroir") or "").strip() + if bz and len(eu_terroir) < 400: + record["link_to_terroir"] = bz + section_roles = dict(record.get("section_roles") or {}) + section_roles["link_to_terroir"] = bz + record["section_roles"] = section_roles + # Surface the BLE PDF as the canonical terroir source so the + # panel + 02d attribute it correctly. + rec_src = dict(record.get("source") or {}) + rec_src["terroir_source_url"] = src.get("url") or "" + rec_src["terroir_source_org"] = "BLE" + record["source"] = rec_src + provenance["terroir_backfilled"] = True + + # For both role-split methods, fold the sidecar's variety roster + # into the in-memory record. The default fold tags non-sidecar + # EU slugs as `accessory` (the BLE PDF's §3.2 is the principal + # allowlist); the flat-no-split method instead tags everything + # as principal because the document doesn't enumerate a split. + if method in ("section-3.2-principal", "section-8-flat-no-split"): + # Build slug → role from the sidecar. + sidecar_role: dict[str, str] = {} + sidecar_details: list[dict] = [] + for g in sidecar.get("grapes") or []: + s = g.get("slug") + if s: + sidecar_role[s] = g.get("role") or "principal" + sidecar_details.append(g) + + # Re-tag the EU-record's existing details. Keep the EU + # record's display name (matches the wine's own + # Einziges-Dokument spelling). When the sidecar has an + # authoritative §3.2 split, slugs NOT named in the sidecar + # default to "accessory" — the BLE PDF's §3.2 is the + # principal-allowlist, so anything outside it is + # implicitly "alle übrigen Rebsorten". When the sidecar is + # flat-no-split, every slug is principal (the regulator + # didn't enumerate a split). + unmatched_default = ( + "principal" if method == "section-8-flat-no-split" else "accessory" + ) + grapes = dict(record.get("grapes") or {}) + details_in = grapes.get("details") or [] + new_details: list[dict] = [] + eu_slugs: set[str] = set() + for d in details_in: + new_d = dict(d) + s = new_d.get("slug") + if s: + new_d["role"] = sidecar_role.get(s, unmatched_default) + new_details.append(new_d) + if s: + eu_slugs.add(s) + + # Fold in any sidecar varieties that the EU record missed + # (the §8 list is more complete than the EU section 7 for + # some Anbaugebiete). + for g in sidecar_details: + s = g.get("slug") + if s and s not in eu_slugs: + new_details.append({ + "slug": s, + "name": g.get("name", s), + "role": g.get("role", "accessory"), + "colour": g.get("colour", ""), + "source": "ble-produktspezifikation", + }) + + # Rebuild principal / accessory lists from the re-tagged + # details (deterministic + dedup-by-first-seen). + principal: list[str] = [] + accessory: list[str] = [] + seen_p: set[str] = set() + seen_a: set[str] = set() + for d in new_details: + s = d.get("slug") + if not s: + continue + if d.get("role") == "accessory": + if s in seen_a: + continue + seen_a.add(s) + accessory.append(s) + else: + if s in seen_p: + continue + seen_p.add(s) + principal.append(s) + grapes["principal"] = principal + grapes["accessory"] = accessory + grapes["details"] = new_details + record["grapes"] = grapes + + record["produktspezifikation"] = provenance + _DE_PRODUKTSPEZIFIKATION_BY_SLUG[slug] = provenance + augmented += 1 + return augmented diff --git a/scripts/_lib/augment/gr.py b/scripts/_lib/augment/gr.py new file mode 100644 index 0000000..5d0175c --- /dev/null +++ b/scripts/_lib/augment/gr.py @@ -0,0 +1,90 @@ +"""GR national-spec (ΥΠΑΑΤ προδιαγραφή / τεχνικός φάκελος) (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. The shared +provenance cache + sidecar dir live in `_shared` (same objects as the +`_sources_for()` reader in stage 04). +""" +from __future__ import annotations + +import json + +from ._shared import _GR_NATIONAL_SPEC_BY_SLUG, NATIONAL_SPECS_GR + + +def augment_gr_records_with_national_specs(records: list[dict]) -> int: + """In-place merge of GR national-spec sidecar data into stub records. + + Sibling of `augment_si_records_with_specifikacija`. 138 of 147 GR + wines ship as content-stubs (no fetchable EU-OJ ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ). + Stage 02f (`scripts/gr/02f_extract_national_specs.py`) parses the + ΥΠΑΑΤ national προδιαγραφή / τεχνικός φάκελος fetched by stage 01c + into `raw/gr/national-specs-extracted/.json` (132 of 138; the + other 6 are unresolved — see CURATOR_TODO.md). + + For each GR stub with a matching sidecar: + - grapes ← §6 ΟΙΝΟΠΟΙΗΣΙΜΕΣ ΠΟΙΚΙΛΙΕΣ (PDF list) or the + grape section's capitalised-token scan (.doc prose) + - link_to_terroir ← §7 ΔΕΣΜΟΣ ΜΕ ΤΗΝ ΓΕΩΓΡΑΦΙΚΗ ΠΕΡΙΟΧΗ + - geo_area_brief / summary / styles ← matching sections + - section_roles ← unified role dict so 02d reads terroir uniformly + - stub_reason ← prefixed `national-spec:` so the audit can tell + EU-OJ-extracted from spec-augmented wines + - national_spec ← provenance block (url, sha256, format, …) + + `record["stub"]` stays True — still NOT an EU-OJ extraction, just + augmented with the canonical ΥΠΑΑΤ source. Returns count augmented. + """ + _GR_NATIONAL_SPEC_BY_SLUG.clear() + if not NATIONAL_SPECS_GR.exists(): + return 0 + augmented = 0 + for record in records: + if record.get("country") != "gr" or not record.get("stub"): + continue + slug = record.get("slug") + if not slug: + continue + sidecar_path = NATIONAL_SPECS_GR / f"{slug}.json" + if not sidecar_path.exists(): + continue + try: + sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + + src = sidecar.get("source") or {} + provenance = { + "url": src.get("source_url") or "", + "sha256": src.get("sha256") or "", + "fetched_at": src.get("fetched_at") or "", + "format": src.get("format") or "", + "source_org": src.get("source_org") or "ypaat", + "filename": src.get("filename") or "", + "parser_template": sidecar.get("parser_template") or "", + } + + if sidecar.get("summary"): + record["summary"] = sidecar["summary"] + if sidecar.get("grapes") and (sidecar["grapes"].get("principal") + or sidecar["grapes"].get("accessory")): + record["grapes"] = sidecar["grapes"] + if sidecar.get("geo_area_brief"): + record["geo_area_brief"] = sidecar["geo_area_brief"] + if sidecar.get("link_to_terroir"): + record["link_to_terroir"] = sidecar["link_to_terroir"] + if sidecar.get("styles"): + record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) + + section_roles = dict(record.get("section_roles") or {}) + for role in ("description", "geo_area", "grape_varieties", "link_to_terroir"): + sidecar_roles = sidecar.get("section_roles") or {} + if sidecar_roles.get(role): + section_roles[role] = sidecar_roles[role] + record["section_roles"] = section_roles + + if record.get("stub_reason") and not record["stub_reason"].startswith("national-spec:"): + record["stub_reason"] = f"national-spec:{record['stub_reason']}" + record["national_spec"] = provenance + _GR_NATIONAL_SPEC_BY_SLUG[slug] = provenance + augmented += 1 + return augmented diff --git a/scripts/_lib/augment/hu.py b/scripts/_lib/augment/hu.py new file mode 100644 index 0000000..1cc2639 --- /dev/null +++ b/scripts/_lib/augment/hu.py @@ -0,0 +1,101 @@ +"""HU national-spec (Agrárminisztérium termékleírás) (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. The shared +provenance cache + sidecar dir live in `_shared` (same objects as the +`_sources_for()` reader in stage 04). +""" +from __future__ import annotations + +import json + +from ._shared import _HU_NATIONAL_SPEC_BY_SLUG, NATIONAL_SPECS_HU + + +def augment_hu_records_with_national_specs(records: list[dict]) -> int: + """In-place merge of HU national-spec sidecar data into stub records. + + Sibling of `augment_ro_records_with_national_specs`. The 15 + grandfathered HU wines (eAmbrosia carries only a non-fetchable + `Ares(...)` reference — no EU-OJ EGYSÉGES DOKUMENTUM) ship as + content-stubs. Stage 02f (`scripts/hu/02f_extract_national_specs.py`) + parses the Agrárminisztérium termékleírás PDF fetched by stage 01c + into `raw/hu/national-specs-extracted/.json`. + + For each HU stub with a matching sidecar: + - grapes ← VI. ENGEDÉLYEZETT SZŐLŐFAJTÁK + - link_to_terroir ← VII. KAPCSOLAT A FÖLDRAJZI TERÜLETTEL + - geo_communes ← IV. KÖRÜLHATÁROLT TERÜLET (commune-precision; + geometry still prefers the Bétard polygon + these wines already have, so this is a record) + - geo_area_brief / summary / styles ← matching sections + - section_roles ← unified role dict so 02d reads terroir uniformly + - stub_reason ← prefixed `national-spec:` + - national_spec ← provenance block (url, sha256, format, …) + + `record["stub"]` stays True. Returns count augmented. + """ + _HU_NATIONAL_SPEC_BY_SLUG.clear() + if not NATIONAL_SPECS_HU.exists(): + return 0 + augmented = 0 + for record in records: + if record.get("country") != "hu": + continue + slug = record.get("slug") + if not slug: + continue + sidecar_path = NATIONAL_SPECS_HU / f"{slug}.json" + if not sidecar_path.exists(): + continue + try: + sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + + src = sidecar.get("source") or {} + provenance = { + "url": src.get("source_url") or "", + "sha256": src.get("sha256") or "", + "fetched_at": src.get("fetched_at") or "", + "format": src.get("format") or "", + "source_org": src.get("source_org") or "agrarminiszterium", + "filename": src.get("filename") or "", + "parser_template": sidecar.get("parser_template") or "", + } + + # Fill-if-empty: a stub is fully empty so this fills everything; + # a non-stub with a thin EU extraction (e.g. Badacsony, whose + # awkward doc structure left the grape section unrouted) gets only + # its EMPTY fields filled — good EUR-Lex data is never clobbered. + cur_grapes = record.get("grapes") or {} + if (sidecar.get("summary") and not record.get("summary")): + record["summary"] = sidecar["summary"] + if (sidecar.get("grapes") + and (sidecar["grapes"].get("principal") or sidecar["grapes"].get("accessory")) + and not (cur_grapes.get("principal") or cur_grapes.get("accessory"))): + record["grapes"] = sidecar["grapes"] + if sidecar.get("geo_area_brief") and not record.get("geo_area_brief"): + record["geo_area_brief"] = sidecar["geo_area_brief"] + if sidecar.get("geo_communes") and not record.get("geo_communes"): + record["geo_communes"] = sidecar["geo_communes"] + if sidecar.get("dulok") and not record.get("dulok"): + record["dulok"] = sidecar["dulok"] + if sidecar.get("link_to_terroir") and not record.get("link_to_terroir"): + record["link_to_terroir"] = sidecar["link_to_terroir"] + if sidecar.get("styles"): + record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) + + section_roles = dict(record.get("section_roles") or {}) + sidecar_roles = sidecar.get("section_roles") or {} + for role in ("geo_area", "grape_varieties", "link_to_terroir"): + if sidecar_roles.get(role) and not section_roles.get(role): + section_roles[role] = sidecar_roles[role] + record["section_roles"] = section_roles + + if record.get("stub") and record.get("stub_reason") \ + and not record["stub_reason"].startswith("national-spec:"): + record["stub_reason"] = f"national-spec:{record['stub_reason']}" + record["national_spec"] = provenance + _HU_NATIONAL_SPEC_BY_SLUG[slug] = provenance + augmented += 1 + return augmented diff --git a/scripts/_lib/augment/it.py b/scripts/_lib/augment/it.py new file mode 100644 index 0000000..5efb239 --- /dev/null +++ b/scripts/_lib/augment/it.py @@ -0,0 +1,283 @@ +"""IT MASAF disciplinare + regional-register + sottozona augmentation (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. The shared +provenance caches + sidecar dirs live in `_shared` (same objects as the +`_sources_for()` reader in stage 04). `_backfill_it_nonstub_from_masaf` is a +private helper of `augment_it_records_with_masaf` and moves with it; +`synthesize_it_sottozone_records` keeps its call order in stage 04 main(). +""" +from __future__ import annotations + +import json + +from _lib.it.sottozona import extract_sottozone as extract_it_sottozone + +from ._shared import ( + _IT_MASAF_BY_SLUG, + _IT_REGISTER_BY_SLUG, + IT_REGIONAL_REGISTERS, + MASAF_DISCIPLINARI_IT, +) + + +def _backfill_it_nonstub_from_masaf(record: dict, sidecar: dict) -> bool: + """Fill ONLY the empty fields of a non-stub IT record from its MASAF + sidecar — the documento unico is canonical, but some OJ docs omit the + geo area or variety list, and the national disciplinare carries them. + Never overwrites populated docunico data. Returns True if anything + was filled.""" + filled = False + g = record.get("grapes") or {} + if sidecar.get("grapes") and not (g.get("principal") or g.get("accessory")): + record["grapes"] = sidecar["grapes"] + filled = True + if sidecar.get("menzioni") and not record.get("menzioni"): + record["menzioni"] = sidecar["menzioni"] + filled = True + section_roles = dict(record.get("section_roles") or {}) + if sidecar.get("geo_area_brief") and not (record.get("geo_area_brief") or "").strip(): + record["geo_area_brief"] = sidecar["geo_area_brief"] + section_roles["geo_area"] = sidecar["geo_area_brief"] + filled = True + if sidecar.get("link_to_terroir") and not (record.get("link_to_terroir") or "").strip(): + record["link_to_terroir"] = sidecar["link_to_terroir"] + section_roles["link_to_terroir"] = sidecar["link_to_terroir"] + filled = True + if filled: + record["section_roles"] = section_roles + record["masaf_backfill"] = True + return filled + + +def augment_it_records_with_masaf(records: list[dict]) -> int: + """In-place merge of MASAF disciplinare sidecar data into IT stub + records. Only stubs are touched — wines whose documento unico was + extracted in stage 02 already carry canonical EUR-Lex data and + shouldn't be overwritten. + + For each IT stub with a matching sidecar at + raw/it/masaf-disciplinari-extracted/.json the following + fields are merged: + - summary ← Article 1 first paragraph + - regione ← derived from Article 3 / 9 text + - grapes ← parsed from Article 2 (principal-only) + - geo_area_brief ← Article 3 body + - link_to_terroir ← Article 9 body + - section_roles ← {grape_varieties, geo_area, link_to_terroir, ...} + - stub_reason ← prefixed "masaf:" so the audit can tell + doc-unico-extracted from masaf-augmented + - masaf ← provenance block (url, sha256, fetched_at, + parser_template, bundle_key, archive_path) + + `record["stub"]` stays True — the record is still NOT a documento + unico extraction, just augmented. Stage 03 / 04 callers use the + `masaf` block to distinguish. + + Returns the number of records augmented. + """ + _IT_MASAF_BY_SLUG.clear() + if not MASAF_DISCIPLINARI_IT.exists(): + return 0 + augmented = 0 + for record in records: + if record.get("country") != "it": + continue + slug = record.get("slug") + if not slug: + continue + sidecar_path = MASAF_DISCIPLINARI_IT / f"{slug}.json" + if not sidecar_path.exists(): + continue + try: + sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + + # Non-stub records carry canonical EUR-Lex documento-unico data — + # only BACKFILL fields the documento unico left empty (some OJ + # docs omit the area or variety list), never overwrite. Stubs get + # the full merge below. + if not record.get("stub"): + if _backfill_it_nonstub_from_masaf(record, sidecar): + augmented += 1 + continue + + # Build the provenance block (also cached for the AOC-blob phase). + src = sidecar.get("source") or {} + match_info = sidecar.get("match") or {} + provenance = { + "filename": src.get("filename") or "", + "sha256": src.get("sha256") or "", + "bytes": src.get("bytes") or 0, + "fetched_at": src.get("fetched_at") or "", + "parser_template": sidecar.get("parser_template") or "", + "bundle_key": src.get("bundle_key") or "", + "archive_path": src.get("archive_path") or "", + "match_how": match_info.get("how") or "", + "pdf_filename": match_info.get("pdf_filename") or "", + # When an override pinned the URL, surface it for the panel. + "override_url": src.get("url") or "", + "override_source_org": src.get("source_org") or "", + } + + # Merge augmented fields onto the record. Replace rather than + # union — the record was a stub so there's nothing to lose. + if sidecar.get("summary"): + record["summary"] = sidecar["summary"] + if sidecar.get("regione") and not record.get("regione"): + record["regione"] = sidecar["regione"] + if sidecar.get("grapes"): + record["grapes"] = sidecar["grapes"] + if sidecar.get("menzioni") and not record.get("menzioni"): + record["menzioni"] = sidecar["menzioni"] + if sidecar.get("geo_area_brief"): + record["geo_area_brief"] = sidecar["geo_area_brief"] + if sidecar.get("link_to_terroir"): + record["link_to_terroir"] = sidecar["link_to_terroir"] + # IT MASAF is the last national-spec layer to carry styles; merge them + # the same way every other augment does (union, never clobber). The + # disciplinare's tipologie + organoleptic articles supply the markers + # (spumante / passito / vin santo / dolce) the grape-colour floor can't + # infer; the floor still backfills any colour the scan missed. + if sidecar.get("styles"): + record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) + section_roles = dict(record.get("section_roles") or {}) + if sidecar.get("grapes"): + section_roles.setdefault("grape_varieties", "") + if sidecar.get("geo_area_brief"): + section_roles["geo_area"] = sidecar["geo_area_brief"] + if sidecar.get("link_to_terroir"): + section_roles["link_to_terroir"] = sidecar["link_to_terroir"] + if sidecar.get("summary"): + section_roles["description"] = sidecar["summary"] + record["section_roles"] = section_roles + + if record.get("stub_reason") and not record["stub_reason"].startswith("masaf:"): + record["stub_reason"] = f"masaf:{record['stub_reason']}" + record["masaf"] = provenance + _IT_MASAF_BY_SLUG[slug] = provenance + augmented += 1 + return augmented + + +def augment_it_records_with_regional_registers(records: list[dict]) -> int: + """Fill the grape roster of regional-IGT records whose disciplinare + defers to the Region's authorised-variety register (the annex is + absent from the MASAF PDF). Each region sidecar at + raw/it/regional-variety-registers/.json lists the IGT slugs + (`igts`) that draw from it. Only applied when the record still has no + grapes, so a varietal IGT (e.g. catalanesca-del-monte-somma, excluded + from the `igts` lists) is never given a whole regional roster. + + Returns the number of records given a roster.""" + _IT_REGISTER_BY_SLUG.clear() + sources = IT_REGIONAL_REGISTERS / "sources.json" + if not sources.exists(): + return 0 + by_slug: dict[str, dict] = {} + for region in json.loads(sources.read_text(encoding="utf-8")): + if region.startswith("_"): + continue + sidecar_path = IT_REGIONAL_REGISTERS / f"{region}.json" + if not sidecar_path.exists(): + continue + try: + sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + for igt in sidecar.get("igts", []): + by_slug[igt] = sidecar + + augmented = 0 + for record in records: + if record.get("country") != "it": + continue + slug = record.get("slug") + sidecar = by_slug.get(slug) + if not sidecar: + continue + g = record.get("grapes") or {} + if g.get("principal") or g.get("accessory"): + continue + slugs = [v["slug"] for v in sidecar.get("varieties", [])] + if not slugs: + continue + record["grapes"] = { + "principal": slugs, + "accessory": [], + "observation": [], + "details": [ + {"slug": v["slug"], "name": v["name"], "role": "principal", + "colour": v.get("colour", ""), + "source": "regional-variety-register"} + for v in sidecar["varieties"] + ], + } + src = sidecar.get("source") or {} + provenance = { + "region": sidecar.get("region", ""), + "url": src.get("url", ""), + "source_org": src.get("source_org", ""), + "note": src.get("note", ""), + "sha256": src.get("sha256", ""), + "n_varieties": len(slugs), + } + record["regional_register"] = provenance + _IT_REGISTER_BY_SLUG[slug] = provenance + augmented += 1 + return augmented + + +def synthesize_it_sottozone_records(records: list[dict]) -> int: + """Emit first-class sub-denomination records for Italian sottozone + detected in the MASAF disciplinare (Chianti's 7, Valtellina's 5, + Bardolino's 3, …). The EU documento unico rarely names them, so + stage 02 emits none — they live in the national disciplinare's + Article 1, which 02f cached in the sidecar's `article_bodies`. + + Each sottozona becomes a child record mirroring the ES subzona / + FR DGC model: `is_sub_denomination=True`, `parent_slug`, + `parent_name`, `parent_id_eambrosia`, inheriting the parent's + grapes / styles / terroir / regione. Geometry resolves via the + stage-04 `parent-appellation` inheritance step. Appended to + `records` (processed after every parent, so parent geometry is + available). Returns the number of sottozona records created.""" + if not MASAF_DISCIPLINARI_IT.exists(): + return 0 + existing = {r.get("slug") for r in records if r.get("country") == "it"} + new_records: list[dict] = [] + for record in list(records): + if record.get("country") != "it" or record.get("is_sub_denomination"): + continue + slug = record.get("slug") + sidecar_path = MASAF_DISCIPLINARI_IT / f"{slug}.json" + if not slug or not sidecar_path.exists(): + continue + try: + sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + bodies = sidecar.get("article_bodies") or {} + text = " ".join( + [sidecar.get("geo_area_brief") or "", bodies.get("1", ""), bodies.get("3", "")] + ) + parent_name = record.get("name") or slug + for sz in extract_it_sottozone(text, parent_name): + sz_slug = f"{slug}-{sz['slug']}" + if not sz["slug"] or sz_slug in existing: + continue + existing.add(sz_slug) + child = dict(record) + child.update({ + "slug": sz_slug, + "name": f"{parent_name} {sz['name']}", + "is_sub_denomination": True, + "parent_slug": slug, + "parent_name": parent_name, + "parent_id_eambrosia": record.get("id_eambrosia") or "", + "menzioni": [], + "sottozona_source": "masaf-disciplinare-article-1", + }) + new_records.append(child) + records.extend(new_records) + return len(new_records) diff --git a/scripts/_lib/augment/ro.py b/scripts/_lib/augment/ro.py new file mode 100644 index 0000000..686d30d --- /dev/null +++ b/scripts/_lib/augment/ro.py @@ -0,0 +1,92 @@ +"""RO national-spec (ONVPV caiet de sarcini) (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. The shared +provenance cache + sidecar dir live in `_shared` (same objects as the +`_sources_for()` reader in stage 04). +""" +from __future__ import annotations + +import json + +from ._shared import _RO_NATIONAL_SPEC_BY_SLUG, NATIONAL_SPECS_RO + + +def augment_ro_records_with_national_specs(records: list[dict]) -> int: + """In-place merge of RO national-spec sidecar data into stub records. + + Sibling of `augment_gr_records_with_national_specs`. The 14 + grandfathered RO wines (eAmbrosia carries only a non-fetchable + `Ares(...)` reference — no EU-OJ DOCUMENT UNIC) ship as content-stubs. + Stage 02f (`scripts/ro/02f_extract_national_specs.py`) parses the + ONVPV caiet de sarcini fetched by stage 01c into + `raw/ro/national-specs-extracted/.json`. + + For each RO stub with a matching sidecar: + - grapes ← §IV Soiurile de struguri (colour-grouped) + - link_to_terroir ← §II Legătura cu aria geografică + - geo_communes ← §III Delimitarea geografică (drives the GISCO + commune-union geometry for the 2 grandfathered + IGPs — the RO-specific delta vs. GR/HR) + - geo_area_brief / summary / styles ← matching sections + - section_roles ← unified role dict so 02d reads terroir uniformly + - stub_reason ← prefixed `national-spec:` + - national_spec ← provenance block (url, sha256, format, …) + + `record["stub"]` stays True. Returns count augmented. + """ + _RO_NATIONAL_SPEC_BY_SLUG.clear() + if not NATIONAL_SPECS_RO.exists(): + return 0 + augmented = 0 + for record in records: + if record.get("country") != "ro" or not record.get("stub"): + continue + slug = record.get("slug") + if not slug: + continue + sidecar_path = NATIONAL_SPECS_RO / f"{slug}.json" + if not sidecar_path.exists(): + continue + try: + sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + + src = sidecar.get("source") or {} + provenance = { + "url": src.get("source_url") or "", + "sha256": src.get("sha256") or "", + "fetched_at": src.get("fetched_at") or "", + "format": src.get("format") or "", + "source_org": src.get("source_org") or "onvpv", + "filename": src.get("filename") or "", + "parser_template": sidecar.get("parser_template") or "", + } + + if sidecar.get("summary"): + record["summary"] = sidecar["summary"] + if sidecar.get("grapes") and (sidecar["grapes"].get("principal") + or sidecar["grapes"].get("accessory")): + record["grapes"] = sidecar["grapes"] + if sidecar.get("geo_area_brief"): + record["geo_area_brief"] = sidecar["geo_area_brief"] + if sidecar.get("geo_communes"): + record["geo_communes"] = sidecar["geo_communes"] + if sidecar.get("link_to_terroir"): + record["link_to_terroir"] = sidecar["link_to_terroir"] + if sidecar.get("styles"): + record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) + + section_roles = dict(record.get("section_roles") or {}) + for role in ("geo_area", "grape_varieties", "link_to_terroir"): + sidecar_roles = sidecar.get("section_roles") or {} + if sidecar_roles.get(role): + section_roles[role] = sidecar_roles[role] + record["section_roles"] = section_roles + + if record.get("stub_reason") and not record["stub_reason"].startswith("national-spec:"): + record["stub_reason"] = f"national-spec:{record['stub_reason']}" + record["national_spec"] = provenance + _RO_NATIONAL_SPEC_BY_SLUG[slug] = provenance + augmented += 1 + return augmented diff --git a/scripts/_lib/augment/sk.py b/scripts/_lib/augment/sk.py new file mode 100644 index 0000000..83ab6a8 --- /dev/null +++ b/scripts/_lib/augment/sk.py @@ -0,0 +1,88 @@ +"""SK national-spec (ÚPV SR špecifikácia výrobku) (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. The shared +provenance cache + sidecar dir live in `_shared` (same objects as the +`_sources_for()` reader in stage 04). +""" +from __future__ import annotations + +import json + +from ._shared import _SK_NATIONAL_SPEC_BY_SLUG, NATIONAL_SPECS_SK + + +def augment_sk_records_with_national_specs(records: list[dict]) -> int: + """In-place merge of SK national-spec sidecar data into stub records. + + Sibling of `augment_bg_records_with_national_specs`. 5 of the SK + content-stubs (no fetchable EU-OJ JEDNOTNÝ DOKUMENT) are augmented from + the ÚPV SR (indprop.gov.sk) per-wine špecifikácia výrobku. Stage 02f + (`scripts/sk/02f_extract_national_specs.py`) parses each text-layer PDF + fetched by stage 01c into `raw/sk/national-specs-extracted/.json`. + + For each SK stub with a matching sidecar: + - grapes ← section f) označenie odrody alebo odrôd + - link_to_terroir ← section g) údaje potvrdzujúce spojitosť + - geo_area_brief / summary / styles ← matching sections + - section_roles ← unified role dict so 02d reads terroir uniformly + - stub_reason ← prefixed `national-spec:` so the audit can tell + EU-OJ-extracted from spec-augmented wines + - national_spec ← provenance block (url, sha256, format, …) + + `record["stub"]` stays True — still NOT an EU-OJ extraction, just + augmented with the canonical ÚPV SR source. Returns count augmented. + """ + _SK_NATIONAL_SPEC_BY_SLUG.clear() + if not NATIONAL_SPECS_SK.exists(): + return 0 + augmented = 0 + for record in records: + if record.get("country") != "sk" or not record.get("stub"): + continue + slug = record.get("slug") + if not slug: + continue + sidecar_path = NATIONAL_SPECS_SK / f"{slug}.json" + if not sidecar_path.exists(): + continue + try: + sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + + src = sidecar.get("source") or {} + provenance = { + "url": src.get("url") or "", + "sha256": src.get("sha256") or "", + "fetched_at": src.get("fetched_at") or "", + "format": src.get("format") or "", + "source_org": src.get("source_org") or "upv-sr", + "filename": src.get("filename") or "", + "parser_template": sidecar.get("parser_template") or "", + } + + if sidecar.get("summary"): + record["summary"] = sidecar["summary"] + if sidecar.get("grapes") and (sidecar["grapes"].get("principal") + or sidecar["grapes"].get("accessory")): + record["grapes"] = sidecar["grapes"] + if sidecar.get("geo_area_brief"): + record["geo_area_brief"] = sidecar["geo_area_brief"] + if sidecar.get("link_to_terroir"): + record["link_to_terroir"] = sidecar["link_to_terroir"] + if sidecar.get("styles"): + record["styles"] = sorted(set(record.get("styles") or []) | set(sidecar["styles"])) + + section_roles = dict(record.get("section_roles") or {}) + for role in ("description", "geo_area", "grape_varieties", "link_to_terroir"): + sidecar_roles = sidecar.get("section_roles") or {} + if sidecar_roles.get(role): + section_roles[role] = sidecar_roles[role] + record["section_roles"] = section_roles + + if record.get("stub_reason") and not record["stub_reason"].startswith("national-spec:"): + record["stub_reason"] = f"national-spec:{record['stub_reason']}" + record["national_spec"] = provenance + _SK_NATIONAL_SPEC_BY_SLUG[slug] = provenance + augmented += 1 + return augmented From 865b16986ce63b5eee20f3ce031aecb94a78e1a6 Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Thu, 11 Jun 2026 12:12:29 +0200 Subject: [PATCH 30/41] refactor: extract grape/style lexicon loading to _lib/lexicon_loading.py (no-op) Phase 6.2 part 1. The self-contained lexicon cluster (load/merge grape + style lexicons, VIVC loaders, build_grapes_info + its private corpus/translation helpers, _truncate_extract, _latin_form_or_empty) moves verbatim to _lib/lexicon_loading.py along with its 5 path constants + _DISAMBIG_SUFFIX. No stage-04 function dependencies; 04 imports back the 6 externally-called names (build_grapes_info, load_style_lexicon, merge_style_lexicon, _load_vivc_by_slug, _load_vivc_colour_by_slug, _latin_form_or_empty). The geometry constant DEPT_NAME_TO_CODE, which physically sat between the lexicon and geometry blocks, stays in 04 with the geometry code (ruff F821 caught the greedy slice). Verified move-only: golden comparator vs today's baseline reports 'identical' after a full stage-04 rebuild. 63 tests + ruff clean. Co-Authored-By: Claude Opus 4.8 (1M context) --- scripts/04_build_maps.py | 378 +------------------------------ scripts/_lib/lexicon_loading.py | 385 ++++++++++++++++++++++++++++++++ 2 files changed, 393 insertions(+), 370 deletions(-) create mode 100644 scripts/_lib/lexicon_loading.py diff --git a/scripts/04_build_maps.py b/scripts/04_build_maps.py index 26331f1..73f4216 100644 --- a/scripts/04_build_maps.py +++ b/scripts/04_build_maps.py @@ -121,6 +121,14 @@ from _lib.it.geometry import ITPolygonIndex from _lib.it.region import derive_regione as derive_it_regione from _lib.it.zones import ITZoneIndex +from _lib.lexicon_loading import ( + _latin_form_or_empty, + _load_vivc_by_slug, + _load_vivc_colour_by_slug, + build_grapes_info, + load_style_lexicon, + merge_style_lexicon, +) from _lib.lieu_dit import LieuDitIndex, derive_climat_name from _lib.lu.geometry import LUPolygonIndex from _lib.lu.region import derive_region as derive_lu_region @@ -155,11 +163,9 @@ taxonomy_dfs_order as _taxonomy_dfs_order, ) from _lib.summaries import derive_summary -from _lib.wiki import is_grape_summary from shapely.geometry import mapping, shape from shapely.ops import unary_union from tqdm import tqdm -from unidecode import unidecode ROOT = Path(__file__).resolve().parent.parent EXTRACTED = ROOT / "raw" / "inao" / "cahier-extracted" @@ -208,15 +214,9 @@ PMTILES_OUT = MAP_DATA / "appellations.pmtiles" GEOJSON_VILLAGES_OUT = MAP_DATA / "appellations-villages.geojson" PMTILES_VILLAGES_OUT = MAP_DATA / "appellations-villages.pmtiles" -LEXICON_DIR = ROOT / "raw" / "wikipedia" / "grapes" -GRAPE_TRANSLATIONS_DIR = ROOT / "raw" / "translations" / "grapes" -VIVC_BY_SLUG = ROOT / "raw" / "vivc" / "by-slug" WIKIDATA_QIDS = ROOT / "raw" / "wikidata" / "qids-by-slug.json" -STYLE_LEXICON_DIR = ROOT / "raw" / "wikipedia" / "styles" -STYLE_TRANSLATIONS_DIR = ROOT / "raw" / "translations" / "styles" -_DISAMBIG_SUFFIX = re.compile(r"\s*\([^)]*\)\s*$") # Cross-border PDOs that physically extend across more than one country. @@ -329,368 +329,6 @@ def _base_colour_styles_from_grapes(grapes: dict, existing_styles: set[str]) -> return add -def load_grape_lexicon(lang: str, max_chars: int = 280) -> dict: - """Load Wikipedia grape data for a locale; returns {slug: {name, extract?, - page_url?, revision_id?, thumbnail?}} for any entry that has at least a - `wikipedia_title` (so a localised display name is available even when - the article summary is filtered out). Truncates `extract` to ~max_chars - at the nearest sentence boundary when present. - - Wikipedia titles often include a parenthetical disambiguator — - "Pinot noir (cépage)", "Mauzac (grape)" — which is article-DB hygiene, - not how the variety is referenced in the wine world. Strip it for - display so the chip reads cleanly.""" - lang_dir = LEXICON_DIR / lang - if not lang_dir.exists(): - return {} - out: dict[str, dict] = {} - for f in lang_dir.glob("*.json"): - d = json.loads(f.read_text(encoding="utf-8")) - if d.get("missing") or d.get("error"): - continue - title = (d.get("wikipedia_title") or "").strip() - if not title: - continue - display_name = _DISAMBIG_SUFFIX.sub("", title).strip() or title - entry: dict = { - "name": display_name, - "page_url": d.get("page_url"), - } - extract = (d.get("extract") or "").strip() - if extract and is_grape_summary(lang, d.get("description", ""), extract): - if len(extract) > max_chars: - cut = extract[:max_chars].rsplit(". ", 1)[0] - extract = cut + ("." if not cut.endswith(".") else "") + " […]" - entry["extract"] = extract - entry["revision_id"] = d.get("revision_id") - if d.get("thumbnail"): - entry["thumbnail"] = d.get("thumbnail") - out[d["slug"]] = entry - return out - - -def merge_grape_lexicon(lang_lex: dict, fr_lex: dict) -> dict: - """Legacy FR-fallback merge. Retained for the styles path which still - uses it; the grapes path now goes through `build_grapes_info()`.""" - if lang_lex is fr_lex: - return lang_lex - out: dict[str, dict] = {} - for slug, fr_entry in fr_lex.items(): - local = lang_lex.get(slug) - if local is None: - merged = dict(fr_entry) - merged["lang_fallback"] = True - out[slug] = merged - else: - merged = dict(local) - if "extract" not in merged and "extract" in fr_entry: - merged["extract"] = fr_entry["extract"] - if "thumbnail" not in merged and "thumbnail" in fr_entry: - merged["thumbnail"] = fr_entry["thumbnail"] - merged["lang_fallback"] = True - out[slug] = merged - for slug, local in lang_lex.items(): - out.setdefault(slug, local) - return out - - -_VIVC_BY_SLUG_CACHE: dict[str, dict] | None = None - - -def _load_vivc_by_slug() -> dict[str, dict]: - """`{slug: {canonical_name, vivc_id, vivc_url}}` from raw/vivc/by-slug/.""" - global _VIVC_BY_SLUG_CACHE - if _VIVC_BY_SLUG_CACHE is not None: - return _VIVC_BY_SLUG_CACHE - out: dict[str, dict] = {} - if not VIVC_BY_SLUG.exists(): - _VIVC_BY_SLUG_CACHE = out - return out - for f in VIVC_BY_SLUG.glob("*.json"): - rec = json.loads(f.read_text(encoding="utf-8")) - prime = (rec.get("prime_name") or "").strip() - vid = rec.get("vivc_id") - if not prime or not isinstance(vid, int): - continue - # str.title() handles apostrophes correctly ("D'AUNIS" → "D'Aunis"), - # which a per-token .capitalize() does not ("D'aunis"). - canonical = prime.title() - out[rec["slug"]] = { - "canonical_name": canonical, - "vivc_id": vid, - "vivc_url": rec.get("source_url"), - } - _VIVC_BY_SLUG_CACHE = out - return out - - -_VIVC_COLOUR_BY_SLUG_CACHE: dict[str, str] | None = None - - -def _load_vivc_colour_by_slug() -> dict[str, str]: - """`{slug: 'blanc'|'gris'|'noir'|'rose'}` from raw/vivc/by-slug/.json - `color` (UPPERCASE NOIR/BLANC/GRIS/ROSE). Berry colour — not a wine style. - VIVC carries colours absent from the curated DEFAULT_COLOUR table - (nebbiolo, chasselas, …), so it is the gap-filler for the style floor. - Kept separate from `_load_vivc_by_slug` so that function's return shape - (consumed by `facets`) is untouched.""" - global _VIVC_COLOUR_BY_SLUG_CACHE - if _VIVC_COLOUR_BY_SLUG_CACHE is not None: - return _VIVC_COLOUR_BY_SLUG_CACHE - out: dict[str, str] = {} - _MAP = {"NOIR": "noir", "BLANC": "blanc", "GRIS": "gris", "ROSE": "rose"} - if VIVC_BY_SLUG.exists(): - for f in VIVC_BY_SLUG.glob("*.json"): - rec = json.loads(f.read_text(encoding="utf-8")) - colour = _MAP.get((rec.get("color") or "").strip().upper()) - slug = rec.get("slug") - if colour and slug: - out[slug] = colour - _VIVC_COLOUR_BY_SLUG_CACHE = out - return out - - -def _load_native_grape(lang: str, slug: str, max_chars: int = 280) -> dict | None: - """Native Wikipedia entry for (slug, lang), or None when missing/empty. - Returns the trimmed entry with name, extract, page_url, revision_id, - thumbnail, matched_via.""" - f = LEXICON_DIR / lang / f"{slug}.json" - if not f.exists(): - return None - d = json.loads(f.read_text(encoding="utf-8")) - if d.get("missing") or d.get("error"): - return None - title = (d.get("wikipedia_title") or "").strip() - extract = (d.get("extract") or "").strip() - if not extract or not is_grape_summary(lang, d.get("description", ""), extract): - return None - display = _DISAMBIG_SUFFIX.sub("", title).strip() if title else slug - if len(extract) > max_chars: - cut = extract[:max_chars].rsplit(". ", 1)[0] - extract = cut + ("." if not cut.endswith(".") else "") + " […]" - out: dict = { - "name": display or slug, - "extract": extract, - "page_url": d.get("page_url"), - "revision_id": d.get("revision_id"), - "matched_via": d.get("matched_via") or "primary", - } - if d.get("thumbnail"): - out["thumbnail"] = d["thumbnail"] - return out - - -def _load_translated_grape(lang: str, slug: str, max_chars: int = 280) -> dict | None: - f = GRAPE_TRANSLATIONS_DIR / lang / f"{slug}.json" - if not f.exists(): - return None - d = json.loads(f.read_text(encoding="utf-8")) - extract = (d.get("extract") or "").strip() - if not extract: - return None - if len(extract) > max_chars: - cut = extract[:max_chars].rsplit(". ", 1)[0] - extract = cut + ("." if not cut.endswith(".") else "") + " […]" - return { - "extract": extract, - "source_lang": d.get("source_lang"), - "page_url": d.get("source_page_url"), - "name": (d.get("source_wikipedia_title") or slug).strip() or slug, - "translator": d.get("translator"), - "translator_kind": d.get("translator_kind"), - } - - -def _corpus_grape_names() -> dict[str, str]: - """Per-slug regulator spelling from the FR+ES+PT corpus, e.g. - `cot → 'cot'`, `malbec → 'malbec'`, `mancin → 'mancin'`. Used as the - canonical sidebar / pill label so three different slugs sharing a - Wikipedia article (Cot ↔ Malbec via vivc, plus a misattributed - Mancin) read as their three distinct cahier names — not three - identical "Malbec" rows.""" - from _lib.grape_corpus import collect_grape_slugs as _c # noqa: PLC0415 - return {slug: entry["name"] for slug, entry in _c().items() if entry.get("name")} - - -_CORPUS_GRAPE_NAMES: dict[str, str] | None = None - - -def _corpus_name_for(slug: str) -> str | None: - global _CORPUS_GRAPE_NAMES - if _CORPUS_GRAPE_NAMES is None: - _CORPUS_GRAPE_NAMES = _corpus_grape_names() - return _CORPUS_GRAPE_NAMES.get(slug) - - -def _latin_form_or_empty(name: str) -> str: - # Cyrillic / Greek / other non-Latin display strings get an - # informational ASCII transliteration so the grape-pill renderer can - # fall back to it when no VIVC canonical name is available (e.g. - # native BG varieties like `mavrud` that VIVC hasn't catalogued). - latin = unidecode(name or "").strip() - return latin if latin and latin != (name or "").strip() else "" - - -def _override_name_with_corpus(entry: dict, slug: str) -> None: - """Replace `entry['name']` (which after `_load_native_grape` is the - Wikipedia article title) with the regulator's cahier spelling when - one exists. Keeps `wikipedia_title` intact for the tooltip header so - attribution stays accurate.""" - cahier = _corpus_name_for(slug) - if not cahier: - return - if "wikipedia_title" not in entry and entry.get("name"): - entry["wikipedia_title"] = entry["name"] - entry["name"] = cahier - latin = _latin_form_or_empty(cahier) - if latin: - entry["name_latin"] = latin - - -def build_grapes_info(target_locale: str) -> dict: - """Per-slug grape data for the target locale's map page. - - Resolution per (slug, target_locale): - 1. Native target-locale Wikipedia entry → `is_translated=false`, - `source_lang=target_locale`. - 2. Translated cache (`02b_translate_grapes.py`) → - `is_translated=true`, `source_lang` from the cache record. - 3. Neither → emit `{canonical_name, vivc_id, vivc_url}` only when - a VIVC record exists; the pill still renders (cahier name + - optional canonical bracket + VIVC link), just without a tooltip - body. - - VIVC `canonical_name`/`vivc_id`/`vivc_url` ride alongside the - Wikipedia entry for every slug that has a resolved VIVC record; - unresolved/missed slugs simply lack those fields. - """ - vivc = _load_vivc_by_slug() - slugs: set[str] = set() - if (LEXICON_DIR / target_locale).exists(): - slugs.update(p.stem for p in (LEXICON_DIR / target_locale).glob("*.json")) - if (GRAPE_TRANSLATIONS_DIR / target_locale).exists(): - slugs.update(p.stem for p in (GRAPE_TRANSLATIONS_DIR / target_locale).glob("*.json")) - slugs.update(vivc.keys()) - # Keep only slugs that the *current* corpus actually emits. Stale - # Wikipedia / translation cache entries for slugs that no longer - # appear in the FR/ES/PT extracted JSONs (e.g. `tempranillo-cencibel` - # after the ES EU-OJ splitter fix) otherwise leak into GRAPES_INFO - # and reappear in the chip-filter index as ghost entries. - corpus_slugs = set(_corpus_grape_names().keys()) | set(vivc.keys()) - slugs &= corpus_slugs - - out: dict[str, dict] = {} - for slug in slugs: - vivc_fields = vivc.get(slug) or {} - native = _load_native_grape(target_locale, slug) - if native is not None: - entry = { - **vivc_fields, - **native, - "source_lang": target_locale, - "is_translated": False, - } - _override_name_with_corpus(entry, slug) - out[slug] = entry - continue - translated = _load_translated_grape(target_locale, slug) - if translated is not None: - entry = { - **vivc_fields, - **translated, - "is_translated": True, - "matched_via": "translation", - } - _override_name_with_corpus(entry, slug) - out[slug] = entry - continue - if vivc_fields: - out[slug] = {**vivc_fields, "is_translated": False, "source_lang": None} - return out - - -def _truncate_extract(extract: str, max_chars: int) -> str: - extract = (extract or "").strip() - if not extract or len(extract) <= max_chars: - return extract - cut = extract[:max_chars].rsplit(". ", 1)[0] - return cut + ("." if not cut.endswith(".") else "") + " […]" - - -def load_style_lexicon(lang: str, max_chars: int = 320) -> dict: - """Load wine-style data for a locale; returns - {slug: {extract, page_url, revision_id, thumbnail?, translation?}} - for each curated entry that has usable text. - - Native Wikipedia fetches (raw/wikipedia/styles//) are preferred. - When a slug has no native entry in `lang` but a translated entry exists - (raw/translations/styles//), the translation is used and a - `translation` metadata block is attached so the UI can render the - "translated from Wikipedia" attribution.""" - out: dict[str, dict] = {} - lang_dir = STYLE_LEXICON_DIR / lang - if lang_dir.exists(): - for f in lang_dir.glob("*.json"): - d = json.loads(f.read_text(encoding="utf-8")) - if d.get("missing") or d.get("error"): - continue - extract = _truncate_extract(d.get("extract") or "", max_chars) - if not extract: - continue - entry: dict = { - "extract": extract, - "page_url": d.get("page_url"), - "revision_id": d.get("revision_id"), - } - if d.get("thumbnail"): - entry["thumbnail"] = d.get("thumbnail") - out[d["slug"]] = entry - - tx_dir = STYLE_TRANSLATIONS_DIR / lang - if tx_dir.exists(): - for f in tx_dir.glob("*.json"): - d = json.loads(f.read_text(encoding="utf-8")) - slug = d.get("slug") or f.stem - if slug in out: - continue # native fetch wins - extract = _truncate_extract(d.get("extract") or "", max_chars) - if not extract: - continue - out[slug] = { - "extract": extract, - "page_url": d.get("source_page_url") or "", - "revision_id": d.get("source_revision_id"), - "translation": { - "source_lang": d.get("source_lang") or "", - "source_page_url": d.get("source_page_url") or "", - "source_wikipedia_title": d.get("source_wikipedia_title") or "", - "translator": d.get("translator") or "", - "translator_kind": d.get("translator_kind") or "", - }, - } - return out - - -def merge_style_lexicon(lang_lex: dict, fr_lex: dict) -> dict: - """FR-fallback for slugs the target locale lacks entirely — both as a - native fetch and as a translation. Used as a last resort so the UI still - renders something rather than an empty pill.""" - if lang_lex is fr_lex: - return lang_lex - out: dict[str, dict] = {} - for slug, fr_entry in fr_lex.items(): - local = lang_lex.get(slug) - if local is None: - merged = dict(fr_entry) - merged["lang_fallback"] = True - out[slug] = merged - else: - out[slug] = dict(local) - for slug, local in lang_lex.items(): - out.setdefault(slug, local) - return out - - # INSEE 2-digit département code → canonical name as written in cahiers. # Used for resolving "Côte-d'Or" → "21" so commune lookup stays inside the # correct département (avoids Saint-Pierre homonym collisions). diff --git a/scripts/_lib/lexicon_loading.py b/scripts/_lib/lexicon_loading.py new file mode 100644 index 0000000..fd79a3b --- /dev/null +++ b/scripts/_lib/lexicon_loading.py @@ -0,0 +1,385 @@ +"""Grape + style lexicon loading / merging + VIVC + grapes_info assembly (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. Self-contained: +no stage-04 function dependencies. The public loaders + `_latin_form_or_empty` +(used by the aocs-blob phase for grape_names_latin) are imported back into 04. +""" +from __future__ import annotations + +import json +import re +from pathlib import Path + +from unidecode import unidecode + +from _lib.wiki import is_grape_summary + +ROOT = Path(__file__).resolve().parents[2] +LEXICON_DIR = ROOT / "raw" / "wikipedia" / "grapes" +GRAPE_TRANSLATIONS_DIR = ROOT / "raw" / "translations" / "grapes" +VIVC_BY_SLUG = ROOT / "raw" / "vivc" / "by-slug" +STYLE_LEXICON_DIR = ROOT / "raw" / "wikipedia" / "styles" +STYLE_TRANSLATIONS_DIR = ROOT / "raw" / "translations" / "styles" +_DISAMBIG_SUFFIX = re.compile(r"\s*\([^)]*\)\s*$") + + +def load_grape_lexicon(lang: str, max_chars: int = 280) -> dict: + """Load Wikipedia grape data for a locale; returns {slug: {name, extract?, + page_url?, revision_id?, thumbnail?}} for any entry that has at least a + `wikipedia_title` (so a localised display name is available even when + the article summary is filtered out). Truncates `extract` to ~max_chars + at the nearest sentence boundary when present. + + Wikipedia titles often include a parenthetical disambiguator — + "Pinot noir (cépage)", "Mauzac (grape)" — which is article-DB hygiene, + not how the variety is referenced in the wine world. Strip it for + display so the chip reads cleanly.""" + lang_dir = LEXICON_DIR / lang + if not lang_dir.exists(): + return {} + out: dict[str, dict] = {} + for f in lang_dir.glob("*.json"): + d = json.loads(f.read_text(encoding="utf-8")) + if d.get("missing") or d.get("error"): + continue + title = (d.get("wikipedia_title") or "").strip() + if not title: + continue + display_name = _DISAMBIG_SUFFIX.sub("", title).strip() or title + entry: dict = { + "name": display_name, + "page_url": d.get("page_url"), + } + extract = (d.get("extract") or "").strip() + if extract and is_grape_summary(lang, d.get("description", ""), extract): + if len(extract) > max_chars: + cut = extract[:max_chars].rsplit(". ", 1)[0] + extract = cut + ("." if not cut.endswith(".") else "") + " […]" + entry["extract"] = extract + entry["revision_id"] = d.get("revision_id") + if d.get("thumbnail"): + entry["thumbnail"] = d.get("thumbnail") + out[d["slug"]] = entry + return out + + +def merge_grape_lexicon(lang_lex: dict, fr_lex: dict) -> dict: + """Legacy FR-fallback merge. Retained for the styles path which still + uses it; the grapes path now goes through `build_grapes_info()`.""" + if lang_lex is fr_lex: + return lang_lex + out: dict[str, dict] = {} + for slug, fr_entry in fr_lex.items(): + local = lang_lex.get(slug) + if local is None: + merged = dict(fr_entry) + merged["lang_fallback"] = True + out[slug] = merged + else: + merged = dict(local) + if "extract" not in merged and "extract" in fr_entry: + merged["extract"] = fr_entry["extract"] + if "thumbnail" not in merged and "thumbnail" in fr_entry: + merged["thumbnail"] = fr_entry["thumbnail"] + merged["lang_fallback"] = True + out[slug] = merged + for slug, local in lang_lex.items(): + out.setdefault(slug, local) + return out + + +_VIVC_BY_SLUG_CACHE: dict[str, dict] | None = None + + +def _load_vivc_by_slug() -> dict[str, dict]: + """`{slug: {canonical_name, vivc_id, vivc_url}}` from raw/vivc/by-slug/.""" + global _VIVC_BY_SLUG_CACHE + if _VIVC_BY_SLUG_CACHE is not None: + return _VIVC_BY_SLUG_CACHE + out: dict[str, dict] = {} + if not VIVC_BY_SLUG.exists(): + _VIVC_BY_SLUG_CACHE = out + return out + for f in VIVC_BY_SLUG.glob("*.json"): + rec = json.loads(f.read_text(encoding="utf-8")) + prime = (rec.get("prime_name") or "").strip() + vid = rec.get("vivc_id") + if not prime or not isinstance(vid, int): + continue + # str.title() handles apostrophes correctly ("D'AUNIS" → "D'Aunis"), + # which a per-token .capitalize() does not ("D'aunis"). + canonical = prime.title() + out[rec["slug"]] = { + "canonical_name": canonical, + "vivc_id": vid, + "vivc_url": rec.get("source_url"), + } + _VIVC_BY_SLUG_CACHE = out + return out + + +_VIVC_COLOUR_BY_SLUG_CACHE: dict[str, str] | None = None + + +def _load_vivc_colour_by_slug() -> dict[str, str]: + """`{slug: 'blanc'|'gris'|'noir'|'rose'}` from raw/vivc/by-slug/.json + `color` (UPPERCASE NOIR/BLANC/GRIS/ROSE). Berry colour — not a wine style. + VIVC carries colours absent from the curated DEFAULT_COLOUR table + (nebbiolo, chasselas, …), so it is the gap-filler for the style floor. + Kept separate from `_load_vivc_by_slug` so that function's return shape + (consumed by `facets`) is untouched.""" + global _VIVC_COLOUR_BY_SLUG_CACHE + if _VIVC_COLOUR_BY_SLUG_CACHE is not None: + return _VIVC_COLOUR_BY_SLUG_CACHE + out: dict[str, str] = {} + _MAP = {"NOIR": "noir", "BLANC": "blanc", "GRIS": "gris", "ROSE": "rose"} + if VIVC_BY_SLUG.exists(): + for f in VIVC_BY_SLUG.glob("*.json"): + rec = json.loads(f.read_text(encoding="utf-8")) + colour = _MAP.get((rec.get("color") or "").strip().upper()) + slug = rec.get("slug") + if colour and slug: + out[slug] = colour + _VIVC_COLOUR_BY_SLUG_CACHE = out + return out + + +def _load_native_grape(lang: str, slug: str, max_chars: int = 280) -> dict | None: + """Native Wikipedia entry for (slug, lang), or None when missing/empty. + Returns the trimmed entry with name, extract, page_url, revision_id, + thumbnail, matched_via.""" + f = LEXICON_DIR / lang / f"{slug}.json" + if not f.exists(): + return None + d = json.loads(f.read_text(encoding="utf-8")) + if d.get("missing") or d.get("error"): + return None + title = (d.get("wikipedia_title") or "").strip() + extract = (d.get("extract") or "").strip() + if not extract or not is_grape_summary(lang, d.get("description", ""), extract): + return None + display = _DISAMBIG_SUFFIX.sub("", title).strip() if title else slug + if len(extract) > max_chars: + cut = extract[:max_chars].rsplit(". ", 1)[0] + extract = cut + ("." if not cut.endswith(".") else "") + " […]" + out: dict = { + "name": display or slug, + "extract": extract, + "page_url": d.get("page_url"), + "revision_id": d.get("revision_id"), + "matched_via": d.get("matched_via") or "primary", + } + if d.get("thumbnail"): + out["thumbnail"] = d["thumbnail"] + return out + + +def _load_translated_grape(lang: str, slug: str, max_chars: int = 280) -> dict | None: + f = GRAPE_TRANSLATIONS_DIR / lang / f"{slug}.json" + if not f.exists(): + return None + d = json.loads(f.read_text(encoding="utf-8")) + extract = (d.get("extract") or "").strip() + if not extract: + return None + if len(extract) > max_chars: + cut = extract[:max_chars].rsplit(". ", 1)[0] + extract = cut + ("." if not cut.endswith(".") else "") + " […]" + return { + "extract": extract, + "source_lang": d.get("source_lang"), + "page_url": d.get("source_page_url"), + "name": (d.get("source_wikipedia_title") or slug).strip() or slug, + "translator": d.get("translator"), + "translator_kind": d.get("translator_kind"), + } + + +def _corpus_grape_names() -> dict[str, str]: + """Per-slug regulator spelling from the FR+ES+PT corpus, e.g. + `cot → 'cot'`, `malbec → 'malbec'`, `mancin → 'mancin'`. Used as the + canonical sidebar / pill label so three different slugs sharing a + Wikipedia article (Cot ↔ Malbec via vivc, plus a misattributed + Mancin) read as their three distinct cahier names — not three + identical "Malbec" rows.""" + from _lib.grape_corpus import collect_grape_slugs as _c # noqa: PLC0415 + return {slug: entry["name"] for slug, entry in _c().items() if entry.get("name")} + + +_CORPUS_GRAPE_NAMES: dict[str, str] | None = None + + +def _corpus_name_for(slug: str) -> str | None: + global _CORPUS_GRAPE_NAMES + if _CORPUS_GRAPE_NAMES is None: + _CORPUS_GRAPE_NAMES = _corpus_grape_names() + return _CORPUS_GRAPE_NAMES.get(slug) + + +def _latin_form_or_empty(name: str) -> str: + # Cyrillic / Greek / other non-Latin display strings get an + # informational ASCII transliteration so the grape-pill renderer can + # fall back to it when no VIVC canonical name is available (e.g. + # native BG varieties like `mavrud` that VIVC hasn't catalogued). + latin = unidecode(name or "").strip() + return latin if latin and latin != (name or "").strip() else "" + + +def _override_name_with_corpus(entry: dict, slug: str) -> None: + """Replace `entry['name']` (which after `_load_native_grape` is the + Wikipedia article title) with the regulator's cahier spelling when + one exists. Keeps `wikipedia_title` intact for the tooltip header so + attribution stays accurate.""" + cahier = _corpus_name_for(slug) + if not cahier: + return + if "wikipedia_title" not in entry and entry.get("name"): + entry["wikipedia_title"] = entry["name"] + entry["name"] = cahier + latin = _latin_form_or_empty(cahier) + if latin: + entry["name_latin"] = latin + + +def build_grapes_info(target_locale: str) -> dict: + """Per-slug grape data for the target locale's map page. + + Resolution per (slug, target_locale): + 1. Native target-locale Wikipedia entry → `is_translated=false`, + `source_lang=target_locale`. + 2. Translated cache (`02b_translate_grapes.py`) → + `is_translated=true`, `source_lang` from the cache record. + 3. Neither → emit `{canonical_name, vivc_id, vivc_url}` only when + a VIVC record exists; the pill still renders (cahier name + + optional canonical bracket + VIVC link), just without a tooltip + body. + + VIVC `canonical_name`/`vivc_id`/`vivc_url` ride alongside the + Wikipedia entry for every slug that has a resolved VIVC record; + unresolved/missed slugs simply lack those fields. + """ + vivc = _load_vivc_by_slug() + slugs: set[str] = set() + if (LEXICON_DIR / target_locale).exists(): + slugs.update(p.stem for p in (LEXICON_DIR / target_locale).glob("*.json")) + if (GRAPE_TRANSLATIONS_DIR / target_locale).exists(): + slugs.update(p.stem for p in (GRAPE_TRANSLATIONS_DIR / target_locale).glob("*.json")) + slugs.update(vivc.keys()) + # Keep only slugs that the *current* corpus actually emits. Stale + # Wikipedia / translation cache entries for slugs that no longer + # appear in the FR/ES/PT extracted JSONs (e.g. `tempranillo-cencibel` + # after the ES EU-OJ splitter fix) otherwise leak into GRAPES_INFO + # and reappear in the chip-filter index as ghost entries. + corpus_slugs = set(_corpus_grape_names().keys()) | set(vivc.keys()) + slugs &= corpus_slugs + + out: dict[str, dict] = {} + for slug in slugs: + vivc_fields = vivc.get(slug) or {} + native = _load_native_grape(target_locale, slug) + if native is not None: + entry = { + **vivc_fields, + **native, + "source_lang": target_locale, + "is_translated": False, + } + _override_name_with_corpus(entry, slug) + out[slug] = entry + continue + translated = _load_translated_grape(target_locale, slug) + if translated is not None: + entry = { + **vivc_fields, + **translated, + "is_translated": True, + "matched_via": "translation", + } + _override_name_with_corpus(entry, slug) + out[slug] = entry + continue + if vivc_fields: + out[slug] = {**vivc_fields, "is_translated": False, "source_lang": None} + return out + + +def _truncate_extract(extract: str, max_chars: int) -> str: + extract = (extract or "").strip() + if not extract or len(extract) <= max_chars: + return extract + cut = extract[:max_chars].rsplit(". ", 1)[0] + return cut + ("." if not cut.endswith(".") else "") + " […]" + + +def load_style_lexicon(lang: str, max_chars: int = 320) -> dict: + """Load wine-style data for a locale; returns + {slug: {extract, page_url, revision_id, thumbnail?, translation?}} + for each curated entry that has usable text. + + Native Wikipedia fetches (raw/wikipedia/styles//) are preferred. + When a slug has no native entry in `lang` but a translated entry exists + (raw/translations/styles//), the translation is used and a + `translation` metadata block is attached so the UI can render the + "translated from Wikipedia" attribution.""" + out: dict[str, dict] = {} + lang_dir = STYLE_LEXICON_DIR / lang + if lang_dir.exists(): + for f in lang_dir.glob("*.json"): + d = json.loads(f.read_text(encoding="utf-8")) + if d.get("missing") or d.get("error"): + continue + extract = _truncate_extract(d.get("extract") or "", max_chars) + if not extract: + continue + entry: dict = { + "extract": extract, + "page_url": d.get("page_url"), + "revision_id": d.get("revision_id"), + } + if d.get("thumbnail"): + entry["thumbnail"] = d.get("thumbnail") + out[d["slug"]] = entry + + tx_dir = STYLE_TRANSLATIONS_DIR / lang + if tx_dir.exists(): + for f in tx_dir.glob("*.json"): + d = json.loads(f.read_text(encoding="utf-8")) + slug = d.get("slug") or f.stem + if slug in out: + continue # native fetch wins + extract = _truncate_extract(d.get("extract") or "", max_chars) + if not extract: + continue + out[slug] = { + "extract": extract, + "page_url": d.get("source_page_url") or "", + "revision_id": d.get("source_revision_id"), + "translation": { + "source_lang": d.get("source_lang") or "", + "source_page_url": d.get("source_page_url") or "", + "source_wikipedia_title": d.get("source_wikipedia_title") or "", + "translator": d.get("translator") or "", + "translator_kind": d.get("translator_kind") or "", + }, + } + return out + + +def merge_style_lexicon(lang_lex: dict, fr_lex: dict) -> dict: + """FR-fallback for slugs the target locale lacks entirely — both as a + native fetch and as a translation. Used as a last resort so the UI still + renders something rather than an empty pill.""" + if lang_lex is fr_lex: + return lang_lex + out: dict[str, dict] = {} + for slug, fr_entry in fr_lex.items(): + local = lang_lex.get(slug) + if local is None: + merged = dict(fr_entry) + merged["lang_fallback"] = True + out[slug] = merged + else: + out[slug] = dict(local) + for slug, local in lang_lex.items(): + out.setdefault(slug, local) + return out From b28abd0b7ef7b557d1738cc721767239338f12e6 Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Thu, 11 Jun 2026 12:52:53 +0200 Subject: [PATCH 31/41] refactor: extract commune-index + DGC/ES geometry chain to _lib/geom_chain.py (no-op) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Completes Phase 6.2. The geometry resolution cluster — normalize_commune, load_commune_index, cahier_insee, union_for_appellation, union_from_insee, _find_sibling_umbrella, the DGCGeomResult dataclass, resolve_dgc_geometry, and the ES fallbacks (_resolve_es_igp_fallback, _resolve_es_sigpac) + DEPT_NAME_TO _CODE — moves verbatim to _lib/geom_chain.py. The members are scattered non-contiguously in 04 and interleaved with non-geometry helpers (_join_set, _geojson_bounds, communes_containing), so each is extracted individually by name; the interleaved helpers stay in 04 (the cluster has no 04-function deps). The 11 names the main build loop calls are imported back. Verified move-only: golden comparator vs today's baseline reports 'identical' after a full stage-04 rebuild. 63 tests + ruff clean. Phase 6 done: 04_build_maps.py is ~2300 lines lighter (augmenters + lexicon + geometry now in _lib/). Co-Authored-By: Claude Opus 4.8 (1M context) --- scripts/04_build_maps.py | 496 +----------------------------------- scripts/_lib/geom_chain.py | 505 +++++++++++++++++++++++++++++++++++++ 2 files changed, 515 insertions(+), 486 deletions(-) create mode 100644 scripts/_lib/geom_chain.py diff --git a/scripts/04_build_maps.py b/scripts/04_build_maps.py index 73f4216..581cfc7 100644 --- a/scripts/04_build_maps.py +++ b/scripts/04_build_maps.py @@ -21,13 +21,10 @@ import datetime as dt import hashlib import json -import re import shutil import subprocess import sys -import unicodedata from collections import Counter -from dataclasses import dataclass from pathlib import Path from _lib.aires import load_aires @@ -88,27 +85,22 @@ from _lib.cz.region import derive_region as derive_cz_region from _lib.de.geometry import DEPolygonIndex from _lib.de.region import derive_region as derive_de_region -from _lib.dgc_village_overrides import DGC_VILLAGE_INSEE -from _lib.es.baleares import ines_for_island -from _lib.es.commune_list import ( - parse_ccaa_wide, - parse_commune_list, - parse_island_wide, - parse_province_wide_list, - parse_whole_commune_prefix, -) from _lib.es.geometry import ESPolygonIndex -from _lib.es.pliego_parcels import parse_polygon_inclusions -from _lib.es.region import ( - CCAA_TO_PROVINCE_INES, - PROVINCE_TO_INE, -) from _lib.es.region import ( derive_ccaa as derive_es_ccaa, ) from _lib.es.sigpac import SigpacIndex from _lib.es.zones import MAPA_ZONES_FILE, ESZoneIndex from _lib.fr_wine_region import derive_wine_region as derive_fr_wine_region +from _lib.geom_chain import ( + _resolve_es_igp_fallback, + _resolve_es_sigpac, + cahier_insee, + load_commune_index, + resolve_dgc_geometry, + union_for_appellation, + union_from_insee, +) from _lib.geometry_overrides import ClipResult, GeometryOverrides from _lib.gr.geometry import GRPolygonIndex from _lib.gr.region import derive_region as derive_gr_region @@ -129,7 +121,7 @@ load_style_lexicon, merge_style_lexicon, ) -from _lib.lieu_dit import LieuDitIndex, derive_climat_name +from _lib.lieu_dit import LieuDitIndex from _lib.lu.geometry import LUPolygonIndex from _lib.lu.region import derive_region as derive_lu_region from _lib.map_template import STARTUP_AOCS_FIELDS @@ -164,7 +156,6 @@ ) from _lib.summaries import derive_summary from shapely.geometry import mapping, shape -from shapely.ops import unary_union from tqdm import tqdm ROOT = Path(__file__).resolve().parent.parent @@ -329,77 +320,6 @@ def _base_colour_styles_from_grapes(grapes: dict, existing_styles: set[str]) -> return add -# INSEE 2-digit département code → canonical name as written in cahiers. -# Used for resolving "Côte-d'Or" → "21" so commune lookup stays inside the -# correct département (avoids Saint-Pierre homonym collisions). -DEPT_NAME_TO_CODE: dict[str, str] = { - "Ain": "01", "Aisne": "02", "Allier": "03", "Alpes-de-Haute-Provence": "04", - "Hautes-Alpes": "05", "Alpes-Maritimes": "06", "Ardèche": "07", "Ardennes": "08", - "Ariège": "09", "Aube": "10", "Aude": "11", "Aveyron": "12", - "Bouches-du-Rhône": "13", "Calvados": "14", "Cantal": "15", "Charente": "16", - "Charente-Maritime": "17", "Cher": "18", "Corrèze": "19", "Corse-du-Sud": "2A", - "Haute-Corse": "2B", "Côte-d'Or": "21", "Côte-d’Or": "21", - "Côtes-d'Armor": "22", "Côtes-d’Armor": "22", "Creuse": "23", - "Dordogne": "24", "Doubs": "25", "Drôme": "26", "Eure": "27", "Eure-et-Loir": "28", - "Finistère": "29", "Gard": "30", "Haute-Garonne": "31", "Gers": "32", - "Gironde": "33", "Hérault": "34", "Ille-et-Vilaine": "35", "Indre": "36", - "Indre-et-Loire": "37", "Isère": "38", "Jura": "39", "Landes": "40", - "Loir-et-Cher": "41", "Loire": "42", "Haute-Loire": "43", "Loire-Atlantique": "44", - "Loiret": "45", "Lot": "46", "Lot-et-Garonne": "47", "Lozère": "48", - "Maine-et-Loire": "49", "Manche": "50", "Marne": "51", "Haute-Marne": "52", - "Mayenne": "53", "Meurthe-et-Moselle": "54", "Meuse": "55", "Morbihan": "56", - "Moselle": "57", "Nièvre": "58", "Nord": "59", "Oise": "60", "Orne": "61", - "Pas-de-Calais": "62", "Puy-de-Dôme": "63", "Pyrénées-Atlantiques": "64", - "Hautes-Pyrénées": "65", "Pyrénées-Orientales": "66", "Bas-Rhin": "67", - "Haut-Rhin": "68", "Rhône": "69", "Haute-Saône": "70", "Saône-et-Loire": "71", - "Sarthe": "72", "Savoie": "73", "Haute-Savoie": "74", "Paris": "75", - "Seine-Maritime": "76", "Seine-et-Marne": "77", "Yvelines": "78", - "Deux-Sèvres": "79", "Somme": "80", "Tarn": "81", "Tarn-et-Garonne": "82", - "Var": "83", "Vaucluse": "84", "Vendée": "85", "Vienne": "86", - "Haute-Vienne": "87", "Vosges": "88", "Yonne": "89", "Territoire de Belfort": "90", - "Essonne": "91", "Hauts-de-Seine": "92", "Seine-Saint-Denis": "93", - "Val-de-Marne": "94", "Val-d'Oise": "95", "Val-d’Oise": "95", - "Guadeloupe": "971", "Martinique": "972", "Guyane": "973", - "La Réunion": "974", "Mayotte": "976", -} - - -def normalize_commune(s: str) -> str: - """Loose match key for commune names — strip diacritics, casing, spacing, - leading articles, parenthetical notes.""" - s = re.sub(r"\(.*?\)", "", s) # drop "(uniquement pour la partie ...)" - s = re.sub(r"^(?:Le|La|Les|L['’])\s+", "", s, flags=re.IGNORECASE) - s = unicodedata.normalize("NFKD", s).encode("ascii", "ignore").decode() - return re.sub(r"[\W_]+", "", s).lower() - - -def load_commune_index( - path: Path, -) -> tuple[dict[tuple[str, str], tuple[dict, str]], dict[str, dict], dict[str, str]]: - """Build three indexes over IGN AdminExpress communes. - - The first is `(dept_code, normalized_name) → (geometry, insee)`, used - by the legacy cahier-text resolver. The second is `insee → geometry`, - used when we resolve communes via the INAO authoritative aires CSV - (which gives INSEE codes directly, avoiding name-fuzzy-match work). - The third is `insee → commune name`, used to render attribution - strings for cadastre lieu-dit matches. - """ - print(f"[load] {path.relative_to(ROOT)} ({path.stat().st_size // (1<<20)} MB)", file=sys.stderr) - fc = json.loads(path.read_text(encoding="utf-8")) - name_idx: dict[tuple[str, str], tuple[dict, str]] = {} - insee_idx: dict[str, dict] = {} - insee_to_name: dict[str, str] = {} - for feat in fc["features"]: - p = feat["properties"] - name_idx[(p["codeDepartement"], normalize_commune(p["nom"]))] = ( - feat["geometry"], p["code"], - ) - insee_idx[p["code"]] = feat["geometry"] - insee_to_name[p["code"]] = p["nom"] - return name_idx, insee_idx, insee_to_name - - def _join_set(values: list[str]) -> str: """Encode a slug list as ';value1;value2;' for MapLibre `in` filtering.""" if not values: @@ -407,71 +327,6 @@ def _join_set(values: list[str]) -> str: return ";" + ";".join(values) + ";" -def cahier_insee(record: dict, commune_idx: dict) -> set[str]: - """Resolve the cahier-extracted commune list to INSEE codes. - - Used as a hint for `lookup_aire` to disambiguate aires-CSV name - collisions (Valençay wine vs chèvre): the wine cahier's commune set - overlaps the wine IDA strongly and the cheese IDA barely. Returns - an empty set when the cahier didn't list communes (Champagne and - similar legal-deferred AOCs). - """ - out: set[str] = set() - by_dept = record.get("aire", {}).get("aire_geographique", {}) or {} - for dept_name, communes in by_dept.items(): - dept_code = DEPT_NAME_TO_CODE.get(dept_name.replace("’", "'")) - if not dept_code: - continue - for commune in communes: - hit = commune_idx.get((dept_code, normalize_commune(commune))) - if hit is not None: - out.add(hit[1]) - return out - - -def union_for_appellation(record: dict, commune_idx: dict) -> tuple[object | None, dict]: - """Resolve commune names → polygons → union (cahier-text path). - - Used as a fallback when neither parcellaire nor INAO aires CSV give - us a direct geometry/INSEE list for this appellation. - """ - matched = unmatched = 0 - geoms = [] - by_dept = record["aire"]["aire_geographique"] - for dept_name, communes in by_dept.items(): - dept_code = DEPT_NAME_TO_CODE.get(dept_name.replace("’", "'")) - if not dept_code: - unmatched += len(communes) - continue - for commune in communes: - key = (dept_code, normalize_commune(commune)) - hit = commune_idx.get(key) - if hit is None: - unmatched += 1 - continue - geoms.append(shape(hit[0])) - matched += 1 - if not geoms: - return None, {"matched": matched, "unmatched": unmatched} - return unary_union(geoms), {"matched": matched, "unmatched": unmatched} - - -def union_from_insee(insee_codes: set[str], insee_idx: dict[str, dict]) -> tuple[object | None, dict]: - """Resolve INSEE codes (from the INAO aires CSV) → polygons → union.""" - matched = unmatched = 0 - geoms = [] - for code in insee_codes: - geom = insee_idx.get(code) - if geom is None: - unmatched += 1 - continue - geoms.append(shape(geom)) - matched += 1 - if not geoms: - return None, {"matched": matched, "unmatched": unmatched} - return unary_union(geoms), {"matched": matched, "unmatched": unmatched} - - def _geojson_bounds(g: dict) -> tuple[float, float, float, float]: """Pure-Python bbox over a GeoJSON geometry — avoids parsing into a shapely object just to read its envelope. Used for cheap bbox @@ -518,161 +373,6 @@ def communes_containing(needle, insee_idx: dict[str, dict]) -> set[str]: return out -def _find_sibling_umbrella( - name: str, - siblings: list[tuple[str, object, object, str]] | None, -) -> tuple[object | None, object | None, str, str]: - """Find the longest sibling DGC whose name strictly prefixes `name`. - - Returns (geom, village_geom, sibling_name, sibling_slug) — all blank - when no match. Used to walk a Chablis premier cru lieu-dit up to the - "Chablis premier cru" umbrella DGC's polygon, instead of the entire - Chablis appellation. - """ - if not siblings: - return None, None, "", "" - best: tuple[object, object, str, str] | None = None - best_len = 0 - for sib_name, sib_geom, sib_v_geom, sib_slug in siblings: - prefix = sib_name + " " - if name.startswith(prefix) and len(sib_name) > best_len: - best = (sib_geom, sib_v_geom, sib_name, sib_slug) - best_len = len(sib_name) - if best is None: - return None, None, "", "" - return best - - -@dataclass -class DGCGeomResult: - """Outcome of `resolve_dgc_geometry()`. `source` is the wining strategy's - label (`parcellaire-dgc`, `dgc-village-override`, `cadastre-lieu-dit-dgc`, - `aires-csv-dgc`, `sibling-dgc`, `parent-appellation`, or `none`). - `sib_*` and `cadastre_match` are populated only by their respective - strategies; downstream consumers (v_geom resolution and MVT - fallback_*/cadastre_* properties) read them keyed on `source`.""" - geom: object | None - source: str - stats: dict - sib_v_geom: object | None = None - sib_name: str = "" - sib_slug: str = "" - cadastre_match: dict | None = None - - -def resolve_dgc_geometry( - record: dict, - *, - parcels_by_denom: dict, - aires_by_app: dict, - insee_idx: dict, - commune_idx: dict, - lieu_dit_index: LieuDitIndex, - parent_geom_by_slug: dict, - sibling_geom_by_id_app: dict, -) -> DGCGeomResult: - """Resolve a DGC's detailed geometry by walking the priority chain: - - 1. parcellaire-dgc — parcel-precise polygon keyed on id_denomination_geo - 2. dgc-village-override — hand-curated DGC_VILLAGE_INSEE table - 3. cadastre-lieu-dit-dgc — sub-commune climat in cadastre lieux-dits - (Chablis premier-cru climats, Givry / Santenay premier cru, …) - 4. aires-csv-dgc — DGC's own row in INAO aires-communes CSV - 5. sibling-dgc — longest-prefix sibling DGC umbrella (Chablis premier - cru X → "Chablis premier cru" umbrella, not whole Chablis) - 6. parent-appellation — inherit parent's polygon - 7. none — no geometry available - - Returns the first matching strategy's result; later strategies are - not evaluated. Adding a fallback = inserting a guarded block at the - right priority slot. Behavior matches the original 7-level cascade - in main() exactly. - """ - id_denom = record.get("id_denomination_geo") or "" - parent_name = record.get("parent_name") or "" - cahier_hint = cahier_insee(record, commune_idx) - siblings = sibling_geom_by_id_app.get(record.get("id_appellation")) - sib_geom, sib_v_geom, sib_name, sib_slug = _find_sibling_umbrella(record["name"], siblings) - - # 1. parcellaire-dgc - parcel_feat = parcels_by_denom.get(id_denom) if id_denom else None - if parcel_feat is not None: - return DGCGeomResult( - geom=shape(parcel_feat["geometry"]), - source="parcellaire-dgc", - stats={"matched": -1, "unmatched": 0}, - ) - - # 2. dgc-village-override - override_insee = DGC_VILLAGE_INSEE.get(id_denom) - if override_insee: - geom, stats = union_from_insee(override_insee, insee_idx) - if geom is not None and not geom.is_empty: - return DGCGeomResult(geom=geom, source="dgc-village-override", stats=stats) - - parent_aires_insee = ( - lookup_aire(aires_by_app, parent_name, cahier_hint) if parent_name else None - ) - - # 3. cadastre-lieu-dit-dgc — sub-commune climat resolution. Strip the - # parent / sibling-umbrella prefix before matching so "Chablis premier - # cru Vaillons" looks up as "Vaillons" inside Chablis. - climat_name = derive_climat_name( - record["name"], parent_name=parent_name, umbrella_name=sib_name, - ) - cadastre_match = lieu_dit_index.resolve( - climat_name, parent_aires_insee, id_denom=id_denom, - ) - if cadastre_match is not None: - return DGCGeomResult( - geom=cadastre_match["geom"], - source="cadastre-lieu-dit-dgc", - stats={"matched": -1, "unmatched": 0}, - cadastre_match=cadastre_match, - ) - - # 4. aires-csv-dgc — but first drop substring matches that round-trip - # to the parent or to the sibling umbrella (those are not informative; - # they signal the DGC has no row of its own and the lookup_aire - # len≥6 fallback latched onto a containing row instead). - dgc_aires_insee = lookup_aire(aires_by_app, record["name"], cahier_hint) - if dgc_aires_insee and parent_aires_insee == dgc_aires_insee: - dgc_aires_insee = None - if dgc_aires_insee and sib_name: - sib_aires_insee = lookup_aire(aires_by_app, sib_name, cahier_hint) - if sib_aires_insee == dgc_aires_insee: - dgc_aires_insee = None - if dgc_aires_insee: - geom, stats = union_from_insee(dgc_aires_insee, insee_idx) - if geom is not None and not geom.is_empty: - return DGCGeomResult(geom=geom, source="aires-csv-dgc", stats=stats) - - # 5. sibling-dgc — umbrella DGC's polygon, when one exists. - if sib_geom is not None: - return DGCGeomResult( - geom=sib_geom, - source="sibling-dgc", - stats={"matched": -1, "unmatched": 0}, - sib_v_geom=sib_v_geom, - sib_name=sib_name, - sib_slug=sib_slug, - ) - - # 6. parent-appellation — inherit. Parents are processed before DGCs, - # so parent_geom_by_slug already holds it. - parent_slug = record.get("parent_slug") or "" - parent_geom = parent_geom_by_slug.get(parent_slug) - if parent_geom is not None: - return DGCGeomResult( - geom=parent_geom, - source="parent-appellation", - stats={"matched": -1, "unmatched": 0}, - ) - - # 7. none - return DGCGeomResult(geom=None, source="none", stats={"matched": 0, "unmatched": 0}) - - def main() -> int: if not COMMUNES_GEOJSON.exists(): print("error: IGN communes geojson missing — run scripts/00_fetch_data.py", file=sys.stderr) @@ -2730,182 +2430,6 @@ def main() -> int: return 0 -def _resolve_es_igp_fallback(record: dict, es_polygons: ESPolygonIndex): - """Resolve geometry for ES wines that miss Figshare (mostly IGPs + - a handful of post-Nov-2021 PDOs). Patterns tried in order: - - 1. **Province-wide** — pliego says "todos los términos municipales - de las provincias de X y Y" (Extremadura). Union all GISCO - municipios in those provinces. - 2. **CCAA-wide** — "totalidad de los municipios de la Comunidad - Autónoma de Castilla y León". Union all province INEs of that - CCAA. - 3. **Island-wide** — "toda la isla de Mallorca" (Balearic IGPs - like Mallorca / Menorca / Serra de Tramuntana). - 4. **Commune-list** — pliego enumerates a flat commune list - (Ribeiras do Morrazo, Barbanza e Iria, Bajo Aragón). Union the - matching GISCO municipios. - - Each pattern is tried against a chain of candidate texts: - `geo_area_brief` first (the canonical stage-02 routed field), then - `sections["9"]` (the EU 2024 single-document "Definición breve de - la zona geográfica delimitada" section — sometimes mis-routed by - stage 02 when section titles collide, e.g. Mallorca / Ribeiras do - Morrazo). - - LAST RESORT — **wine-name → province**: when nothing else fires, - look up the wine's `name` (and the bracketed-form fallback strip) - against `PROVINCE_TO_INE`. Spanish-national-format pliegos for - province-named IGPs (Castelló) sometimes describe the geographic - area in pure prose without listing communes or saying "todos los - municipios de la provincia" — the IGP-covers-the-whole-province - relationship is implicit in the name. The wine's name must match - a known Spanish province (or co-official alias) exactly. - - Returns (geom, source_label, stats) or (None, "none", {}) when - nothing fires.""" - geo = record.get("geo_area_brief") or "" - sec9 = (record.get("sections") or {}).get("9") or "" - - # Try the routed field first, then section 9. Stop on the first - # candidate that actually returns a non-empty polygon. - candidates = [c for c in (geo, sec9) if c] - if not candidates: - slug = record.get("slug", "?") - print(f"[no-commune-match] {slug}: empty geo_area_brief and section 9", file=sys.stderr) - return None, "none", {"matched": 0, "unmatched": 0} - - for text in candidates: - provinces = parse_province_wide_list(text) - if provinces: - ines = [PROVINCE_TO_INE.get(p) for p in provinces if PROVINCE_TO_INE.get(p)] - if ines: - geom, stats = es_polygons.union_provinces(ines) - if geom is not None and not geom.is_empty: - return geom, "gisco-province-wide", { - "matched": stats.get("n_municipios", -1), "unmatched": 0, - } - - ccaa = parse_ccaa_wide(text) - if ccaa: - ines = list(CCAA_TO_PROVINCE_INES.get(ccaa, ())) - if ines: - geom, stats = es_polygons.union_provinces(ines) - if geom is not None and not geom.is_empty: - return geom, "gisco-ccaa-wide", { - "matched": stats.get("n_municipios", -1), "unmatched": 0, - } - - # Balearic islands: pliego says "toda la isla de Mallorca" / - # "todos los municipios de la isla de Menorca" / etc. GISCO LAU - # has no per-island metadata, so we lean on the curated INE-list- - # per-island in `_lib/es/baleares.py` (bbox-classified once from - # the LAU geometry). - island = parse_island_wide(text) - if island: - island_ines = list(ines_for_island(island)) - if island_ines: - polys = [] - for ine in island_ines: - cand = es_polygons._munis_by_ine.get(ine) - if cand and not cand.geom.is_empty: - polys.append(cand.geom) - if polys: - return unary_union(polys), "gisco-island-wide", { - "matched": len(polys), "unmatched": 0, - } - - communes = parse_commune_list(text) - if communes: - geom, stats = es_polygons.union_communes(communes) - if geom is not None and not geom.is_empty: - return geom, "gisco-commune-list", stats - - # Last resort: wine-name → province. The IGP/DOP covers the whole - # province by name (Castelló = Castellón province). Matched against - # PROVINCE_TO_INE's full alias list (Spanish + co-official forms). - name = (record.get("name") or "").strip() - ine = PROVINCE_TO_INE.get(name) - if ine: - geom, stats = es_polygons.union_provinces([ine]) - if geom is not None and not geom.is_empty: - slug = record.get("slug", "?") - print( - f"[gisco-province-by-name] {slug}: name={name!r} → INE {ine} " - f"(no commune list anywhere; province-wide by name)", - file=sys.stderr, - ) - return geom, "gisco-province-by-name", { - "matched": stats.get("n_municipios", -1), "unmatched": 0, - } - - slug = record.get("slug", "?") - print( - f"[no-commune-match] {slug}: " - f"geo_area_brief={len(geo)} chars, section9={len(sec9)} chars, " - f"no province/ccaa/island/commune-list/name-province pattern fired", - file=sys.stderr, - ) - return None, "none", {"matched": 0, "unmatched": 0} - - -def _resolve_es_sigpac( - record: dict, sigpac: SigpacIndex, es_polygons: ESPolygonIndex, -): - """Hybrid SIGPAC + GISCO whole-commune resolver for ES wine records - that have polygon-list inclusions in their pliego. - - The hybrid is **only invoked when polygon-list inclusions exist** - (Priorat / Montsant pattern: pliego enumerates SIGPAC polygon - numbers within shared communes). Wines without polygon-list - inclusions fall through to Figshare which is more reliable for the - PDO commune-precision polygon — running our whole-commune-prefix - parser unconditionally would over-trigger on noisy text (Rioja's - subzona ALL-CAPS headers parsed as commune names, etc.). - - When polygon-list inclusions ARE present, two passes union into one - appellation footprint: - - 1. **Whole-commune prefix** (the 9 fully-included Priorat - communes / 12 fully-included Montsant communes) → union of - GISCO LAU commune polygons. - 2. **Polygon-list inclusions** (Falset: polígonos 1, 4, 5, 6, 7, - 21, 25 enteros) → union of SIGPAC vineyard parcels at - polygon-precision. - - Returns None when there are no polygon-list inclusions or when - SIGPAC isn't loaded for the relevant comarca.""" - if not sigpac.n_comarques: - return None - geo = record.get("geo_area_brief") or "" - if not geo: - return None - - inclusions = parse_polygon_inclusions(geo) - if not inclusions: - return None - - polys = [] - - # Whole-commune prefix → GISCO union (supplements polygon-list when - # the pliego mixes both patterns). - whole_communes = parse_whole_commune_prefix(geo) - if whole_communes: - gc_geom, _ = es_polygons.union_communes(whole_communes) - if gc_geom is not None and not gc_geom.is_empty: - polys.append(gc_geom) - - # Polygon-list inclusions → SIGPAC union - for inc in inclusions: - g = sigpac.polygons_in_municipi(inc.municipio_norm, inc.polygon_numbers) - if g is not None and not g.is_empty: - polys.append(g) - - if not polys: - return None - return unary_union(polys) - - def _sources_for(record: dict) -> dict: """Pull authoritative source URLs from the extracted record. The keys are country-specific: FR records carry BO Agri / show_texte / product diff --git a/scripts/_lib/geom_chain.py b/scripts/_lib/geom_chain.py new file mode 100644 index 0000000..71a1bd9 --- /dev/null +++ b/scripts/_lib/geom_chain.py @@ -0,0 +1,505 @@ +"""Commune-index + DGC/ES geometry resolution chain (stage 04). + +Moved verbatim out of 04_build_maps.py — no behaviour change. Self-contained: +no stage-04 function dependencies. The names the main build loop calls +(union_from_insee, DGCGeomResult, resolve_dgc_geometry, union_for_appellation, +cahier_insee, load_commune_index, normalize_commune, _find_sibling_umbrella, +_resolve_es_igp_fallback, _resolve_es_sigpac, DEPT_NAME_TO_CODE) are imported +back into stage 04. +""" +from __future__ import annotations + +import json +import re +import sys +import unicodedata +from dataclasses import dataclass +from pathlib import Path + +from shapely.geometry import shape +from shapely.ops import unary_union + +from _lib.aires import lookup as lookup_aire +from _lib.dgc_village_overrides import DGC_VILLAGE_INSEE +from _lib.es.baleares import ines_for_island +from _lib.es.commune_list import ( + parse_ccaa_wide, + parse_commune_list, + parse_island_wide, + parse_province_wide_list, + parse_whole_commune_prefix, +) +from _lib.es.geometry import ESPolygonIndex +from _lib.es.pliego_parcels import parse_polygon_inclusions +from _lib.es.region import CCAA_TO_PROVINCE_INES, PROVINCE_TO_INE +from _lib.es.sigpac import SigpacIndex +from _lib.lieu_dit import LieuDitIndex, derive_climat_name + +ROOT = Path(__file__).resolve().parents[2] + + +# INSEE 2-digit département code → canonical name as written in cahiers. +# Used for resolving "Côte-d'Or" → "21" so commune lookup stays inside the +# correct département (avoids Saint-Pierre homonym collisions). +DEPT_NAME_TO_CODE: dict[str, str] = { + "Ain": "01", "Aisne": "02", "Allier": "03", "Alpes-de-Haute-Provence": "04", + "Hautes-Alpes": "05", "Alpes-Maritimes": "06", "Ardèche": "07", "Ardennes": "08", + "Ariège": "09", "Aube": "10", "Aude": "11", "Aveyron": "12", + "Bouches-du-Rhône": "13", "Calvados": "14", "Cantal": "15", "Charente": "16", + "Charente-Maritime": "17", "Cher": "18", "Corrèze": "19", "Corse-du-Sud": "2A", + "Haute-Corse": "2B", "Côte-d'Or": "21", "Côte-d’Or": "21", + "Côtes-d'Armor": "22", "Côtes-d’Armor": "22", "Creuse": "23", + "Dordogne": "24", "Doubs": "25", "Drôme": "26", "Eure": "27", "Eure-et-Loir": "28", + "Finistère": "29", "Gard": "30", "Haute-Garonne": "31", "Gers": "32", + "Gironde": "33", "Hérault": "34", "Ille-et-Vilaine": "35", "Indre": "36", + "Indre-et-Loire": "37", "Isère": "38", "Jura": "39", "Landes": "40", + "Loir-et-Cher": "41", "Loire": "42", "Haute-Loire": "43", "Loire-Atlantique": "44", + "Loiret": "45", "Lot": "46", "Lot-et-Garonne": "47", "Lozère": "48", + "Maine-et-Loire": "49", "Manche": "50", "Marne": "51", "Haute-Marne": "52", + "Mayenne": "53", "Meurthe-et-Moselle": "54", "Meuse": "55", "Morbihan": "56", + "Moselle": "57", "Nièvre": "58", "Nord": "59", "Oise": "60", "Orne": "61", + "Pas-de-Calais": "62", "Puy-de-Dôme": "63", "Pyrénées-Atlantiques": "64", + "Hautes-Pyrénées": "65", "Pyrénées-Orientales": "66", "Bas-Rhin": "67", + "Haut-Rhin": "68", "Rhône": "69", "Haute-Saône": "70", "Saône-et-Loire": "71", + "Sarthe": "72", "Savoie": "73", "Haute-Savoie": "74", "Paris": "75", + "Seine-Maritime": "76", "Seine-et-Marne": "77", "Yvelines": "78", + "Deux-Sèvres": "79", "Somme": "80", "Tarn": "81", "Tarn-et-Garonne": "82", + "Var": "83", "Vaucluse": "84", "Vendée": "85", "Vienne": "86", + "Haute-Vienne": "87", "Vosges": "88", "Yonne": "89", "Territoire de Belfort": "90", + "Essonne": "91", "Hauts-de-Seine": "92", "Seine-Saint-Denis": "93", + "Val-de-Marne": "94", "Val-d'Oise": "95", "Val-d’Oise": "95", + "Guadeloupe": "971", "Martinique": "972", "Guyane": "973", + "La Réunion": "974", "Mayotte": "976", +} + + +def normalize_commune(s: str) -> str: + """Loose match key for commune names — strip diacritics, casing, spacing, + leading articles, parenthetical notes.""" + s = re.sub(r"\(.*?\)", "", s) # drop "(uniquement pour la partie ...)" + s = re.sub(r"^(?:Le|La|Les|L['’])\s+", "", s, flags=re.IGNORECASE) + s = unicodedata.normalize("NFKD", s).encode("ascii", "ignore").decode() + return re.sub(r"[\W_]+", "", s).lower() + + +def load_commune_index( + path: Path, +) -> tuple[dict[tuple[str, str], tuple[dict, str]], dict[str, dict], dict[str, str]]: + """Build three indexes over IGN AdminExpress communes. + + The first is `(dept_code, normalized_name) → (geometry, insee)`, used + by the legacy cahier-text resolver. The second is `insee → geometry`, + used when we resolve communes via the INAO authoritative aires CSV + (which gives INSEE codes directly, avoiding name-fuzzy-match work). + The third is `insee → commune name`, used to render attribution + strings for cadastre lieu-dit matches. + """ + print(f"[load] {path.relative_to(ROOT)} ({path.stat().st_size // (1<<20)} MB)", file=sys.stderr) + fc = json.loads(path.read_text(encoding="utf-8")) + name_idx: dict[tuple[str, str], tuple[dict, str]] = {} + insee_idx: dict[str, dict] = {} + insee_to_name: dict[str, str] = {} + for feat in fc["features"]: + p = feat["properties"] + name_idx[(p["codeDepartement"], normalize_commune(p["nom"]))] = ( + feat["geometry"], p["code"], + ) + insee_idx[p["code"]] = feat["geometry"] + insee_to_name[p["code"]] = p["nom"] + return name_idx, insee_idx, insee_to_name + + +def cahier_insee(record: dict, commune_idx: dict) -> set[str]: + """Resolve the cahier-extracted commune list to INSEE codes. + + Used as a hint for `lookup_aire` to disambiguate aires-CSV name + collisions (Valençay wine vs chèvre): the wine cahier's commune set + overlaps the wine IDA strongly and the cheese IDA barely. Returns + an empty set when the cahier didn't list communes (Champagne and + similar legal-deferred AOCs). + """ + out: set[str] = set() + by_dept = record.get("aire", {}).get("aire_geographique", {}) or {} + for dept_name, communes in by_dept.items(): + dept_code = DEPT_NAME_TO_CODE.get(dept_name.replace("’", "'")) + if not dept_code: + continue + for commune in communes: + hit = commune_idx.get((dept_code, normalize_commune(commune))) + if hit is not None: + out.add(hit[1]) + return out + + +def union_for_appellation(record: dict, commune_idx: dict) -> tuple[object | None, dict]: + """Resolve commune names → polygons → union (cahier-text path). + + Used as a fallback when neither parcellaire nor INAO aires CSV give + us a direct geometry/INSEE list for this appellation. + """ + matched = unmatched = 0 + geoms = [] + by_dept = record["aire"]["aire_geographique"] + for dept_name, communes in by_dept.items(): + dept_code = DEPT_NAME_TO_CODE.get(dept_name.replace("’", "'")) + if not dept_code: + unmatched += len(communes) + continue + for commune in communes: + key = (dept_code, normalize_commune(commune)) + hit = commune_idx.get(key) + if hit is None: + unmatched += 1 + continue + geoms.append(shape(hit[0])) + matched += 1 + if not geoms: + return None, {"matched": matched, "unmatched": unmatched} + return unary_union(geoms), {"matched": matched, "unmatched": unmatched} + + +def union_from_insee(insee_codes: set[str], insee_idx: dict[str, dict]) -> tuple[object | None, dict]: + """Resolve INSEE codes (from the INAO aires CSV) → polygons → union.""" + matched = unmatched = 0 + geoms = [] + for code in insee_codes: + geom = insee_idx.get(code) + if geom is None: + unmatched += 1 + continue + geoms.append(shape(geom)) + matched += 1 + if not geoms: + return None, {"matched": matched, "unmatched": unmatched} + return unary_union(geoms), {"matched": matched, "unmatched": unmatched} + + +def _find_sibling_umbrella( + name: str, + siblings: list[tuple[str, object, object, str]] | None, +) -> tuple[object | None, object | None, str, str]: + """Find the longest sibling DGC whose name strictly prefixes `name`. + + Returns (geom, village_geom, sibling_name, sibling_slug) — all blank + when no match. Used to walk a Chablis premier cru lieu-dit up to the + "Chablis premier cru" umbrella DGC's polygon, instead of the entire + Chablis appellation. + """ + if not siblings: + return None, None, "", "" + best: tuple[object, object, str, str] | None = None + best_len = 0 + for sib_name, sib_geom, sib_v_geom, sib_slug in siblings: + prefix = sib_name + " " + if name.startswith(prefix) and len(sib_name) > best_len: + best = (sib_geom, sib_v_geom, sib_name, sib_slug) + best_len = len(sib_name) + if best is None: + return None, None, "", "" + return best + + +@dataclass +class DGCGeomResult: + """Outcome of `resolve_dgc_geometry()`. `source` is the wining strategy's + label (`parcellaire-dgc`, `dgc-village-override`, `cadastre-lieu-dit-dgc`, + `aires-csv-dgc`, `sibling-dgc`, `parent-appellation`, or `none`). + `sib_*` and `cadastre_match` are populated only by their respective + strategies; downstream consumers (v_geom resolution and MVT + fallback_*/cadastre_* properties) read them keyed on `source`.""" + geom: object | None + source: str + stats: dict + sib_v_geom: object | None = None + sib_name: str = "" + sib_slug: str = "" + cadastre_match: dict | None = None + + +def resolve_dgc_geometry( + record: dict, + *, + parcels_by_denom: dict, + aires_by_app: dict, + insee_idx: dict, + commune_idx: dict, + lieu_dit_index: LieuDitIndex, + parent_geom_by_slug: dict, + sibling_geom_by_id_app: dict, +) -> DGCGeomResult: + """Resolve a DGC's detailed geometry by walking the priority chain: + + 1. parcellaire-dgc — parcel-precise polygon keyed on id_denomination_geo + 2. dgc-village-override — hand-curated DGC_VILLAGE_INSEE table + 3. cadastre-lieu-dit-dgc — sub-commune climat in cadastre lieux-dits + (Chablis premier-cru climats, Givry / Santenay premier cru, …) + 4. aires-csv-dgc — DGC's own row in INAO aires-communes CSV + 5. sibling-dgc — longest-prefix sibling DGC umbrella (Chablis premier + cru X → "Chablis premier cru" umbrella, not whole Chablis) + 6. parent-appellation — inherit parent's polygon + 7. none — no geometry available + + Returns the first matching strategy's result; later strategies are + not evaluated. Adding a fallback = inserting a guarded block at the + right priority slot. Behavior matches the original 7-level cascade + in main() exactly. + """ + id_denom = record.get("id_denomination_geo") or "" + parent_name = record.get("parent_name") or "" + cahier_hint = cahier_insee(record, commune_idx) + siblings = sibling_geom_by_id_app.get(record.get("id_appellation")) + sib_geom, sib_v_geom, sib_name, sib_slug = _find_sibling_umbrella(record["name"], siblings) + + # 1. parcellaire-dgc + parcel_feat = parcels_by_denom.get(id_denom) if id_denom else None + if parcel_feat is not None: + return DGCGeomResult( + geom=shape(parcel_feat["geometry"]), + source="parcellaire-dgc", + stats={"matched": -1, "unmatched": 0}, + ) + + # 2. dgc-village-override + override_insee = DGC_VILLAGE_INSEE.get(id_denom) + if override_insee: + geom, stats = union_from_insee(override_insee, insee_idx) + if geom is not None and not geom.is_empty: + return DGCGeomResult(geom=geom, source="dgc-village-override", stats=stats) + + parent_aires_insee = ( + lookup_aire(aires_by_app, parent_name, cahier_hint) if parent_name else None + ) + + # 3. cadastre-lieu-dit-dgc — sub-commune climat resolution. Strip the + # parent / sibling-umbrella prefix before matching so "Chablis premier + # cru Vaillons" looks up as "Vaillons" inside Chablis. + climat_name = derive_climat_name( + record["name"], parent_name=parent_name, umbrella_name=sib_name, + ) + cadastre_match = lieu_dit_index.resolve( + climat_name, parent_aires_insee, id_denom=id_denom, + ) + if cadastre_match is not None: + return DGCGeomResult( + geom=cadastre_match["geom"], + source="cadastre-lieu-dit-dgc", + stats={"matched": -1, "unmatched": 0}, + cadastre_match=cadastre_match, + ) + + # 4. aires-csv-dgc — but first drop substring matches that round-trip + # to the parent or to the sibling umbrella (those are not informative; + # they signal the DGC has no row of its own and the lookup_aire + # len≥6 fallback latched onto a containing row instead). + dgc_aires_insee = lookup_aire(aires_by_app, record["name"], cahier_hint) + if dgc_aires_insee and parent_aires_insee == dgc_aires_insee: + dgc_aires_insee = None + if dgc_aires_insee and sib_name: + sib_aires_insee = lookup_aire(aires_by_app, sib_name, cahier_hint) + if sib_aires_insee == dgc_aires_insee: + dgc_aires_insee = None + if dgc_aires_insee: + geom, stats = union_from_insee(dgc_aires_insee, insee_idx) + if geom is not None and not geom.is_empty: + return DGCGeomResult(geom=geom, source="aires-csv-dgc", stats=stats) + + # 5. sibling-dgc — umbrella DGC's polygon, when one exists. + if sib_geom is not None: + return DGCGeomResult( + geom=sib_geom, + source="sibling-dgc", + stats={"matched": -1, "unmatched": 0}, + sib_v_geom=sib_v_geom, + sib_name=sib_name, + sib_slug=sib_slug, + ) + + # 6. parent-appellation — inherit. Parents are processed before DGCs, + # so parent_geom_by_slug already holds it. + parent_slug = record.get("parent_slug") or "" + parent_geom = parent_geom_by_slug.get(parent_slug) + if parent_geom is not None: + return DGCGeomResult( + geom=parent_geom, + source="parent-appellation", + stats={"matched": -1, "unmatched": 0}, + ) + + # 7. none + return DGCGeomResult(geom=None, source="none", stats={"matched": 0, "unmatched": 0}) + + +def _resolve_es_igp_fallback(record: dict, es_polygons: ESPolygonIndex): + """Resolve geometry for ES wines that miss Figshare (mostly IGPs + + a handful of post-Nov-2021 PDOs). Patterns tried in order: + + 1. **Province-wide** — pliego says "todos los términos municipales + de las provincias de X y Y" (Extremadura). Union all GISCO + municipios in those provinces. + 2. **CCAA-wide** — "totalidad de los municipios de la Comunidad + Autónoma de Castilla y León". Union all province INEs of that + CCAA. + 3. **Island-wide** — "toda la isla de Mallorca" (Balearic IGPs + like Mallorca / Menorca / Serra de Tramuntana). + 4. **Commune-list** — pliego enumerates a flat commune list + (Ribeiras do Morrazo, Barbanza e Iria, Bajo Aragón). Union the + matching GISCO municipios. + + Each pattern is tried against a chain of candidate texts: + `geo_area_brief` first (the canonical stage-02 routed field), then + `sections["9"]` (the EU 2024 single-document "Definición breve de + la zona geográfica delimitada" section — sometimes mis-routed by + stage 02 when section titles collide, e.g. Mallorca / Ribeiras do + Morrazo). + + LAST RESORT — **wine-name → province**: when nothing else fires, + look up the wine's `name` (and the bracketed-form fallback strip) + against `PROVINCE_TO_INE`. Spanish-national-format pliegos for + province-named IGPs (Castelló) sometimes describe the geographic + area in pure prose without listing communes or saying "todos los + municipios de la provincia" — the IGP-covers-the-whole-province + relationship is implicit in the name. The wine's name must match + a known Spanish province (or co-official alias) exactly. + + Returns (geom, source_label, stats) or (None, "none", {}) when + nothing fires.""" + geo = record.get("geo_area_brief") or "" + sec9 = (record.get("sections") or {}).get("9") or "" + + # Try the routed field first, then section 9. Stop on the first + # candidate that actually returns a non-empty polygon. + candidates = [c for c in (geo, sec9) if c] + if not candidates: + slug = record.get("slug", "?") + print(f"[no-commune-match] {slug}: empty geo_area_brief and section 9", file=sys.stderr) + return None, "none", {"matched": 0, "unmatched": 0} + + for text in candidates: + provinces = parse_province_wide_list(text) + if provinces: + ines = [PROVINCE_TO_INE.get(p) for p in provinces if PROVINCE_TO_INE.get(p)] + if ines: + geom, stats = es_polygons.union_provinces(ines) + if geom is not None and not geom.is_empty: + return geom, "gisco-province-wide", { + "matched": stats.get("n_municipios", -1), "unmatched": 0, + } + + ccaa = parse_ccaa_wide(text) + if ccaa: + ines = list(CCAA_TO_PROVINCE_INES.get(ccaa, ())) + if ines: + geom, stats = es_polygons.union_provinces(ines) + if geom is not None and not geom.is_empty: + return geom, "gisco-ccaa-wide", { + "matched": stats.get("n_municipios", -1), "unmatched": 0, + } + + # Balearic islands: pliego says "toda la isla de Mallorca" / + # "todos los municipios de la isla de Menorca" / etc. GISCO LAU + # has no per-island metadata, so we lean on the curated INE-list- + # per-island in `_lib/es/baleares.py` (bbox-classified once from + # the LAU geometry). + island = parse_island_wide(text) + if island: + island_ines = list(ines_for_island(island)) + if island_ines: + polys = [] + for ine in island_ines: + cand = es_polygons._munis_by_ine.get(ine) + if cand and not cand.geom.is_empty: + polys.append(cand.geom) + if polys: + return unary_union(polys), "gisco-island-wide", { + "matched": len(polys), "unmatched": 0, + } + + communes = parse_commune_list(text) + if communes: + geom, stats = es_polygons.union_communes(communes) + if geom is not None and not geom.is_empty: + return geom, "gisco-commune-list", stats + + # Last resort: wine-name → province. The IGP/DOP covers the whole + # province by name (Castelló = Castellón province). Matched against + # PROVINCE_TO_INE's full alias list (Spanish + co-official forms). + name = (record.get("name") or "").strip() + ine = PROVINCE_TO_INE.get(name) + if ine: + geom, stats = es_polygons.union_provinces([ine]) + if geom is not None and not geom.is_empty: + slug = record.get("slug", "?") + print( + f"[gisco-province-by-name] {slug}: name={name!r} → INE {ine} " + f"(no commune list anywhere; province-wide by name)", + file=sys.stderr, + ) + return geom, "gisco-province-by-name", { + "matched": stats.get("n_municipios", -1), "unmatched": 0, + } + + slug = record.get("slug", "?") + print( + f"[no-commune-match] {slug}: " + f"geo_area_brief={len(geo)} chars, section9={len(sec9)} chars, " + f"no province/ccaa/island/commune-list/name-province pattern fired", + file=sys.stderr, + ) + return None, "none", {"matched": 0, "unmatched": 0} + + +def _resolve_es_sigpac( + record: dict, sigpac: SigpacIndex, es_polygons: ESPolygonIndex, +): + """Hybrid SIGPAC + GISCO whole-commune resolver for ES wine records + that have polygon-list inclusions in their pliego. + + The hybrid is **only invoked when polygon-list inclusions exist** + (Priorat / Montsant pattern: pliego enumerates SIGPAC polygon + numbers within shared communes). Wines without polygon-list + inclusions fall through to Figshare which is more reliable for the + PDO commune-precision polygon — running our whole-commune-prefix + parser unconditionally would over-trigger on noisy text (Rioja's + subzona ALL-CAPS headers parsed as commune names, etc.). + + When polygon-list inclusions ARE present, two passes union into one + appellation footprint: + + 1. **Whole-commune prefix** (the 9 fully-included Priorat + communes / 12 fully-included Montsant communes) → union of + GISCO LAU commune polygons. + 2. **Polygon-list inclusions** (Falset: polígonos 1, 4, 5, 6, 7, + 21, 25 enteros) → union of SIGPAC vineyard parcels at + polygon-precision. + + Returns None when there are no polygon-list inclusions or when + SIGPAC isn't loaded for the relevant comarca.""" + if not sigpac.n_comarques: + return None + geo = record.get("geo_area_brief") or "" + if not geo: + return None + + inclusions = parse_polygon_inclusions(geo) + if not inclusions: + return None + + polys = [] + + # Whole-commune prefix → GISCO union (supplements polygon-list when + # the pliego mixes both patterns). + whole_communes = parse_whole_commune_prefix(geo) + if whole_communes: + gc_geom, _ = es_polygons.union_communes(whole_communes) + if gc_geom is not None and not gc_geom.is_empty: + polys.append(gc_geom) + + # Polygon-list inclusions → SIGPAC union + for inc in inclusions: + g = sigpac.polygons_in_municipi(inc.municipio_norm, inc.polygon_numbers) + if g is not None and not g.is_empty: + polys.append(g) + + if not polys: + return None + return unary_union(polys) From 1c0a540084e89f1b857dc5ae80d44be343a1d2c1 Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Thu, 11 Jun 2026 12:53:32 +0200 Subject: [PATCH 32/41] docs: record stage-04 module layout (augment/lexicon/geom_chain) Co-Authored-By: Claude Opus 4.8 (1M context) --- CLAUDE.md | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/CLAUDE.md b/CLAUDE.md index b7b6c12..80dc96e 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -4220,6 +4220,19 @@ author / editorial dates for a generated page). - Python 3.12, ruff line length 100. - Single-purpose scripts; share helpers via `scripts/_lib/`. +- **Stage 04 module layout.** `04_build_maps.py` orchestrates; the bulk of its + logic lives in `scripts/_lib/`. The per-country national-spec augmenters are + in `_lib/augment/.py` (their shared slug-keyed provenance caches + + sidecar dirs in `_lib/augment/_shared.py` — imported by both the augmenter + that writes them and `_sources_for()`/the panel-blob phase that reads them, + so the dict objects stay identical across the split). Grape/style lexicon + loading is in `_lib/lexicon_loading.py`; the commune-index + DGC/ES geometry + resolution chain (incl. the `DGCGeomResult` dataclass) is in + `_lib/geom_chain.py`. These were move-only extractions verified by the + golden comparator (see below); when moving more out of `04_build_maps.py`, + use the `stage04-extraction` skill — stage-04 failures are silent (a country + or feature just vanishes from the map), so a full-build golden diff, not + "it didn't crash", is the proof. - No comments unless the *why* is non-obvious. Identifiers carry the *what*. - Logs to stderr, structured progress (per-AOC) so reruns are debuggable. - **No silent dict-key overrides.** Python keeps the *last* value when a dict From f00c74d8fabcd46080ec45711d3fb7c545e8e655 Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Thu, 11 Jun 2026 18:45:36 +0200 Subject: [PATCH 33/41] =?UTF-8?q?feat:=20B=C3=A9tard-snapshot=20delta=20au?= =?UTF-8?q?dit=20(audit=5Fbetard=5Fdelta.py)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Flags appellations whose geometry is the Bétard 2022 figshare-pdo fallback but whose GI was registered/amended after the dataset's Nov-2021 data snapshot — the polygon may not reflect a post-snapshot boundary change (or a brand-new GI is on the map only via an incidental file-number match). Cross-references the compact aocs startup blob (geom_source) against each raw//eambrosia index's eu_protection_date/modification_date; buckets FLAGGED/REVIEWED/OK/NO-DATE like the sibling geometry audits, with a betard_delta_overrides.json whitelist and --strict / --cutoff flags. Read-only detector. v1 run: 473 Bétard-tier records → 345 OK, 128 FLAGGED, 0 NO-DATE. Phase 7.1; documented in CLAUDE.md. Co-Authored-By: Claude Opus 4.8 (1M context) --- CLAUDE.md | 20 +++ scripts/_lib/betard_delta_overrides.json | 6 + scripts/audit_betard_delta.py | 156 +++++++++++++++++++++++ 3 files changed, 182 insertions(+) create mode 100644 scripts/_lib/betard_delta_overrides.json create mode 100644 scripts/audit_betard_delta.py diff --git a/CLAUDE.md b/CLAUDE.md index 80dc96e..9c9f896 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -321,6 +321,26 @@ correcting the upstream commune list / resolver. Thresholds are CLI-configurable (`--sliver-max`, `--max-sliver-km2`, …); `--strict` exits non-zero on unreviewed slivers. +## Bétard-snapshot delta audit + +[scripts/audit_betard_delta.py](scripts/audit_betard_delta.py) flags +appellations whose geometry comes from Bétard 2022 (`geom_source` +`figshare-pdo` / `figshare-pdo-alias`) but whose GI was **registered or +amended after the dataset's data snapshot** (Nov-2021 cutoff) — those +polygons may not reflect a post-snapshot boundary change, and a brand-new GI +may only be on the map via an incidental file-number match. It reads the +compact `wiki/data/aocs.en.*.js` startup blob (geom_source is a startup +field) cross-referenced against each `raw//eambrosia/index.json`'s +`eu_protection_date` / `modification_date`, and buckets each record +**FLAGGED** / **REVIEWED** / **OK** / **NO-DATE** (the sibling-audit pattern). +Curator-confirmed-unchanged boundaries go in +[scripts/_lib/betard_delta_overrides.json](scripts/_lib/betard_delta_overrides.json) +(slug → `{reason, source}`) and report as REVIEWED. Read-only detector — +changes no geometry; a real FLAGGED finding is fixed upstream (a newer Bétard +release, a regional zone layer, or a commune-list resolver). `--strict` exits +non-zero on any unreviewed FLAGGED finding; `--cutoff` overrides the snapshot +date. + ## Page format (per-AOC pages) ``` diff --git a/scripts/_lib/betard_delta_overrides.json b/scripts/_lib/betard_delta_overrides.json new file mode 100644 index 0000000..b61f445 --- /dev/null +++ b/scripts/_lib/betard_delta_overrides.json @@ -0,0 +1,6 @@ +{ + "_example-slug": { + "reason": "Curator confirmed the post-snapshot amendment did not move the boundary (e.g. a name/variety change, not a perimeter change).", + "source": "https://public-source-url" + } +} diff --git a/scripts/audit_betard_delta.py b/scripts/audit_betard_delta.py new file mode 100644 index 0000000..2b3507f --- /dev/null +++ b/scripts/audit_betard_delta.py @@ -0,0 +1,156 @@ +#!/usr/bin/env python3 +"""Audit — Bétard-fallback geometries that may post-date the source snapshot. + +Every appellation whose resolved geometry comes from the Bétard 2022 +`EU_PDO.gpkg` (geom_source `figshare-pdo` / `figshare-pdo-alias`) inherits that +dataset's **data snapshot**, which predates publication (CLAUDE.md: the +Nov-2021 cutoff). A GI that was *registered or amended after* the snapshot may +have a boundary the snapshot can't reflect — its map polygon could be stale (an +old perimeter) or, for a brand-new GI, only present because a same-file-number +match happened to land. This audit cross-references each Bétard-tier record's +eAmbrosia registration/amendment dates against the snapshot and buckets it: + + FLAGGED — Bétard geometry, but the GI's latest eAmbrosia date is AFTER the + snapshot: verify the current boundary against the polygon. + (exit != 0 under --strict) + REVIEWED — slug in betard_delta_overrides.json: a curator checked the + post-snapshot amendment did not move the boundary (cite source). + OK — Bétard geometry, latest date <= snapshot: contemporaneous. + NO-DATE — Bétard geometry but no eAmbrosia entry / no usable date (FR INAO + records, or a slug that doesn't match the register). Listed so + the gap is visible, never silently dropped. + +The audit is read-only — it changes no geometry. It reads the compact +`wiki/data/aocs.en.*.js` startup blob (geom_source is a startup field) rather +than the multi-GB geojson, plus every `raw//eambrosia/index.json`. Fixing a +real FLAGGED finding means re-verifying the boundary upstream (a newer Bétard +release, the regional zone layer, or a commune-list resolver) — not editing +this audit. + +Usage: + uv run scripts/audit_betard_delta.py + uv run scripts/audit_betard_delta.py --strict # exit!=0 on FLAGGED + uv run scripts/audit_betard_delta.py --cutoff 2022-01-01 +""" +from __future__ import annotations + +import argparse +import glob +import json +import re +import sys +from collections import defaultdict +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +OVERRIDES = ROOT / "scripts" / "_lib" / "betard_delta_overrides.json" + +# Bétard 2022 EU_PDO.gpkg data-snapshot cutoff (see CLAUDE.md — "the dataset's +# Nov-2021 cutoff"). A GI registered or amended after this date may carry a +# boundary the snapshot does not reflect. +BETARD_SNAPSHOT = "2021-11-01" +BETARD_SOURCES = {"figshare-pdo", "figshare-pdo-alias"} + + +def load_aocs() -> dict[str, dict]: + files = sorted((ROOT / "wiki" / "data").glob("aocs.en.*.js")) + if not files: + sys.exit("error: wiki/data/aocs.en.*.js not found — run stage 04 first") + txt = files[0].read_text(encoding="utf-8") + m = re.match(r"window\.__OWM_DATA=(.*);\s*$", txt, re.S) + if not m: + sys.exit(f"error: could not parse {files[0].name}") + return json.loads(m.group(1))["aocs"] + + +def load_eambrosia() -> dict[str, dict]: + """slug -> eAmbrosia wine record (across every country index).""" + out: dict[str, dict] = {} + for idx in sorted(glob.glob(str(ROOT / "raw" / "*" / "eambrosia" / "index.json"))): + data = json.loads(Path(idx).read_text(encoding="utf-8")) + for wine in data.get("wines") or []: + slug = wine.get("slug") + if slug: + out.setdefault(slug, wine) + return out + + +def latest_date(wine: dict) -> str: + dates = [ + (wine.get(k) or "")[:10] + for k in ("eu_protection_date", "modification_date") + ] + dates = [d for d in dates if d] + return max(dates) if dates else "" + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--cutoff", default=BETARD_SNAPSHOT, + help=f"snapshot date; GIs dated after it are FLAGGED (default {BETARD_SNAPSHOT})") + ap.add_argument("--strict", action="store_true", + help="exit non-zero if any unreviewed FLAGGED finding exists") + args = ap.parse_args() + + aocs = load_aocs() + eam = load_eambrosia() + overrides = json.loads(OVERRIDES.read_text(encoding="utf-8")) if OVERRIDES.exists() else {} + + flagged: list[tuple] = [] + reviewed: list[tuple] = [] + ok = 0 + nodate: list[tuple] = [] + + for slug, rec in aocs.items(): + if rec.get("geom_source", "") not in BETARD_SOURCES: + continue + wine = eam.get(slug) + latest = latest_date(wine) if wine else "" + if not latest: + nodate.append((slug, rec.get("country", ""), rec.get("name", slug))) + continue + if latest > args.cutoff: + row = (slug, rec.get("country", ""), rec.get("name", slug), latest, wine) + (reviewed if slug in overrides else flagged).append(row) + else: + ok += 1 + + total = ok + len(flagged) + len(reviewed) + len(nodate) + print(f"Bétard-fallback geometries (geom_source in {sorted(BETARD_SOURCES)}): {total}") + print(f" snapshot cutoff: {args.cutoff}\n") + + if flagged: + print(f"FLAGGED — amended/registered after the snapshot ({len(flagged)}):") + by_country: dict[str, list] = defaultdict(list) + for slug, country, name, latest, wine in flagged: + by_country[country].append((latest, slug, name, wine)) + for country in sorted(by_country): + print(f" [{country}]") + for latest, slug, name, wine in sorted(by_country[country], reverse=True): + reg = (wine.get("eu_protection_date") or "")[:10] or "?" + mod = (wine.get("modification_date") or "")[:10] or "—" + print(f" {slug:<40} {name[:34]:<34} reg={reg} mod={mod}") + print() + + if reviewed: + print(f"REVIEWED — post-snapshot but curator-confirmed OK ({len(reviewed)}):") + for slug, country, name, latest, _ in sorted(reviewed): + note = overrides.get(slug, {}) + print(f" {slug:<40} [{country}] {note.get('reason', '')}") + print() + + if nodate: + print(f"NO-DATE — Bétard geometry, no eAmbrosia date ({len(nodate)}):") + for slug, country, name in sorted(nodate): + print(f" {slug:<40} [{country}] {name[:40]}") + print() + + print(f"summary: OK={ok} FLAGGED={len(flagged)} REVIEWED={len(reviewed)} NO-DATE={len(nodate)}") + if args.strict and flagged: + print(f"\n--strict: {len(flagged)} unreviewed FLAGGED finding(s)", file=sys.stderr) + return 1 + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From a8416d8a4aaa7b8926667a7a3082d2d0cb5c9b26 Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Thu, 11 Jun 2026 18:46:54 +0200 Subject: [PATCH 34/41] docs: draft VIVC/JKI synonym-republication licence query (Phase 7.3) Stages a ready-to-send email asking JKI whether Open Wine Map may display VIVC synonym strings verbatim under attribution (today it ships only IDs + prime names). Human sends; reply recorded in the draft + folded into CLAUDE.md's VIVC section. CLAUDE.md points at the draft. Co-Authored-By: Claude Opus 4.8 (1M context) --- CLAUDE.md | 5 ++- docs/vivc-jki-licence-query.md | 57 ++++++++++++++++++++++++++++++++++ 2 files changed, 61 insertions(+), 1 deletion(-) create mode 100644 docs/vivc-jki-licence-query.md diff --git a/CLAUDE.md b/CLAUDE.md index 9c9f896..5b8277c 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -95,7 +95,10 @@ details and "Hard rules" for invariants that apply to every country. in the tooltip source-block alongside the Wikipedia attribution. JKI publishes no explicit data licence — the codebase therefore ships VIVC IDs + prime names (factual citation) and does **not** republish - verbatim synonym strings in the UI pending JKI confirmation. Citation: + verbatim synonym strings in the UI pending JKI confirmation (a draft + licence-query email to JKI is staged at + [docs/vivc-jki-licence-query.md](docs/vivc-jki-licence-query.md) for a + human to send; record the reply there). Citation: Röckel et al., Vitis International Variety Catalogue — www.vivc.de. Ambiguous slugs (multiple candidate VIVC entries) get pinned via `raw/vivc/slug_overrides.json` (template at `slug_overrides.example.json`). diff --git a/docs/vivc-jki-licence-query.md b/docs/vivc-jki-licence-query.md new file mode 100644 index 0000000..0d567bc --- /dev/null +++ b/docs/vivc-jki-licence-query.md @@ -0,0 +1,57 @@ +# Draft licence query to JKI / VIVC + +**Status:** draft, awaiting send by a human (see CLAUDE.md "VIVC is a +third-party grape-taxonomy reference"). Record the reply here when it arrives. + +**To:** the VIVC team, Julius Kühn-Institut (JKI), Institute for Grapevine +Breeding Geilweilerhof — contact via https://www.vivc.de/ (Imprint / Kontakt). + +**Suggested subject:** Re-use of VIVC synonym strings under attribution — Open +Wine Map (openwinemap.com) + +--- + +Dear VIVC team, + +I maintain **Open Wine Map** (https://www.openwinemap.com), a free, open +reference map of European wine appellations generated mechanically from +public regulator data (INAO, the EU eAmbrosia register, national +specifications). For grape varieties I rely on the **Vitis International +Variety Catalogue** to reconcile the many regional spellings the regulator +documents use to a single canonical variety, and I cite it as: + +> Röckel, F., Maul, E., et al. — Vitis International Variety Catalogue, +> Julius Kühn-Institut, Geilweilerhof — www.vivc.de + +Today the site uses VIVC **factually and conservatively**: per grape it shows +the VIVC variety number, the VIVC prime name (e.g. *Aragonez (Tempranillo +Tinto)*), berry colour, and a link to the variety's VIVC page. I have +deliberately **not** republished the verbatim synonym strings or the +per-country "Official name in X" flags in the user interface, pending your +guidance on re-use. + +My question: **may the site display VIVC synonym strings (and the +country-official-name flags) verbatim in its UI, with clear attribution to +VIVC and the citation above?** The synonym list would materially help users +recognise a variety under the local name on a label. + +I could not find an explicit data licence on vivc.de, which is why I am +asking directly rather than assuming. If verbatim re-use is not possible, I +am happy to keep the current factual-citation approach (IDs, prime names, +links); if it is possible under specific attribution wording, please let me +know the exact wording you would like shown. + +Thank you for maintaining such a valuable resource for the wine-science and +viticulture community. + +Kind regards, +Boris De Vloed +Open Wine Map — https://www.openwinemap.com + +--- + +## Reply / outcome + +_(record JKI's response here, then update CLAUDE.md's VIVC section +accordingly — if verbatim re-use is granted, the synonym strings can be +surfaced in the grape-pill tooltip with the agreed attribution.)_ From c422b47f2ce2bd131fcd24055c1577f83df52f5c Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Thu, 11 Jun 2026 18:52:37 +0200 Subject: [PATCH 35/41] test: RO + IT parser fixture regression tests (Phase 5) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds fixture-based regression tests for the RO (document_unic + ONVPV caiet + commune) and IT (sottozona + menzione + MASAF article carving) parsers, written by the parser-fixture-writer agent against redacted, licence-clear excerpts of cached regulator docs. Suite 63 → 103 (+19 RO, +21 IT); ruff clean. A .gitignore negation lets the redacted OJ-page .html fixtures commit past the blanket *.html ignore. The tests pin CURRENT parser behaviour (not idealised), so two latent data bugs they surfaced are now locked in + will flip-flag when fixed: - RO: the iana caiet '- soiuri roşii/roze:' colour header glues 'roze:' onto the first red variety, dropping Cabernet Sauvignon. - IT: menzione's _NAME_TOKEN_RE rejects a lowercase connector inside a name, so Chianti Classico's 11 UGAs parse as 10 (San Donato in Poggio dropped). Co-Authored-By: Claude Opus 4.8 (1M context) --- .gitignore | 3 + tests/fixtures/it_masaf_articles_barolo.txt | 31 ++ tests/fixtures/it_menzioni_barolo_comma.txt | 29 ++ .../it_menzioni_chianti_classico_numbered.txt | 35 ++ .../it_menzioni_comma_with_stray_artn.txt | 10 + .../it_sottozona_chianti_preamble_list.txt | 15 + tests/fixtures/it_sottozona_prefix_header.txt | 18 + tests/fixtures/ro_caiet_iana.txt | 48 +++ ...o_document_unic_2024_terasele-dunarii.html | 52 +++ .../fixtures/ro_document_unic_dragasani.html | 51 +++ tests/test_it_parser.py | 350 +++++++++++++++++ tests/test_ro_parser.py | 362 ++++++++++++++++++ 12 files changed, 1004 insertions(+) create mode 100644 tests/fixtures/it_masaf_articles_barolo.txt create mode 100644 tests/fixtures/it_menzioni_barolo_comma.txt create mode 100644 tests/fixtures/it_menzioni_chianti_classico_numbered.txt create mode 100644 tests/fixtures/it_menzioni_comma_with_stray_artn.txt create mode 100644 tests/fixtures/it_sottozona_chianti_preamble_list.txt create mode 100644 tests/fixtures/it_sottozona_prefix_header.txt create mode 100644 tests/fixtures/ro_caiet_iana.txt create mode 100644 tests/fixtures/ro_document_unic_2024_terasele-dunarii.html create mode 100644 tests/fixtures/ro_document_unic_dragasani.html create mode 100644 tests/test_it_parser.py create mode 100644 tests/test_ro_parser.py diff --git a/.gitignore b/.gitignore index eac913c..dfdaf7d 100644 --- a/.gitignore +++ b/.gitignore @@ -42,6 +42,9 @@ raw/wikipedia/* *.pmtiles *.zip *.html +# …except test fixtures (short, redacted, licence-clear OJ-page HTML +# excerpts) — these ARE committed; see tests/fixtures/README.md. +!tests/fixtures/*.html # stage 02c round-trip work files (todo.json emitted by --emit-todo, # done*.json returned by the translator) — ephemeral diff --git a/tests/fixtures/it_masaf_articles_barolo.txt b/tests/fixtures/it_masaf_articles_barolo.txt new file mode 100644 index 0000000..88908e0 --- /dev/null +++ b/tests/fixtures/it_masaf_articles_barolo.txt @@ -0,0 +1,31 @@ +Redacted excerpt — Barolo DOCG national disciplinare (MASAF), +pdftotext -layout output, Articoli 1-3. Public source: MASAF +consolidated disciplinare di produzione. Exercises extract_articles +(article carving on the "Articolo N" header, last-occurrence-wins) and +the Article-2 vitigno-prose grape candidate scan (Nebbiolo from +"...esclusivamente dal vitigno Nebbiolo."). The Article-3 commune list +is truncated. Form-feeds and right-aligned page numbers from the layout +output are preserved on a couple of lines so the header anchor's +[ \t\x0c] leading-whitespace tolerance is exercised. + + Articolo 1 + Denominazione e vini + +1. La denominazione di origine controllata e garantita “Barolo” è riservata ai vini rossi che +rispondono alle condizioni ed ai requisiti stabiliti dal presente disciplinare di produzione, per le +seguenti tipologie: +- «Barolo», +- «Barolo» riserva. + + Articolo 2 + Base ampelografica + +1. I vini a denominazione di origine controllata e garantita «Barolo», devono essere ottenuti da uve +provenienti dai vigneti composti esclusivamente dal vitigno Nebbiolo. + + Articolo 3 + Zona di produzione delle uve + +1. Le uve destinate alla produzione dei vini di cui all'articolo 1 devono essere prodotte nella zona +appresso indicata che comprende, in tutto o in parte, il territorio dei comuni di Barolo, Castiglione +Falletto e Serralunga d'Alba, in provincia di Cuneo. diff --git a/tests/fixtures/it_menzioni_barolo_comma.txt b/tests/fixtures/it_menzioni_barolo_comma.txt new file mode 100644 index 0000000..6c1440a --- /dev/null +++ b/tests/fixtures/it_menzioni_barolo_comma.txt @@ -0,0 +1,29 @@ +Redacted excerpt — Barolo DOCG national disciplinare (MASAF), Article 8, +pdftotext -layout output. Public source: MASAF consolidated disciplinare +di produzione. The full 181-MGA roster is enumerated as a COMMA list +(not a numbered one) carrying "del comune di X" prose entries — the +"shape chosen by yield" guard must route this to the comma parser even +though a marker-count heuristic could be misled. Truncated to the first +~70 names plus the tail for size; behaviour is identical at any length. + +La denominazione di origine controllata e garantita dei vini «Barolo» e «Barolo» riserva può essere +seguita da una delle seguenti «menzioni geografiche aggiuntive», amministrativamente definite +nell’allegato al presente disciplinare di produzione: +Albarella, Altenasso o Garblet Suè o Garbelletto Superiore, Annunziata, Arborina, Arione, Ascheri, +Bablino, Badarina, Baudana, Bergeisa, Bergera-Pezzole, Berri, Bettolotti, Boiolo, Borzone, +Boscareto, Boscatto, Boschetti, Brandini, Brea, Breri, Bricco Ambrogio, Bricco Boschis, Bricco +Chiesa, Bricco Cogni, Bricco delle Viole, Bricco Luciani, Bricco Manescotto, Bricco Manzoni, +Bricco Rocca, Bricco Rocche, Bricco San Biagio, Bricco San Giovanni, Bricco San Pietro, Bricco +Voghera, Briccolina, Broglio, Brunate, Brunella, Bussia, Campasso, Cannubi, Cannubi Boschis o +Cannubi, Cannubi Muscatel o Cannubi, Cannubi San Lorenzo o Cannubi, Cannubi Valletta o +Cannubi, Canova, Capalot, Cappallotto, Carpegna, Case Nere, Castagni, Castellero, Castelletto, +Castello, Cerequio, Cerrati, Cerretta, Cerviano- Merli, Ciocchini, Ciocchini-Loschetto, Codana, +Collaretto, Colombaro, Conca, Corini-Pallaretta, Costabella, Coste di Rose, Coste di Vergne, Crosia, +Damiano, del comune di Barolo, del comune di Castiglione Falletto, del comune di Cherasco, del +comune di Diano d'Alba, del comune di Grinzane Cavour, del comune di La Morra, del comune di +Manforte d'Alba, del comune di Novello, del comune di Roddi, del comune di Serralunga d’Alba, del +comune di Verduno, Drucà, Falletto, Fiasco, Fontanafredda, Fossati, Francia, Gabutti, Galina, +Monprivato, Monrobiolo di Bussia, Montanello, Monvigliero, Mosconi, Neirane, Ornato, +Ravera, Ravera di Monforte, Sarmassa, Vignarionda, Villero, Zoccolaio, Zonchetta, Zuncai. +Le suddette menzioni geografiche aggiuntive, possono essere accompagnate dalla menzione «vigna» +seguita dal relativo toponimo o nome tradizionale, alle condizioni previste al successivo comma 4. diff --git a/tests/fixtures/it_menzioni_chianti_classico_numbered.txt b/tests/fixtures/it_menzioni_chianti_classico_numbered.txt new file mode 100644 index 0000000..d8e2595 --- /dev/null +++ b/tests/fixtures/it_menzioni_chianti_classico_numbered.txt @@ -0,0 +1,35 @@ +Redacted excerpt — Chianti Classico DOCG documento unico (EUR-Lex single +document), the additional-conditions section, flattened to the exact text +shape stage 02 feeds extract_menzioni (geo_area + "\n" + additional). The +11 UGAs are a NUMBERED list: EUR-Lex's table renderer puts the "N." +marker on its own line and the name on the next, separated by single +newlines (no blank lines). The "Link al disciplinare del prodotto" line +is the real list terminator (matched by _LIST_END_RE). The trailing +ELI/ISSN furniture is real and must not be harvested. + +Le seguenti Unità Geografiche Aggiuntive riferite ad aree dalle quali provengono effettivamente le uve da cui il vino è stato ottenuto e la cui delimitazione territoriale è definita nell’allegato 3 del disciplinare di produzione: +1. +Castellina +2. +Castelnuovo Berardenga +3. +Gaiole +4. +Greve +5. +Lamole +6. +Montefioralle +7. +Panzano +8. +Radda +9. +San Casciano +10. +San Donato in Poggio +11. +Vagliagli +Link al disciplinare del prodotto +https://www.politicheagricole.it/flex/cm/pages/ServeBLOB.php/L/IT/IDPagina/20089 +ELI: http://data.europa.eu/eli/C/2024/1036/oj diff --git a/tests/fixtures/it_menzioni_comma_with_stray_artn.txt b/tests/fixtures/it_menzioni_comma_with_stray_artn.txt new file mode 100644 index 0000000..80bcfea --- /dev/null +++ b/tests/fixtures/it_menzioni_comma_with_stray_artn.txt @@ -0,0 +1,10 @@ +# synthetic +# Directly targets the "shape chosen by yield, not marker count" guard in +# menzione.py: a SHORT comma list of UGAs that also carries a stray +# numbered cross-reference ("di cui all'art. 5 ...") in the same block. +# A marker-count heuristic could be tempted toward the numbered parser by +# the "5." token; the comma parser must win because it recovers more +# names. Modelled on the small-DOP comma form documented in menzione.py. + +È consentito l'uso in etichetta di una delle seguenti unità geografiche aggiuntive, di cui all'art. 5 +comma 2 del presente disciplinare: Pian d'Albola, Vistarenni e Monteluco. diff --git a/tests/fixtures/it_sottozona_chianti_preamble_list.txt b/tests/fixtures/it_sottozona_chianti_preamble_list.txt new file mode 100644 index 0000000..48f9f95 --- /dev/null +++ b/tests/fixtures/it_sottozona_chianti_preamble_list.txt @@ -0,0 +1,15 @@ +Redacted excerpt — Chianti DOCG national disciplinare (MASAF), Article 1 +("Denominazione e vini"), pdftotext -layout output. Public source: MASAF +consolidated disciplinare di produzione. Pattern B: a preamble phrase +("e le seguenti sottozone:") followed by a guillemet-wrapped, comma-and- +"e"-separated list where every sottozona name is prefixed with the +parent name ("«Chianti Colli Aretini»"). Exercises guillemet stripping, +parent-name-prefix stripping, and the final "e" conjunction. + +(Denominazione e vini) + 1.1 La denominazione di origine controllata e garantita «Chianti» è riservata ai vini «Chianti», già + riconosciuti a denominazione di origine controllata con decreto del Presidente della Repubblica 9 + agosto 1967, che rispondono alle condizioni ed ai requisiti stabiliti dal presente disciplinare di + produzione per le seguenti tipologie: «Chianti» e «Chianti Superiore» e le seguenti sottozone: + «Chianti Colli Aretini», «Chianti Colli Fiorentini», «Chianti Colli Senesi», «Chianti Colline Pisane», + «Chianti Montalbano», «Chianti Montespertoli» e «Chianti Rufina». diff --git a/tests/fixtures/it_sottozona_prefix_header.txt b/tests/fixtures/it_sottozona_prefix_header.txt new file mode 100644 index 0000000..8f4e2ca --- /dev/null +++ b/tests/fixtures/it_sottozona_prefix_header.txt @@ -0,0 +1,18 @@ +# synthetic +# No document in the current IT corpus fires Pattern A — every real +# sottozona case in raw/it/ is Pattern B (preamble + list). Pattern A is +# the parser's defensive branch for a "Sottozona NAME:" header that +# starts its own line followed by a commune body. This fixture mirrors +# that header form using the real Irpinia "Campi Taurasini" sottozona +# commune content (Irpinia DOC, MASAF disciplinare Article 3) recast to +# the line-start header layout the regex actually matches. The mid- +# sentence form Irpinia really uses ("...con l'indicazione della +# sottozona Campi Taurasini:") does NOT match Pattern A by design. + +Articolo 3 +Zona di produzione delle uve + +Sottozona Campi Taurasini: l'intero territorio amministrativo dei comuni di Taurasi, +Bonito, Castelfranci, Castelvetere sul Calore, Fontanarosa e Lapio. + +Sottozona Serra: comprende l'intero territorio amministrativo del comune di Serra. diff --git a/tests/fixtures/ro_caiet_iana.txt b/tests/fixtures/ro_caiet_iana.txt new file mode 100644 index 0000000..2a58c4c --- /dev/null +++ b/tests/fixtures/ro_caiet_iana.txt @@ -0,0 +1,48 @@ +CAIET DE SARCINI redacted excerpt — IANA (DOC-RO, ONVPV caiet de sarcini). +Public, licence-clear regulator document (onvpv.ro). pdftotext -layout output, +trimmed to sections I-IV plus the V header that terminates section IV. The +"Page 3 of 10" page-furniture line inside the IV. colour list is kept verbatim +because it sits between "Băbească neagră," and "Busuioacă de Bohotin." — i.e. +the wrapped-variety-name case the line-wise colour-segment join must survive. +A form-feed page break precedes "III." to exercise the \x0c→newline fold. + + + I. DEFINIŢIE +Denumirea de origine controlată „IANA” se atribuie vinurilor obţinute din struguri produşi în +arealul delimitat pentru această denumire, cu condiţia respectării tuturor prevederilor din prezentul +caiet de sarcini. + + + + II. LEGĂTURA CU ARIA GEOGRAFICĂ +Calitatea şi caracteristicile vinurilor produse în arealul delimitat pentru denumirea de origine +controlată “IANA” se datorează în mod esenţial mediului geografic, cu factorii săi naturali şi umani. +Climatul temperat continental, cu vădit caracter de excesivitate est-europeană, se reflectă în ierni +aspre şi veri toride. Solurile reprezentative pentru această podgorie sunt: solurile cenuşii, +cernoziomurile cambice, regosolurile şi solurile antropice. + + III. DELIMITAREA TERITORIALĂ PENTRU PRODUCEREA VINULUI CU D.O.C. + „IANA” + 1. Arealul delimitat pentru producerea vinurilor cu denumirea de origine controlată “IANA”, se +întinde pe teritoriul judeţului Vaslui, astfel: +Judeţul Vaslui: +- Comuna Iana, satul Iana; +- Comuna Perieni, satul Perieni; +- Comuna Ciocani, satul Ciocani; +- Comuna Băcani, satele Băcani, Suseni, Vulpăşeni ; +- Comuna Pogana, satul Pogana ; +2. Denumirea de origine controlată “IANA” poate fi completată, în funcţie de interesul +producătorilor, cu una din următoarele denumiri de plai viticol: DEALUL PERIENI, DEALUL +SEACA, DEALUL POGANA. + + + IV. SOIURILE DE STRUGURI +Soiurile de struguri care pot fi folosite pentru obţinerea vinurilor cu D.O.C. „IANA” sunt +următoarele: +- soiuri albe: Aligoté, Fetească regală, Riesling italian, Fetească albă, Sauvignon, Muscat Ottonel; + Page 3 of 10 +- soiuri roşii/roze: Cabernet Sauvignon, Merlot, Pinot noir, Fetească neagră, Băbească neagră, +Busuioacă de Bohotin. + + + V. PRODUCŢIA DE STRUGURI (kg/ha) diff --git a/tests/fixtures/ro_document_unic_2024_terasele-dunarii.html b/tests/fixtures/ro_document_unic_2024_terasele-dunarii.html new file mode 100644 index 0000000..9b63523 --- /dev/null +++ b/tests/fixtures/ro_document_unic_2024_terasele-dunarii.html @@ -0,0 +1,52 @@ + + +

DOCUMENT UNIC

+

1. Denumire/Denumiri +

+

„Terasele Dunării”

+

2. Tipul indicației geografice +

+

IGP

+

3. Țara căreia îi aparține aria geografică delimitată +

+

România

+

5. Categorii de produse viticole +

+

1. Vin

+

8. Indicarea soiului sau a soiurilor de struguri de vinificație din care este produs vinul sau din care sunt produse vinurile +

+

- Aligoté B - Plant de trois, Plant gris, Vert blanc, Troyen blanc

+

- Băbească neagră N - Grossmuttertraube, Hexentraube, Crăcana, Rară neagră

+

- Cabernet Sauvignon N - Petit Vidure, Bourdeos tinto

+

- Chardonnay B - Gentil blanc, Pinot blanc Chardonnay

+

- Fetească albă B - Păsărească albă, Poama fetei, Madchentraube, Leanyka, Leanka

+

- Merlot N - Bigney rouge, Plant Medoc

+

- Pinot Gris G - Affumé, Grau Burgunder, Pinot Grigio, Ruländer

+

- Sauvignon B - Sauvignon Blanc

+

- Traminer Roz Rs - Rosetraminer, Savagnin roz, Gewürztraminer

+

9. Descrierea concisă a arealului geografic delimitat +

+

+ - judeţul Teleorman: +

+

municipiul Zimnicea cu localităţile componente.

+

+ - judeţul Giurgiu: +

+

comuna Daia cu satele Daia, Dăiţa, Plopşoru;

+

comuna Greaca cu satele Greaca, Puţu Grecii;

+

comuna Hotarele cu satele Hotarele, Isvoarele;

+

comuna Prundu cu satele Prundu, Puieni;

+

10. Legătura cu aria geografică +

+

Arealul beneficiază de un climat continental cu influenţe danubiene.

+ diff --git a/tests/fixtures/ro_document_unic_dragasani.html b/tests/fixtures/ro_document_unic_dragasani.html new file mode 100644 index 0000000..91de70a --- /dev/null +++ b/tests/fixtures/ro_document_unic_dragasani.html @@ -0,0 +1,51 @@ + + +

COMUNICAREA UNEI MODIFICĂRI STANDARD CARE MODIFICĂ DOCUMENTUL UNIC

+

DOCUMENT UNIC

+

1. Denumire/denumiri +

+

Drăgăşani

+

2. Tip de indicație geografică: +

+

DOP – Denumire de origine protejată

+

3. Categorii de produse vitivinicole +

+

1. Vin

+

4. Descrierea vinului/vinurilor +

+

DESCRIERE TEXTUALĂ CONCISĂ

+

Vinurile albe/roze

+

Vinuri seci, demiseci, dulci.

+

6. Aria geografică delimitată +

+

Judeţul Vâlcea :

+

Municipiul Drăgăşani - localităţi componente Drăgăşani, Zărneni, Zlătărei, - Valea Caselor;

+

Com. Ştefăneşti - satele Dobruşa, Condoieşti, Şerbăneşti, Ştefăneşti;

+

Com. Prundeni - satele Prundeni, Călina, Zăvideni;

+

7. Soiul/soiurile de struguri de vin principal/principale +

+

Alutus N

+

Cabernet Sauvignon N - Petit Vidure, Bourdeos tinto

+

Chardonnay B - Gentil blanc, Pinot blanc Chardonnay

+

Fetească neagră N - Schwarze Madchentraube, Poama fetei neagră, Păsărească neagră, Coada rândunicii

+

Fetească regală B - Konigliche Madchentraube, Konigsast, Ktralyleanka, Dănăşană, Galbenă de Ardeal

+

Merlot N - Bigney rouge

+

Sauvignon B - Sauvignon verde

+

Tămâioasă românească B - Rumanische Weihrauchtraube, Tamianka

+

8. Descrierea legăturii (legăturilor) +

+

+ Legatura cu aria delimitata +

+

Podgoria Drăgăşani se întinde între Subcarpaţii Getici la nord şi Câmpia Română la sud. Climatul este temperat continental, cu uşoare influenţe mediteraneene.

+

9. Alte condițiile esențiale (ambalarea, etichetarea, alte cerințe) +

+

Cadru juridic: legislaţie naţională.

+ diff --git a/tests/test_it_parser.py b/tests/test_it_parser.py new file mode 100644 index 0000000..1c1d64a --- /dev/null +++ b/tests/test_it_parser.py @@ -0,0 +1,350 @@ +"""Fixture-based regression tests for the Italy (IT) parsers. + +Three parser modules, each with its own documented seam: + + - scripts/_lib/it/sottozona.py — sottozona detection. + Pattern A: a "Sottozona NAME:" header at the start of a line + followed by a commune body. + Pattern B: a preamble ("...le seguenti sottozone:") + a comma-and- + "e"-separated, often guillemet-wrapped and parent-name-prefixed + list. Pattern B is the only shape any real IT document fires in + the current corpus (Chianti 7, Valtellina 5, Bardolino 3); the + prefix-header fixture is therefore `# synthetic`. + + - scripts/_lib/it/menzione.py — MGA/UGA harvesting. + List shape is chosen BY YIELD, not marker count: parse the block + both ways (numbered + comma) and keep whichever recovers more + names. Chianti Classico's 11 UGAs are a numbered list; Barolo's + 181 MGAs are a long comma list carrying stray "del comune di X" + prose + an "art. N" reference that must not divert it to the + numbered parser. + + - scripts/_lib/it/masaf.py — MASAF disciplinare article carving + + Article-2 grape candidate extraction (extract_articles, + article2_candidate_phrases, parse_grapes_with). + +Fixtures are short, redacted excerpts of public regulator documents +(MASAF consolidated disciplinari, EU-OJ documenti unici) under +tests/fixtures/it_*.txt — see tests/fixtures/README.md. Synthetic +fixtures carry a `# synthetic` first-line marker and exist only where no +cached raw/ document exercises the branch (Pattern A; the stray-art.-N +guard in isolation). + +Assertions follow ACTUAL parser behaviour, not the regulator's intent. +One discrepancy is pinned explicitly: +test_menzione_numbered_list_drops_name_with_lowercase_connector documents +that the next-line numbered parser silently drops a UGA whose name +contains a lowercase Italian connector ("San Donato in Poggio"). +""" + +from __future__ import annotations + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.grape_entity import match_variety # noqa: E402 +from _lib.it.masaf import ( # noqa: E402 + article2_candidate_phrases, + extract_articles, + find_article_offsets, + parse_grapes_with, +) +from _lib.it.menzione import extract_menzioni # noqa: E402 +from _lib.it.sottozona import extract_sottozone # noqa: E402 + + +def _names(records: list[dict]) -> list[str]: + return [r["name"] for r in records] + + +def _slugs(records: list[dict]) -> list[str]: + return [r["slug"] for r in records] + + +def _patterns(records: list[dict]) -> set[str]: + return {r["source_pattern"] for r in records} + + +# ========================================================================== +# sottozona — Pattern B (preamble + list) +# ========================================================================== + +def test_sottozona_pattern_b_chianti_seven(fixture_text): + text = fixture_text("it_sottozona_chianti_preamble_list.txt") + out = extract_sottozone(text, "Chianti") + + # All 7 Chianti sottozone, parent prefix stripped, guillemets gone. + assert _names(out) == [ + "Colli Aretini", + "Colli Fiorentini", + "Colli Senesi", + "Colline Pisane", + "Montalbano", + "Montespertoli", + "Rufina", + ] + # Slug derives from the bare (prefix-stripped) name. + assert "colli-aretini" in _slugs(out) + assert "rufina" in _slugs(out) + # Single source pattern, the preamble-list branch. + assert _patterns(out) == {"sottozona-preamble-list"} + + +def test_sottozona_pattern_b_strips_guillemets_and_parent_prefix(fixture_text): + text = fixture_text("it_sottozona_chianti_preamble_list.txt") + out = extract_sottozone(text, "Chianti") + # No guillemet glyph survives in any name. + for name in _names(out): + assert "«" not in name and "»" not in name + # The bare parent name "Chianti" never appears as a standalone + # sottozona (it collapses to empty after prefix strip and is dropped), + # and no name still carries the "Chianti " prefix. + assert "Chianti" not in _names(out) + for name in _names(out): + assert not name.startswith("Chianti ") + + +def test_sottozona_pattern_b_final_e_conjunction(fixture_text): + # The last item ("e «Chianti Rufina»") is separated by the Italian + # final conjunction ` e `, not a comma — it must still be captured. + text = fixture_text("it_sottozona_chianti_preamble_list.txt") + out = extract_sottozone(text, "Chianti") + assert "Rufina" in _names(out) + + +def test_sottozona_parent_name_excluded_from_yield(): + # The parent's own slug is pre-seeded into the seen set, so a list + # that restates the parent does not emit a duplicate record for it. + text = "comprende le seguenti sottozone: Chianti, Colli Aretini e Rufina." + out = extract_sottozone(text, "Chianti") + assert "Chianti" not in _names(out) + assert "Colli Aretini" in _names(out) + assert "Rufina" in _names(out) + + +# ========================================================================== +# sottozona — Pattern A (line-start "Sottozona NAME:" header) — synthetic +# ========================================================================== + +def test_sottozona_pattern_a_prefix_header(fixture_text): + text = fixture_text("it_sottozona_prefix_header.txt") + out = extract_sottozone(text, "Irpinia") + assert _names(out) == ["Campi Taurasini", "Serra"] + assert _patterns(out) == {"sottozona-prefix"} + # Pattern A captures a commune body alongside the name. + campi = next(r for r in out if r["name"] == "Campi Taurasini") + assert campi["communes"] + assert "Taurasi" in campi["communes"][0] + + +def test_sottozona_pattern_a_requires_line_start(): + # The mid-sentence form Irpinia actually uses ("...con l'indicazione + # della sottozona Campi Taurasini:") is NOT a line-start header, so + # Pattern A deliberately does not fire — a parser quirk worth pinning. + text = ( + "«Irpinia» con l'indicazione della sottozona Campi Taurasini: " + "l'intero territorio amministrativo dei comuni di Taurasi e Lapio." + ) + out = extract_sottozone(text, "Irpinia") + assert out == [] + + +def test_sottozona_pattern_b_only_when_pattern_a_absent(fixture_text): + # Pattern B is evaluated only when Pattern A yields nothing — the + # Chianti fixture has no line-start "Sottozona" header, so Pattern B + # runs; assert it does not accidentally also classify under Pattern A. + text = fixture_text("it_sottozona_chianti_preamble_list.txt") + out = extract_sottozone(text, "Chianti") + assert "sottozona-prefix" not in _patterns(out) + + +# ========================================================================== +# menzione — numbered vs comma list shape (chosen by yield) +# ========================================================================== + +def test_menzione_numbered_list_chianti_classico(fixture_text): + text = fixture_text("it_menzioni_chianti_classico_numbered.txt") + out = extract_menzioni(text, "Chianti Classico") + + # ACTUAL behaviour (not the regulator's full 11-UGA roster): the + # numbered parser drops "San Donato in Poggio" — see + # test_menzione_numbered_list_drops_name_with_lowercase_connector for + # why. Single-word and "San X" two-word UGAs all survive; only the one + # with a lowercase connector ("... in ...") is lost. + names = _names(out) + assert names == [ + "Castellina", + "Castelnuovo Berardenga", + "Gaiole", + "Greve", + "Lamole", + "Montefioralle", + "Panzano", + "Radda", + "San Casciano", + "Vagliagli", + ] + assert _patterns(out) == {"numbered-list"} + + +def test_menzione_numbered_list_drops_name_with_lowercase_connector(): + # QUIRK / DISCREPANCY: _NAME_TOKEN_RE in menzione.py does not allow a + # lowercase Italian connector ("in", "di", "del") *inside* a name, so + # the next-line numbered parser silently drops a UGA like "San Donato + # in Poggio" (the regex matches only "San Donato", which != the full + # line, so the line is rejected). Pinned here so the day the regex is + # widened to keep the connector, this test flips and flags the change. + block = "9.\nSan Casciano\n10.\nSan Donato in Poggio\n11.\nVagliagli" + text = "Unità Geografiche Aggiuntive:\n" + block + "\nLink al disciplinare" + out = extract_menzioni(text, "Chianti Classico") + names = _names(out) + assert "San Casciano" in names + assert "Vagliagli" in names + assert "San Donato in Poggio" not in names # the dropped connector name + + +def test_menzione_numbered_list_stops_at_terminator(fixture_text): + # The "Link al disciplinare del prodotto" line + the ELI/ISSN furniture + # after the list must NOT be harvested as menzioni — _LIST_END_RE bounds + # the block at that terminator. + text = fixture_text("it_menzioni_chianti_classico_numbered.txt") + out = extract_menzioni(text, "Chianti Classico") + for name in _names(out): + assert "Link" not in name and "ELI" not in name and "http" not in name + + +def test_menzione_comma_list_barolo_shape_chosen_by_yield(fixture_text): + text = fixture_text("it_menzioni_barolo_comma.txt") + out = extract_menzioni(text, "Barolo") + + # The block has "comma 4" / "successivo comma" style numerics that + # could lure a marker-count heuristic into the numbered parser (which + # yields ~0); the comma parser must win because it recovers far more + # names. + assert _patterns(out) == {"comma-list"} + names = _names(out) + assert len(names) > 50 + for famous in ("Cannubi", "Brunate", "Bussia", "Cerequio", "Sarmassa"): + assert famous in names + + +def test_menzione_comma_list_drops_prose_commune_entries(fixture_text): + text = fixture_text("it_menzioni_barolo_comma.txt") + out = extract_menzioni(text, "Barolo") + # "del comune di Barolo" et al. start lowercase ("del") -> dropped. + for name in _names(out): + assert not name.lower().startswith("del comune") + assert "comune di" not in name.lower() + + +def test_menzione_short_comma_with_stray_art_n(fixture_text): + # Synthetic isolation of the "shape chosen by yield" guard: a short + # comma list carrying a stray "all'art. 5 comma 2" reference must not + # be mis-routed to the numbered parser. + text = fixture_text("it_menzioni_comma_with_stray_artn.txt") + out = extract_menzioni(text, "Chianti") + assert _patterns(out) == {"comma-list"} + assert _names(out) == ["Pian d'Albola", "Vistarenni", "Monteluco"] + + +def test_menzione_no_trigger_returns_empty(): + # No "unità/menzioni geografiche aggiuntive" trigger -> nothing. + out = extract_menzioni( + "La zona di produzione comprende i comuni di Greve e Radda.", "Chianti" + ) + assert out == [] + + +def test_menzione_trigger_without_colon_skipped(): + # A narrative trigger with no following colon introduces no list. + text = ( + "Le menzioni geografiche aggiuntive sono definite nell'allegato 3 " + "del disciplinare e non sono qui elencate." + ) + assert extract_menzioni(text, "Chianti") == [] + + +# ========================================================================== +# masaf — article carving + Article-2 grape extraction +# ========================================================================== + +def test_masaf_extract_articles_carves_bodies(fixture_text): + text = fixture_text("it_masaf_articles_barolo.txt") + bodies = extract_articles(text) + + assert set(bodies) == {1, 2, 3} + # Article 1 keeps its own body, not Article 2's. + assert "Denominazione e vini" in bodies[1] + assert "Base ampelografica" not in bodies[1] + # Article 2 carries the vitigno prose. + assert "Nebbiolo" in bodies[2] + # Article 3 carries the commune list, and Article 2's prose stopped + # at the Article-3 header. + assert "provincia di Cuneo" in bodies[3] + assert "Nebbiolo" not in bodies[3] + + +def test_masaf_find_article_offsets_ordered(fixture_text): + text = fixture_text("it_masaf_articles_barolo.txt") + offsets = find_article_offsets(text) + nums = [n for n, _s, _e in offsets] + assert nums == [1, 2, 3] + # Offsets are sorted by document position. + starts = [s for _n, s, _e in offsets] + assert starts == sorted(starts) + + +def test_masaf_extract_articles_last_occurrence_wins(): + # A TOC line + a real body line share article number 2; the body + # (later occurrence) must win, not the empty TOC entry. + text = ( + "Articolo 2\n" + "Base ampelografica\n" + "\n" + "Articolo 2\n" + "Base ampelografica\n" + "ottenuti dal vitigno Sangiovese.\n" + ) + bodies = extract_articles(text) + assert 2 in bodies + assert "Sangiovese" in bodies[2] + + +def test_masaf_article2_vitigno_prose_yields_grape(fixture_text): + text = fixture_text("it_masaf_articles_barolo.txt") + bodies = extract_articles(text) + phrases = article2_candidate_phrases(bodies[2]) + # The "vitigno Nebbiolo" prose scan surfaces the bare variety name. + assert any(p == "Nebbiolo" for p in phrases) + + +def test_masaf_parse_grapes_barolo_nebbiolo(fixture_text): + text = fixture_text("it_masaf_articles_barolo.txt") + bodies = extract_articles(text) + grapes = parse_grapes_with(match_variety, bodies[2], wine_name="Barolo") + assert "nebbiolo" in grapes["principal"] + # MASAF has no principal/accessory split — everything is principal. + assert grapes["accessory"] == [] + detail = next(d for d in grapes["details"] if d["slug"] == "nebbiolo") + assert detail["role"] == "principal" + assert detail["source"] == "masaf-disciplinare" + + +def test_masaf_article2_candidate_strips_percent_and_index(): + # Numbered, percentage-bearing variety lines: the leading index and + # the trailing share-range must both be stripped so the bare name + # reaches the matcher. + body = ( + "Base ampelografica\n" + "1. Sangiovese: dal 70% al 100%;\n" + "2. Canaiolo nero: da 0 a 30%;\n" + ) + phrases = article2_candidate_phrases(body) + assert "Sangiovese" in phrases + assert "Canaiolo nero" in phrases + # No phrase still carries a percent figure or a leading enumeration. + for p in phrases: + assert "%" not in p + assert not p[:2].strip().rstrip(".").isdigit() diff --git a/tests/test_ro_parser.py b/tests/test_ro_parser.py new file mode 100644 index 0000000..875f224 --- /dev/null +++ b/tests/test_ro_parser.py @@ -0,0 +1,362 @@ +"""Regression + behaviour tests for the Romania (RO) parsers. + +Three target modules, each a seam that has regressed historically (see +commit c4bb2f9 "Romania: complete coverage" and the RO section of CLAUDE.md): + + - scripts/ro/02_extract_pliegos.py — the EU-OJ DOCUMENT UNIC HTML driver + (slice from the DOCUMENT-UNIC anchor, find numbered ti-grseq-1 section + headers, route by Romanian title keyword, parse grapes/communes). + - scripts/_lib/ro/document_unic.py — the Romanian keyword/role tables + + the geo_area title blocklist (the "Țara căreia → România" decoy). + - scripts/_lib/ro/caiet.py — the ONVPV caiet de sarcini PDF parser + (Roman-numeral outline, "Soiurile albe:" / "Soiuri roşii:" colour split, + form-feed folding, line-wise colour-segment join). + - scripts/_lib/ro/commune.py — Romanian commune-list parsing (municipal + tier prefixes, judeţ headers, "cu satele/localităţile componente" tails, + parenthetical sub-village groups). + +Real cached docs live under raw/ro/{oj-pages,national-specs}/ (gitignored). +The fixtures here are short redacted excerpts under tests/fixtures/. + +Assertions are on STRUCTURE (routed roles, slug sets, commune membership, +colour split), not on full-output snapshots. Where a test pins ACTUAL parser +behaviour that diverges from the docstring's ideal (the "/roze" header-suffix +quirk, the cedilla-"şi" split gap), the divergence is called out inline. +""" +from __future__ import annotations + +import importlib +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.ro import caiet, commune # noqa: E402 +from _lib.ro.document_unic import ( # noqa: E402 + _GEO_AREA_TITLE_BLOCKLIST, + SECTION_ROLE_KEYWORDS, +) + +# 02_extract_pliegos starts with a digit, so import it by module path. +extract = importlib.import_module("ro.02_extract_pliegos") + + +# ========================================================================== +# DOCUMENT UNIC HTML driver — section routing +# ========================================================================== + +def _route_html(html: str) -> tuple[dict, dict, dict]: + """Slice → extract numbered sections → route. Returns (sections, titles, + routed) the way build_record drives them.""" + doc = extract.slice_document_unic(html) + assert doc is not None, "DOCUMENT-UNIC anchor must be found" + sections, titles = extract.extract_sections(doc) + routed = extract.route_sections(sections, titles) + return sections, titles, routed + + +def test_anchor_slice_drops_preamble(fixture_text): + html = fixture_text("ro_document_unic_dragasani.html") + doc = extract.slice_document_unic(html) + # The COMUNICAREA… modification preamble before DOCUMENT UNIC is dropped. + assert "COMUNICAREA UNEI MODIFICĂRI" not in doc + assert doc.lstrip().startswith(" str: + """Load the redacted iana caiet excerpt and inject a real form-feed + before the III. header to exercise the \\x0c → newline fold.""" + raw = fixture_text("ro_caiet_iana.txt") + return raw.replace("\n III.", "\x0c III.", 1) + + +def test_caiet_section_split_roman_numerals(fixture_text): + text = _iana_caiet_text(fixture_text) + bodies, titles = caiet.split_sections(text) + # I Definiţie → summary, II Legătura → terroir, III Delimitarea → area, + # IV Soiurile → grapes. + assert set(bodies) >= {"summary", "link_to_terroir", "geo_area", "grape_varieties"} + assert "DEFINIŢIE" in titles["summary"] + assert "SOIURILE DE STRUGURI" in titles["grape_varieties"] + + +def test_caiet_formfeed_fold_does_not_swallow_next_section(fixture_text): + """A form-feed page break right before the "III." header must be folded + to a newline so section II doesn't swallow section III's body.""" + text = _iana_caiet_text(fixture_text) + bodies, _titles = caiet.split_sections(text) + # geo_area (section III) was carved out as its own body, not glued to II. + assert "Judeţul Vaslui" in bodies["geo_area"] + assert "Judeţul Vaslui" not in bodies["link_to_terroir"] + + +def test_caiet_colour_split_white_vs_red(fixture_text): + """"- soiuri albe:" → blanc, "- soiuri roşii/roze:" → noir. Each variety + carries the colour of its header bucket. + + QUIRK pinned here: the red header "soiuri roşii/roze:" has a "/roze" + second-colour suffix that the _COLOUR_HEADER_RE consumes only up to + "roşii"; the leftover "roze: " glues onto the first variety + ("roze: Cabernet Sauvignon"), so Cabernet Sauvignon does NOT resolve in + the red list. This matches the production sidecar + (raw/ro/national-specs-extracted/iana.json — 11 grapes, no + cabernet-sauvignon). The whites and the remaining reds resolve fine.""" + text = _iana_caiet_text(fixture_text) + bodies, _titles = caiet.split_sections(text) + grapes = caiet.parse_grapes(bodies["grape_varieties"]) + by_slug = {d["slug"]: d for d in grapes["details"]} + # Whites + for slug in ("aligote", "feteasca-regala", "welschriesling", + "feteasca-alba", "sauvignon", "muscat-ottonel"): + assert by_slug[slug]["colour"] == "blanc", slug + # Reds that resolve (Cabernet Sauvignon is eaten by the "/roze" suffix). + for slug in ("merlot", "pinot-noir", "feteasca-neagra", + "babeasca-neagra", "busuioaca-de-bohotin"): + assert by_slug[slug]["colour"] == "noir", slug + assert "cabernet-sauvignon" not in by_slug # the documented quirk + # No principal/accessory split — all principal. + assert grapes["accessory"] == [] + assert set(grapes["principal"]) == set(by_slug) + + +def test_regression_caiet_wrapped_variety_name_not_sheared(fixture_text): + """The red list wraps mid-list across a "Page 3 of 10" furniture line: + - soiuri roşii/roze: …, Băbească neagră, + Page 3 of 10 + Busuioacă de Bohotin. + The line-wise colour-segment join (not per-physical-line split) must keep + "Busuioacă de Bohotin" as one token. (commit c4bb2f9 — line-wise join.)""" + text = _iana_caiet_text(fixture_text) + bodies, _titles = caiet.split_sections(text) + grapes = caiet.parse_grapes(bodies["grape_varieties"]) + slugs = set(grapes["principal"]) + # The wrapped tail variety resolves — would be lost if sheared. + assert "busuioaca-de-bohotin" in slugs + by_slug = {d["slug"]: d for d in grapes["details"]} + assert by_slug["busuioaca-de-bohotin"]["colour"] == "noir" + + +def test_caiet_parse_caiet_record_fragment(fixture_text): + """End-to-end: parse_caiet returns the merge-able record fragment with + grapes, communes, styles, link_to_terroir, and the parser template tag.""" + text = _iana_caiet_text(fixture_text) + frag = caiet.parse_caiet(text, "iana") + assert frag["parser_template"] == "onvpv-caiet-de-sarcini-v1" + assert frag["n_grapes"] == 11 + assert "blanc" in frag["styles"] and "rouge" in frag["styles"] + # Communes from section III resolve (head names, satul-tails dropped). + low = {c.lower() for c in frag["geo_communes"]} + assert "perieni" in low and "ciocani" in low and "pogana" in low + # Terroir text is the II. Legătura body. + assert "temperat continental" in frag["link_to_terroir"] From e5151f79e4e8cf7794c76f940fcf03bf183b364c Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Thu, 11 Jun 2026 18:57:24 +0200 Subject: [PATCH 36/41] deploy: catch-all trigger on security-header edge rules Bunny rejects an empty Triggers list ("At least one condition is required"), so each SetResponseHeader edge rule now carries a catch-all Url trigger (Type 0, pattern "*") to apply unconditionally. Co-Authored-By: Claude Opus 4.8 (1M context) --- scripts/deploy.py | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/scripts/deploy.py b/scripts/deploy.py index b5257b5..f1df9f2 100644 --- a/scripts/deploy.py +++ b/scripts/deploy.py @@ -181,7 +181,15 @@ def delete(session: requests.Session, host: str, zone: str, rel: str) -> None: # Security response headers to enforce on every response. # Bunny Edge Rule ActionType 5 = SetResponseHeader. -# Empty Triggers list = apply unconditionally to all requests. +# Bunny rejects an empty Triggers list ("At least one condition is required"), +# so each rule carries a catch-all Url trigger (Type 0, pattern "*") to apply +# unconditionally. +_CATCH_ALL_TRIGGER: dict = { + "Type": 0, + "PatternMatches": ["*"], + "PatternMatchingType": 0, + "Parameter1": "", +} _SECURITY_HEADERS: list[tuple[str, str]] = [ ("Strict-Transport-Security", "max-age=31536000"), ("X-Content-Type-Options", "nosniff"), @@ -224,7 +232,7 @@ def ensure_security_headers(api_key: str, pullzone: str) -> None: "ActionParameter2": header_value, "Description": f"Security: {header_name}", "Enabled": True, - "Triggers": [], + "Triggers": [dict(_CATCH_ALL_TRIGGER)], "TriggerMatchingType": 0, } if current: From 25fc1ca4789e94a2526974fcdab9e5be72f62b1f Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Fri, 12 Jun 2026 07:47:45 +0200 Subject: [PATCH 37/41] fix: recover dropped grapes/UGAs in RO caiet + IT menzione parsers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two latent parser bugs pinned by Phase-5 regression tests; both tests flip to assert the corrected behaviour. - RO `_lib/ro/caiet.py`: the colour header "- soiuri roşii/roze:" left "roze:" glued onto the first red variety, so its name never resolved. `_COLOUR_HEADER_RE` now consumes the "/roze" (and "/rose") second-colour suffix; bucket colour stays the first captured colour (roşii → noir). Recovers Cabernet Sauvignon in iana (11→12) and dealurile-transilvaniei (16→17); no wine loses a grape. - IT `_lib/it/menzione.py`: `_NAME_TOKEN_RE` rejected a lowercase Italian connector inside a name, dropping "San Donato in Poggio" from the numbered-list parser. The token regex now allows a connector (in/di/del/della/…) glued to a following capitalised word. Recovers Chianti Classico's 11th UGA, plus "Corte del Durlo" / "Monte di Colognola" in the Soave family (29→31 each); no name removed or corrupted (connector only accepted before a capitalised word). Verified end-to-end by regenerating the RO 02f sidecars + IT 02 extractions over the real cached regulator documents. Co-Authored-By: Claude Opus 4.8 (1M context) --- scripts/_lib/it/menzione.py | 12 ++++++++++-- scripts/_lib/ro/caiet.py | 7 ++++++- tests/test_it_parser.py | 32 +++++++++++++++----------------- tests/test_ro_parser.py | 26 ++++++++++++-------------- 4 files changed, 43 insertions(+), 34 deletions(-) diff --git a/scripts/_lib/it/menzione.py b/scripts/_lib/it/menzione.py index e4cfbb9..09285b7 100644 --- a/scripts/_lib/it/menzione.py +++ b/scripts/_lib/it/menzione.py @@ -86,9 +86,17 @@ def slugify(s: str) -> str: # A single name in the captured block. Italian proper nouns: starts # with uppercase, may include apostrophes, hyphens, accents. We cap -# length at 60 chars to filter prose fragments. +# length at 60 chars to filter prose fragments. A lowercase Italian +# connector ("in"/"di"/"del"/…) may sit *between* two capitalised words +# — "San Donato in Poggio" — so each continuation segment optionally +# carries a connector glued to the following capitalised word. The +# connector is only accepted when followed by a capitalised word, so a +# trailing/standalone "in" can never extend a name. _NAME_TOKEN_RE = re.compile( - r"[A-ZÀ-Þ][A-Za-zÀ-ÿ'’\-]+(?:[ \-][A-ZÀ-Þ][A-Za-zÀ-ÿ'’\-]+)*" + r"[A-ZÀ-Þ][A-Za-zÀ-ÿ'’\-]+" + r"(?:[ \-]" + r"(?:(?:della|delle|dei|del|di|in|da|de|al|alla|sul|sulla|e)[ \-])?" + r"[A-ZÀ-Þ][A-Za-zÀ-ÿ'’\-]+)*" ) diff --git a/scripts/_lib/ro/caiet.py b/scripts/_lib/ro/caiet.py index 6bdbed8..8a8c768 100644 --- a/scripts/_lib/ro/caiet.py +++ b/scripts/_lib/ro/caiet.py @@ -118,9 +118,14 @@ def split_sections(text: str) -> tuple[dict[str, str], dict[str, str]]: # Grape-section colour headers: "Soiurile albe:", "- Soiuri roşii:", # "soiuri roze:", "soiuri aromate". The variety list may follow on the -# same line after the colon. +# same line after the colon. A header may carry a "/roze" (or "/rose") +# second-colour suffix — "- soiuri roşii/roze:" — which the trailing +# `(?:/colour)*` group consumes so the colon isn't left glued to the +# first variety name (else "roze: Cabernet Sauvignon" never resolves). +# The bucket colour is the FIRST captured colour (roşii → noir). _COLOUR_HEADER_RE = re.compile( r"^[ \t]*[-•·]?\s*soiur?i(?:le)?\s+(albe|ro[șşs-]?ii|ro[șş]ii|roze|rose|aromate)" + r"(?:\s*/\s*(?:albe|ro[șşs-]?ii|ro[șş]ii|roze|rose|aromate))*" r"\s*:?", re.I, ) diff --git a/tests/test_it_parser.py b/tests/test_it_parser.py index 1c1d64a..402ecf4 100644 --- a/tests/test_it_parser.py +++ b/tests/test_it_parser.py @@ -31,10 +31,9 @@ guard in isolation). Assertions follow ACTUAL parser behaviour, not the regulator's intent. -One discrepancy is pinned explicitly: -test_menzione_numbered_list_drops_name_with_lowercase_connector documents -that the next-line numbered parser silently drops a UGA whose name -contains a lowercase Italian connector ("San Donato in Poggio"). +test_menzione_numbered_list_keeps_name_with_lowercase_connector exercises +the widened _NAME_TOKEN_RE that keeps a UGA whose name carries a lowercase +Italian connector ("San Donato in Poggio") intact. """ from __future__ import annotations @@ -168,11 +167,9 @@ def test_menzione_numbered_list_chianti_classico(fixture_text): text = fixture_text("it_menzioni_chianti_classico_numbered.txt") out = extract_menzioni(text, "Chianti Classico") - # ACTUAL behaviour (not the regulator's full 11-UGA roster): the - # numbered parser drops "San Donato in Poggio" — see - # test_menzione_numbered_list_drops_name_with_lowercase_connector for - # why. Single-word and "San X" two-word UGAs all survive; only the one - # with a lowercase connector ("... in ...") is lost. + # All 11 UGAs, including "San Donato in Poggio" whose lowercase Italian + # connector ("... in ...") the widened _NAME_TOKEN_RE now keeps — see + # test_menzione_numbered_list_keeps_name_with_lowercase_connector. names = _names(out) assert names == [ "Castellina", @@ -184,25 +181,26 @@ def test_menzione_numbered_list_chianti_classico(fixture_text): "Panzano", "Radda", "San Casciano", + "San Donato in Poggio", "Vagliagli", ] assert _patterns(out) == {"numbered-list"} -def test_menzione_numbered_list_drops_name_with_lowercase_connector(): - # QUIRK / DISCREPANCY: _NAME_TOKEN_RE in menzione.py does not allow a - # lowercase Italian connector ("in", "di", "del") *inside* a name, so - # the next-line numbered parser silently drops a UGA like "San Donato - # in Poggio" (the regex matches only "San Donato", which != the full - # line, so the line is rejected). Pinned here so the day the regex is - # widened to keep the connector, this test flips and flags the change. +def test_menzione_numbered_list_keeps_name_with_lowercase_connector(): + # _NAME_TOKEN_RE in menzione.py allows a lowercase Italian connector + # ("in", "di", "del", …) *inside* a name, glued to a following + # capitalised word, so the next-line numbered parser keeps a UGA like + # "San Donato in Poggio" intact. (Previously the regex matched only + # "San Donato" != the full line and dropped it; the widened token + # regex fixed that.) block = "9.\nSan Casciano\n10.\nSan Donato in Poggio\n11.\nVagliagli" text = "Unità Geografiche Aggiuntive:\n" + block + "\nLink al disciplinare" out = extract_menzioni(text, "Chianti Classico") names = _names(out) assert "San Casciano" in names assert "Vagliagli" in names - assert "San Donato in Poggio" not in names # the dropped connector name + assert "San Donato in Poggio" in names # connector name kept intact def test_menzione_numbered_list_stops_at_terminator(fixture_text): diff --git a/tests/test_ro_parser.py b/tests/test_ro_parser.py index 875f224..6ce2062 100644 --- a/tests/test_ro_parser.py +++ b/tests/test_ro_parser.py @@ -20,8 +20,8 @@ Assertions are on STRUCTURE (routed roles, slug sets, commune membership, colour split), not on full-output snapshots. Where a test pins ACTUAL parser -behaviour that diverges from the docstring's ideal (the "/roze" header-suffix -quirk, the cedilla-"şi" split gap), the divergence is called out inline. +behaviour that diverges from the docstring's ideal (the cedilla-"şi" split +gap), the divergence is called out inline. """ from __future__ import annotations @@ -305,13 +305,11 @@ def test_caiet_colour_split_white_vs_red(fixture_text): """"- soiuri albe:" → blanc, "- soiuri roşii/roze:" → noir. Each variety carries the colour of its header bucket. - QUIRK pinned here: the red header "soiuri roşii/roze:" has a "/roze" - second-colour suffix that the _COLOUR_HEADER_RE consumes only up to - "roşii"; the leftover "roze: " glues onto the first variety - ("roze: Cabernet Sauvignon"), so Cabernet Sauvignon does NOT resolve in - the red list. This matches the production sidecar - (raw/ro/national-specs-extracted/iana.json — 11 grapes, no - cabernet-sauvignon). The whites and the remaining reds resolve fine.""" + The red header "soiuri roşii/roze:" carries a "/roze" second-colour + suffix; _COLOUR_HEADER_RE now consumes it (and the trailing colon), so + the first red variety — Cabernet Sauvignon — resolves instead of being + glued to a leftover "roze: " prefix. The bucket colour is the FIRST + captured colour (roşii → noir).""" text = _iana_caiet_text(fixture_text) bodies, _titles = caiet.split_sections(text) grapes = caiet.parse_grapes(bodies["grape_varieties"]) @@ -320,11 +318,11 @@ def test_caiet_colour_split_white_vs_red(fixture_text): for slug in ("aligote", "feteasca-regala", "welschriesling", "feteasca-alba", "sauvignon", "muscat-ottonel"): assert by_slug[slug]["colour"] == "blanc", slug - # Reds that resolve (Cabernet Sauvignon is eaten by the "/roze" suffix). - for slug in ("merlot", "pinot-noir", "feteasca-neagra", - "babeasca-neagra", "busuioaca-de-bohotin"): + # Reds — Cabernet Sauvignon (the first, formerly eaten by "/roze") now + # resolves alongside the rest of the red list. + for slug in ("cabernet-sauvignon", "merlot", "pinot-noir", + "feteasca-neagra", "babeasca-neagra", "busuioaca-de-bohotin"): assert by_slug[slug]["colour"] == "noir", slug - assert "cabernet-sauvignon" not in by_slug # the documented quirk # No principal/accessory split — all principal. assert grapes["accessory"] == [] assert set(grapes["principal"]) == set(by_slug) @@ -353,7 +351,7 @@ def test_caiet_parse_caiet_record_fragment(fixture_text): text = _iana_caiet_text(fixture_text) frag = caiet.parse_caiet(text, "iana") assert frag["parser_template"] == "onvpv-caiet-de-sarcini-v1" - assert frag["n_grapes"] == 11 + assert frag["n_grapes"] == 12 assert "blanc" in frag["styles"] and "rouge" in frag["styles"] # Communes from section III resolve (head names, satul-tails dropped). low = {c.lower() for c in frag["geo_communes"]} From 935d7876d38f100d3c6a4a97e8deec2766ec92bc Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Fri, 12 Jun 2026 17:23:49 +0200 Subject: [PATCH 38/41] =?UTF-8?q?feat:=20Umbria=20regional=20wine-zone=20h?= =?UTF-8?q?arvest=20(geoportal-zone:umbria)=20=E2=80=94=20Phase=207.2?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds the 6th official Italian regional production-zone layer, moving 20 Umbrian appellations off the Bétard whole-municipality fallback onto the Regione Umbria delimited zones (CC-BY 4.0). - zone_sources.py: promote `umbria` from todo → active with a new `fetch_type: "ckan_shapefiles"` shape — the region publishes one per-appellation shapefile per CKAN dataset (19 of them, a mix of .zip and .7z) instead of a single WFS/zip layer. Carries the CKAN catalog endpoint, the assigned CRS (EPSG:3004 — the shapefiles ship no .prj), dotted tier-prefix stripping, an "X o Y" alternate-name split, and a curated extra_names map (the combined DOC+DOCG datasets cover the DOCG sub-name on the shared delimited area). - stage 00 `fetch_ckan_shapefiles()`: enumerate the catalog, download + extract each archive into raw/it/regional-zones/umbria//. py7zr (the `bootstrap` group, used by MASAF 02f) is imported lazily and gated like the Playwright WAF bootstrap — absent → the .7z datasets are skipped with a loud warning, the .zip ones still harvest. - ITZoneIndex: split the loader into `_load_layers` (the existing WFS/zip path, unchanged) + `_load_ckan` (glob extracted .shp, assign the declared CRS, strip the dotted ZONE tier prefix, index every name variant). The `ZONE` field is per-feature, so the existing connector- stripped normalisation matching still applies. Geometry verified: an isolated with-vs-without-Umbria rebuild differs on exactly 20 slugs — all Umbrian (14 figshare-pdo + 5 gisco-* + Orvieto lazio→lazio+umbria), zero collateral. Sagrantino, Torgiano Rosso Riserva and Rosso Orvietano all resolve; only Narni (no published shapefile) stays on its commune-union fallback. audit_it_regions green; overlap audit shows only thin border slivers (Italian zones overlap by design). tests/test_it_zones.py locks in the tier-strip / alt-name / DOCG-alias matching. Co-Authored-By: Claude Opus 4.8 (1M context) --- CLAUDE.md | 17 ++-- CURATOR_TODO.md | 8 +- GEOMETRY_SOURCES.md | 8 +- scripts/_lib/it/zone_sources.py | 47 +++++++++-- scripts/_lib/it/zones.py | 135 +++++++++++++++++++++++++------- scripts/it/00_fetch_data.py | 88 +++++++++++++++++++++ tests/test_it_zones.py | 84 ++++++++++++++++++++ 7 files changed, 339 insertions(+), 48 deletions(-) create mode 100644 tests/test_it_zones.py diff --git a/CLAUDE.md b/CLAUDE.md index 5b8277c..c785804 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -939,13 +939,16 @@ Per IT record, in priority order (each step records the chosen source in `geom_source` so the panel can attribute correctly): 1. **`geoportal-zone:`** — official regional-geoportal - production-zone polygon, matched by appellation name. Five regions - are harvested (Piemonte, Veneto, Lazio, Lombardia, Toscana — all - CC-BY 4.0 / IODL 2.0); ~218 of 531 IT wines resolve here, including - every flagship (Barolo, Soave, Valpolicella, Chianti, Brunello, - Bolgheri, Franciacorta, Frascati). An appellation spanning regions - is the union of its per-region pieces. Umbria + Puglia are tracked - to-dos (see CURATOR_TODO.md). + production-zone polygon, matched by appellation name. Six regions + are harvested (Piemonte, Veneto, Lazio, Lombardia, Toscana, Umbria — + all CC-BY 4.0 / IODL 2.0); ~237 of 531 IT wines resolve here, + including every flagship (Barolo, Soave, Valpolicella, Chianti, + Brunello, Bolgheri, Franciacorta, Frascati, Sagrantino di + Montefalco). An appellation spanning regions is the union of its + per-region pieces (e.g. Orvieto = `lazio+umbria`). Umbria is harvested + via a bespoke CKAN per-appellation `.zip`/`.7z` shapefile fetch + (`fetch_type: ckan_shapefiles` in `zone_sources.py`); Puglia is the + remaining tracked to-do (see CURATOR_TODO.md). 2. **`parent-appellation`** — sottozone (sub-denominations) inherit the parent's polygon. 3. **`figshare-pdo`** — exact `file_number` (`PDO-IT-A*` / diff --git a/CURATOR_TODO.md b/CURATOR_TODO.md index 6810e8f..e39f4bb 100644 --- a/CURATOR_TODO.md +++ b/CURATOR_TODO.md @@ -805,12 +805,12 @@ Region tracker: | Toscana | ✅ active | CC-BY 4.0 (GEOscopio; download page links CC-BY) | direct zip, `zo_vin_nom_zon` layer; 55 wines | | Lazio | ✅ active | CC-BY 4.0 | GeoServer WFS, DOC+DOCG+IGT; 29 wines | | Lombardia | ✅ active | CC-BY 4.0 | ArcGIS MapServer, DOC+DOCG+IGT; 34 wines | -| Umbria | ⏳ todo | CC-BY 4.0 | needs a bespoke fetch — ~23 separate per-appellation `.7z` shapefiles via the dati.regione.umbria.it CKAN API | +| Umbria | ✅ active | CC-BY 4.0 | CKAN `package_search` → 19 per-appellation `.zip`/`.7z` shapefiles (`fetch_type: ckan_shapefiles`); 20 wines matched (all but Narni, which publishes no shapefile) | | Puglia | ⏳ todo | IODL 2.0 | endpoint not reachable (SIT Puglia WFS/ArcGIS hosts 404 / login-gated) — needs the live WFS layer name | -**5 of 7 regions harvested → 218 IT wines on official zone polygons** -(`geoportal-zone`); the rest fall back to Bétard. Umbria + Puglia are -real to-dos, not skips — see the per-region notes and +**6 of 7 regions harvested → ~237 IT wines on official zone polygons** +(`geoportal-zone`); the rest fall back to Bétard. Puglia is the one +remaining to-do, not a skip — see the per-region notes and [scripts/_lib/it/zone_sources.py](scripts/_lib/it/zone_sources.py). | Abruzzo | ❌ fallback | custom, unconfirmed | portal SSL cert expired; stays on Bétard | | Campania | ❌ fallback | unconfirmed | dataset page 404s; stays on Bétard | diff --git a/GEOMETRY_SOURCES.md b/GEOMETRY_SOURCES.md index 5f0a727..d920ceb 100644 --- a/GEOMETRY_SOURCES.md +++ b/GEOMETRY_SOURCES.md @@ -46,7 +46,7 @@ wines stay on Bétard). ### Active — wired into `scripts/_lib/it/zone_sources.py` -5 regions covering 218 of 531 IT wines, including most flagships. +6 regions covering ~237 of 531 IT wines, including most flagships. | Regione | Layer | Endpoint | Licence | Attribution | |---|---|---|---|---| @@ -55,6 +55,7 @@ wines stay on Bétard). | **Lazio** | Vini DOC / DOCG / IGT (ARSIAL, 3 layers) | `geoportale.regione.lazio.it/geoserver/wfs` — `geonode:Vini_{DOC,DOCG,IGT}_Regione_Lazio` | CC-BY 4.0 | Regione Lazio — ARSIAL | | **Lombardia** | Aree di pregio vitivinicolo (3 sub-layers, ArcGIS MapServer) | `cartografia.servizirl.it/expo/rest/services/gpt/Aree_Pregio_Viti_Vinicolo/MapServer/{0,1,2}/query` | CC-BY 4.0 | Regione Lombardia | | **Toscana** | Zone di produzione vitivinicola DOP/IGP | [regione.toscana.it/geoscopio shapefile zip](https://www502.regione.toscana.it/geoscopio/download/tematici/zone_prod_vini/zone_prod_vini.zip) (sub-layer `zo_vin_nom_zon_2026_05`) | CC-BY 4.0 | Regione Toscana — GEOscopio | +| **Umbria** | Zone di produzione vini (per-appellation) | `dati.regione.umbria.it` CKAN `api/3/action/package_search` — 19 per-appellation `.zip`/`.7z` shapefiles, `fetch_type: ckan_shapefiles`; `ZONE` field, EPSG:3004 (no `.prj`), dotted tier-prefix strip | CC-BY 4.0 | Regione Umbria | Name-field varies (`denominazi` for most; `NOME_ZONA` for Lombardia with `strip_kind_prefix=True`; `NOM_ZON` for Toscana). Matching uses connector- @@ -65,7 +66,6 @@ stripped + saint-folded normalisation in | Regione | What's there | Blocker | |---|---|---| -| **Umbria** | dati.regione.umbria.it CKAN — ~23 separate per-appellation datasets, each a .7z shapefile | Bespoke CKAN-enumerate + 7z-extract fetch. Endpoint: `api/3/action/package_search?q=vini`. CC-BY 4.0. | | **Puglia** | SIT Puglia (WFS / ArcGIS) | Endpoint not reachable as of 2026-05-22 — WFS/ArcGIS hosts probed returned 404 / empty; cartography page is login-gated. Need the live layer name. IODL 2.0 expected. | ### Fallback — no licence-clear open layer @@ -253,8 +253,8 @@ ES: IT: simple = Bétard 2022 - advanced = 5 regional geoportals → Bétard (current default; Umbria + Puglia - would land here when their fetches are unblocked) + advanced = 6 regional geoportals → Bétard (current default; Puglia would + land here when its fetch is unblocked) PT: simple = Bétard 2022 (DOP only; IGPs invisible) diff --git a/scripts/_lib/it/zone_sources.py b/scripts/_lib/it/zone_sources.py index c4bef3e..caa1041 100644 --- a/scripts/_lib/it/zone_sources.py +++ b/scripts/_lib/it/zone_sources.py @@ -21,6 +21,20 @@ - `layers` — list of `{url, filename, layer?}` to download; `layer` names the sub-layer for multi-layer files. +Two fetch shapes: + - The default (`layers` list): each layer is one WFS/ArcGIS/zip download + with a per-feature `name_field`. Stage 00 downloads the files flat into + `raw/it/regional-zones/`; `ITZoneIndex` reads them by `layer["filename"]`. + - `fetch_type: "ckan_shapefiles"` (Umbria): the region publishes one + per-appellation shapefile per CKAN dataset (a mix of `.zip` and `.7z`). + Stage 00 enumerates the CKAN catalog, downloads + extracts each archive + into `raw/it/regional-zones///`, and `ITZoneIndex` + globs the extracted `.shp` files. These shapefiles carry no `.prj`, so a + `crs` must be declared; the `ZONE` field carries a dotted tier prefix + ("D.O.C. e D.O.C.G. Montefalco") stripped via `tier_prefix: "dotted"`, + may pack an "X o Y" alternate name (`alt_name_split`), and a combined + DOC+DOCG dataset covers the DOCG too (`extra_names`). + Only "active" entries are fetched and indexed. """ @@ -119,15 +133,38 @@ "layer": "zo_vin_nom_zon_2026_05", # appellation-name zones (not subzones) }], }, - # ─────────────── to-do: layer exists, harvesting still needs work ─────────────── "umbria": { - "status": "todo", + "status": "active", "label": "Regione Umbria — Zone di produzione vini (per-appellation)", "licence": "CC-BY 4.0", - "note": "dati.regione.umbria.it CKAN — ~23 separate per-appellation " - "datasets, each a .7z shapefile. Needs a CKAN-enumerate + " - "7z-extract fetch (api/3/action/package_search?q=vini).", + "licence_url": "https://creativecommons.org/licenses/by/4.0/", + "attribution": "Regione Umbria — dati.regione.umbria.it", + "fetch_type": "ckan_shapefiles", + # CKAN catalog endpoint; stage 00 merges these queries and keeps + # every dataset whose title is a "Zona/Zone di produzione vin…" + # carrying a .zip/.7z shapefile resource (19 appellations). + "ckan_base": "https://dati.regione.umbria.it/api/3/action/package_search", + "ckan_queries": ["vini", "produzione vini"], + "extract_dir": "umbria", + "name_field": "ZONE", + # The shapefiles ship with no .prj — coordinates are Monte Mario / + # Italy zone 2 (Gauss-Boaga Est); verified by reprojecting Orvieto + # to its true 12.14°E / 42.72°N centroid. + "crs": "EPSG:3004", + # ZONE = "D.O.C. e D.O.C.G. Montefalco" / "I.G.T. Umbria" — strip the + # dotted tier prefix before name matching. + "tier_prefix": "dotted", + # "DOC Rosso Orvietano o Orvietano Rosso" packs an alternate name. + "alt_name_split": True, + # The two combined DOC+DOCG datasets carry one polygon for the shared + # delimited area; the DOCG sub-name is enumerated here so it resolves + # to the same zone (keyed by the stripped+normalised base name). + "extra_names": { + "montefalco": ["Montefalco Sagrantino"], + "torgiano": ["Torgiano Rosso Riserva"], + }, }, + # ─────────────── to-do: layer exists, harvesting still needs work ─────────────── "puglia": { "status": "todo", "label": "Regione Puglia — Vini DOC/DOCG/IGP (SIT Puglia)", diff --git a/scripts/_lib/it/zones.py b/scripts/_lib/it/zones.py index feee551..6596037 100644 --- a/scripts/_lib/it/zones.py +++ b/scripts/_lib/it/zones.py @@ -44,6 +44,39 @@ def _norm(s: str) -> str: return "".join(w for w in s.split() if w not in _CONNECTORS) +# A leading Italian quality-tier abbreviation in the Umbria `ZONE` field — +# "D.O.C.", "D.O.C.G.", "I.G.T.", or a combined "D.O.C. e D.O.C.G." — with +# the dataset's irregular dot/space placement ("D.O.C .", "D.O.C.G.Torgiano", +# undotted "DOC Rosso Orvietano"). Anchored at the start and repeated to eat +# the combined form; every Umbria appellation name begins with a non-D/I +# letter, so this never bites into a real name. +_DOTTED_TIER_RE = re.compile( + r"^\s*(?:[DI][.\s]*[OG][.\s]*[CT][.\s]*(?:G[.\s]*)?(?:e[.\s]+)?)+", + re.IGNORECASE, +) +# Italian "o" (alternate-name separator) as a standalone word, e.g. +# "Rosso Orvietano o Orvietano Rosso". +_ALT_NAME_RE = re.compile(r"\s+o\s+", re.IGNORECASE) + + +def _strip_tier(raw: str, spec: dict) -> str: + if spec.get("tier_prefix") == "dotted": + return _DOTTED_TIER_RE.sub("", raw).strip() + return raw + + +def _name_variants(base: str, spec: dict) -> list[str]: + """Expand one stripped `ZONE` name into every appellation name it should + index under: the alternate-name halves of an "X o Y" string, plus any + curated `extra_names` (a combined DOC+DOCG dataset covers the DOCG too).""" + parts = _ALT_NAME_RE.split(base) if spec.get("alt_name_split") else [base] + extra = spec.get("extra_names") or {} + out = list(parts) + for p in parts: + out.extend(extra.get(_norm(p), [])) + return out + + class ITZoneIndex: def __init__(self, zones_dir: Path, target_crs: str = "EPSG:4326") -> None: self._by_name: dict[str, list[tuple[BaseGeometry, str]]] = {} @@ -51,36 +84,82 @@ def __init__(self, zones_dir: Path, target_crs: str = "EPSG:4326") -> None: self._regions: list[str] = [] for region, spec in active_sources().items(): - name_field = spec["name_field"] - strip_prefix = spec.get("strip_kind_prefix", False) - region_used = False - for layer in spec["layers"]: - path = zones_dir / layer["filename"] - if not path.exists(): - continue - read_kwargs = {"layer": layer["layer"]} if layer.get("layer") else {} - try: - gdf = gpd.read_file(path, **read_kwargs) - except Exception: # noqa: BLE001 - continue - if gdf.crs is None or name_field not in gdf.columns: - continue - if gdf.crs.to_string() != target_crs: - gdf = gdf.to_crs(target_crs) - region_used = True - for _, row in gdf.iterrows(): - geom = row.geometry - raw = str(row.get(name_field) or "") - if strip_prefix: - raw = re.sub(r"^\s*(DOCG|DOC|IGT)\s+", "", raw, flags=re.I) - name = _norm(raw) - if not name or geom is None or geom.is_empty: - continue - self._by_name.setdefault(name, []).append((geom, region)) - self._n_zones += 1 - if region_used: + if spec.get("fetch_type") == "ckan_shapefiles": + if self._load_ckan(region, spec, zones_dir, target_crs): + self._regions.append(region) + continue + if self._load_layers(region, spec, zones_dir, target_crs): self._regions.append(region) + def _add(self, name: str, geom: BaseGeometry, region: str) -> None: + if not name or geom is None or geom.is_empty: + return + self._by_name.setdefault(name, []).append((geom, region)) + self._n_zones += 1 + + def _load_layers( + self, region: str, spec: dict, zones_dir: Path, target_crs: str + ) -> bool: + """Default WFS/ArcGIS/zip layers with a per-feature `name_field`.""" + name_field = spec["name_field"] + strip_prefix = spec.get("strip_kind_prefix", False) + region_used = False + for layer in spec["layers"]: + path = zones_dir / layer["filename"] + if not path.exists(): + continue + read_kwargs = {"layer": layer["layer"]} if layer.get("layer") else {} + try: + gdf = gpd.read_file(path, **read_kwargs) + except Exception: # noqa: BLE001 + continue + if gdf.crs is None or name_field not in gdf.columns: + continue + if gdf.crs.to_string() != target_crs: + gdf = gdf.to_crs(target_crs) + region_used = True + for _, row in gdf.iterrows(): + raw = str(row.get(name_field) or "") + if strip_prefix: + raw = re.sub(r"^\s*(DOCG|DOC|IGT)\s+", "", raw, flags=re.I) + self._add(_norm(raw), row.geometry, region) + return region_used + + def _load_ckan( + self, region: str, spec: dict, zones_dir: Path, target_crs: str + ) -> bool: + """CKAN per-appellation shapefiles (Umbria): glob the extracted `.shp`, + assign the declared CRS (the files carry no `.prj`), strip the dotted + tier prefix off `ZONE`, and index every name variant.""" + name_field = spec["name_field"] + src_crs = spec.get("crs") + base_dir = zones_dir / spec.get("extract_dir", region) + if not base_dir.exists(): + return False + region_used = False + for shp in sorted(base_dir.glob("**/*.shp")): + try: + gdf = gpd.read_file(shp) + except Exception: # noqa: BLE001 + continue + if name_field not in gdf.columns: + continue + if gdf.crs is None and src_crs: + gdf = gdf.set_crs(src_crs) + if gdf.crs is None: + continue + if gdf.crs.to_string() != target_crs: + gdf = gdf.to_crs(target_crs) + region_used = True + for _, row in gdf.iterrows(): + geom = row.geometry + if geom is None or geom.is_empty: + continue + base = _strip_tier(str(row.get(name_field) or ""), spec) + for variant in _name_variants(base, spec): + self._add(_norm(variant), geom, region) + return region_used + @property def n_zones(self) -> int: return self._n_zones diff --git a/scripts/it/00_fetch_data.py b/scripts/it/00_fetch_data.py index 64fd2a5..25efe5d 100644 --- a/scripts/it/00_fetch_data.py +++ b/scripts/it/00_fetch_data.py @@ -53,8 +53,10 @@ import re import sys import unicodedata +import zipfile from datetime import datetime, timezone from pathlib import Path +from urllib.parse import quote import requests @@ -381,6 +383,89 @@ def fetch_istat_comuni() -> dict: return manifest +def fetch_ckan_shapefiles(region: str, spec: dict) -> dict: + """Harvest a CKAN per-appellation shapefile catalog (Umbria): enumerate + the catalog, then download + extract each "Zona/Zone di produzione vin…" + dataset's `.zip`/`.7z` shapefile into + `raw/it/regional-zones///`. + + `.zip` is extracted with the stdlib; `.7z` needs `py7zr` (the `bootstrap` + dependency group). py7zr is imported lazily — if it is absent the `.7z` + datasets are skipped with a loud warning (the `.zip` ones still harvest), + mirroring the Playwright-gated WAF bootstrap. Cached by presence of an + extracted `.shp`; delete the dataset dir to re-fetch.""" + extract_root = REGIONAL_ZONES_DIR / spec.get("extract_dir", region) + extract_root.mkdir(parents=True, exist_ok=True) + try: + import py7zr + have_7z = True + except ImportError: + py7zr = None + have_7z = False + + datasets: dict[str, dict] = {} + for q in spec.get("ckan_queries") or ["vini"]: + url = f"{spec['ckan_base']}?q={quote(q)}&rows=500" + r = requests.get(url, headers={"User-Agent": UA, "Accept": "application/json"}, + timeout=120) + r.raise_for_status() + for p in r.json().get("result", {}).get("results", []): + title = p.get("title") or "" + if not title.lower().startswith( + ("zona di produzione vin", "zone di produzione vin") + ): + continue + res = next((rs for rs in p.get("resources", []) + if (rs.get("url") or "").lower().endswith((".zip", ".7z"))), None) + if res: + datasets[p["name"]] = {"title": title, "url": res["url"]} + + files: list[dict] = [] + skipped_7z: list[str] = [] + for name, meta in sorted(datasets.items()): + url = meta["url"] + is_7z = url.lower().endswith(".7z") + dest_dir = extract_root / name + if list(dest_dir.glob("**/*.shp")): + print(f"[zones] cache hit {region}/{name}", file=sys.stderr) + files.append({"dataset": name, "title": meta["title"], "url": url, + "from_cache": True}) + continue + if is_7z and not have_7z: + skipped_7z.append(name) + continue + print(f"[zones] fetch {region}/{name}", file=sys.stderr) + resp = requests.get(url, headers={"User-Agent": UA}, timeout=300) + resp.raise_for_status() + archive = extract_root / (name + (".7z" if is_7z else ".zip")) + archive.write_bytes(resp.content) + dest_dir.mkdir(parents=True, exist_ok=True) + try: + if is_7z: + with py7zr.SevenZipFile(archive) as z: + z.extractall(dest_dir) + else: + with zipfile.ZipFile(archive) as z: + z.extractall(dest_dir) + finally: + archive.unlink(missing_ok=True) + print(f"[zones] saved {region}/{name} ({len(resp.content):,} b)", file=sys.stderr) + files.append({"dataset": name, "title": meta["title"], "url": url, + "from_cache": False}) + + if skipped_7z: + print( + f"[zones] WARNING py7zr missing — skipped {len(skipped_7z)} .7z " + f"{region} datasets; install the `bootstrap` group to harvest them: " + f"{', '.join(skipped_7z)}", + file=sys.stderr, + ) + return {"licence": spec.get("licence", ""), + "attribution": spec.get("attribution", ""), + "fetch_type": "ckan_shapefiles", + "files": files, "skipped_7z": skipped_7z} + + def fetch_regional_zones() -> dict: """Download each active regional wine production-zone layer. A region may have several layers (DOC / DOCG / IGT). Cached by presence — @@ -388,6 +473,9 @@ def fetch_regional_zones() -> dict: REGIONAL_ZONES_DIR.mkdir(parents=True, exist_ok=True) out: dict[str, dict] = {} for region, spec in active_sources().items(): + if spec.get("fetch_type") == "ckan_shapefiles": + out[region] = fetch_ckan_shapefiles(region, spec) + continue files = [] for layer in spec["layers"]: dest = REGIONAL_ZONES_DIR / layer["filename"] diff --git a/tests/test_it_zones.py b/tests/test_it_zones.py new file mode 100644 index 0000000..aa7feff --- /dev/null +++ b/tests/test_it_zones.py @@ -0,0 +1,84 @@ +"""Regression tests for the Italy (IT) regional wine-zone matcher. + +Focus: the Umbria CKAN `ckan_shapefiles` source in +scripts/_lib/it/zone_sources.py and its name-matching helpers in +scripts/_lib/it/zones.py. The Umbria shapefiles carry the appellation in a +`ZONE` field with a dotted tier prefix ("D.O.C. e D.O.C.G. Montefalco"), +sometimes an "X o Y" alternate name, and a combined DOC+DOCG dataset that +covers the DOCG too — so the stripped+normalised name must resolve to the +right eAmbrosia wine slug. These are the exact `ZONE` strings observed in the +live dati.regione.umbria.it catalog (2026-06). + +Pure-function tests — no network, no shapefile read. +""" +from __future__ import annotations + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.it.zone_sources import ZONE_SOURCES # noqa: E402 +from _lib.it.zones import _name_variants, _norm, _strip_tier # noqa: E402 + +UMBRIA = ZONE_SOURCES["umbria"] + + +def test_umbria_source_is_active_ckan(): + assert UMBRIA["status"] == "active" + assert UMBRIA["fetch_type"] == "ckan_shapefiles" + assert UMBRIA["name_field"] == "ZONE" + # Shapefiles ship without a .prj, so a CRS must be declared. + assert UMBRIA["crs"] == "EPSG:3004" + + +def test_strip_tier_handles_dotted_prefixes(): + strip = lambda z: _strip_tier(z, UMBRIA) # noqa: E731 + # Plain DOC / IGT, including the dataset's irregular "D.O.C ." spacing. + assert strip("D.O.C . Orvieto Classico") == "Orvieto Classico" + assert strip("D.O.C. Amelia") == "Amelia" + assert strip("I.G.T. Umbria") == "Umbria" + # Combined "D.O.C. e D.O.C.G." with and without a space before the name. + assert strip("D.O.C. e D.O.C.G. Montefalco") == "Montefalco" + assert strip("D.O.C. e D.O.C.G.Torgiano") == "Torgiano" + # Undotted "DOC ". + assert strip("DOC Rosso Orvietano o Orvietano Rosso") == "Rosso Orvietano o Orvietano Rosso" + + +def test_strip_tier_never_bites_into_a_real_name(): + # Every Umbria appellation name begins with a non-D/I letter, so the + # anchored tier regex must leave names that merely contain D/I/O/G/C/T + # intact (no prefix to strip means the input is returned unchanged). + assert _strip_tier("Colli del Trasimeno", UMBRIA) == "Colli del Trasimeno" + assert _strip_tier("Todi", UMBRIA) == "Todi" + + +def test_alt_name_split_recovers_both_halves(): + # "Rosso Orvietano o Orvietano Rosso" — the Italian " o " separates two + # spellings; both are indexed so the eAmbrosia "Rosso Orvietano" resolves. + variants = _name_variants("Rosso Orvietano o Orvietano Rosso", UMBRIA) + assert "Rosso Orvietano" in variants + assert "Orvietano Rosso" in variants + assert _norm("Rosso Orvietano") in {_norm(v) for v in variants} + + +def test_combined_docg_alias_covers_the_docg(): + # The single "Montefalco" / "Torgiano" polygon (DOC + DOCG share the + # delimited area) must also resolve the DOCG sub-name. + assert "Montefalco Sagrantino" in _name_variants("Montefalco", UMBRIA) + assert "Torgiano Rosso Riserva" in _name_variants("Torgiano", UMBRIA) + # A name with no curated extra is returned as-is. + assert _name_variants("Amelia", UMBRIA) == ["Amelia"] + + +def test_stripped_names_normalise_to_eambrosia_slugs(): + # End-to-end: the ZONE string, once stripped + variant-expanded + normed, + # must equal the eAmbrosia wine's normalised name for the four cases that + # needed special handling. + def norms(zone: str) -> set[str]: + return {_norm(v) for v in _name_variants(_strip_tier(zone, UMBRIA), UMBRIA)} + + assert _norm("Rosso Orvietano") in norms("DOC Rosso Orvietano o Orvietano Rosso") + assert _norm("Montefalco Sagrantino") in norms("D.O.C. e D.O.C.G. Montefalco") + assert _norm("Torgiano Rosso Riserva") in norms("D.O.C. e D.O.C.G.Torgiano") + assert _norm("Orvieto") in norms("D.O.C . Orvieto") From 74284de649a67df1fdd2223c5e03e1938f1afb20 Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Fri, 12 Jun 2026 19:48:08 +0200 Subject: [PATCH 39/41] test: parser fixture regression tests for the remaining 10 countries (Phase 5) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Completes Phase 5 — FR/ES/RO/IT were already done; this adds redacted- fixture regression tests for every remaining country parser, written by the parser-fixture-writer agent (one country each) and mirroring the existing tests/test_{ro,it}_parser.py style: STRUCTURE-only assertions, the shared fixture_text conftest fixture, and inline pinning of ACTUAL parser behaviour (so a test flips and flags the day a quirk is fixed). +196 tests (109 → 305), all green, ruff clean. Coverage: - AT (19): einziges_dokument section guard + dash-synonym variety split. - DE (28): einziges_dokument + all four BLE Produktspezifikation templates (A/B/C/D) incl. the §3.2-principal vs §8-flat role split. - HU (16): the monotonic-number + nested-subsection guard (section-4 "Bor – …" decoys must not shadow sections 5–9). - GR (16): greek_norm final-sigma fold (ς→σ) + OIV colour-letter codes. - PT (20): caderno_sections 3 variants + subregião patterns A & B + commune_list concelho shapes. - BG (24): edinen_dokument nested-subsection guard + IAVV specifikacija colour split + Cyrillic-preserving commune casefold. - HR (18): jedinstveni_dokument + the lettered-section specifikacija (form-feed fold, forward-only letter guard, colour-adjective fallback). - SI (19): enotni_dokument em-dash split + all three specifikacija templates incl. the 2007 priporočene/dovoljene role split. - CZ (22): national_spec variety table + rowspan commune-tree walker + SZPI CHZO terroir/style parser. - SK (14): jednotny_dokument old+new title variants + the ÚPV §f two-column left-Odroda-only extraction (synonym never leaks). Fixtures are short redacted excerpts (≤103 lines) with source-file provenance comments; nothing from raw/ is committed. The agents also pinned several pre-existing latent bugs as current behaviour (GR 2024 country decoy routes as geo_area; BG section-9 drop; PT distrito leak) — each test flips when the parser is fixed. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../fixtures/at_einziges_dokument_wachau.html | 41 ++ tests/fixtures/at_section7_carnuntum.html | 20 + tests/fixtures/at_styles_wien.html | 17 + tests/fixtures/bg_edinen_dokument_melnik.html | 36 ++ ...bg_edinen_dokument_nested_subsections.html | 29 + tests/fixtures/bg_specifikacija_sliven.txt | 27 + tests/fixtures/cz_chzo_moravske_layout.txt | 48 ++ .../cz_jednotny_dokument_synthetic.html | 26 + .../fixtures/cz_vyhlaska254_commune_tree.html | 38 ++ tests/fixtures/cz_vyhlaska88_varieties.html | 17 + tests/fixtures/de_ble_templateA_mosel.txt | 31 ++ tests/fixtures/de_ble_templateB_ahr.txt | 22 + tests/fixtures/de_ble_templateC_rheingau.txt | 28 + tests/fixtures/de_ble_templateD_baden.txt | 46 ++ ...nziges_dokument_wurzburger_stein_berg.html | 43 ++ .../gr_eniaio_engrafo_2024_makedonia.html | 43 ++ .../fixtures/gr_eniaio_engrafo_mantinia.html | 40 ++ .../fixtures/gr_eniaio_engrafo_santorini.html | 44 ++ .../hr_jedinstveni_dokument_ponikve.html | 36 ++ tests/fixtures/hr_specifikacija_dingac.txt | 45 ++ .../hr_specifikacija_primorska_docx.txt | 24 + .../fixtures/hu_egyseges_dokumentum_eger.html | 86 +++ .../hu_egyseges_dokumentum_soltvadkerti.html | 50 ++ tests/fixtures/pt_area_concelho_patterns.txt | 12 + tests/fixtures/pt_caderno_variantA_douro.txt | 40 ++ .../pt_caderno_variantB_vinho-verde.txt | 43 ++ tests/fixtures/pt_caderno_variantC_dao.txt | 38 ++ .../pt_subregiao_patternA_vinho-verde.txt | 55 ++ tests/fixtures/si_enotni_dokument_cvicek.html | 103 ++++ tests/fixtures/si_mkgp_doc_bizeljcan.txt | 57 ++ tests/fixtures/si_mkgp_doc_teran.txt | 36 ++ .../si_pravilnik_2007_bela-krajina.html | 28 + .../si_pravilnik_2022_belokranjec.html | 55 ++ ...jednotny_dokument_new_stredoslovenska.html | 27 + ..._jednotny_dokument_old_skalicky-rubin.html | 30 ++ .../sk_jednotny_dokument_tokaj_predikat.html | 44 ++ .../sk_specifikacija_karpatska_prihlaska.txt | 33 ++ ...sk_specifikacija_nitrianska_two_column.txt | 58 ++ tests/test_at_parser.py | 359 +++++++++++++ tests/test_bg_parser.py | 427 +++++++++++++++ tests/test_cz_parser.py | 373 +++++++++++++ tests/test_de_parser.py | 496 ++++++++++++++++++ tests/test_gr_parser.py | 339 ++++++++++++ tests/test_hr_parser.py | 344 ++++++++++++ tests/test_hu_parser.py | 333 ++++++++++++ tests/test_pt_parser.py | 376 +++++++++++++ tests/test_si_parser.py | 343 ++++++++++++ tests/test_sk_parser.py | 359 +++++++++++++ 48 files changed, 5245 insertions(+) create mode 100644 tests/fixtures/at_einziges_dokument_wachau.html create mode 100644 tests/fixtures/at_section7_carnuntum.html create mode 100644 tests/fixtures/at_styles_wien.html create mode 100644 tests/fixtures/bg_edinen_dokument_melnik.html create mode 100644 tests/fixtures/bg_edinen_dokument_nested_subsections.html create mode 100644 tests/fixtures/bg_specifikacija_sliven.txt create mode 100644 tests/fixtures/cz_chzo_moravske_layout.txt create mode 100644 tests/fixtures/cz_jednotny_dokument_synthetic.html create mode 100644 tests/fixtures/cz_vyhlaska254_commune_tree.html create mode 100644 tests/fixtures/cz_vyhlaska88_varieties.html create mode 100644 tests/fixtures/de_ble_templateA_mosel.txt create mode 100644 tests/fixtures/de_ble_templateB_ahr.txt create mode 100644 tests/fixtures/de_ble_templateC_rheingau.txt create mode 100644 tests/fixtures/de_ble_templateD_baden.txt create mode 100644 tests/fixtures/de_einziges_dokument_wurzburger_stein_berg.html create mode 100644 tests/fixtures/gr_eniaio_engrafo_2024_makedonia.html create mode 100644 tests/fixtures/gr_eniaio_engrafo_mantinia.html create mode 100644 tests/fixtures/gr_eniaio_engrafo_santorini.html create mode 100644 tests/fixtures/hr_jedinstveni_dokument_ponikve.html create mode 100644 tests/fixtures/hr_specifikacija_dingac.txt create mode 100644 tests/fixtures/hr_specifikacija_primorska_docx.txt create mode 100644 tests/fixtures/hu_egyseges_dokumentum_eger.html create mode 100644 tests/fixtures/hu_egyseges_dokumentum_soltvadkerti.html create mode 100644 tests/fixtures/pt_area_concelho_patterns.txt create mode 100644 tests/fixtures/pt_caderno_variantA_douro.txt create mode 100644 tests/fixtures/pt_caderno_variantB_vinho-verde.txt create mode 100644 tests/fixtures/pt_caderno_variantC_dao.txt create mode 100644 tests/fixtures/pt_subregiao_patternA_vinho-verde.txt create mode 100644 tests/fixtures/si_enotni_dokument_cvicek.html create mode 100644 tests/fixtures/si_mkgp_doc_bizeljcan.txt create mode 100644 tests/fixtures/si_mkgp_doc_teran.txt create mode 100644 tests/fixtures/si_pravilnik_2007_bela-krajina.html create mode 100644 tests/fixtures/si_pravilnik_2022_belokranjec.html create mode 100644 tests/fixtures/sk_jednotny_dokument_new_stredoslovenska.html create mode 100644 tests/fixtures/sk_jednotny_dokument_old_skalicky-rubin.html create mode 100644 tests/fixtures/sk_jednotny_dokument_tokaj_predikat.html create mode 100644 tests/fixtures/sk_specifikacija_karpatska_prihlaska.txt create mode 100644 tests/fixtures/sk_specifikacija_nitrianska_two_column.txt create mode 100644 tests/test_at_parser.py create mode 100644 tests/test_bg_parser.py create mode 100644 tests/test_cz_parser.py create mode 100644 tests/test_de_parser.py create mode 100644 tests/test_gr_parser.py create mode 100644 tests/test_hr_parser.py create mode 100644 tests/test_hu_parser.py create mode 100644 tests/test_pt_parser.py create mode 100644 tests/test_si_parser.py create mode 100644 tests/test_sk_parser.py diff --git a/tests/fixtures/at_einziges_dokument_wachau.html b/tests/fixtures/at_einziges_dokument_wachau.html new file mode 100644 index 0000000..8052b3e --- /dev/null +++ b/tests/fixtures/at_einziges_dokument_wachau.html @@ -0,0 +1,41 @@ +

VERÖFFENTLICHUNG EINES ÄNDERUNGSANTRAGS

+

Diese Veröffentlichung verleiht das Recht auf Einspruch.

+

EINZIGES DOKUMENT

+

1.   Name(n) +

+

Wachau

+

2.   Art der geografischen Angabe +

+

g.U. – Geschützte Ursprungsbezeichnung

+

3.   Kategorien von Weinbauerzeugnissen +

+

1. Wein

+

4.   Beschreibung des Weins / der Weine +

+

+ Wachau g.U. +

+

KURZE TEXTBESCHREIBUNG

+

In der g.U. Wachau werden größtenteils Weißweine produziert; Rotweine spielen eine sehr untergeordnete Rolle.

+

5.   Weinbereitungsverfahren +

+

5.1.   Spezifische önologische Verfahren +

+

Keine.

+

5.2.   Höchsterträge +

+

9000 Kilogramm Trauben je Hektar

+

6.   Abgegrenztes geografisches Gebiet +

+

Die Ursprungsbezeichnung „Wachau“ umfasst die niederösterreichischen Gemeinden Aggsbach, Dürnstein, Mautern an der Donau, Spitz und Weißenkirchen.

+

7.   Keltertraubensorte(n) +

+

Grüner Veltliner - Weißgipfler

+

Weißer Riesling - Rheinriesling

+

Weißer Riesling - Riesling

+

8.   Beschreibung des Zusammenhangs bzw. der Zusammenhänge +

+

Das Weinbaugebiet Wachau ist gekennzeichnet durch steile Terrassenanlagen am Ufer der Donau.

+

9.   Weitere wesentliche Bedingungen (Verpackung, Etikettierung, sonstige Anforderungen) +

+

Keine.

diff --git a/tests/fixtures/at_section7_carnuntum.html b/tests/fixtures/at_section7_carnuntum.html new file mode 100644 index 0000000..8ae86c8 --- /dev/null +++ b/tests/fixtures/at_section7_carnuntum.html @@ -0,0 +1,20 @@ +

EINZIGES DOKUMENT

+

1.   Name(n) +

+

Carnuntum

+

6.   Abgegrenztes geografisches Gebiet +

+

Die Ursprungsbezeichnung „Carnuntum“ umfasst den politischen Bezirk Bruck an der Leitha und den Gerichtsbezirk Schwechat.

+

7.   Wichtigste Keltertraubensorte(n) +

+

Blaufränkisch - Frankovka

+

Chardonnay - Morillon

+

Grüner Veltliner - Weißgipfler

+

Weißer Burgunder - Klevner

+

Weißer Burgunder - Pinot Blanc

+

Weißer Burgunder - Weißburgunder

+

Zweigelt - Blauer Zweigelt

+

Zweigelt - Rotburger

+

8.   Beschreibung des Zusammenhangs bzw. der Zusammenhänge +

+

Pannonisches Klima mit warmen Tagen und kühlen Nächten.

diff --git a/tests/fixtures/at_styles_wien.html b/tests/fixtures/at_styles_wien.html new file mode 100644 index 0000000..0eac3ed --- /dev/null +++ b/tests/fixtures/at_styles_wien.html @@ -0,0 +1,17 @@ +

EINZIGES DOKUMENT

+

1.   Name des Erzeugnisses +

+

Wien

+

3.   Kategorien des Weinbauerzeugnisses +

+

1. Wein

+

4. Schaumwein

+

4.   Beschreibung des Weins/der Weine +

+

In Wien werden überwiegend Weißweine erzeugt; Rotweine spielen eine geringere Rolle.

+

Trauben, die für die Erzeugung von Sekt verwendet werden, unterliegen eigenen Vorschriften.

+

Die traditionellen Bezeichnungen „Kabinett“, „Spätlese“, „Eiswein“ etc. dürfen verwendet werden.

+

7.   Wichtigste Keltertraubensorte(n) +

+

Grüner Veltliner - Weißgipfler

+

Weißer Riesling - Riesling

diff --git a/tests/fixtures/bg_edinen_dokument_melnik.html b/tests/fixtures/bg_edinen_dokument_melnik.html new file mode 100644 index 0000000..40779a3 --- /dev/null +++ b/tests/fixtures/bg_edinen_dokument_melnik.html @@ -0,0 +1,36 @@ +

COMMUNICATION preamble that must be dropped

+

Заявление за изменение, преди ЕДИНЕН ДОКУМЕНТ — да се отреже.

+

ЕДИНЕН ДОКУМЕНТ

+

1. Наименование на продукта

+

Мелник

+

2. Вид на географското означение

+

ЗНП — Защитено наименование за произход

+

3. Категории лозаро-винарски продукти

+

1. Вино

+

4. Описание на виното или вината

+

Белите вина се произвеждат от бели сортове грозде. Червените вина се произвеждат от червени сортове грозде. Произвеждат се и розе вина.

+

5. Винопроизводствени практики

+

Без обогатяване.

+

6. Определен географски район

+

Районът за производство на виното със ЗНП „Мелник“ е очертан при следните граници на землищата на населените места:

+

—

+

в община Сандански — с. Лехово, с. Ново Ходжово, с. Петрово, гр. Сандански , с. Плоски,

+

в община Петрич — с. Кърналово, с. Михнево,

+

в община Струмяни — с. Микрево,

+

в община Кресна — гр. Кресна.

+

7. Винен(и) сорт(ове) грозде

+

Viognier

+

Гренаш

+

Каберне Совиньон

+

Каберне Фран

+

Керацуда

+

Мерло

+

Мискет сандански - Мускат сандански

+

Мускат отонел

+

Тамянка - Теменуга

+

Шардоне

+

Широка мелнишка лоза - Мелник

+

8. Описание на връзката или връзките

+

Лозовите насаждения в землищата, очертаващи района попадат в Югозападна България, в долината на река Струма.

+

9. Други специфични изисквания (опаковане, етикетиране, други изисквания)

+

Етикетиране съгласно националното законодателство.

diff --git a/tests/fixtures/bg_edinen_dokument_nested_subsections.html b/tests/fixtures/bg_edinen_dokument_nested_subsections.html new file mode 100644 index 0000000..e73113c --- /dev/null +++ b/tests/fixtures/bg_edinen_dokument_nested_subsections.html @@ -0,0 +1,29 @@ +

ЕДИНЕН ДОКУМЕНТ

+

1. Наименование/наименования

+

Дунавска равнина

+

2. Вид на географското означение

+

ЗГУ — Защитено географско указание

+

3. Категории лозаро-винарски продукти

+

1. Вино

+

4. Описание на виното или вината

+

1. Бели вина

+

Бели вина със свеж и плодов характер.

+

2. Вина розе

+

Розе вина с малинов оттенък.

+

3. Червени вина

+

Червени вина с плътна структура.

+

4. Качествени пенливи вина

+

Качествени пенливи вина по класическия метод.

+

5. Винопроизводствени практики

+

Без обогатяване.

+

6. Определен географски район

+

в община Свищов, в община Плевен, в община Русе.

+

7. Винен сорт грозде или винени сортове грозде

+

Каберне Совиньон

+

Мерло

+

Памид

+

Гъмза

+

8. Описание на връзката или връзките

+

Районът обхваща Северна България, между река Дунав и Стара планина.

+

9. Други основни условия (опаковане, етикетиране, други изисквания)

+

Етикетиране съгласно националното законодателство.

diff --git a/tests/fixtures/bg_specifikacija_sliven.txt b/tests/fixtures/bg_specifikacija_sliven.txt new file mode 100644 index 0000000..47ee3c3 --- /dev/null +++ b/tests/fixtures/bg_specifikacija_sliven.txt @@ -0,0 +1,27 @@ + 1. Вино със ЗНП, традиционно наименование – гарантирано наименование за произход +(ГНП) “Сливен”. + 2. Виното се произвежда по традиционната технология за производство на бели и червени +вина. Допуска се отлежаване в дъбови бъчви за сортовете, подходящи за отлежаване. + 3. Районът за производство на вино със ЗНП “Сливен” е очертан при следните граници на +землищата на населените места – гр. Сливен, с. Близнец, с. Кермен, с. Дядово, находящи +се в област Сливен. + 4. Максималният добив на грозде допустим за производство на вино със ЗНП “Сливен” е +9000 kg/ha. + 5. Винените сортове грозде разрешени за производство на вино със ЗНП “Сливен” са: + - за бели вина: Ркацители, Шардоне, Юни блан, Мускат отонел, Мискет червен, Димят +и Алиготе; + - за червени вина и розе: Каберне совиньон, Мерло, Пино ноар, Памид и Шевка; + 6. Връзка с географския район. + а) Природни фактори: + Лозовите насаждения са разположени в подножието на южните склонове на планината +Гребенец, от където започва Източна Стара планина. Районът попада към подбалканските +полета с умереноконтинентален климат. Зимата е мека, а лятото сравнително горещо. +Характерен за района е местният вятър „бора”. Основните почвени различия са делувиалните +почви и канелено-горски почви. + + б) Човешки фактори: + Винарството в Сливенския район има дългогодишна традиция. Десетилетия наред +лозарството и винопроизводството са основен поминък на местното население. + 7. Приложими изисквания. + Етикетиране съгласно националното законодателство. + 8. Контролен орган, който проверява спазването на разпоредбите за спецификация. diff --git a/tests/fixtures/cz_chzo_moravske_layout.txt b/tests/fixtures/cz_chzo_moravske_layout.txt new file mode 100644 index 0000000..2964cb7 --- /dev/null +++ b/tests/fixtures/cz_chzo_moravske_layout.txt @@ -0,0 +1,48 @@ + CHZO „moravské“ + + +1 Popis vinařského regionu +Ve vinařské oblasti Morava je vyšší produkce bílých vín, protože plocha osázená bílými moštovými +odrůdami představuje zhruba dvě třetiny z celkové plochy vinic v podoblasti. Modré odrůdy +představují zbývající část výměry a používají se k výrobě červených a růžových vín. + + +1.1 Zhodnocení oblasti po stránce meteorologické + +Meteorologická data pochází z měření Českého hydrometeorologického ústavu (ČHMÚ). +Hodnota heliotermického indexu podle HUGLIN (1978) je 1773,9 a zařazuje oblast do kategorie H-1. + + +1.2 Zhodnocení podoblasti po stránce geologické a půdní + +Bradlo Pavlovských vrchů je tvořeno jurskými vápenci, obklopeno křídovitými sedimenty. +Půdy na vápenatém podloží jsou vhodné pro pěstování Chardonnay a Ryzlinku vlašského. + + +2 Druhy výrobků z révy vinné - popis vín + +2.1 Moravské zemské víno + +Bílé moravské zemské víno – organoleptické vlastnosti. +Růžové moravské zemské víno – organoleptické vlastnosti. +Červené moravské zemské víno – organoleptické vlastnosti. + + +2.2 Likérové víno + +Likérové víno je vyrobeno z hroznů sklizených ve stejné vinařské oblasti. + + +2.3 Šumivé víno + +Šumivé víno musí splňovat jakostní požadavky. + + +2.5 Perlivé víno dosycené oxidem uhličitým + +Perlivé víno je dosyceno oxidem uhličitým. + + +3 Základní enologické postupy + +Enologické postupy musí být v souladu s předpisy EU. diff --git a/tests/fixtures/cz_jednotny_dokument_synthetic.html b/tests/fixtures/cz_jednotny_dokument_synthetic.html new file mode 100644 index 0000000..1b47bc0 --- /dev/null +++ b/tests/fixtures/cz_jednotny_dokument_synthetic.html @@ -0,0 +1,26 @@ + + +

KOMUNIKACE O ZMĚNĚ — preamble before the single document

+

JEDNOTNÝ DOKUMENT

+

1. Název

+

Mělník

+

3. Kategorie výrobků z révy vinné

+

1. Víno

+

6. Vymezená zeměpisná oblast

+

Vinařská podoblast mělnická ve vinařské oblasti Čechy.

+

7. Hlavní moštové odrůdy

+

Ryzlink rýnský - Rheinriesling

+

Svatovavřinecké - Saint Laurent

+

Rulandské modré - Pinot noir

+

8. Popis souvislostí

+

Oblast má kontinentální podnebí, spraše a opukové podloží, což dává vínům jejich charakter.

+

9. Další základní podmínky

+

Označování v souladu s předpisy.

+ diff --git a/tests/fixtures/cz_vyhlaska254_commune_tree.html b/tests/fixtures/cz_vyhlaska254_commune_tree.html new file mode 100644 index 0000000..5b2feed --- /dev/null +++ b/tests/fixtures/cz_vyhlaska254_commune_tree.html @@ -0,0 +1,38 @@ + +

Příloha

+

Vinařské obce a viniční tratě v jednotlivých vinařských podoblastech

+

A. VINAŘSKÁ OBLAST ČECHY

+

1. Vinařská podoblast mělnická

+
+ + + + + + + +
Vinařská obecKatastrální územíNázev viniční trati
1. Benátky nad Jizerou1. Nové Benátky1. Pod zámkem
2. Stráň nad Jizerou
2. Obodř1. Nad vinicí
2. Cítov1. Cítov1. Na vinici
2. Velký kus
3. Kuks1. Kuks1. Nad zámkem
+

2. Vinařská podoblast litoměřická

+
+ + + + + +
Vinařská obecKatastrální územíNázev viniční trati
1. Bělušice1. Bělušice u Mostu1. Vinice Lenka
2. Vinice Klára
3. Vinice velká Anna
2. Most1. Čepirohy1. Vinice Barbora
+

B. VINAŘSKÁ OBLAST MORAVA

+

1. Vinařská podoblast mikulovská

+
+ + + + + +
Vinařská obecKatastrální územíNázev viniční trati
1. Bavory1. Bavory1. Pod Pálavou
2. Slunečná
3. Růžová
2. Březí1. Březí u Mikulova1. Liščí kopec
diff --git a/tests/fixtures/cz_vyhlaska88_varieties.html b/tests/fixtures/cz_vyhlaska88_varieties.html new file mode 100644 index 0000000..31414f5 --- /dev/null +++ b/tests/fixtures/cz_vyhlaska88_varieties.html @@ -0,0 +1,17 @@ + +

Příloha č. 2 k vyhlášce č. 88/2017 Sb.

+

Zkratky moštových odrůd révy vinné zapsaných ve Státní odrůdové knize České republiky a odrůd révy vinné pro výrobu zemských vín a zkratky některých tradičních výrazů, které lze používat v průvodním dokladu a v evidenčních knihách

+

I. Bílé moštové odrůdy

+
Název odrůdyZkratka
1. AureliusAu
2. ChardonnayCh
3. Műller ThurgauMT
4. Rulandské šedéRŠ, RS
5. Ryzlink rýnskýRR
6. SauvignonSg
7. Veltlínské zelenéVZ
+

II. Modré moštové odrůdy

+
Název odrůdyZkratka
1. AndréAn
2. Cabernet SauvignonCS
3. FrankovkaFr
4. MerlotMe
5. Modrý PortugalMP
6. SvatovavřineckéSv
+

III. Odrůdy pro výrobu zemských vín

+
Název odrůdyZkratka
1. Bílý PortugalBP
2. Modrý JanekMJ
3. Tramín žlutýTŽ, TZ
+

IV. Seznam zkratek pro některé tradiční výrazy

+
Tradiční výrazZkratka
1. Jakostní vínoJAK
2. Pozdní sběrPS
+

Příloha č. 3 k vyhlášce č. 88/2017 Sb.

+

Seznam chorob a vad vína

diff --git a/tests/fixtures/de_ble_templateA_mosel.txt b/tests/fixtures/de_ble_templateA_mosel.txt new file mode 100644 index 0000000..91452b1 --- /dev/null +++ b/tests/fixtures/de_ble_templateA_mosel.txt @@ -0,0 +1,31 @@ +# Redacted excerpt — BLE Produktspezifikation "Mosel" (Template A). +# Source: raw/de/produktspezifikationen/mosel.pdf via `pdftotext -layout`. +# Public, licence-clear: Amtliches Werk §5 UrhG (BLE). Trimmed to the +# §3.2 Mindestmostgewicht block (per-variety thresholds → de-facto +# principal) + §8 Zugelassene Keltertraubensorten white/red lists + +# the §9 Zusammenhang header that bounds §8. +3.2 Natürlicher Mindestalkoholgehalt und Mindestmostgewichte (Angabe in % vol Alkohol +und Grad Öchsle) +Qualitätswein: +Rebsorten Elbling (Weißer/Blauer/Roter/Schwarzer Elbling) 6,7 % vol und 55° Öchsle +Riesling (Weißer/Roter/Schwarzblauer Riesling) 6,7 % vol und 55° Öchsle +Rebsorte Müller Thurgau 7,2 % vol und 58° Öchsle +Rebsorte Dornfelder 8,8 % vol und 68° Öchsle +alle übrigen Rebsorten 7,5 % vol und 60° Öchsle + +Kabinett: +Rebsorte Elbling (Weißer/Blauer/Roter/Schwarzer Elbling) 9,1 % vol und 70° Öchsle +alle übrigen Rebsorten 9,5 % vol und 73° Öchsle + +3.3 Organoleptische Beschreibung + +8 Zugelassene Keltertraubensorten +Weiße Rebsorten: +Auxerrois, Bacchus, Chardonnay, Elbling, Kerner, Müller Thurgau, Riesling, +Ruländer, Scheurebe, Weißer Burgunder, Weißer Riesling. +Rote Rebsorten: +Blauer Spätburgunder, Dornfelder, Merlot, Müllerrebe, Regent, St. Laurent, +Tempranillo. + +9 Angaben, aus denen sich der Zusammenhang gemäß Verordnung (EU) Nr. 1308/2013 + Artikel 93 Absatz 1 Buchstabe a Ziffer i ergibt diff --git a/tests/fixtures/de_ble_templateB_ahr.txt b/tests/fixtures/de_ble_templateB_ahr.txt new file mode 100644 index 0000000..81f72c2 --- /dev/null +++ b/tests/fixtures/de_ble_templateB_ahr.txt @@ -0,0 +1,22 @@ +# Redacted excerpt — BLE Produktspezifikation "Ahr" (Template B). +# Source: raw/de/produktspezifikationen/ahr.pdf via `pdftotext -layout`. +# Public, licence-clear: Amtliches Werk §5 UrhG (BLE). Trimmed to the +# un-numbered "Zugelassene Keltertraubensorten:" anchor + the bullet +# "• Weißwein" / "• Rot- und Roséwein" colour blocks. Ahr has no §3.2 +# per-variety Mostgewicht split, so the role split is flat-no-split. +Zugelassene Keltertraubensorten: + +Keltertraubensorten der Art Vitis vinifera, aus denen die Weine des bestimmten +Anbaugebietes „Ahr“ gewonnen werden: + • Weißwein +Bacchus, Chardonnay, Huxelrebe, Johanniter, Kerner, +Müller-Thurgau, Optima, Ortega, Riesling, Ruländer, Solaris, +Weißer Burgunder; + • Rot- und Roséwein +Acolon, Blauer Frühburgunder, Blauer Portugieser, Blauer Spätburgunder, +Cabernet Franc, Cabernet Sauvignon, Domina, Dornfelder, Merlot, +Müllerrebe, Regent, St. Laurent. + +Zusammenhang mit dem geografischen Gebiet: +Das Gebiet liegt im Rheinischen Schiefergebirge und gehört landschaftlich zur +Eifel. diff --git a/tests/fixtures/de_ble_templateC_rheingau.txt b/tests/fixtures/de_ble_templateC_rheingau.txt new file mode 100644 index 0000000..abcc4e7 --- /dev/null +++ b/tests/fixtures/de_ble_templateC_rheingau.txt @@ -0,0 +1,28 @@ +# Redacted excerpt — BLE Produktspezifikation "Rheingau" (Template C). +# Source: raw/de/produktspezifikationen/rheingau.pdf via `pdftotext -layout`. +# Public, licence-clear: Amtliches Werk §5 UrhG (BLE). Trimmed to the +# §5.1 named-Mostgewicht row ("Spätburgunder Rotwein 8,4 66°"), the +# "7. Rebsorten" section with "• Rebsorten für Weißwein" / "• Rebsorten +# für Rot- und Roséwein" bullets, the inline "insbes. Weißer Riesling +# mit rd. 80 %" principal, and the §8 Zusammenhang header bound. +Rotweinsorten: +Spätburgunder Rotwein 8,4 66° +Sonstige Sorten Rotwein 7,8 62° + +7. Rebsorten +Angabe der Keltertraubensorten, aus denen der Wein / die Weine des Rheingaus +gewonnen wird / werden: + +• Rebsorten für Weißwein + insbes. Weißer Riesling mit rd. 80 % auf der Rebfläche im Rheingau vertreten sowie + für Rheingau zugelassene, klassifizierte Rebsorten: + Albalonga, Auxerrois, Bacchus, Chardonnay, Ehrenfelser, Kerner, Müller-Thurgau, + Weißer Riesling, Scheurebe, Grüner Silvaner. + +• Rebsorten für Rot- und Roséwein + Blauer Spätburgunder mit rd. 13 % auf der Rebfläche im Rheingau vertreten sowie + für Rheingau zugelassene, klassifizierte Rebsorten: + Acolon, Cabernet Sauvignon, Dornfelder, Merlot, Regent, Blauer Spätburgunder. + +8. Angaben, aus denen sich der Zusammenhang gemäß Artikel 118b Absatz 1 + Buchstabe a Ziffer i der VO (EG) Nr. 1234/2007 ergibt diff --git a/tests/fixtures/de_ble_templateD_baden.txt b/tests/fixtures/de_ble_templateD_baden.txt new file mode 100644 index 0000000..807c1f6 --- /dev/null +++ b/tests/fixtures/de_ble_templateD_baden.txt @@ -0,0 +1,46 @@ +# Redacted excerpt — BLE Produktspezifikation "Baden" (Template D). +# Source: raw/de/produktspezifikationen/baden.pdf via `pdftotext -layout`. +# Public, licence-clear: Amtliches Werk §5 UrhG (BLE). Trimmed to one +# §3.2.X Bereich block with "Weiße Rebsorten" / "Rote Rebsorten" +# colour subheaders and the tiered "- Variety ... 8,X % vol und YY°Oe" +# Mostgewicht rows (lowest tier = Leitsorten → principal). Includes the +# "alle übrigen Rebsorten" + "als Versuch angebaute" accessory/skip +# rows, AND the flat §8 list that Template A's §8 parser harvests. +3.2.1. Bereich: Bereiche Markgräflerland, Tuniberg, Kaiserstuhl, Breisgau, Ortenau, Kraichgau +und Badische Bergstraße + +Qualitätswein + +Weiße Rebsorten +Rebsorten: + + - Gutedel, Riesling, 8,0 % vol und 63°Oe + + - Merzling, Müller Thurgau, Muskateller, Muskat Ottonel, + + Nobling, Perle, Silvaner 8,4 % vol und 66°Oe + + - alle übrigen Rebsorten und Weine ohne Sortenangabe 8,9 % vol und 69°Oe + + - als Versuch angebaute, nicht in das Rebsortenverzeichnis + + nach Nr. 8 eingetragene Rebsorten 9,4 % vol und 72°Oe + +Rote Rebsorten +Rebsorten: + + - Tempranillo, Trollinger 8,0 % vol und 63°oe + + - Deckrot, Tauberschwarz 8,4 % vol und 66°Oe + + - alle übrigen Rebsorten und Weine ohne Sortenangabe 8,9 % vol und 69°Oe + +3.3 Organoleptische Beschreibung + +8. Zugelassene Keltertraubensorten +Weiße Rebsorten: +Auxerrois, Bacchus, Chardonnay, Gutedel, Merzling, Müller Thurgau, Nobling, +Perle, Riesling, Silvaner, Weißburgunder. +Rote Rebsorten: +Deckrot, Dornfelder, Merlot, Regent, Spätburgunder, Tauberschwarz, +Tempranillo, Trollinger. diff --git a/tests/fixtures/de_einziges_dokument_wurzburger_stein_berg.html b/tests/fixtures/de_einziges_dokument_wurzburger_stein_berg.html new file mode 100644 index 0000000..a30494c --- /dev/null +++ b/tests/fixtures/de_einziges_dokument_wurzburger_stein_berg.html @@ -0,0 +1,43 @@ + + + +

Veröffentlichung eines Antrags

+

EINZIGES DOKUMENT

+

1. Einzutragender Name +

+

Würzburger Stein-Berg

+

3. Art der geografischen Angabe +

+

g.U. — geschützte Ursprungsbezeichnung

+

4. Kategorien von Weinbauerzeugnissen +

+

1. Wein

+

5. Beschreibung des Weins/der Weine +

+

Qualitätswein

+

Weißweine aus den Rebsorten Silvaner, Riesling und Weißer Burgunder. Qualitätsweine vom Würzburger Stein-Berg sind trockene Weißweine.

+

7. Abgegrenztes geografisches Gebiet +

+

Die g. U. Würzburger Stein-Berg beinhaltet ausschließlich Flächen der in der Weinbergsrolle eingetragenen Einzellage Würzburger Stein in der kreisfreien Stadt Würzburg, Bayern.

+

8. Wichtigste Keltertraubensorte(n) +

+

Weißer Riesling — Riesling, Riesling renano, Rheinriesling, Klingelberger

+

Weißer Burgunder — Pinot Blanc, Pinot Bianco, Weißburgunder

+

Grüner Silvaner — Silvaner, Sylvaner

+

9. Beschreibung des Zusammenhangs bzw. der Zusammenhänge +

+

Die Einzellage Würzburger Stein prägt einen Muschelkalkboden in Steillage über dem Maintal; die Südexposition begünstigt vollreife Rieslinge.

+

10. Weitere wesentliche Bedingungen +

+

Etikettierung gemäß den nationalen Vorschriften.

+ + diff --git a/tests/fixtures/gr_eniaio_engrafo_2024_makedonia.html b/tests/fixtures/gr_eniaio_engrafo_2024_makedonia.html new file mode 100644 index 0000000..77b5f0f --- /dev/null +++ b/tests/fixtures/gr_eniaio_engrafo_2024_makedonia.html @@ -0,0 +1,43 @@ + + +

ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ

+

1.   Ονομασία(ες)

+

Μακεδονία (Makedonia)

+

2.   Τύπος γεωγραφικής ένδειξης

+

ΠΓΕ — Προστατευόμενη Γεωγραφική Ένδειξη

+

4.   Χώρα στην οποία ανήκει η γεωγραφική περιοχή

+

Ελλάδα

+

6.   Περιγραφή του οίνου ή των οίνων

+

Λευκοί, ερυθροί και ροζέ οίνοι.

+

8.   Ένδειξη της/των οινοποιήσιμης/-ων ποικιλίας/-ών αμπέλου από την/τις οποία/-ες παράγεται ο οίνος ή οι οίνοι

+

—

+

Cabernet Sauvignon N

+

—

+

Chardonnay B

+

—

+

Gewurtztraminer Rs

+

—

+

Ugni Blanc B - Trebbiano

+

—

+

Αγιωργίτικο Ν

+

—

+

Ξινόμαυρο Ν

+

10.   Δεσμός με τη γεωγραφική περιοχή

+

Η Μακεδονία χαρακτηρίζεται από ηπειρωτικό και μεσογειακό κλίμα.

+ diff --git a/tests/fixtures/gr_eniaio_engrafo_mantinia.html b/tests/fixtures/gr_eniaio_engrafo_mantinia.html new file mode 100644 index 0000000..9aac568 --- /dev/null +++ b/tests/fixtures/gr_eniaio_engrafo_mantinia.html @@ -0,0 +1,40 @@ + + +

ΑΙΤΗΣΗ ΓΙΑ ΤΡΟΠΟΠΟΙΗΣΗ ΤΩΝ ΠΡΟΔΙΑΓΡΑΦΩΝ ΤΟΥ ΠΡΟΪΟΝΤΟΣ

+

1.   Κανόνες που διέπουν την τροποποίηση

+

ΠΡΟΟΙΜΙΟ — δεν αποτελεί μέρος του ενιαίου εγγράφου.

+

2.1.   Διόρθωση στον αφρώδη λευκό οίνο

+

ΠΡΟΟΙΜΙΟ — τμήμα τροποποίησης, εκτός ενιαίου εγγράφου.

+ +

ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ

+

1.   Καταχωρισμένη ονομασία

+

Μαντινεία (Mantinia)

+

2.   Τύπος γεωγραφικής ένδειξης

+

ΠΟΠ — Προστατευόμενη Ονομασία Προέλευσης

+

4.   Περιγραφή του οίνου (των οίνων)

+

Οίνος λευκός ξηρός. Οίνος λευκός αφρώδης ποιότητας.

+

6.   Οριοθετημένη γεωγραφική περιοχή

+

Η οριοθετημένη ζώνη βρίσκεται στην επαρχία Μαντινείας.

+

7.   Κυριότερες οινοποιήσιμες ποικιλίες

+ + +

 

Ασπρούδες Β

+ + +

 

Μοσχοφίλερο N - Μαυροφίλερο

+

8.   Περιγραφή του δεσμού (των δεσμών)

+

Το γεωγραφικό περιβάλλον του οροπεδίου της Μαντινείας.

+

9.   Άλλες ουσιώδεις προϋποθέσεις

+

Πρόσθετες διατάξεις που αφορούν στην επισήμανση.

+ diff --git a/tests/fixtures/gr_eniaio_engrafo_santorini.html b/tests/fixtures/gr_eniaio_engrafo_santorini.html new file mode 100644 index 0000000..d01d39c --- /dev/null +++ b/tests/fixtures/gr_eniaio_engrafo_santorini.html @@ -0,0 +1,44 @@ + + +

ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ

+

1.   Ονομασία(εσ)

+

Σαντορίνη (Santorini)

+

3.   Κατηγορίεσ αμπελοοινικών προϊόντων

+

1. Οίνος. 3. Οίνος λικέρ. 15. Οίνος από λιαστά σταφύλια.

+

4.   Περιγραφή του (των) οίνου(-ων)

+

Λευκός ξηρός οίνος. Οίνος λικέρ από λιασμένα σταφύλια.

+

6.   Οριοθετημένη γεωγραφική περιοχή

+

Η νήσος Θήρα (Σαντορίνη) και η νήσος Θηρασία.

+

7.   Κύρια(εσ) οινοποιήσιμη(εσ) ποικιλία(εσ) σταφυλιού

+ + +

 

Αηδάνι άσπρο Β

+ + +

 

Αθήρι Β

+ + +

 

Ασύρτικο Β

+ + +

 

Μονεμβασιά Β - Μονοβασιά, Μονομβασίτικο

+ + +

 

Ροδίτης Rs - Αλεπού

+

8.   Περιγραφη του/των δεσμου/-ων

+

Το ηφαιστειακό έδαφος και το αμμοχαλικώδες υπέδαφος της καλντέρας.

+

9.   Αλλεσ ουσιωδεισ προϋποθεσεισ

+

Πρόσθετες διατάξεις επισήμανσης.

+ diff --git a/tests/fixtures/hr_jedinstveni_dokument_ponikve.html b/tests/fixtures/hr_jedinstveni_dokument_ponikve.html new file mode 100644 index 0000000..019de4d --- /dev/null +++ b/tests/fixtures/hr_jedinstveni_dokument_ponikve.html @@ -0,0 +1,36 @@ + + +

PROIZVODA

+

JEDINSTVENI DOKUMENT

+

„PONIKVE”

+

PDO-HR-02000

+

1.   Naziv koji je potrebno upisati u registar

+

Ponikve

+

2.   Vrsta oznake zemljopisnog podrijetla

+

ZOI – zaštićena oznaka izvornosti

+

3.   Kategorije proizvoda od vinove loze

+

1. Vino

+

4.   Opis vina

+

Bijela vina su svježa i mineralna; crna vina su puna i strukturirana; ružičasta vina su laganija.

+

5.   Prakse proizvodnje vina

+

Berba grožđa obavlja se ručno.

+

6.   Razgraničeno zemljopisno područje

+

Zaštićena oznaka izvornosti „Ponikve” odnosi se na vinogradarski položaj Ponikve, koji se nalazi na području katastarske općine Ponikve.

+

7.   Glavne sorte vinove loze

+

Maraština – Rukatac, Maraškin, Mareština, Krizol, Višana, Malvasia del Chianti, Malvasia lunga, Pavlos

+

Plavac mali crni – Plavac mali, Plavac veliki, Crljenak mali, Crljenac, Pagadebit crni, Zelenka

+

Pošip bijeli – Pošip, Pošipak, Pošipica

+

8.   Opis povezanosti

+

Položaj Ponikve obilježava sredozemna klima s velikim brojem sunčanih sati i kamenita tla koja vinima daju prepoznatljiva sortna svojstva.

+

9.   Osnovni dodatni uvjeti

+

Pakiranje se obavlja unutar razgraničenog područja.

+ diff --git a/tests/fixtures/hr_specifikacija_dingac.txt b/tests/fixtures/hr_specifikacija_dingac.txt new file mode 100644 index 0000000..99b3ba1 --- /dev/null +++ b/tests/fixtures/hr_specifikacija_dingac.txt @@ -0,0 +1,45 @@ +# synthetic +# Mirrors the MPS SPECIFIKACIJA PROIZVODA lettered a)-h) outline (the +# antiword/.doc -> text and pdftotext shape) so scripts/_lib/hr/specifikacija.py +# can be exercised without running the antiword Docker converter. Real +# Croatian section titles + a redacted, representative variety roster. +# Cross-checked against raw/hr/specifikacije-extracted/{dingac,sjeverna-dalmacija}.json +# so the asserted grape slugs match what parse_specifikacija actually emits. +# +# Exercised seams: +# - lettered a)-h) slicer (_lettered_sections) + role routing (_route_sections) +# - colour markers "Bijele sorte:" -> blanc / "Crne sorte:" -> noir +# - trailing-colour-adjective fallback (Chardonnay crni / Riesling zuti) +# - the FORWARD-ONLY letter guard: the "...prema tocki e) ..." cross-reference +# inside section c) is a backward jump and must NOT register as a heading. +# - the \x0c form-feed normalisation is injected by the test before g). +a) Naziv koji se zaštićuje: + Dingač + +b) Opis najznačajnijih fizikalno kemijskih i organoleptičnih (senzornih) svojstava vina: + Vino je tamno rubin crvene boje, punog okusa i izraženih sortnih svojstava. + +c) Specifični enološki postupci i ograničenja: + Svi enološki postupci za proizvodnju moraju biti u skladu s propisima, a najveći + dozvoljeni urod naveden je prema točki e) Maksimalni urod po hektaru. + +d) Granice područja: + Područje ZOI „Dingač" smješteno je na strminama uz južnu morsku obalu poluotoka + Pelješac, unutar katastarskih općina Potomje, Pijavičino i Kuna. + +e) Maksimalni urod po ha: + Najveći dozvoljeni urod iznosi 9 000 kg grožđa po hektaru. + +f) Sorte vinove loze: + Bijele sorte: + Chardonnay crni, Maraština, Pošip bijeli, Debit, Riesling žuti. + + Crne sorte: + Plavac mali crni, Babić, Plavina, Merlot, Croatina crna, Syrah. + +g) Pojedinosti koje se odnose na kakvoću ili svojstva vina povezane sa zemljopisnim uvjetima: + Karakteristike tla i klime južne obale Pelješca, uz jaku insolaciju i utjecaj mora, + daju grožđu visoku koncentraciju šećera i bojila. + +h) Prihvatljivi zahtjevi: + Sukladno Uredbi (EU) br. 1308/2013. diff --git a/tests/fixtures/hr_specifikacija_primorska_docx.txt b/tests/fixtures/hr_specifikacija_primorska_docx.txt new file mode 100644 index 0000000..2e21428 --- /dev/null +++ b/tests/fixtures/hr_specifikacija_primorska_docx.txt @@ -0,0 +1,24 @@ +# synthetic +# Mirrors the Primorska Hrvatska .docx shape: Word auto-numbering strips the +# lettered a)-j) prefixes, so the lettered slicer (_lettered_sections) finds +# < 5 sections and parse_specifikacija falls through to the keyword-title +# slicer (_keyword_sections). The role-keyword heading lines must START the +# line for the fallback to anchor. parser_template => "mps-specifikacija-docx". +# Cross-checked against raw/hr/specifikacije-extracted/primorska-hrvatska.json. +Naziv koji se zaštićuje: +Primorska Hrvatska + +Opis najznačajnijih svojstava vina: +Vina su svježa i mineralna. + +Granice područja: +Obuhvaća priobalno područje i otoke. + +Sorte vinove loze: +Bijele sorte: +Maraština, Pošip bijeli, Debit. +Crne sorte: +Plavac mali crni, Babić, Plavina. + +Pojedinosti koje se odnose na kakvoću ili svojstva vina povezane sa zemljopisnim uvjetima: +Sredozemna klima i kamenita tla daju prepoznatljiva sortna vina. diff --git a/tests/fixtures/hu_egyseges_dokumentum_eger.html b/tests/fixtures/hu_egyseges_dokumentum_eger.html new file mode 100644 index 0000000..f8d5d01 --- /dev/null +++ b/tests/fixtures/hu_egyseges_dokumentum_eger.html @@ -0,0 +1,86 @@ + +

COMMUNICATION — STANDARD AMENDMENT

+

Bejegyzett elnevezés módosítása

+

Preamble text that precedes the single document and must be dropped.

+ +

EGYSÉGES DOKUMENTUM

+ +

1.   A termék elnevezése +

+

Eger

+

Egri

+ +

2.   A földrajzi árujelző típusa +

+

OEM – oltalom alatt álló eredetmegjelölés

+ +

3.   A szőlőből készült termékek kategóriái +

+

1. Bor

+ +

4.   A bor(ok) leírása +

+

1.   Bor – Rozé fajta és küvé +

+

Rozébor leírása — redacted description of the rosé wine type.

+

2.   Bor – Siller fajta és küvé +

+

Siller bor leírása — redacted description of the siller wine type.

+ +

5.   Borkészítési eljárások +

+

Redacted vinification practices and maximum yields.

+ +

6.   Körülhatárolt földrajzi terület +

+

1.   CLASSICUS BOROK: +

+

Aldebrő, Andornaktálya, Demjén, Eger, Egerbakta, Egerszalók, Egerszólát.

+

2.   SUPERIOR ÉS GRAND SUPERIOR BOROK: +

+

Feldebrő, Felsőtárkány, Kerecsend, Maklár, Nagytálya, Noszvaj, Novaj.

+ +

7.   Fontosabb borszőlőfajták +

+

kadarka – jenei fekete

+

furmint – zapfner

+

tramini – traminer

+

kékfrankos – blaufränkisch

+

olasz rizling – nemes rizling

+

királyleányka – feteasca regale

+

leányka – leányszőlő

+

cabernet franc – carbonet

+

pinot noir – kék rulandi

+

syrah – serine noir

+ +

8.   A KAPCSOLAT(OK) LEÍRÁSA +

+

Bor (1)

+

1.   Körülhatárolt terület bemutatása +

+

Természeti tényezők. Eger a Mátra és a Bükk-hegység között fekszik.

+

2.   A borok leírása +

+

Redacted wine description subsection.

+

3.   Az okszerű kapcsolat bemutatása és bizonyítása +

+

Redacted causal-link subsection.

+ +

9.   További alapvető feltételek (csomagolás, címkézés, egyéb követelmények) +

+

Redacted additional conditions: packaging and labelling.

diff --git a/tests/fixtures/hu_egyseges_dokumentum_soltvadkerti.html b/tests/fixtures/hu_egyseges_dokumentum_soltvadkerti.html new file mode 100644 index 0000000..afae996 --- /dev/null +++ b/tests/fixtures/hu_egyseges_dokumentum_soltvadkerti.html @@ -0,0 +1,50 @@ + +

EGYSÉGES DOKUMENTUM

+ +

1.   Bejegyzendő elnevezés +

+

Soltvadkerti

+ +

2.   A földrajzi árujelző típusa +

+

OEM – oltalom alatt álló eredetmegjelölés

+ +

3.   A szőlőből készült termékek kategóriái +

+

1. Bor

+ +

4.   A bor(ok) leírása +

+

Fehérbor és pezsgő — redacted wine description.

+ +

5.   Borkészítési eljárások +

+

Redacted vinification practices.

+ +

6.   Körülhatárolt földrajzi terület +

+

Soltvadkert településnek közigazgatási határain belüli szőlőterületek.

+ +

7.   Fontosabb borszőlőfajták +

+

Ezerjó – Kolmreifler

+ +

8.   Kapcsolat a földrajzi területtel +

+

Természeti tényezők (bor és pezsgő). A lehatárolt termőterület redacted.

+ +

9.   További alapvető feltételek +

+

Redacted additional conditions.

diff --git a/tests/fixtures/pt_area_concelho_patterns.txt b/tests/fixtures/pt_area_concelho_patterns.txt new file mode 100644 index 0000000..c7795f6 --- /dev/null +++ b/tests/fixtures/pt_area_concelho_patterns.txt @@ -0,0 +1,12 @@ +# synthetic +Minimal one-line samples of each "Área Delimitada" concelho/distrito pattern the +commune_list parser handles. Each line is a verbatim-shape (but trimmed) excerpt of +the area-section grammar seen across PT cadernos (Dão municípios list, Vinho Verde +distrito-all, Algarve whole-distrito, Península de Setúbal bare distrito, Açores +archipelago). Kept minimal — no full document. + +Do distrito de Coimbra, os municípios de Arganil, Oliveira do Hospital e Tábua. +Todos os municípios dos distritos de Braga e de Viana do Castelo. +A área geográfica abrange todo o distrito de Faro. +Distrito de Setúbal. +A área geográfica de produção da IG "Açores" abrange todas as ilhas do Arquipélago dos Açores. diff --git a/tests/fixtures/pt_caderno_variantA_douro.txt b/tests/fixtures/pt_caderno_variantA_douro.txt new file mode 100644 index 0000000..d862ecd --- /dev/null +++ b/tests/fixtures/pt_caderno_variantA_douro.txt @@ -0,0 +1,40 @@ +CADERNO DE ESPECIFICAÇÕES DO "DOURO" — PDO-PT-A1539 (redacted excerpt). +Variant A: "Roman + Arabic" template — the Roman preamble (I..IV) then "V. DOCUMENTO ÚNICO" +wrapping numbered Arabic interior sections (1..8). The keyword anchors must carve the +numbered interior bodies (last-write-wins), not the Roman preamble. + +I. NOME(S) A REGISTAR: Douro + +II. DADOS RELATIVOS AO REQUERENTE: + +III. CADERNO DE ESPECIFICAÇÕES + +IV. DECISÃO NACIONAL DE APROVAÇÃO: + +V. DOCUMENTO ÚNICO: + +2. CATEGORIAS DOS PRODUTOS VITIVINÍCOLAS +Vinho; Vinho licoroso (Moscatel do Douro). + +5. ÁREA DELIMITADA +A área geográfica correspondente à Denominação de Origem "Porto" é a mesma que se +encontra demarcada para a produção do vinho do Douro e abrange os seguintes distritos, +concelhos e freguesias, tradicionalmente agrupadas em três áreas geográficas mais restritas: +Baixo Corgo: no distrito de Vila Real abrange os concelhos de Mesão Frio, de Peso da Régua e +de Santa Marta de Penaguião; as freguesias de Abaças, Ermida e Folhadela, do concelho de Vila Real. +Cima Corgo: no distrito de Vila Real abrange as freguesias de Alijó, Amieiro e Carlão, do concelho +de Alijó; as freguesias de Candedo, Murça e Noura, do concelho de Murça. +Douro Superior: no distrito de Bragança abrange a freguesia de Vilarelhos, do concelho de +Alfândega da Fé; as freguesias de Freixo de Espada à Cinta, Ligares e Mazouco; o concelho de +Vila Nova de Foz Côa. + +6. UVAS DE VINHO +a. Inventário das principais castas de uvas de vinho +b. Castas de uvas de vinho recomendadas: Touriga Nacional, Touriga Franca, Tinta Roriz. + +7. RELAÇÃO COM A ÁREA GEOGRÁFICA +Elementos relativos à área geográfica: +Situada no nordeste de Portugal, na bacia hidrográfica do Douro, rodeada de montanhas que a +isolam das influências atlânticas, a região beneficia de solos xistosos. + +VI. OUTRAS INFORMAÇÕES diff --git a/tests/fixtures/pt_caderno_variantB_vinho-verde.txt b/tests/fixtures/pt_caderno_variantB_vinho-verde.txt new file mode 100644 index 0000000..17aa4cc --- /dev/null +++ b/tests/fixtures/pt_caderno_variantB_vinho-verde.txt @@ -0,0 +1,43 @@ +CADERNO DE ESPECIFICAÇÕES DO "VINHO VERDE" — PDO-PT-A1545 (redacted excerpt) +Variant B: Arabic-only / documento-único-first template (no "V. DOCUMENTO ÚNICO" wrapper). + +2. CATEGORIAS DOS PRODUTOS VITIVINÍCOLAS + +Vinho («Vinho Verde») +Vinho espumante («Espumante de Vinho Verde») + +3. DESCRIÇÃO DO(S) VINHO(S) + +Brancos: límpidos ou ligeiramente opalinos, com cor entre citrino descorado e ligeiramente dourado. +Tintos: límpidos ou ligeiramente opalinos, com cor entre rubi e vermelho retinto. + +5. ZONA GEOGRÁFICA DEMARCADA + +A área geográfica de produção da DO "Vinho Verde" abrange as seguintes divisões +administrativas: +a) Todos os municípios dos distritos de Braga e de Viana do Castelo; +b) Do distrito de Aveiro, os municípios de Arouca, Castelo de Paiva e Vale de Cambra e a freguesia +de Ossela, do município de Oliveira de Azeméis; +c) Do distrito do Porto, os municípios de Amarante, Baião, Felgueiras, Gondomar e Lousada; +d) Do distrito de Vila Real, os municípios de Mondim de Basto e Ribeira de Pena; +e) Do distrito de Viseu, os municípios de Cinfães e Resende, com exceção da freguesia de Barrô. + +6. PRINCIPAL(IS) CASTA(S) DE UVA + +As castas utilizadas na produção de vinho com DO «Vinho Verde» são as que constam do quadro +seguinte. +Alvarinho +Arinto; Pedernã +Loureiro +Trajadura; Treixadura +Vinhão; Sousão + +7. RELAÇÃO COM A ZONA GEOGRÁFICA + +Elementos relativos à área geográfica: +A região do Vinho Verde situa-se no noroeste de Portugal, beneficiando de um clima atlântico +húmido e de solos graníticos que conferem aos vinhos a sua frescura característica. + +8. OUTRAS CONDIÇÕES + +8.1. Disposições aplicáveis ao rótulo. diff --git a/tests/fixtures/pt_caderno_variantC_dao.txt b/tests/fixtures/pt_caderno_variantC_dao.txt new file mode 100644 index 0000000..901541b --- /dev/null +++ b/tests/fixtures/pt_caderno_variantC_dao.txt @@ -0,0 +1,38 @@ +CADERNO DE ESPECIFICAÇÕES "DÃO" — PDO-PT-A1542 (redacted excerpt). +Variant C: Arabic-short / older format — numbered 1..9 with lowercase mixed-case headers +("Delimitação da Área Geográfica", "Relação com o Meio Geográfico"), no DOCUMENTO ÚNICO +wrapper. Tests the "MEIO GEOGRÁFICO" link variant + the "DELIMITAÇÃO DA ÁREA GEOGRÁFICA" +area variant + the enumerated "os municípios de X, Y e Z" concelho list. + +1. Identificação do Nome: DÃO + +2. Descrição do Vinho + +2.1. Características do Produto + +4. Delimitação da Área Geográfica +A área da Região Demarcada do Dão, conforme representada cartograficamente no anexo do +Decreto-Lei n.º 376/93 de 5 de Novembro, compreende: +a) Do distrito de Coimbra, os municípios de Arganil, Oliveira do Hospital e Tábua; +b) Do distrito da Guarda, os municípios de Aguiar da Beira, Fornos de Algodres, Gouveia e Seia; +c) Do distrito de Viseu, os municípios de Carregal do Sal, Mangualde, Mortágua, Nelas, +Penalva do Castelo, Santa Comba Dão, Sátão e Tondela. + +5. Rendimentos Máximos por Hectare +O Rendimento máximo por hectare das vinhas é de: +a) Vinhos Tintos: 80 hl +b) Vinhos Brancos: 100 hl + +6. Castas Utilizadas: +a) Castas tintas: +1 - Alfrocheiro, Aragonês, Jaen, Tinto-Cão, Touriga-Nacional e Trincadeira. +b) Castas brancas: +1 - Bical, Cerceal-Branco, Encruzado, Malvasia-Fina e Verdelho. + +7. Relação com o Meio Geográfico + +7.1 Factores Naturais +As vinhas destinadas à elaboração dos vinhos DOP Dão devem ser instaladas em terrenos +predominantemente graníticos, abrigados pelas serras envolventes. + +8. Exigências Aplicáveis diff --git a/tests/fixtures/pt_subregiao_patternA_vinho-verde.txt b/tests/fixtures/pt_subregiao_patternA_vinho-verde.txt new file mode 100644 index 0000000..58115ad --- /dev/null +++ b/tests/fixtures/pt_subregiao_patternA_vinho-verde.txt @@ -0,0 +1,55 @@ +VINHO VERDE — PDO-PT-A1545 — section 6 sub-região casta tables (redacted excerpt). +Pattern A: "Sub-região [de|do|da] NAME" line headers, each followed by a casta-list body. + +Os vinhos e produtos vitivinícolas com indicação de sub-região devem ser exclusivamente obtidos +a partir das castas enumeradas nos quadros seguintes para a respetiva sub-região. + +Sub-região de Amarante +Amaral +Arinto; Pedernã +Avesso +Trajadura; Treixadura +Vinhão; Sousão + +Sub-região de Ave +Amaral +Loureiro +Padeiro +Vinhão; Sousão + +Sub-região de Baião +Alvarelhão Brancelho +Avesso +Azal + +Sub-região de Basto +Batoca; Alvaraça +Rabo-de-Anho +Trajadura; Treixadura + +Sub-região do Cávado +Loureiro +Padeiro +Trajadura; Treixadura + +Sub-região do Lima +Borraçal +Loureiro +Padeiro + +Sub-região de Monção e Melgaço +Alvarinho +Pedral +Loureiro + +Sub-região do Paiva +Amaral +Avesso +Loureiro + +Sub-região do Sousa +Amaral +Azal +Espadeiro +Trajadura; Treixadura +Vinhão; Sousão diff --git a/tests/fixtures/si_enotni_dokument_cvicek.html b/tests/fixtures/si_enotni_dokument_cvicek.html new file mode 100644 index 0000000..959e2f5 --- /dev/null +++ b/tests/fixtures/si_enotni_dokument_cvicek.html @@ -0,0 +1,103 @@ + + + +

PUBLICATION OF A SINGLE DOCUMENT

+

ENOTNI DOKUMENT

+

„Cviček“

+

1. Ime ali imena

+

Cviček

+ +

2. Vrsta geografske označbe

+

ZOP – Zaščitena označba porekla

+ +

6. Opis vina ali vin

+

Cviček je suho vino rdečkaste barve, svežega in pitkega okusa.

+ +

8. Sorta ali sorte vinske trte, iz katerih se pridobi vino oziroma vina

+ + + +

—

bela žlahtnina

+ + + +

—

beli pinot - weissburgunder

+ + + +

—

chardonay

+ + + +

—

gamay

+ + + +

—

kraljevina

+ + + +

—

laški rizling

+ + + +

—

modra frankinja - frankinja

+ + + +

—

portugalka

+ + + +

—

ranfol - štajerska belina

+ + + +

—

ranina - radgonska ranina

+ + + +

—

rdeča žlahtnina

+ + + +

—

rumeni plavec

+ + + +

—

sivi pinot

+ + + +

—

zeleni silvanec

+ + + +

—

zweigelt

+ + + +

—

šentlovrenka

+ + + +

—

žametovka

+ +

9. Jedrnat opis razmejenega geografskega območja

+

Cviček se prideluje le na območju vinorodnega okoliša Dolenjska, na vinogradniških legah, ki ležijo nad 210 metri nadmorske višine.

+ +

10. Povezava z geografskim območjem

+

Dolenjska je gričevnata pokrajina s pretežno karbonatnimi kamninami; celinsko podnebje z razmeroma nizkimi povprečnimi temperaturami daje vinu Cviček značilno svežino in nizko stopnjo alkohola.

+ +

11. Dodatne veljavne zahteve

+

Označevanje v skladu z nacionalno zakonodajo.

+ + diff --git a/tests/fixtures/si_mkgp_doc_bizeljcan.txt b/tests/fixtures/si_mkgp_doc_bizeljcan.txt new file mode 100644 index 0000000..49bb24b --- /dev/null +++ b/tests/fixtures/si_mkgp_doc_bizeljcan.txt @@ -0,0 +1,57 @@ +# synthetic +# Synthetic minimal fixture mirroring the MKGP per-wine ".doc" SPECIFIKACIJA +# PROIZVODA layout (parser_template mkgp-doc-v1). The real inputs are MS Word +# 97-2003 binary .doc files that need antiword (Docker) to convert — not run +# here. Section numbering (1..9 with trailing colon), the §6 "Sorte" colour +# split (bele:/rdeče:), and the "Tradicionalna imena" predikat-roster boiler- +# plate in §2 are reproduced from the real bizeljcan/teran extracts under +# raw/si/specifikacije-extracted/ so the asserted slugs + style truncation +# match the live parser output. + +SPECIFIKACIJA PROIZVODA v skladu s 118 c členom Uredbe Sveta 1234/2007 + +1. Ime, ki naj se zaščiti: + +Bizeljčan + +2. Opis vin: + +Bela vina so sveža, sortna, z izrazito kislino. Rdeča vina so lahkotna. +Pridelujejo se mirna bela in rdeča vina. + +Tradicionalna imena: vrhunsko vino ZGP, pozna trgatev, izbor, jagodni izbor, +suhi jagodni izbor, ledeno vino, slamno vino, penina, vrhunsko peneče vino. + +3. Posebni enološki postopki: + +Brez posebnosti. + +4. Opredelitev geografskega območja: + +Vino se prideluje v vinorodnem okolišu Bizeljsko Sremič, na območju občin +Brežice in Krško. + +5. Največji donos: + +Bela: 10 000 kg grozdja na hektar. Rdeča: 9 000 kg grozdja na hektar. + +6. Sorte: + +bele: Laški rizling, Beli pinot, Sivi pinot, Chardonnay, Sauvignon, +Rumeni plavec, Zeleni silvanec, Renski rizling, Dišeči traminec, Šipon, +Ranina, Ranfol, Rumeni muškat, Rizvanec, Kraljevina, Bela žlahtnina, Kerner; +rdeče: Modra frankinja, Žametovka, Modri pinot, Portugalka, Šentlovrenka, +Gamay, Zweigelt, Rdeča žlahtnina; + +7. Povezava z geografskim območjem: + +Gričevnata pokrajina ob reki Sotli z meljasto-glinastimi tlemi in toplim +celinskim podnebjem daje vinom polnost in svežino. + +8. Veljavne zahteve: + +Označevanje v skladu z nacionalno zakonodajo. + +9. Pregledi (preverjanje skladnosti s specifikacijo proizvoda): + +Nadzor izvaja pooblaščena organizacija. diff --git a/tests/fixtures/si_mkgp_doc_teran.txt b/tests/fixtures/si_mkgp_doc_teran.txt new file mode 100644 index 0000000..2ffb290 --- /dev/null +++ b/tests/fixtures/si_mkgp_doc_teran.txt @@ -0,0 +1,36 @@ +# synthetic +# Synthetic minimal fixture mirroring the MKGP per-wine ".doc" SPECIFIKACIJA +# PROIZVODA for a SINGLE-variety wine (Teran). The real input is a binary .doc +# needing antiword. §6 "Sorte" here has NO colour prefix (bele:/rdeče:) — it +# exercises the no-colour-marker fallback that treats the whole §6 body as one +# comma list. Cross-checked against raw/si/specifikacije-extracted/teran.json +# (refošk -> refosco-dal-peduncolo-rosso, colour noir, style rouge). + +SPECIFIKACIJA PROIZVODA v skladu s 118 c členom Uredbe Sveta 1234/2007 + +1. Ime, ki naj se zaščiti: + +Teran + +2. Opis vin: + +Teran je rdeče vino temno rubinaste barve z izrazito kislino in nizko stopnjo +alkohola. + +4. Opredelitev geografskega območja: + +Vino teran se prideluje le na območju podokoliša Kraške planote, znotraj +vinorodnega okoliša Kras. + +6. Sorte: + +refošk + +7. Povezava z geografskim območjem: + +Rdeča jerovica (terra rossa) na apnenčasti kraški planoti in vpliv burje dajeta +teranu značilno barvo, kislino in mineralnost. + +8. Veljavne zahteve: + +Označevanje v skladu z nacionalno zakonodajo. diff --git a/tests/fixtures/si_pravilnik_2007_bela-krajina.html b/tests/fixtures/si_pravilnik_2007_bela-krajina.html new file mode 100644 index 0000000..9adb9d4 --- /dev/null +++ b/tests/fixtures/si_pravilnik_2007_bela-krajina.html @@ -0,0 +1,28 @@ + + + +

Pravilnik o seznamu geografskih označb za vina in trsnem izboru, stran 6732.

+

Na podlagi 8. člena Zakona o vinu in drugih proizvodih iz grozdja in vina.

+ +

1. člen

+

Ta pravilnik določa seznam geografskih označb za vina in trsni izbor.

+ +

PRILOGA 2

+

Priporočene in dovoljene sorte po posameznih vinorodnih okoliših:

+

5. v vinorodnem okolišu Bela krajina:

+

a) priporočene sorte: Laški rizling, Beli pinot, Sauvignon, Sivi pinot, +Chardonnay, Rumeni muškat, Modra frankinja, Žametovka;

+

b) dovoljene sorte: Zeleni silvanec, Renski rizling, Ranina, Kraljevina, +Traminec, Dišeči traminec, Kerner, Bela žlahtnina, Modri pinot, Gamay, +Zweigelt, Portugalka, Šentlovrenka, Rdeča žlahtnina.

+ + diff --git a/tests/fixtures/si_pravilnik_2022_belokranjec.html b/tests/fixtures/si_pravilnik_2022_belokranjec.html new file mode 100644 index 0000000..e4ed5c2 --- /dev/null +++ b/tests/fixtures/si_pravilnik_2022_belokranjec.html @@ -0,0 +1,55 @@ + + + +

Pravilnik o vinu s priznanim tradicionalnim poimenovanjem Metliška črnina in +vinu s priznanim tradicionalnim poimenovanjem Belokranjec

+ +

1. člen (vsebina)

+

Ta pravilnik določa pogoje pridelave vina PTP Metliška črnina in vina PTP +Belokranjec v skladu z določbami 5. člena in 9. člena tega pravilnika.

+ +

2. člen (značilnosti vina)

+

(1) Vino PTP Metliška črnina je suho mirno rdeče vino.

+

(2) Vino PTP Belokranjec je suho mirno belo vino rumenkaste barve z +zelenkastimi odtenki, s primarno aromo belih sort in prijetno blago svežega +okusa, kot je opredeljeno v skladu s prejšnjim odstavkom in 9. členu tega +pravilnika.

+ +

3. člen (geografska označba)

+

Geografska označba je Bela krajina, kot je določeno v 4. členu.

+ +

4. člen (področje pridelave)

+

Grozdje za vino PTP Metliška črnina in grozdje za vino PTP Belokranjec ter +vino PTP Metliška črnina in vino PTP Belokranjec se pridelujejo le na območju +vinorodnega okoliša Bela krajina.

+ +

5. člen (sorte vinske trte in gojitvena oblika)

+

(1) Vino PTP Metliška črnina se prideluje iz rdečih sort, navedenih v 6. členu.

+

(2) Sorte vinske trte, iz katerih se prideluje vino PTP Belokranjec, so:

+

1. Kraljevina;

+

2. Laški rizling;

+

3. Beli pinot;

+

4. Chardonnay;

+

5. Zeleni silvanec;

+

6. Sauvignon;

+

7. Renski rizling;

+

8. Rumeni muškat;

+

9. Kerner;

+

10. Sivi pinot.

+ +

6. člen (število trsov na hektar)

+

Najmanjše število trsov na hektar je določeno v skladu s 5. členom.

+ + diff --git a/tests/fixtures/sk_jednotny_dokument_new_stredoslovenska.html b/tests/fixtures/sk_jednotny_dokument_new_stredoslovenska.html new file mode 100644 index 0000000..b2f9f15 --- /dev/null +++ b/tests/fixtures/sk_jednotny_dokument_new_stredoslovenska.html @@ -0,0 +1,27 @@ + +
+

PRÍLOHA

+

JEDNOTNÝ DOKUMENT

+

1. Názov výrobku

+

Stredoslovenská

+

2. Druh zemepisného označenia

+

CHZO – chránené zemepisné označenie

+

3. Kategórie vinohradníckych/vinárskych výrobkov

+

1. Víno

+

4. Opis vína (vín)

+

Biele, ružové a červené vína.

+

5. Vinárske výrobné postupy

+

a. Základné enologické postupy

+

6. Vymedzená zemepisná oblasť

+

Stredoslovenská vinohradnícka oblasť je vinohradnícka oblasť, ktorá je ohraničená hranicami katastrálnych území obcí: Abovce, Bátka, Bátorová.

+

7. Hlavné muštové odrody

+

Furmint

+

8. Opis súvislostí

+

Údaje o zemepisnej oblasti. Stredoslovenská vinohradnícka oblasť sa rozkladá pozdĺž južnej hranice Slovenskej republiky.

+

9. Ďalšie základné podmienky (balenie, označovanie, iné požiadavky)

+

Označovanie v zmysle platnej legislatívy.

+
diff --git a/tests/fixtures/sk_jednotny_dokument_old_skalicky-rubin.html b/tests/fixtures/sk_jednotny_dokument_old_skalicky-rubin.html new file mode 100644 index 0000000..7fbc987 --- /dev/null +++ b/tests/fixtures/sk_jednotny_dokument_old_skalicky-rubin.html @@ -0,0 +1,30 @@ + +
+

PRÍLOHA

+

JEDNOTNÝ DOKUMENT

+

1. Názov

+

Skalický rubín

+

2. Druh zemepisného označenia

+

CHOP – chránené označenie pôvodu

+

3. Kategórie vinárskych výrobkov

+

1. Víno

+

4. Opis vína (vín)

+

Červené víno s rubínovou farbou.

+

5. Enologické postupy

+

a. Základné enologické postupy

+

6. Vymedzená oblasť

+

Zemepisná jednotka na výrobu vína Skalický rubín je ohraničená hranicami katastrálneho územia mesta Skalica a hranicami katastrálnych území obcí Mokrý Háj, Popudinské Močidľany.

+

7. Hlavné muštové odrody

+

Svätovavrinecké

+

Frankovka modrá

+

Modrý Portugal

+

8. Údaje potvrdzujúce spojitosť

+

Územie sa nachádza na úpätí Bielych Karpát s geologickou deformáciou zemskej kôry vytvorenej povodím rieky Morava. Pôda je prevažne černozem.

+

9. Ďalšie základné podmienky

+

Balenie a označovanie v zmysle platnej legislatívy.

+
diff --git a/tests/fixtures/sk_jednotny_dokument_tokaj_predikat.html b/tests/fixtures/sk_jednotny_dokument_tokaj_predikat.html new file mode 100644 index 0000000..b6275ee --- /dev/null +++ b/tests/fixtures/sk_jednotny_dokument_tokaj_predikat.html @@ -0,0 +1,44 @@ + +
+

PRÍLOHA

+

JEDNOTNÝ DOKUMENT

+

1. Názov (názvy)

+

Vinohradnícka oblasť Tokaj

+

2. Druh zemepisného označenia

+

CHOP – chránené označenie pôvodu

+

3. Kategórie vinohradníckych/vinárskych výrobkov

+

1. Víno · 3. Likérové víno · 4. Šumivé víno

+

4. Opis vín

+

Tokajské vína rôznych kategórií a predikátov.

+

5. Tokajský výber päťputňový

+

STRUČNÝ SLOVNÝ OPIS. Organoleptické vlastnosti: typický tokajský charakter z hrozna napadnutého ušľachtilou plesňou.

+

6. Tokajská esencia

+

STRUČNÝ SLOVNÝ OPIS. Esencia z najsladšieho cibébového hrozna.

+

7. Muštová odroda (muštové odrody)

+

Furmint

+

Kabar

+

Kövérszőlő - Tučné hrozno

+

Lipovina

+

Muškát žltý

+

Zéta - Zeta

+

8. Opis súvislostí

+

Údaje o zemepisnej oblasti. Tokajská vinohradnícka oblasť leží na juhovýchode Slovenska.

+

20. Akostné víno s prívlastkom bobuľový výber

+

STRUČNÝ SLOVNÝ OPIS. Hrozno prešlo hrozienkovatením.

+

23. Akostné víno s prívlastkom ľadové víno

+

STRUČNÝ SLOVNÝ OPIS. Hrozno zberané za mrazu.

+

24. Akostné víno s prívlastkom slamové víno

+

STRUČNÝ SLOVNÝ OPIS. Hrozno sušené na slame.

+

25. Likérové víno

+

STRUČNÝ SLOVNÝ OPIS. Organoleptické vlastnosti: jantárová farba.

+

26. Sekt V. O.

+

STRUČNÝ SLOVNÝ OPIS. Šumivé víno vyrobené tradičnou metódou.

+
diff --git a/tests/fixtures/sk_specifikacija_karpatska_prihlaska.txt b/tests/fixtures/sk_specifikacija_karpatska_prihlaska.txt new file mode 100644 index 0000000..6ce37fe --- /dev/null +++ b/tests/fixtures/sk_specifikacija_karpatska_prihlaska.txt @@ -0,0 +1,33 @@ +Redacted excerpt of the 1996-era UPV SR "Prihlaska oznacenia povodu" +for Karpatska perla (application 0005-96; pdftotext -layout of +raw/sk/national-specs/karpatska-perla.pdf). This is the OLDER numbered +03.N template (not the modern lettered a-i one) with a FLAT inline +variety list in 03.5 ("z ... odrod vinnej revy: ...") rather than a +two-column Odroda/Synonymum table. The PDF is an OCR-scanned image, so +the variety blob is noisy ("C hardonnay", "Mu ~kat Ottonel", "MUller +Thurgau", "Dievc i c hrozno") — the fuzzy matcher + targeted OCR repairs +recover it. Public source (uradne dielo), licence-clear. + +03.1 Názov výrobku vrátane znenia označenia pôvodu: + + Karpatská perla + +03.2 Zemepisné vymedzenie územia, na ktorom sa uskutočňuje výroba: + + Územie sa nachádza na svahoch Malých Karpát s pôdami vzniknutými + zvetrávaním žuly. Podnebie je teplé, kontinentálne. + +03.4 Opis vlastností výrobku daných zemepisným prostredím: + + Vína sú charakteristické vyššou extraktívnosťou a buketnosťou, + ktorú podmieňuje terroir a história karpatských Nemcov. + +03.5 Opis spôsobu získavania, prípadne opis originálnych miestnych spôsobov výroby: + + Karpatská perla sa vyrába z tradič n ých odrôd vinnej révy: Aurelius, Bouvierovo hrozno, Devín, + Dievč í c hrozno, Feteasca regala, C hardonnay, Irsai Oliver, Muškát moravský, Mu ~kát Ottonel, + MUller Thurgau, Neubu rské, Pálava, Rizling rýnsky, Rizling vlašský, Rulandské biele, Rulandské + šedé, Sauvignon, Silvánske zelené, Tramín červený, Veltlinske červené skoré, Veltllnske zelené, + Alibernet, André, Cabernet Sauvignon, Dunaj, Frankovka modr á, Modrý Portugal, Neronet, + Rulandské modré, Svätovavrinecké, Zweigeltrebe z hrozna, ktorého cukornatosť dosiahla najmenej + 16 °NM s naj vyšším hektá rovým výnosom 14 000 kg. diff --git a/tests/fixtures/sk_specifikacija_nitrianska_two_column.txt b/tests/fixtures/sk_specifikacija_nitrianska_two_column.txt new file mode 100644 index 0000000..f97010e --- /dev/null +++ b/tests/fixtures/sk_specifikacija_nitrianska_two_column.txt @@ -0,0 +1,58 @@ +Redacted excerpt of the UPV SR national specifikacia for Nitrianska +vinohradnicka oblast (pdftotext -layout of +raw/sk/national-specs/nitrianska.pdf). Kept to the b / d / f / g lettered +sections. The seam is the f-section: a two-column Odroda / Synonymum +table grouped under MUSTOVE BIELE (white) and MUSTOVE MODRE (blue-black). +The parser must take ONLY the left Odroda column (the >=2-space gutter): +the "Feteasca regala" row carries the synonym "Pesecka leanka" on the +right, which on its own resolves to feteasca-ALBA (a different variety) — +that synonym must never reach the matcher. Column gutter spacing is +preserved verbatim. Public source (uradne dielo), licence-clear. + +b) Opis vína + +Biele, ružové a červené vína. Vyrábajú sa aj sekty a likérové vína. + +d) vymedzenie príslušnej zemepisnej oblasti + +Nitrianska vinohradnícka oblasť zahŕňa katastrálne územia obcí Nitra, +Zlaté Moravce, Vráble. + +f) označenie odrody alebo odrôd viniča, z ktorého sa víno vyrába: + +V Nitrianskej vinohradníckej oblasti je povolené pestovať všetky odrody podľa aktuálnej Listiny +registrovaných odrôd. Pre túto oblasť je povolené používať nasledujúce synonymá pre +označenie odrôd: + + Odroda Synonymum +MUŠTOVÉ Aurelius + + BIELE Bouvierovo hrozno + Devín + Dievčie hrozno Leányka, Mädchentraube, Dívčí hrozen, Feteasca alba + Feteasca regala Pesecká leánka, Pesecké dievčie hrozno + Chardonnay Chardonnay blanc, Pinot blanc Chardonnay, Pinot + Irsai Oliver Irsay, Muskat Oliver + Muškát Ottonel Ottonel muskotály, Muscat Ottonel, Miszket Otonel + Müller – Thurgau Rizling szilváni, Rizvanac, Rivaner + Rizling vlašský Olasz rizling, Welschriesling, Taljanska grasevina, Ryzlink + vlašský + Rulandské šedé Szürkebarát, Pinot gris, Ruländer, Pinot grigio, Klevner + Sauvignon Sauvignon blanc, Sauvignon biely + Tramín červený Tramín, Tramini, Roter traminer, Gewürtztraminer + Alibernet + Frankovka modrá Frankovka + Modrý Portugal Kékoportó, Blauer Portugieser, Portugalské modré +MUŠTOVÉ + Nitranka + MODRÉ + Rudava + Rulandské modré Pinot noir, Červený klevner, Blauer klewner, Pinot nero + Svätovavrinecké Szentlörinc, Saint Laurent, Svatovařinecké + Zweigeltrebe Rotburger, Blauer Zweigeltrebe, Zweigelt + +g) údaje potvrdzujúce spojitosť + +Údaje o zemepisnej oblasti. Nitrianska vinohradnícka oblasť patrí medzi +najrozmanitejšie spomedzi všetkých slovenských. Vinohrady sú vysadené na +sprašových pahorkatinách. V oblasti vládne veľmi teplé a suché klíma. diff --git a/tests/test_at_parser.py b/tests/test_at_parser.py new file mode 100644 index 0000000..a25f0da --- /dev/null +++ b/tests/test_at_parser.py @@ -0,0 +1,359 @@ +"""Fixture-based regression tests for the Austria (AT) parser. + +Two target modules, both landed in a single commit (38387ab "austria and +slovenia") — there are no later "fix"-flavoured commits to mine, so the +cases below pin the documented seams of the parser instead of past bugs: + + - scripts/at/02_extract_pliegos.py — the EUR-Lex "EINZIGES DOKUMENT" + HTML driver: slice from the anchor, find numbered `ti-grseq-1` + section headers (the SECTION_NUM_RE guard drops the numberless decoy + headers "KURZE TEXTBESCHREIBUNG" / italic "Wachau g.U." the same way + the RO parser drops "DOCUMENT UNIC"), route bodies by German title + keyword, and parse the section-7 "Offizieller Name - Synonym, …" + variety lines (canonical name = segment before the dash; the synonym + blob is a fallback). + - scripts/_lib/at/einziges_dokument.py — the German keyword/role tables, + the geo_area title blocklist ("Art der geografischen Angabe" carries + "geografische" but its body is just "g.U."), the German colour + vocabulary, and the Prädikat/Schaumwein STYLE_MARKERS. + - scripts/_lib/at/region.py — Bundesland derivation: the curated + file_number map (authoritative) + a free-text scan fallback. + +Real cached docs live under raw/at/oj-pages/ (gitignored). The fixtures +here are short redacted excerpts under tests/fixtures/at_*.html. + +Assertions are on STRUCTURE (routed roles, slug sets, colour split, +section keys), not full-output snapshots. Several tests pin ACTUAL +parser behaviour that diverges from the docstring's ideal — the empty +`colour` field for VIVC-folded varieties (Blaufränkisch→lemberger, +Weißer Burgunder→pinot-blanc, Weißer Riesling→riesling) and the +adjective-vs-noun gap in find_bundesland_in_text. The divergence is +called out inline at each such test. +""" + +from __future__ import annotations + +import importlib +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.at import region # noqa: E402 +from _lib.at.einziges_dokument import ( # noqa: E402 + _GEO_AREA_TITLE_BLOCKLIST, + COLOUR_BY_KEYWORD, + SECTION_ROLE_KEYWORDS, + STYLE_MARKERS, +) + +# 02_extract_pliegos starts with a digit, so import it by module path. +extract = importlib.import_module("at.02_extract_pliegos") + + +def _route_html(html: str) -> tuple[dict, dict, dict]: + """Slice → extract numbered sections → route. Returns (sections, + titles, routed) the way build_record drives them.""" + doc = extract.slice_einziges_dokument(html) + assert doc is not None, "EINZIGES-DOKUMENT anchor must be found" + sections, titles = extract.extract_sections(doc) + routed = extract.route_sections(sections, titles) + return sections, titles, routed + + +# ========================================================================== +# EINZIGES DOKUMENT HTML driver — anchor slice + section routing +# ========================================================================== + +def test_anchor_slice_drops_preamble(fixture_text): + html = fixture_text("at_einziges_dokument_wachau.html") + doc = extract.slice_einziges_dokument(html) + # The "VERÖFFENTLICHUNG EINES ÄNDERUNGSANTRAGS" modification preamble + # before EINZIGES DOKUMENT is dropped. + assert "ÄNDERUNGSANTRAGS" not in doc + assert doc.lstrip().startswith(" curated file_number map > text scan. + assert region.derive_bundesland( + {"bundesland": "Wien", "file_number": "PDO-AT-A0205"} + ) == "Wien" + # Curated map wins over a misleading text candidate. + assert region.derive_bundesland( + {"file_number": "PDO-AT-A0228"}, "irgendwo in Niederösterreich" + ) == "Steiermark" + # Text scan is the fallback only when the file_number is unknown. + assert region.derive_bundesland( + {"file_number": "PDO-AT-ZZZZZ"}, "das Gebiet liegt in der Steiermark" + ) == "Steiermark" + # Nothing resolves → empty string. + assert region.derive_bundesland({"file_number": "PDO-AT-ZZZZZ"}, "nichts") == "" + + +def test_derive_bundesland_end_to_end_wachau(fixture_text): + """End-to-end through the real build path: Wachau's file_number + PDO-AT-A0205 resolves to Niederösterreich via the curated map even + though the section-6 text only carries the inflected adjective.""" + _sections, _titles, routed = _route_html( + fixture_text("at_einziges_dokument_wachau.html") + ) + bl = region.derive_bundesland( + {"file_number": "PDO-AT-A0205"}, + routed.get("geo_area", ""), + routed.get("link_to_terroir", ""), + "Wachau", + ) + assert bl == "Niederösterreich" diff --git a/tests/test_bg_parser.py b/tests/test_bg_parser.py new file mode 100644 index 0000000..2305cc6 --- /dev/null +++ b/tests/test_bg_parser.py @@ -0,0 +1,427 @@ +"""Fixture-based regression tests for the Bulgaria (BG) parsers. + +Bulgaria is the first CYRILLIC-script corpus. The load-bearing invariant +that distinguishes BG from every Latin-script country: string handling +never goes through NFKD-ASCII (which collapses Cyrillic to the empty +string) — slugs route through `unidecode`, and commune matching uses +`.casefold()`. See the Cyrillic-handling note in CLAUDE.md. + +Three target modules, each a seam that has regressed historically (and +that the BG section of CLAUDE.md enumerates): + + - scripts/bg/02_extract_pliegos.py + scripts/_lib/bg/edinen_dokument.py + — the EU-OJ "ЕДИНЕН ДОКУМЕНТ" HTML driver. Slice from the anchor, + find numbered `ti-grseq-1` headers, route by Bulgarian title keyword. + The HU/BG monotonic-number + role-keyword guard in `extract_sections` + filters per-style / per-variety subsections nested inside section 4 + (the same nested-`ti-grseq-1` decoy issue as Hungary) so the real + sections 5–9 are not shadowed. + - scripts/_lib/bg/specifikacija.py — the IAVV national-spec template + (numbered 1–8). §5 colour split (за бели вина / за червени вина и + розе / розе), §6 terroir (а) Природни / б) Човешки). + - scripts/_lib/bg/commune.py — Cyrillic-preserving obshtina matching: + `.casefold()` (NOT NFKD-ASCII), settlement-tier prefixes (с./гр./ + село/град) dropped, област markers consumed with their trailing name. + +Real cached docs live under raw/bg/{oj-pages,national-specs}/ (gitignored). +The HTML/text fixtures here are short redacted excerpts under +tests/fixtures/bg_* (the melnik / specifikacija slices preserve real +load-bearing Cyrillic; the nested-subsections HTML mirrors the +dunavska-ravnina section-4 decoy structure). + +Assertions are on STRUCTURE (routed roles incl. the nested-subsection +guard, Cyrillic→slug variety sets, colour split, commune casefold), not +on full-output snapshots. Where a test pins ACTUAL behaviour that diverges +from the docstring's ideal (the section-9 "Други основни условия" drop, +the "находящи се" participle leak), the divergence is called out inline. +""" +from __future__ import annotations + +import importlib +import sys +import unicodedata +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.bg import commune # noqa: E402 +from _lib.bg.commune import _normalise_commune, parse_commune_list # noqa: E402 +from _lib.bg.edinen_dokument import ( # noqa: E402 + _GEO_AREA_TITLE_BLOCKLIST, + SECTION_ROLE_KEYWORDS, +) +from _lib.bg.specifikacija import parse_specifikacija # noqa: E402 + +# 02_extract_pliegos starts with a digit, so import it by module path. +extract = importlib.import_module("bg.02_extract_pliegos") + + +def _route_html(html: str) -> tuple[dict, dict, dict]: + """Slice → extract numbered sections → route, the way build_record + drives them. Returns (sections, titles, routed).""" + doc = extract.slice_document_unic(html) + assert doc is not None, "ЕДИНЕН ДОКУМЕНТ anchor must be found" + sections, titles = extract.extract_sections(doc) + routed = extract.route_sections(sections, titles) + return sections, titles, routed + + +# ========================================================================== +# ЕДИНЕН ДОКУМЕНТ HTML driver — anchor slice + section routing +# ========================================================================== + +def test_anchor_slice_drops_preamble(fixture_text): + html = fixture_text("bg_edinen_dokument_melnik.html") + doc = extract.slice_document_unic(html) + # The modification/communication preamble before ЕДИНЕН ДОКУМЕНТ is dropped. + assert "COMMUNICATION preamble" not in doc + assert "да се отреже" not in doc + # The slice starts at the anchor paragraph itself. + assert doc.lstrip().startswith("` + headers numbered 1.→4. (Бели вина / Вина розе / Червени вина / + Качествени пенливи вина — per-wine-type subsections that restart the + numbering at 1). A naive first-occurrence dedupe would let those decoy + bodies shadow the real sections 5–9. `extract_sections`'s guard — + monotonic top-level number (must not go backwards) + the title must + contain a section-role keyword — filters them. (Same nested-ti-grseq-1 + decoy issue as Hungary.)""" + sections, titles, routed = _route_html( + fixture_text("bg_edinen_dokument_nested_subsections.html") + ) + # The nested 1.→4. colour-bucket decoys never registered as top-level + # sections: section 4's title is the real "Описание …", not a colour. + assert titles["4"].startswith("Описание на виното") + for decoy in ("Бели вина", "Вина розе", "Червени вина", "Качествени пенливи вина"): + assert decoy not in titles.values(), decoy + # Sections 5–8 survived with their real roles (not shadowed by section + # 4's per-style subsection bodies). + assert titles["5"].startswith("Винопроизводствени практики") + assert titles["6"].startswith("Определен географски район") + assert titles["7"].startswith("Винен сорт") + assert titles["8"].startswith("Описание на връзката") + assert "viticultural_practices" in routed + assert "geo_area" in routed + assert "grape_varieties" in routed + assert "link_to_terroir" in routed + # The geo_area body is section 6's obshtina list — NOT a wine-style + # description body that a shadowing bug would have left there. + assert "община Свищов" in routed["geo_area"] + assert "Бели вина" not in routed["geo_area"] + # Grapes come from the real section 7 (the colour-bucket decoy bodies + # in section 4 carry no variety names). + assert "kadarka" in set(routed and extract.parse_grapes( + routed["grape_varieties"])["principal"]) + + +def test_actual_section9_other_conditions_title_dropped(fixture_text): + """DISCREPANCY pin: the nested fixture's section 9 is titled "Други + основни условия (…)". The additional_conditions keyword table carries + "други условия" / "други специфични изисквания" / "други съществени + условия" — but "Други ОСНОВНИ условия" has "основни" between the two + words, so the substring "други условия" misses and the section is + silently dropped (no additional_conditions role). This mirrors the + real dunavska-ravnina extraction, which also keeps only sections 1–8. + The melnik fixture (title "Други специфични изисквания") does match — + so the gap is the specific "основни условия" wording, not all of + section 9.""" + _sections, titles, routed = _route_html( + fixture_text("bg_edinen_dokument_nested_subsections.html") + ) + assert "9" not in titles + assert "additional_conditions" not in routed + + +# ========================================================================== +# Grape parsing — Cyrillic → slug, em-dash synonym split, colour +# ========================================================================== + +def test_grape_parsing_cyrillic_to_slug_and_synonym_split(fixture_text): + """Section 7 is a flat per-line list. `Name - synonym` (plain hyphen + with spaces) keeps the canonical Bulgarian name and drops the Latin + synonym blob. Cyrillic names resolve to English-canonical slugs via the + shared lexicon; the Latin-script "Viognier" line resolves too.""" + _sections, _titles, routed = _route_html( + fixture_text("bg_edinen_dokument_melnik.html") + ) + grapes = extract.parse_grapes(routed["grape_varieties"]) + slugs = set(grapes["principal"]) + # Cyrillic → English-canonical slugs. + assert {"grenache", "cabernet-sauvignon", "cabernet-franc", "merlot", + "chardonnay", "muscat-ottonel"} <= slugs + # Latin-script line in a Cyrillic doc still resolves. + assert "viognier" in slugs + # Bulgarian native varieties keep their own slugs. + assert "kerasuda" in slugs # Керацуда + assert "shiroka-melnishka-loza" in slugs # Широка мелнишка лоза + # `Тамянка` folds to muscat-blanc-a-petits-grains (VIVC synonym chain). + assert "muscat-a-petits-grains" in slugs + # Display name is the segment BEFORE " - " (synonym dropped). + by_slug = {d["slug"]: d for d in grapes["details"]} + assert by_slug["sandanski-misket"]["name"] == "Мискет сандански" + assert "Мускат" not in by_slug["sandanski-misket"]["name"] + assert by_slug["shiroka-melnishka-loza"]["name"] == "Широка мелнишка лоза" + # Colour comes from the lexicon match. + assert by_slug["cabernet-sauvignon"]["colour"] == "noir" + assert by_slug["chardonnay"]["colour"] == "blanc" + # BG single document has no principal/accessory split — all principal. + assert grapes["accessory"] == [] + + +def test_grape_parsing_gamza_folds_to_kadarka(fixture_text): + """`Гъмза` (same DNA as Kadarka) folds to the `kadarka` slug — pinned + because it is one of the BG → international folds listed in CLAUDE.md.""" + _sections, _titles, routed = _route_html( + fixture_text("bg_edinen_dokument_nested_subsections.html") + ) + grapes = extract.parse_grapes(routed["grape_varieties"]) + by_slug = {d["slug"]: d for d in grapes["details"]} + assert "kadarka" in grapes["principal"] + assert by_slug["kadarka"]["name"] == "Гъмза" + + +# ========================================================================== +# geo_area blocklist — section 2 ("Вид на географското означение") decoy +# ========================================================================== + +def test_geo_area_blocklist_table_present(): + # Section 2's title carries "географско" but its body is just ЗНП/ЗГУ; + # the blocklist must keep it out of geo_area. A missing entry re-opens + # the regression. + assert "вид на географското означение" in _GEO_AREA_TITLE_BLOCKLIST + assert "вид на географското указание" in _GEO_AREA_TITLE_BLOCKLIST + + +def test_regression_section2_kind_not_routed_to_geo_area(fixture_text): + """Section 2 "Вид на географското означение" (body "ЗНП — …") carries + the inflected "географското" so it would otherwise shadow the real + area in section 6. The blocklist keeps geo_area on section 6 (commune + list), not the ЗНП/ЗГУ kind decoy.""" + _sections, _titles, routed = _route_html( + fixture_text("bg_edinen_dokument_melnik.html") + ) + geo = routed.get("geo_area", "") + assert geo.strip() != "ЗНП — Защитено наименование за произход" + assert not geo.startswith("ЗНП") + # The real section-6 area landed. + assert "Районът за производство" in geo + + +def test_geo_area_keyword_table_most_specific_first(): + # The most-specific area title is listed first so it wins over the + # bare "географски район" / "географска зона" forms. + geo_keywords = SECTION_ROLE_KEYWORDS["geo_area"] + assert geo_keywords[0] == "определен географски район" + + +# ========================================================================== +# geo_area → commune list (Cyrillic-preserving obshtina parse) +# ========================================================================== + +def test_section6_geo_communes_settlements_dropped(fixture_text): + """The section-6 area body lists obshtini (`в община NAME`) each + followed by their settlements (`с. X`, `гр. X`). parse_commune_list + keeps only the 4 obshtini and drops every settlement.""" + _sections, _titles, routed = _route_html( + fixture_text("bg_edinen_dokument_melnik.html") + ) + communes = parse_commune_list(routed["geo_area"]) + assert communes == ["Сандански", "Петрич", "Струмяни", "Кресна"] + # No settlement (с./гр. prefixed) leaked through as a commune. + for c in communes: + assert not c.startswith("с.") and not c.startswith("гр.") + + +# ========================================================================== +# commune.py — Cyrillic-preserving normalisation + tier handling +# ========================================================================== + +def test_casefold_preserves_cyrillic_unlike_nfkd_ascii(): + """The core BG invariant: NFKD-ASCII folding (the RO/Latin path) erases + Cyrillic entirely; the BG normaliser uses .casefold(), which keeps it. + A regression to the NFKD path would silently produce empty keys → zero + commune matches → every BG wine falls off the obshtina-union geometry.""" + name = "Сливен" + nfkd_ascii = unicodedata.normalize("NFKD", name).encode("ascii", "ignore").decode() + assert nfkd_ascii == "", "NFKD-ASCII must erase Cyrillic (the trap)" + # The BG normaliser keeps Cyrillic, just casefolded + tier-stripped. + assert _normalise_commune("община " + name) == "сливен" + assert _normalise_commune(name) == "сливен" + + +def test_normalise_strips_settlement_and_tier_prefix(): + # Both община (obshtina tier) and гр./с. (settlement tier) prefixes are + # stripped; the bare Cyrillic name (two words preserved) remains. + assert _normalise_commune("община Велико Търново") == "велико търново" + assert _normalise_commune("гр. Сандански") == "сандански" + assert _normalise_commune("с. Лехово") == "лехово" + + +def test_commune_oblast_marker_consumed_with_trailing_name(): + # `в област NAME` is consumed together with its province name so the + # province does not bleed in as an obshtina candidate. + out = parse_commune_list( + "землищата на гр. Сливен в община Сливен, находящи се в област Сливен." + ) + # The obshtina "Сливен" survives (from `в община Сливен`); both the + # settlement `гр. Сливен` and the province `област Сливен` are stripped. + assert "Сливен" in out + + +def test_commune_plural_obshtini_sublist_keeps_two_word_name(): + # `в общините NAME, NAME и NAME` introduces a sub-list; a two-word + # obshtina (Велико Търново) survives intact. + out = parse_commune_list("в общините Велико Търново, Свищов и Павликени.") + assert out == ["Велико Търново", "Свищов", "Павликени"] + + +def test_commune_province_name_obshtina_survives(): + # 28 of 265 obshtini share their name with their parent province. The + # province occurrence (`област Пловдив`) is consumed, but the obshtina + # of the same name after `общините` survives. + out = parse_commune_list("в област Пловдив, в общините Пловдив, Асеновград и Сопот.") + assert "Пловдив" in out + assert "Асеновград" in out and "Сопот" in out + + +def test_commune_settlement_only_chunk_dropped(): + # A list of bare settlements (с. X) with no parent obshtina names yields + # nothing — settlements are LAU3, not in the GISCO LAU obshtina index. + out = parse_commune_list("с. Лехово, с. Петрово, с. Яново.") + assert out == [] + + +def test_actual_participle_leak_находящи_се(): + """DISCREPANCY pin: `находящи се` ("located in", a participle phrase + preceding `в област …`) is NOT in _PROSE_TOKENS, so when the area body + is just `гр. X, находящи се в област Y` the participle survives as a + spurious commune candidate (matching the real nova-zagora extraction, + whose geo_communes are ['Шивачево', 'находящи се']). Pinned so a future + _PROSE_TOKENS edit that filters it is a conscious change, not an + accident.""" + out = parse_commune_list("землищата на гр. Сливен, находящи се в област Сливен.") + assert "находящи се" in out + + +def test_commune_oblast_names_table_casefolded(): + # The province-name table is casefolded Cyrillic (not ASCII), so it + # actually matches the casefolded commune keys. + assert "пловдив" in commune._OBLAST_NAMES + assert "велико търново" in commune._OBLAST_NAMES + # No entry is empty (which an NFKD-ASCII fold would have produced). + assert all(n for n in commune._OBLAST_NAMES) + + +# ========================================================================== +# specifikacija.py — IAVV national-spec template (numbered 1–8) +# ========================================================================== + +def test_specifikacija_numbered_sections_and_template(fixture_text): + text = fixture_text("bg_specifikacija_sliven.txt") + out = parse_specifikacija(text, "sliven") + assert out["parser_template"] == "iavv-specifikacija-v1" + # 8 numbered sections sliced on the `N.` line anchors. + assert out["n_sections"] == 8 + # Roles routed by leading-text keyword scan. + for role in ("description", "geo_area", "grape_varieties", "link_to_terroir"): + assert role in out["section_roles"], role + + +def test_specifikacija_section5_colour_split(fixture_text): + """§5 groups varieties under Bulgarian colour markers. `за бели вина:` + → blanc; `за червени вина и розе:` → noir (the whole marker — incl. the + `и розе` second-colour suffix — is consumed so `розе` is NOT later + treated as a variety separator). Every variety carries its bucket + colour; there is no principal/accessory split (all principal).""" + text = fixture_text("bg_specifikacija_sliven.txt") + out = parse_specifikacija(text, "sliven") + by_slug = {d["slug"]: d for d in out["grapes"]["details"]} + # Whites + for slug in ("rkatsiteli", "chardonnay", "ugni-blanc", "muscat-ottonel", + "cherven-misket", "dimyat", "aligote"): + assert by_slug[slug]["colour"] == "blanc", slug + # Reds (incl. the Bulgarian crossing Шевка → shevka). The `и розе` + # suffix did not produce a spurious "розе" variety. + for slug in ("cabernet-sauvignon", "merlot", "pinot-noir", "pamid", "shevka"): + assert by_slug[slug]["colour"] == "noir", slug + assert "rose" not in by_slug # "и розе" suffix not parsed as a variety + # No principal/accessory split. + assert out["grapes"]["accessory"] == [] + assert set(out["grapes"]["principal"]) == set(by_slug) + # 7 white + 5 red. + assert len([d for d in out["grapes"]["details"] if d["colour"] == "blanc"]) == 7 + assert len([d for d in out["grapes"]["details"] if d["colour"] == "noir"]) == 5 + + +def test_specifikacija_section6_terroir_natural_and_human(fixture_text): + """§6 "Връзка с географския район." carries the `а) Природни фактори` / + `б) Човешки фактори` subsections — both must land in link_to_terroir.""" + text = fixture_text("bg_specifikacija_sliven.txt") + out = parse_specifikacija(text, "sliven") + link = out["link_to_terroir"] + assert "Природни фактори" in link + assert "Човешки фактори" in link + assert "умереноконтинентален климат" in link + # The terroir text did NOT swallow the next section (§7 requirements). + assert "Приложими изисквания" not in link + + +def test_specifikacija_geo_area_is_section3(fixture_text): + # §3 (Районът за производство …) routes to geo_area, distinct from the + # §6 terroir text. + text = fixture_text("bg_specifikacija_sliven.txt") + out = parse_specifikacija(text, "sliven") + assert "Районът за производство" in out["geo_area_brief"] + assert "Сливен" in out["geo_area_brief"] + + +def test_specifikacija_styles_from_grape_colours(fixture_text): + # With both white and red varieties present, colour-derived styles are + # blanc + rouge (the §2 sparkling marker is redacted out of this + # excerpt, so only the colour bases assert). + text = fixture_text("bg_specifikacija_sliven.txt") + out = parse_specifikacija(text, "sliven") + assert "blanc" in out["styles"] + assert "rouge" in out["styles"] diff --git a/tests/test_cz_parser.py b/tests/test_cz_parser.py new file mode 100644 index 0000000..02e9497 --- /dev/null +++ b/tests/test_cz_parser.py @@ -0,0 +1,373 @@ +"""Fixture-based regression tests for the Czech Republic (CZ) parsers. + +CZ is the worst single-document corpus: all 13 EU-OJ wines ship as +content-stubs (no fetchable JEDNOTNÝ DOKUMENT), so the data — and the +parser regression surface — lives in the NATIONAL-SPEC layer, not in the +EU-OJ HTML driver. Three target modules: + + - scripts/_lib/cz/national_spec.py — the two Czech wine-law decrees. + `parse_varieties` → Vyhláška 88/2017 Sb. Příloha 2: 3 Roman- + numeral colour blocks (I. white / II. red / III. zemské-víno), + each a flat `. Name ` table. The block-terminator + slicing (cut BEFORE the next "II."/"III."/"IV." marker) keeps + the marker from leaking into the prior block's last entry and + keeps the IV. abbreviations table out of the variety roster. + `parse_commune_tree` → Vyhláška 254/2010 Sb. Příloha: a 3-column + rowspan table (Vinařská obec / Katastrální území / Název viniční + trati) walked by a rowspan-tracking state machine that yields + the column-0 obec name exactly once per obec regardless of how + many KÚ / trať rows it spans, switching the active macro-region + on the "A. … ČECHY" / "B. … MORAVA" headers. + + - scripts/_lib/cz/chzo_spec.py — the SZPI CHZO (PGI) product-spec + parser (`szpi-chzo-specifikace-v1`) over `pdftotext -layout`: + Section 1 (Popis vinařského regionu) → region_terroir_text + (sliced from the "1 …" header to the next top-level "3 …", + so section 3 Enologické postupy does NOT bleed in). + Section 2 (Druhy výrobků) → style roster (still colours + + Likérové→vin-de-liqueur / Šumivé→sparkling / Perlivé→ + semi-sparkling). + + - scripts/_lib/cz/jednotny_dokument.py — Czech keyword/role tables for + the (never-yet-exercised) EU-OJ JEDNOTNÝ DOKUMENT. A small unit test + of the role tables + a synthetic single document driven through the + cz/02_extract_pliegos extractor, since no real CZ document fires it. + +Real cached docs live under raw/cz/national-specs/ (gitignored — the +decree HTMLs + the SZPI PDFs). Fixtures here are short redacted excerpts; +the JEDNOTNÝ DOKUMENT fixture is `# synthetic` (no real doc exists). + +Assertions are on STRUCTURE (per-colour slug sets, the rowspan state +machine's obec membership per podoblast, CHZO terroir-text presence + +style roster, JD role routing), not full-output snapshots. The +`name`-vs-lexicon-slug distinction is pinned explicitly: parse_varieties +emits its own NFKD slug (`frankovka`), while stage-04's augmenter resolves +the `name` through match_variety to the canonical lexicon slug +(`blaufrankisch`) — both layers are exercised. +""" +from __future__ import annotations + +import importlib +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.cz import chzo_spec, national_spec # noqa: E402 +from _lib.cz.jednotny_dokument import ( # noqa: E402 + _GEO_AREA_TITLE_BLOCKLIST, + SECTION_ROLE_KEYWORDS, +) +from _lib.grape_entity import match_variety # noqa: E402 + +# cz/02_extract_pliegos starts with a digit → import by module path. +extract = importlib.import_module("cz.02_extract_pliegos") + + +# ========================================================================== +# national_spec.parse_varieties — Vyhláška 88/2017 Sb. variety table +# ========================================================================== + +def _varieties(fixture_text) -> dict: + return national_spec.parse_varieties( + fixture_text("cz_vyhlaska88_varieties.html") + ) + + +def _by_colour(out: dict, colour: str) -> list[dict]: + return [v for v in out["varieties"] if v["colour"] == colour] + + +def test_varieties_three_colour_blocks_counted(fixture_text): + out = _varieties(fixture_text) + # The redacted fixture keeps 7 white / 6 red / 3 zemské rows; the + # block counters reflect the colour split, not one flat list. + assert out["n_white"] == 7 + assert out["n_red"] == 6 + assert out["n_zemske"] == 3 + assert len(out["varieties"]) == 16 + # The Příloha č. 2 anchor is recorded so the panel cites the decree. + assert out["source_anchor"] == "Příloha č. 2 k vyhlášce č. 88/2017 Sb." + + +def test_varieties_per_block_slug_sets(fixture_text): + out = _varieties(fixture_text) + # parse_varieties emits its OWN NFKD-ASCII slug (`_slug`), not the + # grape-lexicon slug — pin the structure per colour block. + white = {v["slug"] for v in _by_colour(out, "blanc")} + red = {v["slug"] for v in _by_colour(out, "noir")} + zemske = {v["slug"] for v in _by_colour(out, "zemske")} + assert white == { + "aurelius", "chardonnay", "muller-thurgau", "rulandske-sede", + "ryzlink-rynsky", "sauvignon", "veltlinske-zelene", + } + assert red == { + "andre", "cabernet-sauvignon", "frankovka", "merlot", + "modry-portugal", "svatovavrinecke", + } + assert zemske == {"bily-portugal", "modry-janek", "tramin-zluty"} + # Every variety carries its block colour. + for v in _by_colour(out, "blanc"): + assert v["colour"] == "blanc" + for v in _by_colour(out, "noir"): + assert v["colour"] == "noir" + + +def test_varieties_name_resolves_to_lexicon_slug(fixture_text): + """The augmenter (scripts/_lib/augment/cz.py) feeds the parser's + `name` — NOT its `slug` — through match_variety to get the canonical + grape-lexicon slug. The parser's own slug and the lexicon slug + diverge for renamed natives, so pin the lexicon mapping on the + load-bearing ones.""" + out = _varieties(fixture_text) + by_name = {v["name"]: v for v in out["varieties"]} + # Parser NFKD slug vs lexicon slug diverge here: + assert by_name["Frankovka"]["slug"] == "frankovka" + assert match_variety("Frankovka").slug == "blaufrankisch" + assert match_variety("Svatovavřinecké").slug == "sankt-laurent" + assert match_variety("Modrý Portugal").slug == "blauer-portugieser" + assert match_variety("Ryzlink rýnský").slug == "riesling" + assert match_variety("Veltlínské zelené").slug == "gruner-veltliner" + + +def test_varieties_preserves_source_typo_muller_thurgau(fixture_text): + """The decree mis-types "Müller Thurgau" with U+0170 Ű + ("Műller Thurgau"). The parser preserves the source spelling + verbatim (never silently corrects it), and the lexicon still folds + both spellings to muller-thurgau.""" + out = _varieties(fixture_text) + mt = next(v for v in out["varieties"] if "ller Thurgau" in v["name"]) + assert mt["name"] == "Műller Thurgau" # source typo kept + assert mt["slug"] == "muller-thurgau" + assert match_variety(mt["name"]).slug == "muller-thurgau" + + +def test_varieties_multi_abbreviation_kept_off_name(fixture_text): + """A comma-separated abbreviation ("RŠ, RS") must be stripped off the + name entirely — the trailing-abbreviation regex consumes both + alternates so the name is the bare "Rulandské šedé".""" + out = _varieties(fixture_text) + rs = next(v for v in out["varieties"] if v["name"].startswith("Rulandské šedé")) + assert rs["name"] == "Rulandské šedé" + assert "RŠ" not in rs["name"] and "RS" not in rs["name"] + + +def test_regression_block_terminator_no_marker_leak(fixture_text): + """The Roman-numeral block headers (II./III./IV.) are part of the + literal terminator, so block slicing cuts BEFORE the marker. The + last white variety must be clean "Veltlínské zelené" (not glued to a + trailing "II"), and the last red "Svatovavřinecké" (not "… III").""" + out = _varieties(fixture_text) + names = {v["name"] for v in out["varieties"]} + assert "Veltlínské zelené" in names + assert "Svatovavřinecké" in names + for n in names: + assert not n.endswith(" II") and not n.endswith(" III") + + +def test_regression_abbreviation_table_not_parsed_as_varieties(fixture_text): + """The III. (zemské) block ends at the IV. abbreviations table + ("Seznam zkratek pro některé tradiční výrazy"), which is a + traditional-term list, NOT a variety list. Its rows + ("Jakostní víno", "Pozdní sběr") must never enter the roster.""" + out = _varieties(fixture_text) + names = {v["name"] for v in out["varieties"]} + assert "Jakostní víno" not in names + assert "Pozdní sběr" not in names + # And the Příloha č. 3 wine-defects appendix below it is sliced off too. + assert not any("chorob" in n.lower() for n in names) + + +# ========================================================================== +# national_spec.parse_commune_tree — Vyhláška 254/2010 Sb. rowspan table +# ========================================================================== + +def _commune_tree(fixture_text) -> dict: + return national_spec.parse_commune_tree( + fixture_text("cz_vyhlaska254_commune_tree.html") + ) + + +def test_commune_tree_podoblast_keys_and_anchor(fixture_text): + out = _commune_tree(fixture_text) + # One entry per podoblast heading the rowspan walker saw a table for. + assert set(out["podoblasti"]) == {"melnicka", "litomericka", "mikulovska"} + assert out["source_anchor"] == "Příloha k vyhlášce č. 254/2010 Sb." + + +def test_commune_tree_macro_region_switch(fixture_text): + """The active macro-region flips on the "A. … ČECHY" / "B. … MORAVA" + headers. The two Bohemian podoblasti carry Čechy; the Moravian one + carries Morava — proving the state machine switched mid-document.""" + out = _commune_tree(fixture_text) + assert out["podoblasti"]["melnicka"]["macro_region"] == "Čechy" + assert out["podoblasti"]["litomericka"]["macro_region"] == "Čechy" + assert out["podoblasti"]["mikulovska"]["macro_region"] == "Morava" + + +def test_commune_tree_rowspan_yields_obec_once(fixture_text): + """The core of the rowspan state machine: an obec cell with + rowspan="3" spans 3 physical rows (one per KÚ / trať), but the + walker must yield the column-0 obec name exactly ONCE — never once + per spanned row, and never the column-1 KÚ name.""" + out = _commune_tree(fixture_text) + mel = out["podoblasti"]["melnicka"] + # "Benátky nad Jizerou" spans 3 rows, "Cítov" spans 2 — each once. + assert mel["communes"] == ["Benátky nad Jizerou", "Cítov", "Kuks"] + # The KÚ-column cells ("Nové Benátky", "Obodř", "Cítov") and the + # trať-column cells ("Pod zámkem", …) must NOT appear as obce. + assert "Nové Benátky" not in mel["communes"] + assert "Obodř" not in mel["communes"] + assert "Pod zámkem" not in mel["communes"] + + +def test_commune_tree_obec_name_strips_leading_ordinal(fixture_text): + """The obec cell carries a "1. Benátky nad Jizerou" ordinal prefix; + _LEADING_ORDINAL_RE strips it so the bare obec name remains (and a + multi-word name like "Benátky nad Jizerou" stays intact).""" + out = _commune_tree(fixture_text) + for pod in out["podoblasti"].values(): + for obec in pod["communes"]: + assert not obec[:3].strip().rstrip(".").isdigit() + assert "Benátky nad Jizerou" in out["podoblasti"]["melnicka"]["communes"] + + +def test_commune_tree_second_podoblast_communes(fixture_text): + out = _commune_tree(fixture_text) + lit = out["podoblasti"]["litomericka"] + # Bělušice spans 3 rows (3 traťs), Most one — each once. + assert lit["communes"] == ["Bělušice", "Most"] + mik = out["podoblasti"]["mikulovska"] + assert mik["communes"] == ["Bavory", "Březí"] + + +# ========================================================================== +# chzo_spec.parse_chzo_spec — SZPI CHZO product spec +# ========================================================================== + +def _chzo(fixture_text) -> dict: + return chzo_spec.parse_chzo_spec( + fixture_text("cz_chzo_moravske_layout.txt"), "moravske" + ) + + +def test_chzo_region_and_pgi_file_number(fixture_text): + out = _chzo(fixture_text) + assert out["region"] == "Morava" + assert out["pgi_file_number"] == "PGI-CZ-A0902" + assert out["source_anchor"] == "1 Popis vinařského regionu" + + +def test_chzo_terroir_text_is_section_1_body(fixture_text): + """Section-1 terroir text spans the region intro + 1.1 climate + 1.2 + geology/soils. It is the regulator-grounded terroir source for every + CZ wine in the region (tier-agnostic), so its presence is the gating + assertion for CZ terroir-fact extraction.""" + out = _chzo(fixture_text) + terroir = out["region_terroir_text"] + assert terroir.startswith("1 Popis vinařského regionu") + # 1.1 + 1.2 subsection content is included. + assert "HUGLIN" in terroir + assert "vápenci" in terroir + assert "Chardonnay" in terroir + + +def test_regression_chzo_section_1_stops_before_section_3(fixture_text): + """_slice_top_section cuts section 1 at the next top-level header + (sub=None). Section 2 (Druhy výrobků) sits between, then section 3 + (Základní enologické postupy). The terroir slice must NOT swallow + section 2 or 3 — pin that the enology prose stays out.""" + out = _chzo(fixture_text) + terroir = out["region_terroir_text"] + assert "Enologické postupy" not in terroir + assert "Likérové víno" not in terroir # section 2 stays out of terroir + + +def test_chzo_style_roster(fixture_text): + """Section 2 yields the style roster: still colours (white/red/rose + from the 2.1 subsection headers) + the three special wine types.""" + out = _chzo(fixture_text) + styles = set(out["styles"]) + # Still-wine colours. + assert {"white", "rose", "red"} <= styles + # Likérové → vin-de-liqueur, Šumivé → sparkling, Perlivé → semi-sparkling. + assert "vin-de-liqueur" in styles + assert "sparkling" in styles + assert "semi-sparkling" in styles + # The list is sorted + deduped. + assert out["styles"] == sorted(set(out["styles"])) + + +def test_chzo_unknown_slug_yields_empty_region(fixture_text): + # A slug not in CHZO_REGION / CHZO_PGI_FILE_NUMBER yields empty + # region + pgi but still parses the section bodies. + out = chzo_spec.parse_chzo_spec( + fixture_text("cz_chzo_moravske_layout.txt"), "bogus" + ) + assert out["region"] == "" + assert out["pgi_file_number"] == "" + # Styles still resolve (they come from the text, not the slug map). + assert "sparkling" in out["styles"] + + +# ========================================================================== +# jednotny_dokument — role tables + synthetic single-document routing +# ========================================================================== + +def test_jd_role_keyword_tables_cover_the_four_core_roles(): + """The four semantic roles every downstream consumer reads must be + present with their canonical Czech section titles first.""" + assert SECTION_ROLE_KEYWORDS["name"][0] == "název" + assert SECTION_ROLE_KEYWORDS["geo_area"][0] == "vymezená zeměpisná oblast" + assert SECTION_ROLE_KEYWORDS["grape_varieties"][0] == "hlavní moštové odrůdy" + assert SECTION_ROLE_KEYWORDS["link_to_terroir"][0] == "popis souvislostí" + + +def test_jd_geo_area_blocklist_excludes_category_titles(): + """"Druh zeměpisného označení" and "Kategorie výrobků z révy vinné" + both contain "zeměpis"/"výrobk" tokens that could lure the geo_area + matcher; the blocklist keeps them off the geo_area role.""" + assert "druh zeměpisného označení" in _GEO_AREA_TITLE_BLOCKLIST + assert "kategorie výrobků z révy vinné" in _GEO_AREA_TITLE_BLOCKLIST + + +def _route_jd(html: str) -> tuple[dict, dict, dict]: + doc = extract.slice_jednotny_dokument(html) + assert doc is not None, "JEDNOTNÝ DOKUMENT anchor must be found" + sections, titles = extract.extract_sections(doc) + routed = extract.route_sections(sections, titles) + return sections, titles, routed + + +def test_jd_synthetic_anchor_and_section_routing(fixture_text): + """Synthetic single document driven through the cz/02 extractor: the + anchor regex finds JEDNOTNÝ DOKUMENT, the numbered ti-grseq-1 headers + parse, and the Czech keyword tables route each body to its role.""" + html = fixture_text("cz_jednotny_dokument_synthetic.html") + _sections, _titles, routed = _route_jd(html) + # The four core roles all routed. + assert {"name", "geo_area", "grape_varieties", "link_to_terroir"} <= set(routed) + # Section 6 body → geo_area (the area sentence, not the category). + assert "podoblast mělnická" in routed["geo_area"] + # Section 8 body → link_to_terroir (the terroir prose). + assert "spraše" in routed["link_to_terroir"] + + +def test_jd_synthetic_preamble_dropped(fixture_text): + # slice_jednotny_dokument starts AT the anchor → the modification + # preamble before "JEDNOTNÝ DOKUMENT" is dropped. + html = fixture_text("cz_jednotny_dokument_synthetic.html") + doc = extract.slice_jednotny_dokument(html) + assert "KOMUNIKACE O ZMĚNĚ" not in doc + + +def test_jd_synthetic_grape_section_resolves_lexicon_slugs(fixture_text): + """The grape section is `Name - synonym` per line; the canonical + Czech name (before " - ") resolves via the shared lexicon. No + principal/accessory split in the CZ single document → all principal.""" + html = fixture_text("cz_jednotny_dokument_synthetic.html") + _sections, _titles, routed = _route_jd(html) + grapes = extract.parse_grapes(routed["grape_varieties"]) + slugs = set(grapes["principal"]) + assert {"riesling", "sankt-laurent", "pinot-noir"} <= slugs + assert grapes["accessory"] == [] diff --git a/tests/test_de_parser.py b/tests/test_de_parser.py new file mode 100644 index 0000000..94d4353 --- /dev/null +++ b/tests/test_de_parser.py @@ -0,0 +1,496 @@ +"""Fixture-based regression + behaviour tests for the Germany (DE) parsers. + +Two parser surfaces, each a documented seam (see the DE section of +CLAUDE.md and the module docstrings): + + - scripts/de/02_extract_pliegos.py + scripts/_lib/de/einziges_dokument.py + — the EU-OJ "EINZIGES DOKUMENT" HTML driver: slice from the + `

EINZIGES DOKUMENT

` anchor, find numbered + `ti-grseq-1` section headers, route by German title keyword, and + parse the section-7/8 variety list whose lines are + `Canonical Name — Synonym, Synonym` (canonical kept, synonyms dropped). + The geo_area role carries a title BLOCKLIST so section 3 "Art der + geografischen Angabe" (which contains "geografisch") does not shadow + the real abgegrenztes Gebiet. + + - scripts/_lib/de/produktspezifikation.py — the BLE Produktspezifikation + PDF parser with FOUR template branches the BLE drafted across agency + eras (the highest-value target — a tweak for one template has + historically broken another): + A numbered "8 Zugelassene Keltertraubensorten" + §3.2 per-variety + Mindestmostgewicht (de-facto principal). Mosel, Pfalz, Nahe, … + (tolerates the Saale-Unstrut "Kellertraubensorten" PDF typo). + B un-numbered "Zugelassene Keltertraubensorten:" + bullet + "• Weißwein" / "• Rot- und Roséwein". Ahr, Sachsen. + C "7. Rebsorten" (NOT §8) + "• Rebsorten für Weißwein" bullets + and inline "insbes. {variety} mit rd. X %" / §5.1 named- + Mostgewicht principals. Rheingau, Hessische Bergstraße. + D Baden multi-Bereich §3.2.X tiered Mostgewicht rows (lowest tier + = Leitsorten → principal), with a flat §8 the Template-A parser + harvests for the full authorised set. + The 02f role-split tag is `section-3.2-principal` when §3.2 names + per-variety principals, else `section-8-flat-no-split`. + +Fixtures are short redacted excerpts of public, licence-clear regulator +documents (BLE Produktspezifikationen = Amtliches Werk §5 UrhG; EU-OJ +documents) under tests/fixtures/de_*.{txt,html}. The HTML fixture is a +`# synthetic` faithful reconstruction of the real OJ markup (the real +page is 45 KB; the marker is in an HTML comment on line 1). + +Assertions are on STRUCTURE (routed roles, variety NAME/slug sets, the +template letter, the principal/accessory split + its role_split_method +tag), not full-output snapshots. Where a test pins ACTUAL behaviour that +diverges from the docstring ideal (the `\\btrocken\\b` inflection gap; +the trailing `;` on a §8 name) the divergence is called out inline. +""" + +from __future__ import annotations + +import importlib +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.de import produktspezifikation as ps # noqa: E402 +from _lib.de.einziges_dokument import ( # noqa: E402 + _GEO_AREA_TITLE_BLOCKLIST, + SECTION_ROLE_KEYWORDS, +) +from _lib.grape_entity import match_variety # noqa: E402 + +# 02_extract_pliegos starts with a digit, so import it by module path. +extract = importlib.import_module("de.02_extract_pliegos") + + +# ========================================================================== +# Einziges Dokument HTML driver — anchor slice + section routing +# ========================================================================== + +def _route_html(html: str) -> tuple[dict, dict, dict]: + """Slice → extract numbered sections → route, the way build_record + drives them. Returns (sections, titles, routed).""" + doc = extract.slice_einziges_dokument(html) + assert doc is not None, "EINZIGES DOKUMENT anchor must be found" + sections, titles = extract.extract_sections(doc) + routed = extract.route_sections(sections, titles) + return sections, titles, routed + + +def test_anchor_slice_drops_preamble(fixture_text): + html = fixture_text("de_einziges_dokument_wurzburger_stein_berg.html") + doc = extract.slice_einziges_dokument(html) + # The "Veröffentlichung eines Antrags" preamble before the anchor is gone. + assert "Veröffentlichung eines Antrags" not in doc + # The slice begins at the EINZIGES DOKUMENT anchor paragraph. + assert doc.lstrip().startswith(" tuple[str, set[str], set[str]]: + """Mirror scripts/de/02f_extract_produktspezifikation.py:_build_record's + role-split logic over already-text-extracted Produktspezifikation + content. Returns (role_split_method, principal_slugs, accessory_slugs). + + §3.2 principal names → principal; every other §8 name → accessory. + When §3.2 is empty, everything in §8 is principal (flat-no-split).""" + principal_raw = ps.parse_section_3_2_principal_names(text) + section_8 = ps.parse_section_8_authorised(text) + all_names = [*section_8["white"], *section_8["red"]] + + principal_slugs: set[str] = set() + for name in principal_raw: + m = match_variety(name) + if m is not None: + principal_slugs.add(m.slug) + + accessory_slugs: set[str] = set() + if principal_slugs: + role_split_method = "section-3.2-principal" + for name in all_names: + m = match_variety(name) + if m is None or m.slug in principal_slugs: + continue + accessory_slugs.add(m.slug) + else: + role_split_method = "section-8-flat-no-split" + for name in all_names: + m = match_variety(name) + if m is not None: + principal_slugs.add(m.slug) + + return role_split_method, principal_slugs, accessory_slugs diff --git a/tests/test_gr_parser.py b/tests/test_gr_parser.py new file mode 100644 index 0000000..984f844 --- /dev/null +++ b/tests/test_gr_parser.py @@ -0,0 +1,339 @@ +"""Fixture-based regression tests for the Greece (GR) ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ parser. + +Target modules: + + - scripts/_lib/gr/eniaio_engrafo.py — the Greek keyword/role tables plus + the `greek_norm` comparator (casefold + polytonic/monotonic diacritic + strip + FINAL SIGMA fold ς→σ + short parenthetical/slash inflection + drop). The final-sigma fold is the critical seam: a Greek section + title typed with a final ς (e.g. "Κυριότερες οινοποιήσιμες ποικιλίες") + casefolds to a medial σ via capital Σ, so without the ς→σ fold the + keyword table — typed with ς — would never match and the grape + section would never route. See the GR section of CLAUDE.md and the + module docstring. + - scripts/gr/02_extract_pliegos.py — the EU-OJ HTML driver + (slice_document_unic → extract_sections → route_sections → + parse_grapes / parse_styles). Reuses the ES/IT/RO idiom: walk the + `ti-grseq-1` headers, find the ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ anchor, keep a monotonic + 1→N run of role-titled top-level sections, route by Greek title + keyword. + - scripts/_lib/grape_entity.py — `_COLOUR_LETTER_TO_NAME` and + `match_variety`. Greek section-7/8 variety lines carry an OIV colour + code (Greek capital Β/Ν/Γ — glyph-identical to but distinct code + points from Latin B/N/G — or Latin B/N/G/Rs/Rg). The colour-letter + suffix overrides the variety's natural colour. + +Real cached docs live under raw/gr/oj-pages/ (gitignored); the fixtures +here are short redacted excerpts under tests/fixtures/gr_*.html. + +Assertions are on STRUCTURE (greek_norm behaviour, routed roles, slug set ++ colour split), not full-output snapshots. + +DISCREPANCY (pinned to ACTUAL behaviour, flagged inline at +test_regression_2024_country_decoy_routes_as_geo_area): the post-2024 +template's section "Χώρα στην οποία ανήκει η γεωγραφική περιοχή" (body +"Ελλάδα") carries the substring "γεωγραφική περιοχή", so it collides with +the geo_area keyword and — because it is NOT in _GEO_AREA_TITLE_BLOCKLIST — +the parser routes the "Ελλάδα" country decoy as geo_area. This is the GR +analogue of RO's "Țara căreia → România" decoy, which RO *does* blocklist +(see tests/test_ro_parser.py). The GR blocklist has the gap. +""" +from __future__ import annotations + +import importlib +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.gr.eniaio_engrafo import ( # noqa: E402 + _GEO_AREA_TITLE_BLOCKLIST, + DOC_ANCHOR_NORM, + SECTION_ROLE_KEYWORDS, + greek_norm, +) +from _lib.grape_entity import _COLOUR_LETTER_TO_NAME # noqa: E402 + +# 02_extract_pliegos starts with a digit, so import it by module path. +extract = importlib.import_module("gr.02_extract_pliegos") + + +def _route_html(html: str) -> tuple[dict, dict, dict]: + """Slice → extract numbered sections → route, the way build_record drives + them. Returns (sections, titles, routed).""" + doc = extract.slice_document_unic(html) + assert doc is not None, "ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ anchor must be found" + sections, titles = extract.extract_sections(doc) + routed = extract.route_sections(sections, titles) + return sections, titles, routed + + +# ========================================================================== +# greek_norm — the comparator seam (final sigma + diacritics + inflection) +# ========================================================================== + +def test_greek_norm_folds_final_sigma(): + """ς (U+03C2) and medial σ (U+03C3) collapse to the same key. This is + THE critical fold: `.casefold()` of capital Σ yields medial σ, so a + title rendered in capitals ("…ΠΟΙΚΙΛΙΕΣ") casefolds to a trailing σ + while the keyword table is typed with ς — without the fold they never + compare equal.""" + assert greek_norm("ποικιλίες") == greek_norm("ποικιλίεσ") == "ποικιλιεσ" + # No final sigma survives in any normalised output. + assert "ς" not in greek_norm("ΟΙΝΟΠΟΙΗΣΙΜΕΣ ΠΟΙΚΙΛΙΕΣ") + # A capital-Σ title and a final-ς keyword normalise to the same key. + assert greek_norm("Κυριότερες οινοποιήσιμες ποικιλίες") == greek_norm( + "κυριοτερεσ οινοποιησιμεσ ποικιλιεσ" + ) + + +def test_greek_norm_strips_polytonic_and_monotonic_diacritics(): + # Polytonic accents (older OJ pages) collapse to the monotonic / bare + # base, so the anchor matches regardless of accent style. + assert greek_norm("ΕΝΙΑΊΟ ΈΓΓΡΑΦΟ") == greek_norm("ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ") + assert greek_norm("ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ") == DOC_ANCHOR_NORM == "ενιαιο εγγραφο" + + +def test_greek_norm_drops_short_inflection_groups(): + # The combined singular/plural template splices `(εσ)` into a heading; + # greek_norm drops short (≤5-char) parens so the canonical wording is + # contiguous again. + assert greek_norm("Κύρια(εσ) ποικιλία(εσ)") == greek_norm("Κύρια ποικιλία") + # The post-2024 slash variant `ποικιλίας/-ών` drops the same way. + assert greek_norm("ποικιλίας/-ών αμπέλου") == "ποικιλιασ αμπελου" + # But a real long alternation (`λευκός/ερυθρός`) is preserved (the + # trailing-letter lookahead leaves the slash in place). + assert "/" in greek_norm("λευκός/ερυθρός") + + +def test_regression_final_sigma_section7_routes_to_grapes(fixture_text): + """The Mantinia section-7 title "Κυριότερες οινοποιήσιμες ποικιλίες" + ends in a final ς. Its body must route to grape_varieties — this is + the exact title that the final-sigma fold first unblocked (per the + module docstring + CLAUDE.md GR section). Without the ς→σ fold the + title would never match the `ποικιλίες`/`ποικιλιεσ` keyword.""" + _sections, titles, routed = _route_html( + fixture_text("gr_eniaio_engrafo_mantinia.html") + ) + # The section-7 title carries a final sigma. + assert titles["7"].endswith("ποικιλίες") + assert "grape_varieties" in routed + # Its body (the variety table), not section 6's area body, landed. + assert "Μοσχοφίλερο" in routed["grape_varieties"] + assert "επαρχία Μαντινε" not in routed["grape_varieties"] + + +# ========================================================================== +# HTML driver — anchor slice + section routing +# ========================================================================== + +def test_anchor_slice_drops_modification_preamble(fixture_text): + """The modification-preamble template carries an outer + ΑΙΤΗΣΗ-ΓΙΑ-ΤΡΟΠΟΠΟΙΗΣΗ block (numbered 1 / 2.1) BEFORE the inner + ΕΝΙΑΙΟ ΕΓΓΡΑΦΟ anchor; slice_document_unic must drop it so the + preamble's own numbered headers don't pollute the section run.""" + html = fixture_text("gr_eniaio_engrafo_mantinia.html") + doc = extract.slice_document_unic(html) + assert doc is not None + # The preamble marker text is gone — slice starts at the anchor. + assert "ΠΡΟΟΙΜΙΟ" not in doc + assert "ΑΙΤΗΣΗ ΓΙΑ ΤΡΟΠΟΠΟΙΗΣΗ" not in doc + assert doc.lstrip().startswith(" tuple[dict, dict, dict]: + """Slice → extract numbered sections → route, the way build_record drives + them. Returns (sections, titles, routed).""" + doc = extract.slice_jedinstveni_dokument(html) + assert doc is not None, "JEDINSTVENI-DOKUMENT anchor must be found" + sections, titles = extract.extract_sections(doc) + routed = extract.route_sections(sections, titles) + return sections, titles, routed + + +# ========================================================================== +# JEDINSTVENI DOKUMENT — HTML driver: anchor + section routing +# ========================================================================== + +def test_anchor_slice_drops_preamble(fixture_text): + html = fixture_text("hr_jedinstveni_dokument_ponikve.html") + doc = extract.slice_jedinstveni_dokument(html) + assert doc is not None + # The "PROIZVODA" doc-title line before the anchor is dropped; the slice + # begins at the JEDINSTVENI DOKUMENT header tag itself. + assert "oj-doc-ti" not in doc + assert "JEDINSTVENI" in doc.lstrip()[:60] + + +def test_section_routing_ponikve(fixture_text): + _sections, _titles, routed = _route_html( + fixture_text("hr_jedinstveni_dokument_ponikve.html") + ) + # The four semantic roles the downstream consumers depend on. + for role in ("name", "geo_area", "grape_varieties", "link_to_terroir"): + assert role in routed, role + # Section 1 body → name. + assert routed["name"] == "Ponikve" + # Section 6 body (area) lands in geo_area, NOT terroir prose. + assert "vinogradarski položaj Ponikve" in routed["geo_area"] + # Section 7 body lands in grape_varieties. + assert "Maraština" in routed["grape_varieties"] + # Section 8 body lands in link_to_terroir. + assert "sredozemna klima" in routed["link_to_terroir"] + + +def test_section_keys_are_number_prefixed(fixture_text): + """Only the numbered top-level headers register as sections; the bare + decoy lines below the anchor ("„PONIKVE”", the file-number "PDO-HR-…") + carry no leading "N." and must NOT become sections.""" + html = fixture_text("hr_jedinstveni_dokument_ponikve.html") + doc = extract.slice_jedinstveni_dokument(html) + sections, titles = extract.extract_sections(doc) + assert set(sections) == set(titles) + assert set(sections) == {"1", "2", "3", "4", "5", "6", "7", "8", "9"} + for num in sections: + assert num[0].isdigit(), f"section key {num!r} should be number-prefixed" + + +# ========================================================================== +# JEDINSTVENI DOKUMENT — grape parsing (em-dash synonym split + colour) +# ========================================================================== + +def test_grape_parsing_em_dash_synonym_split(fixture_text): + """`Maraština – Rukatac, Maraškin, …` → the canonical name before the + ` – ` em-dash separator resolves; the synonym blob is only a fallback. + Display name is the segment BEFORE the dash (synonyms dropped).""" + _sections, _titles, routed = _route_html( + fixture_text("hr_jedinstveni_dokument_ponikve.html") + ) + grapes = extract.parse_grapes(routed["grape_varieties"]) + assert grapes["principal"] == ["marastina", "plavac-mali", "posip"] + by_slug = {d["slug"]: d for d in grapes["details"]} + # Display name is the head segment; synonyms after the em-dash are dropped. + assert by_slug["marastina"]["name"] == "Maraština" + assert "Rukatac" not in by_slug["marastina"]["name"] + # "Plavac mali crni – …" keeps the trailing colour word on the head name + # but still resolves to the plavac-mali slug (noir). + assert by_slug["plavac-mali"]["name"] == "Plavac mali crni" + assert by_slug["plavac-mali"]["colour"] == "noir" + assert by_slug["marastina"]["colour"] == "blanc" + assert by_slug["posip"]["colour"] == "blanc" + # No principal/accessory split in the HR single document → all principal. + assert grapes["accessory"] == [] + + +def test_styles_colour_detection(fixture_text): + """parse_styles scans the description ("Opis vina") + category bodies for + Croatian colour adjectives: bijela vina → blanc, crna vina → noir, + ružičasta vina → rose.""" + sections, titles, _routed = _route_html( + fixture_text("hr_jedinstveni_dokument_ponikve.html") + ) + styles = extract.parse_styles(sections, titles) + assert "blanc" in styles + assert "noir" in styles + assert "rose" in styles + + +# ========================================================================== +# Regression: the "Vrsta oznake zemljopisnog podrijetla" geo_area decoy +# ========================================================================== + +def test_geo_area_blocklist_table_present(): + # Section 2's title carries "zemljopisnog" but its body is just the + # ZOI/ZOZP label — the blocklist must disqualify it from geo_area, or it + # would shadow the real area in section 6. + assert "vrsta oznake zemljopisnog podrijetla" in _GEO_AREA_TITLE_BLOCKLIST + + +def test_regression_section2_zemljopisnog_decoy_not_routed_to_geo_area(fixture_text): + """Section 2 ("Vrsta oznake zemljopisnog podrijetla") contains the + keyword "zemljopisnog" that the geo_area keyword scan matches, but its + body is "ZOI – zaštićena oznaka izvornosti", not an area. The blocklist + must keep geo_area on section 6 (the real area).""" + _sections, _titles, routed = _route_html( + fixture_text("hr_jedinstveni_dokument_ponikve.html") + ) + geo = routed["geo_area"] + assert "zaštićena oznaka izvornosti" not in geo.split(".")[0] + # The real area (section 6) is what landed. + assert "vinogradarski položaj Ponikve" in geo + + +def test_geo_area_keyword_ordering_most_specific_first(): + """The geo_area keyword list is ordered most-specific-first so the full + "razgraničeno zemljopisno područje" wins over the bare + "zemljopisno područje" fragment.""" + geo_keywords = SECTION_ROLE_KEYWORDS["geo_area"] + assert geo_keywords[0] == "razgraničeno zemljopisno područje" + + +# ========================================================================== +# specifikacija.py — lettered a)–h) outline (Dingač-shape PDF) +# ========================================================================== + +def _dingac_text(fixture_text) -> str: + """Load the synthetic lettered fixture (drop the `#` header lines) and + inject a real form-feed before the g) header to exercise the + \\x0c → newline fold.""" + raw = fixture_text("hr_specifikacija_dingac.txt") + body = "".join(line for line in raw.splitlines(keepends=True) + if not line.startswith("#")) + return body.replace("\ng)", "\x0cg)", 1) + + +def test_lettered_section_split(fixture_text): + text = _dingac_text(fixture_text) + sections = spec._lettered_sections(text) + # All eight lettered sections a)–h) carve out. + assert set(sections) == set("abcdefgh") + # f) (grape varieties) holds the colour-marker block. + assert "Bijele sorte:" in sections["f"] + assert "Crne sorte:" in sections["f"] + + +def test_lettered_role_routing(fixture_text): + text = _dingac_text(fixture_text) + sections = spec._lettered_sections(text) + routed = spec._route_sections(sections) + for role in ("name", "description", "geo_area", "grape_varieties", + "link_to_terroir"): + assert role in routed, role + assert "Dingač" in routed["name"] + assert "poluotoka" in routed["geo_area"] + # g) terroir text routes to link_to_terroir. + assert "insolaciju" in routed["link_to_terroir"] + + +def test_regression_formfeed_fold_before_g_section(fixture_text): + """A `\\x0c` form-feed page break right before the g) header must be + folded to a newline so the `^`-anchored g) letter is still matched and + section f) doesn't swallow g)'s terroir body.""" + text = _dingac_text(fixture_text) + assert "\x0cg)" in text # the test injected a real form-feed + sections = spec._lettered_sections(text) + # g) carved out as its own section despite the form-feed. + assert "g" in sections + assert "insolaciju" in sections["g"] + # f) (grapes) did not absorb g)'s terroir prose. + assert "insolaciju" not in sections["f"] + + +def test_regression_forward_only_letter_guard(fixture_text): + """Section c) contains a backward cross-reference "…prema točki e) + Maksimalni urod…". The forward-only guard (an anchor whose letter does + not advance past the last accepted one is a reference, not a heading) + must keep that inline "e)" out of the section map — section e) stays the + real "Maksimalni urod po ha" heading.""" + text = _dingac_text(fixture_text) + sections = spec._lettered_sections(text) + # The cross-reference text survives inside section c)'s body. + assert "prema točki e)" in sections["c"] + # Section e) is the real heading body, not the c) cross-reference fragment. + assert "Najveći dozvoljeni urod" in sections["e"] + + +def test_specifikacija_colour_markers_white_vs_red(fixture_text): + """`Bijele sorte:` → blanc, `Crne sorte:` → noir. Each variety carries + the colour of its marker bucket (the matcher's own colour wins when it + has one, but the bucket is the fallback).""" + text = _dingac_text(fixture_text) + frag = spec.parse_specifikacija(text, "dingac") + by_slug = {d["slug"]: d for d in frag["grapes"]["details"]} + for slug in ("chardonnay", "marastina", "posip", "debit", "riesling"): + assert by_slug[slug]["colour"] == "blanc", slug + for slug in ("plavac-mali", "babic", "plavina", "merlot", "croatina", + "syrah"): + assert by_slug[slug]["colour"] == "noir", slug + # No principal/accessory split — every variety is principal. + assert frag["grapes"]["accessory"] == [] + assert set(frag["grapes"]["principal"]) == set(by_slug) + + +def test_regression_trailing_colour_adjective_fallback(fixture_text): + """`Chardonnay crni` / `Riesling žuti` carry a trailing Croatian colour + adjective that makes the bare matcher reject the adjective-bearing form. + The `_COLOUR_ADJ_RE` fallback in `_add` retries without the trailing + adjective so both still resolve. (Verified: match_variety('Chardonnay + crni') / ('Riesling žuti') return None, while the stripped forms match.) + `Croatina crna` is the documented marker case — it resolves to + `croatina`.""" + text = _dingac_text(fixture_text) + frag = spec.parse_specifikacija(text, "dingac") + slugs = set(frag["grapes"]["principal"]) + # These two only resolve via the trailing-adjective retry. + assert "chardonnay" in slugs + assert "riesling" in slugs + # The Croatina crna marker case. + assert "croatina" in slugs + + +def test_specifikacija_record_fragment_shape(fixture_text): + text = _dingac_text(fixture_text) + frag = spec.parse_specifikacija(text, "dingac") + # Lettered docs (≥ 5 sections) carry the v1 template tag. + assert frag["parser_template"] == "mps-specifikacija-v1" + assert frag["n_sections"] == 8 + # Styles: white + red grapes → blanc + rouge colour-derived styles. + assert "blanc" in frag["styles"] + assert "rouge" in frag["styles"] + # Terroir text from g) is preserved. + assert "insolaciju" in frag["link_to_terroir"] + # geo area brief from d) is preserved. + assert "poluotoka" in frag["geo_area_brief"] + + +# ========================================================================== +# specifikacija.py — .docx keyword-title fallback (no a)–j) prefixes) +# ========================================================================== + +def _primorska_text(fixture_text) -> str: + raw = fixture_text("hr_specifikacija_primorska_docx.txt") + return "".join(line for line in raw.splitlines(keepends=True) + if not line.startswith("#")) + + +def test_docx_lettered_slicer_finds_no_sections(fixture_text): + """Word auto-numbering strips the a)–j) prefixes, so the lettered slicer + finds < 5 sections → parse_specifikacija falls through to the keyword + slicer.""" + text = _primorska_text(fixture_text) + assert len(spec._lettered_sections(text)) < 5 + + +def test_regression_docx_keyword_section_fallback(fixture_text): + """`_keyword_sections` anchors on the role-keyword heading lines + themselves (no letters) and slices each body to the next heading, so the + grape roster + terroir text are still recovered for the docx.""" + text = _primorska_text(fixture_text) + routed = spec._keyword_sections(text) + for role in ("name", "description", "geo_area", "grape_varieties", + "link_to_terroir"): + assert role in routed, role + assert "Maraština" in routed["grape_varieties"] + assert "Sredozemna" in routed["link_to_terroir"] + assert "priobalno" in routed["geo_area"] + + +def test_docx_parser_template_and_grapes(fixture_text): + text = _primorska_text(fixture_text) + frag = spec.parse_specifikacija(text, "primorska-hrvatska") + # The docx path is tagged distinctly from the lettered v1 path. + assert frag["parser_template"] == "mps-specifikacija-docx" + slugs = set(frag["grapes"]["principal"]) + assert {"marastina", "posip", "debit"} <= slugs # whites + assert {"plavac-mali", "babic", "plavina"} <= slugs # reds + by_slug = {d["slug"]: d for d in frag["grapes"]["details"]} + assert by_slug["marastina"]["colour"] == "blanc" + assert by_slug["plavac-mali"]["colour"] == "noir" diff --git a/tests/test_hu_parser.py b/tests/test_hu_parser.py new file mode 100644 index 0000000..9aafaf9 --- /dev/null +++ b/tests/test_hu_parser.py @@ -0,0 +1,333 @@ +"""Fixture-based regression tests for the Hungary (HU) parser. + +Target module: scripts/_lib/hu/egyseges_dokumentum.py — the keyword/role +tables for the EUR-Lex "EGYSÉGES DOKUMENTUM" template — driven through the +HTML machinery in scripts/hu/02_extract_pliegos.py (the slice-from-anchor + +numbered-section extractor + keyword router + grape parser). + +The CRITICAL seam is `extract_sections`: a monotonic-number state machine +that must NOT be fooled by wine-type subsections nested inside section 4 +(description of wines) — and the analogous link subsections inside section 8 +— which re-use `

` and RESTART numbering at 1 +("1. Bor – Rozé fajta és küvé", "2. Bor – Siller…", "1. CLASSICUS BOROK:"). +A naive first-occurrence dedupe would let those nested decoys shadow the +real top-level sections 5–9. Two guards collaborate: (a) accept a top-level +header only when it is the next expected integer; (b) drop a candidate whose +title matches a known nested-subsection prefix (`Bor –`, `Pezsgő`, +`Classicus`, …) even when its number would otherwise fit. + +Real cached docs live under raw/hu/oj-pages/*.html (gitignored). The fixtures +here are short redacted excerpts under tests/fixtures/: + - hu_egyseges_dokumentum_eger.html — Eger, the canonical document carrying + BOTH nested-subsection decoy families (the `Bor –` rows inside §4 and + the `1./2.` `CLASSICUS`/`SUPERIOR` rows between §6 and §7, and the + `1./2./3.` link rows inside §8). + - hu_egyseges_dokumentum_soltvadkerti.html — the older-template variant + whose §8 link title "Kapcsolat a földrajzi területtel" carries the + "földrajzi terület" geo_area keyword and must be blocklisted out of + geo_area. + +Assertions are on STRUCTURE (accepted section keys, routed roles, the +monotonic-number guard dropping the nested decoys, variety slug set), not +full-output snapshots. Where a test pins ACTUAL parser behaviour that +diverges from the obvious expectation (Furmint resolving with an EMPTY +colour string, the Hungarian display names folding to international slugs), +the divergence is called out inline. +""" +from __future__ import annotations + +import importlib +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.hu.egyseges_dokumentum import ( # noqa: E402 + _GEO_AREA_TITLE_BLOCKLIST, + SECTION_ROLE_KEYWORDS, +) + +# 02_extract_pliegos starts with a digit, so import it by module path. +extract = importlib.import_module("hu.02_extract_pliegos") + + +def _route_html(html: str) -> tuple[dict, dict, dict]: + """Slice → extract numbered sections → route. Returns + (sections, titles, routed) the way build_record drives them.""" + doc = extract.slice_egyseges_dokumentum(html) + assert doc is not None, "EGYSÉGES DOKUMENTUM anchor must be found" + sections, titles = extract.extract_sections(doc) + routed = extract.route_sections(sections, titles) + return sections, titles, routed + + +def _ordered_keys(sections: dict) -> list[str]: + return sorted(sections, key=lambda k: [int(p) for p in k.split(".")]) + + +# ========================================================================== +# Anchor slice + section routing +# ========================================================================== + +def test_anchor_slice_drops_preamble(fixture_text): + html = fixture_text("hu_egyseges_dokumentum_eger.html") + doc = extract.slice_egyseges_dokumentum(html) + assert doc is not None + # The STANDARD-AMENDMENT modification preamble before the anchor is dropped. + assert "STANDARD AMENDMENT" not in doc + assert doc.lstrip().lower().startswith("` tag and RESTART numbering at 1. They must NOT be + accepted as top-level sections — otherwise their bodies would shadow the + real sections 5–9. (egyseges_dokumentum nested-subsection guard.)""" + html = fixture_text("hu_egyseges_dokumentum_eger.html") + doc = extract.slice_egyseges_dokumentum(html) + _sections, titles = extract.extract_sections(doc) + # No accepted title is one of the nested `Bor –` wine-type decoys. + for t in titles.values(): + assert not t.startswith("Bor –"), t + # The real description section 4 kept its own title, not a decoy's. + assert titles["4"] == "A bor(ok) leírása" + + +def test_regression_classicus_superior_decoys_between_section_6_and_7(fixture_text): + """Eger restarts numbering again between §6 and §7 with + `1. CLASSICUS BOROK:` / `2. SUPERIOR ÉS GRAND SUPERIOR BOROK:`. These + nested rows must not be accepted as top-level sections; their commune + text lands as section 6's BODY (geo_area), and the real section 7 + (grapes) still wins.""" + html = fixture_text("hu_egyseges_dokumentum_eger.html") + doc = extract.slice_egyseges_dokumentum(html) + sections, titles = extract.extract_sections(doc) + assert "CLASSICUS BOROK:" not in titles.values() + assert "SUPERIOR ÉS GRAND SUPERIOR BOROK:" not in titles.values() + # The decoy commune text is captured as part of section 6's body. + assert "CLASSICUS" in sections["6"] + assert "Andornaktálya" in sections["6"] + # Section 7 is the real grape section, not a CLASSICUS decoy. + assert titles["7"] == "Fontosabb borszőlőfajták" + + +def test_regression_link_subsection_decoys_inside_section_8(fixture_text): + """Section 8 ("A KAPCSOLAT(OK) LEÍRÁSA") nests `1. Körülhatárolt terület + bemutatása`, `2. A borok leírása`, `3. Az okszerű kapcsolat…` rows that + restart numbering at 1. A naive first-occurrence dedupe would let + "2. A borok leírása" (an innocuous-looking title) be accepted as a + top-level section and shadow the real section 9. The monotonic guard + (number 1/2/3 ≠ last_top+1 once §8 is accepted) drops them all.""" + html = fixture_text("hu_egyseges_dokumentum_eger.html") + doc = extract.slice_egyseges_dokumentum(html) + sections, titles = extract.extract_sections(doc) + assert "A borok leírása" not in titles.values() + assert "Az okszerű kapcsolat bemutatása és bizonyítása" not in titles.values() + # The link subsection text is captured as part of section 8's body. + assert "okszerű kapcsolat" in sections["8"].lower() or \ + "Körülhatárolt terület bemutatása" in sections["8"] + # The real section 9 (additional conditions) survived. + assert titles["9"].startswith("További alapvető feltételek") + + +def test_regression_prefix_guard_drops_decoy_when_number_would_fit(): + """Isolates the title-prefix guard (`_looks_like_nested_subsection`) from + the monotonic-number guard: a `Bor –` decoy numbered as the NEXT expected + integer (5, right after §4) would pass the monotonic check, so only the + prefix guard can drop it. The real section 5 (Borkészítési eljárások) + must then win the `5` slot.""" + def H(n: str, t: str) -> str: + return ( + f'

{n}. {t}

' + f"

body of {n} {t[:8]}

" + ) + doc = ( + H("1", "Elnevezés") + + H("2", "A földrajzi árujelző típusa") + + H("3", "A szőlőből készült termékek kategóriái") + + H("4", "A bor(ok) leírása") + + H("5", "Bor – Rozé fajta és küvé") # prefix decoy, number fits + + H("5", "Borkészítési eljárások") # the real section 5 + ) + _sections, titles = extract.extract_sections(doc) + assert titles["5"] == "Borkészítési eljárások" + assert not any(t.startswith("Bor –") for t in titles.values()) + + +def test_looks_like_nested_subsection_table(): + """Direct unit on the prefix predicate: the documented decoy prefixes + fire; the real top-level section titles do not.""" + f = extract._looks_like_nested_subsection + assert f("Bor – Rozé fajta és küvé") + assert f("Bor - Siller") # plain hyphen variant + assert f("Pezsgő") + assert f("CLASSICUS BOROK:") # `classicus` prefix + assert f("Likőrbor") + # Real top-level section titles are NOT nested decoys. + assert not f("A bor(ok) leírása") + assert not f("Körülhatárolt földrajzi terület") + assert not f("Fontosabb borszőlőfajták") + assert not f("A kapcsolat(ok) leírása") + + +# ========================================================================== +# Grape parsing — "Canonical name – Synonym" en-dash split + alias folding +# ========================================================================== + +def test_grape_parsing_endash_synonym_split(fixture_text): + """Section 7 lists `Canonical name – Synonym, …`. The canonical name is + the segment BEFORE the en-dash; the synonym tail is only a fallback. The + display name keeps the canonical Hungarian spelling, sans synonyms.""" + _sections, _titles, routed = _route_html( + fixture_text("hu_egyseges_dokumentum_eger.html") + ) + grapes = extract.parse_grapes(routed["grape_varieties"]) + slugs = set(grapes["principal"]) + # Direct-name resolutions. + assert {"kadarka", "furmint", "syrah", "cabernet-franc", "pinot-noir"} <= slugs + # Display name is the segment before the en-dash (synonyms dropped). + by_slug = {d["slug"]: d for d in grapes["details"]} + assert by_slug["kadarka"]["name"] == "kadarka" + assert "jenei fekete" not in by_slug["kadarka"]["name"] + # No principal/accessory split in the HU single document → all principal. + assert grapes["accessory"] == [] + assert set(grapes["principal"]) == set(by_slug) + + +def test_grape_parsing_hungarian_names_fold_to_international_slugs(fixture_text): + """The GRAPE_ALIAS / matcher folds Hungarian variety names onto the + shared international canonical slugs: tramini → gewurztraminer, + kékfrankos → blaufrankisch, olasz rizling → welschriesling, + királyleányka → feteasca-regala, leányka → feteasca-alba.""" + _sections, _titles, routed = _route_html( + fixture_text("hu_egyseges_dokumentum_eger.html") + ) + grapes = extract.parse_grapes(routed["grape_varieties"]) + slugs = set(grapes["principal"]) + assert "gewurztraminer" in slugs # tramini + assert "blaufrankisch" in slugs # kékfrankos + assert "welschriesling" in slugs # olasz rizling + assert "feteasca-regala" in slugs # királyleányka + assert "feteasca-alba" in slugs # leányka + + +def test_grape_colour_actual_behaviour(fixture_text): + """Per-grape colour comes from the matcher. ACTUAL behaviour pinned: + a noir variety (kadarka / syrah) carries colour 'noir', but Furmint + resolves with an EMPTY colour string here (the lexicon entry carries no + default colour for the bare Furmint match in this context) — pinned so a + future lexicon edit is a conscious change, not an accident.""" + _sections, _titles, routed = _route_html( + fixture_text("hu_egyseges_dokumentum_eger.html") + ) + grapes = extract.parse_grapes(routed["grape_varieties"]) + by_slug = {d["slug"]: d for d in grapes["details"]} + assert by_slug["kadarka"]["colour"] == "noir" + assert by_slug["syrah"]["colour"] == "noir" + # Divergence: Furmint resolves with no colour string. + assert by_slug["furmint"]["colour"] == "" + + +def test_item_candidates_endash_and_hyphen_split(): + """`_item_candidates` splits a name/synonym item on a spaced en-dash OR a + spaced hyphen, returning the canonical head first then the synonyms.""" + assert extract._item_candidates("kadarka – jenei fekete") == [ + "kadarka", "jenei fekete"] + assert extract._item_candidates("cabernet franc - carbonet") == [ + "cabernet franc", "carbonet"] + # Multiple comma-separated synonyms after the dash all become candidates. + assert extract._item_candidates("furmint – zapfner, posipel, som") == [ + "furmint", "zapfner", "posipel", "som"] + + +# ========================================================================== +# Style detection +# ========================================================================== + +def test_style_detection_rose_from_description(fixture_text): + """parse_styles scans the description + additional-conditions bodies for + Hungarian colour adjectives + Tokaji-ladder markers. The Eger excerpt's + §4 mentions `Rozébor` → the `rose` style tag is detected.""" + html = fixture_text("hu_egyseges_dokumentum_eger.html") + doc = extract.slice_egyseges_dokumentum(html) + sections, titles = extract.extract_sections(doc) + styles = extract.parse_styles(sections, titles) + assert "rose" in styles + + +# ========================================================================== +# Regression: older-template link title carries the geo_area keyword +# ========================================================================== + +def test_geo_area_blocklist_table_present(): + # The blocklist must keep both diacritic + ASCII-folded forms of the + # "Kapcsolat a földrajzi…" link-title decoy. A missing entry re-opens the + # regression where the terroir section steals the geo_area role. + assert "kapcsolat a földrajzi" in _GEO_AREA_TITLE_BLOCKLIST + assert "kapcsolat a foldrajzi" in _GEO_AREA_TITLE_BLOCKLIST + + +def test_regression_kapcsolat_foldrajzi_title_not_routed_to_geo_area(fixture_text): + """Older template: section 8 is titled "Kapcsolat a földrajzi területtel", + which contains the "földrajzi terület" geo_area keyword and would + otherwise shadow the real area in section 6. The blocklist must keep + geo_area on section 6 and route section 8 to link_to_terroir.""" + _sections, _titles, routed = _route_html( + fixture_text("hu_egyseges_dokumentum_soltvadkerti.html") + ) + assert "geo_area" in routed and "link_to_terroir" in routed + assert routed["geo_area"] != routed["link_to_terroir"] + # The real area (section 6) is what landed in geo_area. + assert "Soltvadkert" in routed["geo_area"] + assert "Természeti" not in routed["geo_area"] + # Section 8's terroir prose landed in link_to_terroir. + assert "Természeti" in routed["link_to_terroir"] + + +def test_grape_keywords_routing_table_priority(): + """The grape-variety keyword tuple lists the most specific inflected + forms first; the bare "szőlőfajták" stem is last so a more specific + title wins. Pin the head of the tuple so a re-order is deliberate.""" + grape_keywords = SECTION_ROLE_KEYWORDS["grape_varieties"] + assert grape_keywords[0] == "fontosabb borszőlőfajták" + # geo_area's most specific keyword leads its tuple too. + geo_keywords = SECTION_ROLE_KEYWORDS["geo_area"] + assert geo_keywords[0] == "körülhatárolt földrajzi terület" diff --git a/tests/test_pt_parser.py b/tests/test_pt_parser.py new file mode 100644 index 0000000..123057a --- /dev/null +++ b/tests/test_pt_parser.py @@ -0,0 +1,376 @@ +"""Fixture-based regression tests for the Portugal (PT) parsers. + +Three parser modules, each a documented seam in the IVV-caderno pipeline +(see the PT section of CLAUDE.md and the module docstrings): + + - scripts/_lib/pt/caderno_sections.py — keyword-anchor section finder. + PT cadernos come in three structural variants and the parser must + carve the same semantic roles (`area` / `grapes` / `link` / `yields` + …) out of all of them WITHOUT knowing which variant it is reading: + Variant A — "Roman + Arabic" with a "V. DOCUMENTO ÚNICO" wrapper + (Douro, Porto, Alentejo). Repeating roles → last-write + wins (the numbered interior copy, not the Roman + preamble). + Variant B — Arabic-only / documento-único-first (Vinho Verde). + Variant C — Arabic-short / older format (Dão), with mixed-case + headers and the "Meio Geográfico" link variant. + Anchors require a numeric prefix ("5. ÁREA…") so in-prose mentions + don't become false section boundaries. + + - scripts/_lib/pt/subregiao.py — sub-região (DGC analogue) detection. + Pattern A: "Sub-região [de|do|da] NAME" line headers + a body + (Vinho Verde lists them in the grapes section, one casta table + each). + Pattern B: Douro-style "NAME: no distrito de X …" colon prefix, + gated behind a "três áreas geográficas" / "sub-regiões" preamble + so captions don't false-fire. + extract_subregioes tries A on (area + grapes), then B on area, + and returns whichever fires with ≥ 2 matches (else parent-only). + + - scripts/_lib/pt/commune_list.py — "Área Delimitada" concelho/distrito + parsing. Handles the enumerated "os municípios de X, Y e Z" list, + the "Todos os municípios dos distritos de …" distrito-all form, the + "abrange todo o distrito de X" whole-distrito form, the bare + "Distrito de X." sentence, and the "Arquipélago dos Açores" macro + token. + +Real cached docs live under raw/pt/ivv/cadernos/*.pdf (gitignored). The +fixtures here are short redacted excerpts under tests/fixtures/, with +expected output cross-checked against raw/pt/cadernos-extracted/*.json. + +Assertions are on STRUCTURE (carved role keys, sub-região names + pattern +tag, concelho / distrito / macro sets), not full-output snapshots. Where a +test pins ACTUAL parser behaviour that diverges from the docstring's ideal +(the distrito name leaking into the concelho list; the Pattern A header +form vs. the section-1 bullet list), the divergence is called out inline. +""" + +from __future__ import annotations + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.pt.caderno_sections import extract_sections # noqa: E402 +from _lib.pt.commune_list import parse_commune_list # noqa: E402 +from _lib.pt.subregiao import ( # noqa: E402 + detect_pattern_a, + detect_pattern_b, + extract_subregioes, + slugify, +) + + +def _names(records: list[dict]) -> list[str]: + return [r["name"] for r in records] + + +def _slugs(records: list[dict]) -> list[str]: + return [r["slug"] for r in records] + + +def _patterns(records: list[dict]) -> set[str]: + return {r["source_pattern"] for r in records} + + +# ========================================================================== +# caderno_sections — keyword-anchor section routing across variants +# ========================================================================== + +def test_caderno_variant_b_roles_routed(fixture_text): + """Variant B (Vinho Verde, Arabic-only). The three roles the downstream + consumers depend on land in the right bodies.""" + sec = extract_sections(fixture_text("pt_caderno_variantB_vinho-verde.txt")) + assert {"area", "grapes", "link"} <= set(sec) + # "5. ZONA GEOGRÁFICA DEMARCADA" body → area (commune list prose), + # NOT the link narrative. + assert sec["area"].startswith("A área geográfica de produção da DO") + assert "Todos os municípios dos distritos de Braga" in sec["area"] + # "6. PRINCIPAL(IS) CASTA(S) DE UVA" body → grapes. + assert "As castas utilizadas na produção" in sec["grapes"] + assert "Alvarinho" in sec["grapes"] + # "7. RELAÇÃO COM A ZONA GEOGRÁFICA" body → link (terroir prose). + assert sec["link"].startswith("Elementos relativos à área geográfica") + assert "clima atlântico" in sec["link"] + # Section boundaries are clean: the grape names did not bleed into area, + # and the area commune prose did not bleed into grapes. + assert "Alvarinho" not in sec["area"] + assert "Todos os municípios" not in sec["grapes"] + + +def test_caderno_variant_c_roles_routed(fixture_text): + """Variant C (Dão, Arabic-short older format). Exercises the + "Delimitação da Área Geográfica" area-header variant, the "Relação com + o Meio Geográfico" link variant, and a distinct `yields` role from + "Rendimentos Máximos por Hectare".""" + sec = extract_sections(fixture_text("pt_caderno_variantC_dao.txt")) + assert {"area", "grapes", "link", "yields"} <= set(sec) + # "4. Delimitação da Área Geográfica" → area. + assert sec["area"].startswith("A área da Região Demarcada do Dão") + assert "os municípios de Arganil" in sec["area"] + # "5. Rendimentos Máximos por Hectare" → yields, carved out as its own + # body (not glued onto the area commune list). + assert "Rendimento máximo por hectare" in sec["yields"] + assert "Arganil" not in sec["yields"] + # "6. Castas Utilizadas:" → grapes. + assert "Castas tintas" in sec["grapes"] + assert "Touriga-Nacional" in sec["grapes"] + # "7. Relação com o Meio Geográfico" → link (the "Meio Geográfico" + # variant, not "Área/Zona Geográfica"). + assert "Factores Naturais" in sec["link"] + assert "graníticos" in sec["link"] + + +def test_caderno_variant_a_last_write_wins(fixture_text): + """Variant A (Douro) wraps numbered interior sections inside + "V. DOCUMENTO ÚNICO". For a role that appears both in the Roman + preamble and the numbered interior, the LAST (interior, content-rich) + copy must win — the parser's last-write-wins rule.""" + sec = extract_sections(fixture_text("pt_caderno_variantA_douro.txt")) + assert {"area", "grapes", "link"} <= set(sec) + # "5. ÁREA DELIMITADA" interior body → area (the Pattern-B sub-region + # colon-prefix prose). + assert "três áreas geográficas mais restritas" in sec["area"] + assert "Baixo Corgo: no distrito de Vila Real" in sec["area"] + # "6. UVAS DE VINHO" → grapes; "7. RELAÇÃO COM A ÁREA GEOGRÁFICA" → link. + assert "Inventário das principais castas" in sec["grapes"] + assert sec["link"].startswith("Elementos relativos à área geográfica") + assert "bacia hidrográfica do Douro" in sec["link"] + + +def test_caderno_anchor_requires_numeric_prefix(): + """A bare in-prose mention ("a área delimitada da DOP é …") without a + "N." numeric prefix must NOT become a section boundary — otherwise the + real section bodies would be sheared at the false anchor.""" + text = ( + "1. NOME E TIPO\n" + "O nome a registar é Exemplo. A área delimitada da DOP é pequena " + "e a sua zona geográfica demarcada está bem definida.\n" + "5. ÁREA DELIMITADA\n" + "Os municípios de Arganil e Tábua.\n" + ) + sec = extract_sections(text) + # Only the numbered "5. ÁREA DELIMITADA" anchored the area; the in-prose + # "área delimitada" mention in section 1's body did not split it. + assert "Os municípios de Arganil e Tábua" in sec["area"] + assert "O nome a registar" not in sec["area"] + + +def test_caderno_no_anchors_returns_empty(): + # Defensive: a document with no recognisable section header yields {}. + assert extract_sections("Texto livre sem cabeçalhos numerados.") == {} + + +# ========================================================================== +# subregiao — Pattern A ("Sub-região NAME" headers, Vinho Verde) +# ========================================================================== + +def test_subregiao_pattern_a_vinho_verde_nine(fixture_text): + text = fixture_text("pt_subregiao_patternA_vinho-verde.txt") + out = detect_pattern_a(text) + # All 9 Vinho Verde sub-regiões, article ("de/do/da") stripped from the + # captured name. + assert _names(out) == [ + "Amarante", + "Ave", + "Baião", + "Basto", + "Cávado", + "Lima", + "Monção e Melgaço", + "Paiva", + "Sousa", + ] + assert _patterns(out) == {"A"} + # Multi-word names survive ("de Monção e Melgaço" → "Monção e Melgaço"). + assert "Monção e Melgaço" in _names(out) + + +def test_subregiao_pattern_a_slug_diacritic_fold(fixture_text): + text = fixture_text("pt_subregiao_patternA_vinho-verde.txt") + out = detect_pattern_a(text) + slugs = _slugs(out) + # Slug folds diacritics + spaces. + assert "cavado" in slugs + assert "moncao-e-melgaco" in slugs + assert slugify("Cávado") == "cavado" + assert slugify("Monção e Melgaço") == "moncao-e-melgaco" + + +def test_subregiao_pattern_a_captures_casta_body(fixture_text): + """Each "Sub-região NAME" header carries the casta-list lines beneath it + as its body, up to the next header. The body is what stage 02 stores as + the sub-region's geo_area_brief / inherited grape context.""" + text = fixture_text("pt_subregiao_patternA_vinho-verde.txt") + out = detect_pattern_a(text) + amarante = next(r for r in out if r["name"] == "Amarante") + assert "Amaral" in amarante["body"] + assert "Vinhão; Sousão" in amarante["body"] + # The body stops before the next header — Ave's castas don't leak in. + assert "Padeiro" not in amarante["body"] + + +def test_subregiao_pattern_a_ignores_section1_bullet_list(): + """The section-1 NOME block lists the sub-regiões as a guillemet bullet + list ("- «Amarante»;"), NOT as "Sub-região NAME" headers. Pattern A is + deliberately keyed on the header form, so the bullet list must NOT fire + — that is why stage 02 routes Pattern A over the grapes section, not + the name block.""" + bullets = "Sub-regiões:\n- «Amarante»;\n- «Ave»;\n- «Baião»;\n- «Sousa»." + assert detect_pattern_a(bullets) == [] + + +# ========================================================================== +# subregiao — Pattern B (Douro colon prefix, preamble-gated) +# ========================================================================== + +def test_subregiao_pattern_b_douro_three(fixture_text): + """End-to-end via the variant-A Douro caderno fixture: the area section + carries the "três áreas geográficas" preamble + the three colon-prefix + items, so extract_subregioes returns the 3 sub-regiões tagged B.""" + sec = extract_sections(fixture_text("pt_caderno_variantA_douro.txt")) + out = extract_subregioes(sec.get("area", ""), sec.get("grapes", "")) + assert _names(out) == ["Baixo Corgo", "Cima Corgo", "Douro Superior"] + assert _patterns(out) == {"B"} + assert "douro-superior" in _slugs(out) + + +def test_subregiao_pattern_b_requires_preamble(): + """Pattern B is conservative: the colon-prefix items alone are NOT + enough — a "três áreas geográficas" / "sub-regiões" preamble must + appear, otherwise a stray "Name: no distrito …" caption would + false-fire.""" + items = ( + "Baixo Corgo: no distrito de Vila Real abrange os concelhos de " + "Mesão Frio e Peso da Régua.\n" + "Cima Corgo: no distrito de Vila Real abrange as freguesias de " + "Alijó e Amieiro.\n" + "Douro Superior: no distrito de Bragança abrange a freguesia de " + "Vilarelhos.\n" + ) + # No preamble → no detection. + assert detect_pattern_b(items) == [] + # Same items WITH the preamble → 3 detected. + with_preamble = "agrupadas em três áreas geográficas mais restritas:\n" + items + assert _names(detect_pattern_b(with_preamble)) == [ + "Baixo Corgo", + "Cima Corgo", + "Douro Superior", + ] + + +def test_subregiao_extract_threshold_two(): + """extract_subregioes returns the empty list (parent-only DOP) when a + pattern yields fewer than 2 matches — a single "Sub-região NAME" header + is not enough to model a sub-denominated wine.""" + one_header = "Sub-região de Amarante\nAmaral\nAzal\nLoureiro\n" + assert extract_subregioes("", one_header) == [] + + +def test_subregiao_extract_empty_inputs(): + assert extract_subregioes("", "") == [] + + +# ========================================================================== +# commune_list — Área Delimitada concelho / distrito / macro parsing +# ========================================================================== + +def test_commune_list_enumerated_municipios(fixture_text): + """Variant-C Dão area: "os municípios de Arganil, Oliveira do Hospital e + Tábua" → the concelho names, list separators (comma + final " e ") + split, articles stripped.""" + sec = extract_sections(fixture_text("pt_caderno_variantC_dao.txt")) + cl = parse_commune_list(sec["area"]) + concelhos = cl["concelhos"] + assert "Arganil" in concelhos + assert "Oliveira do Hospital" in concelhos + assert "Tábua" in concelhos + assert "Seia" in concelhos + assert "Tondela" in concelhos + # Multi-word concelho survived the " e " split (it is one item, not two). + assert "Oliveira do Hospital" in concelhos + # No distrito names leaked into the concelho list for the Dão "Do + # distrito de X, os municípios de …" form. + for d in ("Coimbra", "Guarda", "Viseu"): + assert d not in concelhos + # Dão enumerates municípios, so distritos stays empty. + assert cl["distritos"] == [] + + +def test_commune_list_distrito_all_form(fixture_text): + """Variant-B Vinho Verde area: "Todos os municípios dos distritos de + Braga e de Viana do Castelo" → both distritos, expanded by the caller.""" + sec = extract_sections(fixture_text("pt_caderno_variantB_vinho-verde.txt")) + cl = parse_commune_list(sec["area"]) + assert "Braga" in cl["distritos"] + assert "Viana do Castelo" in cl["distritos"] + # The enumerated municípios in the b)/c)/d)/e) clauses are also captured. + for c in ("Arouca", "Amarante", "Baião", "Mondim de Basto", "Cinfães"): + assert c in cl["concelhos"] + + +def test_commune_list_distrito_name_leaks_into_concelhos(fixture_text): + """ACTUAL behaviour pin (divergence from the ideal): the + "distritos de Braga e de Viana do Castelo" phrase is also matched by the + município regex, and the " e " split surfaces "Viana do Castelo" as a + concelho candidate. It is harmless downstream (a same-named município + does exist and resolves), but the leak is real — pinned so a future + tightening of the município regex is a conscious choice.""" + sec = extract_sections(fixture_text("pt_caderno_variantB_vinho-verde.txt")) + cl = parse_commune_list(sec["area"]) + assert "Viana do Castelo" in cl["concelhos"] + + +def test_commune_list_pattern_samples(fixture_text): + """The focused pattern fixture exercises every area-section shape in one + pass: enumerated municípios, distrito-all, whole-distrito, bare + "Distrito de X.", and the Açores archipelago macro token.""" + cl = parse_commune_list(fixture_text("pt_area_concelho_patterns.txt")) + # Enumerated município list. + assert {"Arganil", "Oliveira do Hospital", "Tábua"} <= set(cl["concelhos"]) + # "Todos os municípios dos distritos de Braga e de Viana do Castelo". + assert "Braga" in cl["distritos"] + # "abrange todo o distrito de Faro" — whole-distrito. + assert "Faro" in cl["distritos"] + # Bare "Distrito de Setúbal." standalone sentence. + assert "Setúbal" in cl["distritos"] + # "Arquipélago dos Açores" → macro token (caller expands to the ilhas). + assert cl["macro_regions"] == ["acores"] + + +def test_commune_list_macro_archipelago_only(): + # An archipelago-only area section emits the macro token and no concelhos. + text = 'A IG "Açores" abrange todas as ilhas do Arquipélago dos Açores.' + cl = parse_commune_list(text) + assert cl["macro_regions"] == ["acores"] + assert cl["concelhos"] == [] + cl_m = parse_commune_list("A área abrange a Região Autónoma da Madeira.") + assert cl_m["macro_regions"] == ["madeira"] + + +def test_commune_list_empty_input(): + cl = parse_commune_list("") + assert cl == { + "concelhos": [], + "distritos": [], + "macro_regions": [], + "raw_hits": 0, + } + + +def test_commune_list_rejects_boundary_prose(): + """Alentejo-style boundary prose ("Estremoz até à ribeira da Fonte Boa") + is full of noise words; a município captured with such a tail must be + rejected by the _NOISE_WORDS filter, not admitted with a long noise + string.""" + text = ( + "os municípios de Estremoz até à ribeira da Fonte Boa onde prossegue " + "pela estrada até ao limite do concelho." + ) + cl = parse_commune_list(text) + # The single captured token carries noise words ("até", "ribeira", + # "estrada", "limite") → discarded; nothing survives. + for c in cl["concelhos"]: + assert "ribeira" not in c.lower() + assert "estrada" not in c.lower() diff --git a/tests/test_si_parser.py b/tests/test_si_parser.py new file mode 100644 index 0000000..a1f48b7 --- /dev/null +++ b/tests/test_si_parser.py @@ -0,0 +1,343 @@ +"""Fixture-based regression tests for the Slovenia (SI) parsers. + +Two parser surfaces, each a documented seam that historically regresses when +a tweak for one Slovenian source shape breaks another: + + - scripts/_lib/si/enotni_dokument.py (+ the HTML driver in + scripts/si/02_extract_pliegos.py) — the EUR-Lex "ENOTNI DOKUMENT" EU-OJ + template. Section-keyword role routing (Ime ali imena / Razmejeno + geografsko območje / Sorta ali sorte vinske trte / Povezava z geografskim + območjem), the "Vrsta geografske označbe" geo_area blocklist decoy, and + the grape-variety section's single em-dash-bulleted line + ("— bela žlahtnina — beli pinot - weissburgunder — …"): split on the + em-dash bullets, then on a plain hyphen for "Name - synonym" (the head + resolves, the synonym blob is a fallback). + + - scripts/_lib/si/specifikacija.py — the national-spec parser, three + branches: + * mkgp-doc-v1: numbered SPECIFIKACIJA-PROIZVODA sections 1–9; §6 Sorte + split by bele:/rdeče:/rose:; the "Tradicionalna imena" predikat-roster + boilerplate truncated out of the style scan. + * uradni-list-pravilnik-2007: Priloga 2 per-okoliš + priporočene sorte -> principal / dovoljene sorte -> accessory — the + ONLY SI source carrying a real principal/accessory split. + * uradni-list-pravilnik-2022-ptp: strict `\\b\\d+\\. člen\\b` + word-boundary article slicer that must NOT false-positive on the + genitive/locative `5. člena` / `9. členu` cross-references. + +Real cached docs live under raw/si/{oj-pages,specifikacije}/ (gitignored). +The Cviček ENOTNI DOKUMENT and the two pravilnik HTML branches are redacted +excerpts of those real documents; the .doc-sourced mkgp fixtures are +`# synthetic` (the binary .doc -> text path needs the antiword Docker image, +not run here) but their asserted slugs/styles are cross-checked against +raw/si/specifikacije-extracted/*.json so they match live parser output. + +Assertions are on STRUCTURE (routed roles, the em-dash variety split, the +2007 priporočene/dovoljene role split, the 2022 člen word-boundary guard), +not on full-output snapshots. Where a test pins ACTUAL behaviour (the +synonym-head resolution, the matched-subset of varieties), the divergence +from the regulator's full roster is called out inline. +""" +from __future__ import annotations + +import importlib +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.si import specifikacija as spec # noqa: E402 +from _lib.si.enotni_dokument import _GEO_AREA_TITLE_BLOCKLIST # noqa: E402 + +# 02_extract_pliegos starts with a digit, so import it by module path. +extract = importlib.import_module("si.02_extract_pliegos") + + +# ========================================================================== +# ENOTNI DOKUMENT HTML driver — section routing +# ========================================================================== + +def _route_html(html: str) -> tuple[dict, dict, dict]: + """Slice → extract numbered sections → route, the way build_record does.""" + doc = extract.slice_enotni_dokument(html) + assert doc is not None, "ENOTNI-DOKUMENT anchor must be found" + sections, titles = extract.extract_sections(doc) + routed = extract.route_sections(sections, titles) + return sections, titles, routed + + +def test_anchor_slice_drops_preamble(fixture_text): + html = fixture_text("si_enotni_dokument_cvicek.html") + doc = extract.slice_enotni_dokument(html) + # The "PUBLICATION OF A SINGLE DOCUMENT" preamble before the anchor drops. + assert "PUBLICATION OF A SINGLE DOCUMENT" not in doc + assert doc.lstrip().startswith(" all principal).""" + _sections, _titles, routed = _route_html( + fixture_text("si_enotni_dokument_cvicek.html") + ) + grapes = extract.parse_grapes(routed["grape_varieties"]) + slugs = set(grapes["principal"]) + # Spot the load-bearing members across white + red. + assert {"chasselas", "pinot-blanc", "chardonnay", "gamay", "kraljevina", + "welschriesling", "lemberger", "zametovka", "zweigelt", + "sankt-laurent"} <= slugs + # 17 distinct varieties recovered from the 17 em-dash bullets. + assert len(grapes["principal"]) == 17 + # No accessory split in the EU single document. + assert grapes["accessory"] == [] + + +def test_grape_name_synonym_hyphen_split_head_resolves(fixture_text): + """"beli pinot - weissburgunder" splits on the plain hyphen: the HEAD + ("beli pinot") resolves to pinot-blanc and is kept as the display name; + the synonym blob ("weissburgunder") is only a fallback. Likewise + "modra frankinja - frankinja" -> lemberger via the head.""" + _sections, _titles, routed = _route_html( + fixture_text("si_enotni_dokument_cvicek.html") + ) + grapes = extract.parse_grapes(routed["grape_varieties"]) + by_slug = {d["slug"]: d for d in grapes["details"]} + # Head resolved, synonym dropped from the display name. + assert by_slug["pinot-blanc"]["name"] == "beli pinot" + assert "weissburgunder" not in by_slug["pinot-blanc"]["name"].lower() + assert by_slug["lemberger"]["name"] == "modra frankinja" + assert "frankinja -" not in by_slug["lemberger"]["name"] + + +def test_grape_typo_chardonay_still_resolves(fixture_text): + """The cahier source carries a real spelling typo "chardonay" (one 'n'); + the lexicon matcher must still fold it to chardonnay (display name keeps + the verbatim typo).""" + _sections, _titles, routed = _route_html( + fixture_text("si_enotni_dokument_cvicek.html") + ) + grapes = extract.parse_grapes(routed["grape_varieties"]) + by_slug = {d["slug"]: d for d in grapes["details"]} + assert "chardonnay" in grapes["principal"] + assert by_slug["chardonnay"]["name"] == "chardonay" + + +def test_grape_items_split_unit(): + """_grape_items splits the single bulleted line on em-dash + newline.""" + items = extract._grape_items("— bela žlahtnina — beli pinot - weissburgunder — gamay") + assert items == ["bela žlahtnina", "beli pinot - weissburgunder", "gamay"] + + +def test_item_candidates_head_then_synonyms(): + """_item_candidates returns the canonical head first, then the comma-split + synonyms after the plain hyphen.""" + cands = extract._item_candidates("beli pinot - weissburgunder, pinot bianco") + assert cands[0] == "beli pinot" + assert "weissburgunder" in cands and "pinot bianco" in cands + + +# ========================================================================== +# specifikacija.py — mkgp-doc-v1 (numbered SPECIFIKACIJA PROIZVODA) +# ========================================================================== + +def test_mkgp_section_split_numbered(fixture_text): + text = fixture_text("si_mkgp_doc_bizeljcan.txt") + out = spec.parse_mkgp_doc(text, "bizeljcan") + assert out["parser_template"] == "mkgp-doc-v1" + # Sections 1..9 sliced by the "N. Title:" headers. + titles = out["section_titles"] + assert set(titles) >= {"1", "2", "4", "6", "7"} + assert "Sorte" in titles["6"] + assert "Povezava z geografskim območjem" in titles["7"] + + +def test_mkgp_grape_colour_split_bele_vs_rdece(fixture_text): + """§6 Sorte splits by colour header bele: -> blanc, rdeče: -> noir. Each + variety carries the colour of its bucket. Asserts the stable matched + subset (some Slovenian name forms don't resolve via the lexicon — that + recall gap is the parser's ACTUAL behaviour, pinned here).""" + text = fixture_text("si_mkgp_doc_bizeljcan.txt") + out = spec.parse_mkgp_doc(text, "bizeljcan") + by_slug = {d["slug"]: d for d in out["grapes"]["details"]} + # White-bucket members resolve as blanc. + for slug in ("welschriesling", "pinot-blanc", "chardonnay", "sauvignon"): + assert by_slug[slug]["colour"] == "blanc", slug + # Red-bucket members resolve as noir. + for slug in ("lemberger", "zametovka", "sankt-laurent"): + assert by_slug[slug]["colour"] == "noir", slug + # mkgp has no principal/accessory split -> all principal. + assert out["grapes"]["accessory"] == [] + + +def test_regression_mkgp_tradicionalna_imena_truncated_from_styles(fixture_text): + """The §2 description carries a "Tradicionalna imena: …" predikat roster + (pozna trgatev / jagodni izbor / penina / …) listing every designation + AUTHORISED for the okoliš, not styles actually produced. _parse_mkgp_styles + must slice that boilerplate off before scanning, else every wine ends up + tagged sparkling-quality + vendanges-tardives.""" + text = fixture_text("si_mkgp_doc_bizeljcan.txt") + out = spec.parse_mkgp_doc(text, "bizeljcan") + # Only the grape-colour-derived base styles survive. + assert set(out["styles"]) == {"blanc", "rouge"} + assert "sparkling-quality" not in out["styles"] + assert "vendanges-tardives" not in out["styles"] + # Sanity: WITHOUT the truncation the predikat ladder leaks in — proving the + # slice is load-bearing, not incidental. + desc = out["section_roles"]["description"] + roster = desc[desc.index("Tradicionalna"):].replace("Tradicionalna imena", "", 1) + leaked = spec._parse_mkgp_styles("opis " + roster, out["grapes"]) + assert "sparkling-quality" in leaked and "vendanges-tardives" in leaked + + +def test_mkgp_single_variety_no_colour_prefix_fallback(fixture_text): + """A §6 Sorte body with NO colour prefix (Teran's lone "refošk") falls + through to the whole-body-as-one-comma-list path. refošk folds to + refosco-dal-peduncolo-rosso (noir), style rouge.""" + text = fixture_text("si_mkgp_doc_teran.txt") + out = spec.parse_mkgp_doc(text, "teran") + assert out["grapes"]["principal"] == ["refosco-dal-peduncolo-rosso"] + by_slug = {d["slug"]: d for d in out["grapes"]["details"]} + assert by_slug["refosco-dal-peduncolo-rosso"]["name"] == "refošk" + assert by_slug["refosco-dal-peduncolo-rosso"]["colour"] == "noir" + assert out["styles"] == ["rouge"] + assert out["link_to_terroir"].startswith("Rdeča jerovica") + + +# ========================================================================== +# specifikacija.py — uradni-list-pravilnik-2007 (the real role split) +# ========================================================================== + +def test_pravilnik_2007_dispatches_by_title(fixture_text): + html = fixture_text("si_pravilnik_2007_bela-krajina.html") + out = spec.parse_uradni_list_pravilnik(html, "bela-krajina") + assert out is not None + assert out["parser_template"] == "uradni-list-pravilnik-2007" + assert out["matched_okoliši"] == ["Bela krajina"] + + +def test_regression_pravilnik_2007_priporocene_dovoljene_role_split(fixture_text): + """Priloga 2: "a) priporočene sorte: …;" -> principal, + "b) dovoljene sorte: …." -> accessory. This is the ONLY SI source with a + real principal/accessory split — the split must survive verbatim. Slugs + cross-checked against raw/si/specifikacije-extracted/bela-krajina.json.""" + html = fixture_text("si_pravilnik_2007_bela-krajina.html") + out = spec.parse_uradni_list_pravilnik(html, "bela-krajina") + principal = set(out["grapes"]["principal"]) + accessory = set(out["grapes"]["accessory"]) + # priporočene -> principal + assert {"welschriesling", "pinot-blanc", "sauvignon", "pinot-gris", + "chardonnay", "muscat-a-petits-grains", "lemberger", + "zametovka"} == principal + # dovoljene -> accessory + assert {"sylvaner", "riesling", "bouvier", "kraljevina", "gewurztraminer", + "kerner", "chasselas", "pinot-noir", "gamay", "zweigelt", + "portugais-bleu", "sankt-laurent", "chasselas-rose"} == accessory + # The two buckets are disjoint (no variety counted twice). + assert principal.isdisjoint(accessory) + + +def test_pravilnik_2007_wrong_title_returns_none(): + # A document that is not the 2007/2022 pravilnik dispatches to None. + out = spec.parse_uradni_list_pravilnik( + "

Some unrelated regulation.

", + "bela-krajina", + ) + assert out is None + + +# ========================================================================== +# specifikacija.py — uradni-list-pravilnik-2022-ptp (člen word-boundary guard) +# ========================================================================== + +def test_pravilnik_2022_dispatches_and_parses(fixture_text): + html = fixture_text("si_pravilnik_2022_belokranjec.html") + out = spec.parse_uradni_list_pravilnik(html, "belokranjec") + assert out is not None + assert out["parser_template"] == "uradni-list-pravilnik-2022-ptp" + # Article 5 ¶(2) enumerated Belokranjec list -> 10 principal varieties, + # all white (matches raw/si/specifikacije-extracted/belokranjec.json). + assert len(out["grapes"]["principal"]) == 10 + assert {"kraljevina", "welschriesling", "pinot-blanc", "chardonnay", + "sylvaner", "sauvignon", "riesling", "muscat-a-petits-grains", + "kerner", "pinot-gris"} == set(out["grapes"]["principal"]) + assert out["grapes"]["accessory"] == [] + assert out["styles"] == ["blanc"] + + +def test_regression_2022_clen_word_boundary_ignores_genitive(fixture_text): + """_PRAVILNIK_CLEN_RE uses `\\bčlen\\b`, so the genitive "5. člena" and + locative "9. členu" cross-references inside the article bodies must NOT be + parsed as new article headers. The fixture injects both forms; only the + six real `N. člen` headers (1..6) may register, or article slicing breaks + and the Belokranjec variety list (article 5) is lost.""" + html = fixture_text("si_pravilnik_2022_belokranjec.html") + text = spec._html_to_text(html) + # The inflected forms are present in the body... + assert "5. člena" in text and "9. členu" in text + # ...but only the six nominative `N. člen` headers are detected. + bodies = spec._articles_2022(text) + assert sorted(bodies) == [1, 2, 3, 4, 5, 6] + # Header-number matches from the strict regex never include 5/9 from the + # genitive forms beyond the real headers. + nums = [m.group(1) for m in spec._PRAVILNIK_CLEN_RE.finditer(text)] + assert nums == ["1", "2", "3", "4", "5", "6"] + + +def test_pravilnik_2022_only_belokranjec_slug(fixture_text): + """The 2022 PTP branch is slug-gated: it parses only for "belokranjec". + For any other slug (e.g. the sibling metliska-crnina) it returns None so + the caller falls through.""" + html = fixture_text("si_pravilnik_2022_belokranjec.html") + assert spec.parse_uradni_list_pravilnik(html, "metliska-crnina") is None diff --git a/tests/test_sk_parser.py b/tests/test_sk_parser.py new file mode 100644 index 0000000..79d6224 --- /dev/null +++ b/tests/test_sk_parser.py @@ -0,0 +1,359 @@ +"""Fixture-based regression tests for the Slovakia (SK) parsers. + +Two parser surfaces, each with its own documented seam: + + - scripts/_lib/sk/jednotny_dokument.py (+ the HTML driver in + scripts/sk/02_extract_pliegos.py) — the EU-OJ "JEDNOTNÝ DOKUMENT" + template. Section-keyword role routing has to handle BOTH title + variants the corpus actually ships: + * geo_area: "Vymedzená zemepisná oblasť" (newer) AND the older, + shorter "Vymedzená oblasť". + * link_to_terroir: "Opis súvislostí" (newer) AND the older + "Údaje potvrdzujúce spojitosť". + Regression (commit b869b1b): the Tokaj document gives each wine TYPE + its own numbered ti-grseq-1 section (Tokajský výber, ľadové víno, + slamové víno, Likérové víno, Sekt …) instead of one "Opis vín" + section, so style detection has to scan the section TITLES + per-type + "STRUČNÝ SLOVNÝ OPIS" bodies, not just an "opis vín" title. + + - scripts/_lib/sk/specifikacija.py — the ÚPV SR national spec, two + templates: + * upv-sr-specifikacia-v1: lettered a–i outline. The critical seam is + section f), a two-column Odroda (canonical, LEFT) / Synonymum + (foreign synonyms, RIGHT) table grouped under MUŠTOVÉ BIELE + (→blanc) / MUŠTOVÉ MODRÉ (→noir). The parser must take ONLY the + left Odroda column (the ≥2-space gutter), so the Pesecká leánka ↔ + Feteasca regala synonym confusion never reaches the matcher. + * upv-sr-prihlaska-v1: the older numbered 03.N template with a FLAT + inline §03.5 variety list (OCR-scanned, noisy). + +Real cached docs live under raw/sk/{oj-pages,national-specs}/ (gitignored). +The fixtures here are short, redacted excerpts under tests/fixtures/. + +Assertions are on STRUCTURE (routed roles for both title variants, the +left-Odroda-column-only grape set + colour split, the right-column synonym +NOT leaking, the Tokaj predikát style set), not on full-output snapshots. +Where a test pins ACTUAL parser behaviour that diverges from the module +docstring's ideal (the §f geo/link bodies still carrying their wrapped +title prefix because they have no early colon; the glued-bucket white block +never receiving a bucket colour fallback), the divergence is called out +inline. +""" +from __future__ import annotations + +import importlib +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.sk import specifikacija # noqa: E402 +from _lib.sk.jednotny_dokument import ( # noqa: E402 + SECTION_ROLE_KEYWORDS, +) + +# 02_extract_pliegos starts with a digit, so import it by module path. +extract = importlib.import_module("sk.02_extract_pliegos") + + +# ========================================================================== +# JEDNOTNÝ DOKUMENT HTML driver — section routing helper +# ========================================================================== + +def _route_html(html: str) -> tuple[dict, dict, dict]: + """Slice → extract numbered sections → route, the way build_record + drives them. Returns (sections, titles, routed).""" + doc = extract.slice_jednotny_dokument(html) + assert doc is not None, "JEDNOTNÝ DOKUMENT anchor must be found" + sections, titles = extract.extract_sections(doc) + routed = extract.route_sections(sections, titles) + return sections, titles, routed + + +# ========================================================================== +# Section routing — the NEWER template variant +# ========================================================================== + +def test_routing_new_template_titles(fixture_text): + """The newer template: geo titled "Vymedzená zemepisná oblasť", link + titled "Opis súvislostí". Both must route to the right semantic role.""" + sections, titles, routed = _route_html( + fixture_text("sk_jednotny_dokument_new_stredoslovenska.html") + ) + # Title variants present in the doc. + assert "Vymedzená zemepisná oblasť" in titles.values() + assert "Opis súvislostí" in titles.values() + # The four roles downstream consumers depend on. + for role in ("name", "geo_area", "grape_varieties", "link_to_terroir"): + assert role in routed, role + # Section 6 body (commune list) lands in geo_area, not terroir prose. + assert "katastrálnych území" in routed["geo_area"] + assert "Abovce" in routed["geo_area"] + # Section 8 body lands in link_to_terroir. + assert "južnej hranice" in routed["link_to_terroir"] + # The category section 2 ("Druh zemepisného označenia") body is CHZO, + # and it must NOT have shadowed the real area despite carrying + # "zemepisného" — the geo_area blocklist keeps it out. + assert routed["geo_area"].strip() != "CHZO – chránené zemepisné označenie" + + +# ========================================================================== +# Section routing — the OLDER template variant (the same roles, other titles) +# ========================================================================== + +def test_routing_old_template_titles(fixture_text): + """The older template (Skalický rubín): geo titled the shorter + "Vymedzená oblasť", link titled "Údaje potvrdzujúce spojitosť". Both + must route to the SAME semantic roles as the newer titles, so + downstream consumers stay template-agnostic.""" + sections, titles, routed = _route_html( + fixture_text("sk_jednotny_dokument_old_skalicky-rubin.html") + ) + # The older, shorter title variants are what this doc carries. + assert "Vymedzená oblasť" in titles.values() + assert "Vymedzená zemepisná oblasť" not in titles.values() + assert "Údaje potvrdzujúce spojitosť" in titles.values() + assert "Opis súvislostí" not in titles.values() + # …yet they route to the identical roles. + for role in ("name", "geo_area", "grape_varieties", "link_to_terroir"): + assert role in routed, role + assert "katastrálneho územia mesta Skalica" in routed["geo_area"] + assert "Bielych Karpát" in routed["link_to_terroir"] + # Section 7 (the grape list) lands in grape_varieties, not geo/terroir. + assert "Svätovavrinecké" in routed["grape_varieties"] + + +def test_both_title_variants_are_wired_in_keyword_table(): + """Pin that BOTH the newer and older title forms live in the keyword + table — a future trim of either re-opens the routing regression.""" + geo = SECTION_ROLE_KEYWORDS["geo_area"] + assert "vymedzená zemepisná oblasť" in geo # newer + assert "vymedzená oblasť" in geo # older + link = SECTION_ROLE_KEYWORDS["link_to_terroir"] + assert "opis súvislostí" in link # newer + assert "údaje potvrdzujúce spojitosť" in link # older + + +# ========================================================================== +# Grape parsing — flat list + the `Name - synonym` split +# ========================================================================== + +def test_grape_parsing_flat_red_list(fixture_text): + """Skalický rubín section 7 is a flat 3-variety red list (one per + line). All resolve to principal (no role split in the SK single + document); each carries the noir colour from the lexicon.""" + _sections, _titles, routed = _route_html( + fixture_text("sk_jednotny_dokument_old_skalicky-rubin.html") + ) + grapes = extract.parse_grapes(routed["grape_varieties"]) + assert set(grapes["principal"]) == { + "sankt-laurent", "blaufrankisch", "blauer-portugieser" + } + # No principal/accessory split — everything is principal. + assert grapes["accessory"] == [] + by_slug = {d["slug"]: d for d in grapes["details"]} + # The canonical Slovak names are kept as the display name. + assert by_slug["sankt-laurent"]["name"] == "Svätovavrinecké" + assert by_slug["blaufrankisch"]["name"] == "Frankovka modrá" + for d in grapes["details"]: + assert d["role"] == "principal" + + +def test_grape_parsing_name_synonym_em_dash_split(fixture_text): + """The Tokaj grape section uses the `Name - synonym` form + (`Kövérszőlő - Tučné hrozno`, `Zéta - Zeta`). The canonical name + before the ` - ` separator resolves; the synonym blob is a fallback.""" + _sections, _titles, routed = _route_html( + fixture_text("sk_jednotny_dokument_tokaj_predikat.html") + ) + grapes = extract.parse_grapes(routed["grape_varieties"]) + slugs = set(grapes["principal"]) + # Furmint / Kabar / Lipovina / Muškát žltý plus the hyphen-synonym rows. + assert {"furmint", "kabar", "harslevelu"} <= slugs + assert "koverszolo" in slugs # from "Kövérszőlő - Tučné hrozno" + assert "zeta" in slugs # from "Zéta - Zeta" + # The display name is the segment BEFORE " - " (synonym dropped). + by_slug = {d["slug"]: d for d in grapes["details"]} + assert by_slug["koverszolo"]["name"] == "Kövérszőlő" + assert "Tučné" not in by_slug["koverszolo"]["name"] + + +# ========================================================================== +# Style detection — colour keyword + the Tokaj predikát ladder regression +# ========================================================================== + +def test_styles_colour_keyword_from_description(fixture_text): + """A clean "Opis vína" section body carrying a colour phrase yields the + colour style. Skalický rubín's description ("Červené víno") → noir.""" + sections, titles, routed = _route_html( + fixture_text("sk_jednotny_dokument_old_skalicky-rubin.html") + ) + grapes = extract.parse_grapes(routed["grape_varieties"]) + styles = extract.parse_styles(sections, titles, grapes) + # Red wine → noir; the red grapes also carry noir, reinforcing it. + assert "noir" in styles + # No sparkling / sweet markers in this plain red doc. + assert "sparkling" not in styles + assert "grains-nobles" not in styles + + +def test_regression_tokaj_predikat_ladder_styles(fixture_text): + """Regression (commit b869b1b): the Tokaj doc has NO single "Opis vín" + section — each wine TYPE is its own numbered ti-grseq-1 section + (Tokajský výber, bobuľový výber, ľadové víno, slamové víno, Likérové + víno, Sekt). parse_styles must scan the section TITLES + per-type + bodies, so the full predikát ladder is detected. Before the fix this + document rendered with zero styles.""" + sections, titles, routed = _route_html( + fixture_text("sk_jednotny_dokument_tokaj_predikat.html") + ) + grapes = extract.parse_grapes(routed["grape_varieties"]) + styles = set(extract.parse_styles(sections, titles, grapes)) + # The botrytis / late-harvest / straw / ice / sparkling / liqueur tags + # are recovered from the per-wine-type section titles. + assert "grains-nobles" in styles # bobuľový výber (botrytis) + assert "vendanges-tardives" in styles # ľadové víno + assert "vin-de-paille" in styles # slamové víno + assert "vin-de-liqueur" in styles # Likérové víno + assert "sparkling" in styles # Sekt V. O. + # The white predikát grapes give the base colour style. + assert "blanc" in styles + # Sanity: the fix turned a zero-style document into a rich one. + assert len(styles) >= 5 + + +# ========================================================================== +# specifikacija.py — §f two-column Odroda/Synonymum table (left column only) +# ========================================================================== + +def _nitrianska(fixture_text) -> dict: + text = fixture_text("sk_specifikacija_nitrianska_two_column.txt") + return specifikacija.parse_specifikacija(text, "nitrianska") + + +def test_specifikacija_template_and_routing(fixture_text): + out = _nitrianska(fixture_text) + assert out["parser_template"] == "upv-sr-specifikacia-v1" + # The lettered b/d/f/g sections routed to their roles. + roles = out["section_roles"] + assert roles["grape_varieties"] # f) + assert "katastrálne územia" in roles["geo_area"] # d) + # g) terroir narrative landed in link_to_terroir. + assert "Vinohrady" in out["link_to_terroir"] + + +def test_specifikacija_left_odroda_column_only(fixture_text): + """The §f grape set is built from the LEFT (Odroda) column only. Every + canonical Slovak name resolves; the right-column synonyms do not become + separate varieties.""" + out = _nitrianska(fixture_text) + by_slug = {d["slug"]: d for d in out["grapes"]["details"]} + slugs = set(by_slug) + # Canonical left-column names, white block. + assert {"aurelius", "bouvier", "devin", "chardonnay", "irsai-oliver", + "muscat-ottonel", "muller-thurgau", "welschriesling", "pinot-gris", + "sauvignon", "gewurztraminer"} <= slugs + # Canonical left-column names, red block. + assert {"alibernet", "blaufrankisch", "blauer-portugieser", "nitranka", + "rudava", "pinot-noir", "sankt-laurent", "zweigelt"} <= slugs + # The display name is always the left (Odroda) cell, never a synonym. + assert by_slug["feteasca-regala"]["name"] == "Feteasca regala" + assert by_slug["welschriesling"]["name"] == "Rizling vlašský" + assert by_slug["sankt-laurent"]["name"] == "Svätovavrinecké" + + +def test_regression_synonym_column_not_leaked(fixture_text): + """The CRITICAL seam: the "Feteasca regala" row carries "Pesecká + leánka" in the right (Synonymum) column, and "Dievčie hrozno" carries + "Leányka" / "Feteasca alba". Read as bare names those synonyms resolve + to feteasca-ALBA — the WRONG variety for the Feteasca-regala row. The + parser takes only the left column, so: + * feteasca-regala comes from its own left cell "Feteasca regala", + * feteasca-alba comes from its own left cell "Dievčie hrozno", + and NEITHER comes from a leaked synonym.""" + out = _nitrianska(fixture_text) + by_slug = {d["slug"]: d for d in out["grapes"]["details"]} + # Both varieties present, each pinned to its own left-column name. + assert by_slug["feteasca-regala"]["name"] == "Feteasca regala" + assert by_slug["feteasca-alba"]["name"] == "Dievčie hrozno" + # No detail's display name is a right-column synonym string. + names = {d["name"] for d in out["grapes"]["details"]} + for synonym in ("Pesecká leánka", "Pesecké dievčie hrozno", "Leányka", + "Welschriesling", "Pinot gris", "Gewürtztraminer", + "Olasz rizling"): + assert synonym not in names, synonym + # A wrapped synonym-continuation line ("vlašský" alone, the wrap of the + # Rizling vlašský synonym blob) must not have become a variety either. + assert all(d["name"] != "vlašský" for d in out["grapes"]["details"]) + + +def test_specifikacija_colour_bucket_split(fixture_text): + """MUŠTOVÉ BIELE → white block, MUŠTOVÉ MODRÉ → red block. When the + lexicon assigns no colour, a variety in the standalone-labelled red + block inherits the noir bucket hint. + + ACTUAL-behaviour pin: the WHITE bucket label is GLUED to the first + variety ("MUŠTOVÉ Aurelius", " BIELE Bouvierovo hrozno") rather than + sitting on its own line, so it is never a dedicated bucket-label line + and the white block gets NO bucket colour fallback — lexicon-empty + whites keep colour "". Only the standalone "MUŠTOVÉ" / "MODRÉ" label + lines flip the hint, so the bucket fallback fires for the red block.""" + out = _nitrianska(fixture_text) + by_slug = {d["slug"]: d for d in out["grapes"]["details"]} + # Red block, lexicon gives these no colour → noir bucket fallback. + for slug in ("alibernet", "nitranka", "rudava"): + assert by_slug[slug]["colour"] == "noir", slug + # Lexicon-coloured reds. + assert by_slug["blauer-portugieser"]["colour"] == "noir" + assert by_slug["sankt-laurent"]["colour"] == "noir" + # White block: lexicon colours where known… + assert by_slug["welschriesling"]["colour"] == "blanc" + assert by_slug["chardonnay"]["colour"] == "blanc" + # …and the documented quirk — a lexicon-empty white (Sauvignon) does + # NOT inherit a bucket colour because the white label was glued, so its + # colour stays "". + assert by_slug["sauvignon"]["colour"] == "" + + +def test_specifikacija_styles_from_description(fixture_text): + """Section b) "Opis vína" drives style detection: "Biele, ružové a + červené" → base colours, "sekty" → sparkling, "likérové vína" → + vin-de-liqueur.""" + out = _nitrianska(fixture_text) + styles = set(out["styles"]) + assert "vin-de-liqueur" in styles + # blanc + rouge come from the grape-colour fallback (white + red blocks). + assert "blanc" in styles + assert "rouge" in styles + + +# ========================================================================== +# specifikacija.py — the older numbered 03.N prihláška template +# ========================================================================== + +def test_prihlaska_template_detected_and_flat_list(fixture_text): + """The 1996 Karpatská perla prihláška has no lettered a–i sections and + a flat §03.5 inline variety list, so parse_specifikacija falls through + to the upv-sr-prihlaska-v1 branch.""" + text = fixture_text("sk_specifikacija_karpatska_prihlaska.txt") + out = specifikacija.parse_specifikacija(text, "karpatska-perla") + assert out["parser_template"] == "upv-sr-prihlaska-v1" + # The full flat roster (31 varieties in the real doc) resolves. + assert len(out["grapes"]["principal"]) == 31 + assert out["grapes"]["accessory"] == [] + + +def test_prihlaska_ocr_noisy_names_recovered(fixture_text): + """The OCR scan mangles several names — "C hardonnay", "Mu ~kát + Ottonel", "MUller Thurgau" — but the fuzzy matcher + targeted repairs + recover the canonical slug.""" + text = fixture_text("sk_specifikacija_karpatska_prihlaska.txt") + out = specifikacija.parse_specifikacija(text, "karpatska-perla") + slugs = set(out["grapes"]["principal"]) + assert "chardonnay" in slugs # "C hardonnay" + assert "muscat-ottonel" in slugs # "Mu ~kát Ottonel" + assert "muller-thurgau" in slugs # "MUller Thurgau" + # The flat list carries both Feteasca regala and Dievčie hrozno, so + # both feteasca slugs are present (no two-column gutter here). + assert "feteasca-regala" in slugs + assert "feteasca-alba" in slugs From f058bf01a6ac7b58cf40ac4c033b67bb9ff61f54 Mon Sep 17 00:00:00 2001 From: Boris De Vloed Date: Sat, 20 Jun 2026 22:26:29 +0200 Subject: [PATCH 40/41] misc --- CLAUDE.md | 93 ++- raw/wikipedia/style_overrides.json | 37 +- scripts/02_extract_cahiers.py | 5 +- scripts/04_build_maps.py | 503 +++++++++++++-- scripts/_lib/aging_taxonomy.py | 187 ++++++ scripts/_lib/appellation_urls.json | 24 + scripts/_lib/assets/app.js | 923 +++++++++++++++++++--------- scripts/_lib/ch/reglement.py | 103 +++- scripts/_lib/content_block.py | 53 +- scripts/_lib/grape_lexicon.py | 59 +- scripts/_lib/map_template.py | 598 ++++++++++++++---- scripts/_lib/style_taxonomy.py | 184 +++++- scripts/ch/02_extract_reglements.py | 2 +- scripts/deploy.py | 186 +++++- scripts/es/02_extract_pliegos.py | 8 +- tests/test_content_block.py | 50 ++ tests/test_entity_jsonld.py | 23 + 17 files changed, 2527 insertions(+), 511 deletions(-) create mode 100644 scripts/_lib/aging_taxonomy.py diff --git a/CLAUDE.md b/CLAUDE.md index c785804..140aed8 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -406,7 +406,7 @@ twice with no changes upstream must be a no-op (cache hits). | 02g_fetch_vivc.py | raw/inao/cahier-extracted/*.json + raw/es/pliegos-extracted/ + raw/pt/cadernos-extracted/ + raw/vivc/slug_overrides.json | raw/vivc/{search,passport,by-slug}/*.html\|json + manifest.json + slug_overrides.example.json | | 02i_fetch_wikidata_qids.py | raw/*/*-extracted/*.json (slug + id_eambrosia) + raw/wikipedia/aocs// + raw/wikidata/slug_overrides.json | raw/wikidata/qids-by-slug.json + p9854.json + manifest.json + slug_overrides.example.json | | 03_generate_wiki.py | raw/inao/cahier-extracted/*.json + raw/terroir-facts/ | wiki/*.md, wiki/_index.json | -| 04_build_maps.py | raw/inao/cahier-extracted/*.json + raw/wikipedia/grapes/ + raw/translations/grapes/ + raw/vivc/by-slug/ + raw/wikidata/qids-by-slug.json + raw/wikipedia/styles/ + raw/translations/styles/ + raw/wikipedia/aocs/ + raw/translations/summaries/ + raw/translations/terroir-facts/ + raw/terroir-facts/ + raw/ign/communes.geojson + raw/inao/parcellaire/ + raw/cadastre/lieux-dits/ | wiki/index.html (EN canonical = homepage), wiki/{fr,es,nl}/index.html, wiki/map-data/*.pmtiles, wiki/robots.txt, wiki/sitemap.xml (homepage × 4 locales) | +| 04_build_maps.py | raw/inao/cahier-extracted/*.json + raw/wikipedia/grapes/ + raw/translations/grapes/ + raw/vivc/by-slug/ + raw/wikidata/qids-by-slug.json + raw/wikipedia/styles/ + raw/translations/styles/ + raw/wikipedia/aocs/ + raw/translations/summaries/ + raw/translations/terroir-facts/ + raw/terroir-facts/ + raw/ign/communes.geojson + raw/inao/parcellaire/ + raw/cadastre/lieux-dits/ | wiki/index.html (EN canonical = homepage), wiki/{fr,es,nl}/index.html, wiki/{en,fr,es,nl}//index.html (per-appellation entity pages), wiki/{en,fr,es,nl}/appellations/index.html (browse-index hub linking every indexable slug, grouped by country — fixes the entity-page link-graph orphan problem), wiki/map-data/*.pmtiles, wiki/robots.txt, wiki/sitemap.xml (4 home + 4 browse + 1,638 index slugs × 4 locales), wiki/llms.txt (AI-crawler index), wiki/404.html | ## Spain pipeline (`scripts/es/`) @@ -2738,20 +2738,44 @@ CH-specific notes: - The per-canton règlement parser at [scripts/_lib/ch/reglement.py](scripts/_lib/ch/reglement.py) uses a **whole-document grape-lexicon scan** rather than section-scoped - extraction. The shared `_lib.grape_entity.match_variety` is robust - enough (lexicon-based + per-token rejection) to scan ~50 KB of - règlement text without false positives, and cantonal règlements - frequently bury the variety list in an annex or refer to an - external annex by article — section-scoped extraction missed most - of the recall. Commune extraction stays section-scoped because + extraction, because cantonal règlements frequently bury the variety + list in an annex or refer to an external annex by article — + section-scoped extraction missed most of the recall. Scanning ~50 KB + of regulatory prose is noise-prone, though, so `extract_varieties` + guards the shared `match_variety` three ways (2026-06): a + **commune-name guard** (a chunk that is a Swiss commune — Genève, + Sion, Cortaillod — is area prose, not a grape), a **fuzzy floor** + (`_CH_FUZZY_FLOOR`=95: weak fuzzy hits matched common words / place + fragments to obscure non-Swiss grapes — `vigne`→viognier, + `Nein`→durif), and a small **`_CH_STOP_SURFACES`** set for prose + words that are exact grape aliases out of context (`canton`→chenin, + `Säure`→calitor, `weisse`→valente). Candidate cleaning strips + leading FR/IT/DE determiners (`il Merlot`, `la Bondola`), `a)`/`°` + ordinal markers, and footnote tails, and adds a name-before-number + variant to recover must-weight (Mostgewicht) lines + (`a) Blauburgunder 19,4 °Brix`). The Swiss Agroscope crossings + + natives the scan surfaces (Garanoir, Mara, Galotta, Doral, Divico, + Carminoir, Diolinoir, Bondola, Completer, Gouais→heunisch) live in + the shared `grape_lexicon.py`; Cornalin / Humagne are deliberately + NOT folded yet (Valais Cornalin = Rouge du Pays vs Humagne Rouge = + Cornalin d'Aoste is an identity tangle that needs a VIVC pass). + Commune extraction stays section-scoped because whole-document commune scans generate huge false-positive lists. - Variety extraction recall: 20 of 26 cantons return non-zero - varieties; AG (29), GE (47), VD (16), VS (25), FR (66), NE (18), - TI (14), JU (4), LU (7), BL (8), GR (5), SG (5), OW (4), TG (4), - ZH (3), SH (2), SO (1), BE (1), GL (1), ZG (1). Cantons with 0 - varieties (AI, AR, BS, NW, SZ, UR) defer to federal OVin without - cataloguing varieties locally (BS defers to BL via inter-cantonal - Vereinbarung; the others are tiny corpora ~5 ha each). + Variety extraction recall (post-2026-06 FP cleanup — counts dropped + because the prior figures double-counted prose/place false positives + that the new guards remove, and rose where Mostgewicht recall recovers + real lists): 11 cantons return non-zero cantonale-tier varieties; + AG (65), GE (51), VS (44), VD (38), LU (25), TI (22), NE (15), BL (7), + SG (4), OW (4), GR (2). The other cantons (AI, AR, BE, BS, GL, JU, NW, + SH, SO, SZ, TG, UR, ZG, ZH) now return 0 at the cantonale tier: + some defer to federal OVin without cataloguing varieties locally + (AI/AR/BS/NW/SZ/UR; BS defers to BL via inter-cantonal Vereinbarung; + tiny corpora ~5 ha each), and the rest (BE, GL, JU, SH, SO, TG, ZG, + ZH) DO grow wine but either publish a règlement too sparse to parse + (ZH/SO are ~3 KB with no variety list) or enumerate varieties in a + Mostgewicht / annex form the scan doesn't yet recover — a known + recall gap, not a regression (their prior non-zero counts were + entirely prose/place false positives now removed). - For the 5 multi-AOC cantons (VD has 10 AOCs, GE has 23, TI has 4, BE has 3, FR has 2), the canton-wide règlement body is the v1 default for variety lists. Per-AOC commune-list carving (Phase 2.5) @@ -4242,6 +4266,47 @@ author / editorial dates for a generated page). `{jsonld_html}` `str.format` slot — its JSON braces are data, not format fields, so it must not be double-braced or `esc()`-ed. +## Internal linking & crawlability (browse index, cross-links, llms.txt, 404) + +The map homepage is a JS app shell with no crawlable `` links to entity +pages (the sidebar list is client-rendered), so the per-appellation pages were +link-graph orphans discoverable only via the sitemap. Stage 04 now emits a +static link layer (all in [scripts/_lib/map_template.py](scripts/_lib/map_template.py) ++ [scripts/04_build_maps.py](scripts/04_build_maps.py)): + +- **Browse-index pages** — `wiki//appellations/index.html` per locale, + a self-contained page (`_render_browse_page` + `_BROWSE_TEMPLATE`, NOT the + map shell) listing every **index**-classified appellation as a real ``, + grouped by country and sorted by localized name. The static crawl hub. + A namespace `assert "appellations" not in aocs` guards the path collision. +- **Inbound links** to the browse hub: the sidebar footer (`{browse_path}` + slot, on the homepage AND every entity page) + a paragraph in the About + dialog (`about_browse_html`). +- **Entity cross-link nav** (`_entity_nav_html` → `