diff --git a/.gitignore b/.gitignore index dfdaf7d..e3d2744 100644 --- a/.gitignore +++ b/.gitignore @@ -64,8 +64,8 @@ done*.json .claude/settings.local.json .claude/scheduled_tasks.lock -# gettext catalogs — UI-localisation layer; not committed in the minimal -# bootstrap. Stage 04 falls back to the FR source when these are absent. -locale/ +# gettext catalogs: the .po sources + .pot are committed (a fresh checkout must +# render EN/ES/NL, not the FR msgids); only the compiled .mo is derived. +locale/**/*.mo tmp/* diff --git a/CLAUDE.md b/CLAUDE.md index 65f8729..d55d025 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -149,7 +149,210 @@ details and "Hard rules" for invariants that apply to every country. `--workers N` for concurrent processing (Ollama needs `OLLAMA_NUM_PARALLEL >= N`; Anthropic respects account limits). `scripts/audit_terroir_facts.py` recomputes coverage against the - current sources and flags drift / erosion. + current sources and flags drift / erosion. The grounding test itself is + the shared [scripts/_lib/terroir_coverage.py](scripts/_lib/terroir_coverage.py) + — ellipsis-aware: a quote that joins two source spans with "[…]" is + graded span by span, so it no longer fails the single-contiguous-match + test and flips to `wiki` — and block-aware: the verbatim blocks (≥ 12 + chars) of a quote are summed, so one pdftotext artefact inside a + quote ("gradi- giorno") no longer halves its coverage (2026-09-13: + Montepulciano d'Abruzzo had lost 8 of 9 true facts to it); the 0.6 + threshold is unchanged — and typography-folding (2026-09-14): both + sides are NFKC-normalised, every quote / apostrophe / dash variant + folded (a cahier's `’` against the model's `'`, „low-9" and + « guillemets »), soft hyphens and zero-width characters dropped and + hyphenated line breaks closed before matching — on the r1 corpus 51 % + of the FR quotes and 8.5 % of the others scored higher, none crossed + the threshold downwards. Every 02d script plus the audit import + it. The fourth sub-section is earned deterministically + ([scripts/_lib/terroir_interactions.py](scripts/_lib/terroir_interactions.py)): + after its four calls a 02d script drops an `interactions` fact whose + grounding quote carries no causal connective of the source language + (15-language table; cap 2 per record), and the gate demotes such a + fact to the natural factors — the connective is looked for in the + *quote*, never only in the bullet, because a bullet adding the "thanks + to" the source lacks is the failure mode. No fact is ever promoted into + the sub-section. 02d also normalises bullets at write time + (`normalize_facts`), so the normalise post-pass is a no-op on a fresh + cache. Italy: stage 02f emits the whole MASAF Art. 9 as + `link_to_terroir_full` next to the 4,000-character panel cut + (`link_to_terroir`, unchanged); IT 02d, the gate and the audits read + the full text (469 of 522 disciplinari are longer than the cut). Two **cache post-passes** apply fixes without an LLM call: + [scripts/recompute_terroir_provenance.py](scripts/recompute_terroir_provenance.py) + re-grades the existing caches with that rule (loading each country's + 02d module for the exact source text it graded against; stale caches + are skipped, never dropped) and syncs the per-fact `provenance` into + the 02e caches; [scripts/dedupe_terroir_facts.py](scripts/dedupe_terroir_facts.py) + collapses facts restated across the four sub-section calls + ([scripts/_lib/terroir_dedupe.py](scripts/_lib/terroir_dedupe.py): + near-identical bullets, or same/contained source quote + substantial + bullet overlap; never two bullets with different numbers, never two + bullets leading with different sub-denomination names — stage 04's + sibling filter depends on those) and prunes the same indices from the + index-aligned 02e caches, updating their `source_facts_sha`, so no + translation is lost or redone. Both take `--dry-run` / `--only` and + write a JSON report under `tmp/terroir-facts-review/`. Two more + post-passes of the same shape: [scripts/normalize_terroir_facts.py](scripts/normalize_terroir_facts.py) + applies the deterministic clean-up of + [scripts/_lib/terroir_normalize.py](scripts/_lib/terroir_normalize.py) + (regulatory colour codes after a grape name — "Pinot noir N" — VT / + SGN expansion, terminal period, and in the four target locales the + Latinisation of residual Greek / Cyrillic script: homoglyphs mapped, + gloss tokens transliterated, chemical prefixes such as "α-terpineol" + and predominantly non-Latin bullets left alone) to source and + translation caches in step, and stage 04 applies the same normaliser at + render time; + [scripts/filter_terroir_boilerplate.py](scripts/filter_terroir_boilerplate.py) + drops facts whose quote is a tautology pattern + ([scripts/_lib/terroir_boilerplate.py](scripts/_lib/terroir_boilerplate.py)) + quoted by ≥ 3 records of one country ("the wines' uniqueness is due to + soil, climate and summer winds" — most Greek PGI specs), never a + record's only fact and never a bullet carrying a number. Stage 02d + itself now dedupes after its four sub-section calls, appends one shared + English style block to every country's extraction prompt + ([scripts/_lib/terroir_prompts.py](scripts/_lib/terroir_prompts.py): + full sentences, no arrows, no colour codes, expand VT / SGN, never + mention the document, keep hedges, skip tautologies) and, for a cahier + shared by several appellations (the 51 Alsace grands crus), grades + each record against its own chapter only + ([scripts/_lib/terroir_chapters.py](scripts/_lib/terroir_chapters.py)) + — a cru without an own chapter is skipped, never grounded on another + cru's text. On the translation side every 02e script builds its prompt + through `translation_system_prompt` in the same module: appellation, + commune, vineyard, grape and institution names stay verbatim; common + nouns (generic soils, climates, harvest categories, scheme + abbreviations) are translated; geography takes the target language's + established exonym (Vosges → Vogezen, Rhin → Rijn, Piemonte → + Piedmont; verbatim only as an appellation name); Greek and Bulgarian + sources must come out in Latin script. `scripts/_lib/exonyms.py` holds + the exonym table and + [scripts/detect_untranslated_terroir_facts.py](scripts/detect_untranslated_terroir_facts.py) + lists the (slug, locale) pairs still carrying non-Latin script, a + leaked common noun or a source-form geographic name, with the exact + `02e --refresh --only … --lang …` command per country; every 02e + script takes `--only`. +- **Per-record review feedback is a constraint layer for the next + extraction, never content.** A quality review's verified findings are + kept per record in `raw/terroir-facts-feedback/.json` + ([scripts/_lib/terroir_feedback.py](scripts/_lib/terroir_feedback.py)): + `do_not_claim` (verified misleading claims with the failure mode, the + stage — `extraction` / `translation` / `both` — and the verifier's + source-quoting reason), `capture_if_present` (what the source describes + prominently and no bullet captured), `record_cautions` (sibling text + inside the source, a wrong binding, a source typo, a wrong Wikipedia + article), `reviews` and a `history` list for later gate runs. Every + 02d script calls `with_feedback(system, slug)` right after formatting + its system prompt (live call and `--emit-todo` alike; the wiring lint + in `tests/test_terroir_feedback.py` fails when a script drops it): + the block phrases each negative as "do not assert … unless the source + states it explicitly", never carries the reviewer's corrected text, + and includes only the `extraction` / `both` entries — `translation` + entries are for a 02e back-check. `audit_terroir_facts.py` reports + `feedback_recurrence`: a do-not-claim entry whose source-language + bullet still matches a current bullet (the known error is still there + before a re-run, or came back after one). Sidecars are built and + merged from a review's evidence directory by + [scripts/build_terroir_feedback.py](scripts/build_terroir_feedback.py) + (review 2026-09-12: 384 do-not-claim entries in 346 records, 1,133 + hints, 96 cautions over 1,238 records — see + [docs/review-terroir-facts-2026-09-12.md](docs/review-terroir-facts-2026-09-12.md)). + Like the other `raw/` caches the sidecars are gitignored; a checkout + without them extracts exactly as before. +- **Every terroir-fact write is snapshotted; the chain is one rollback + unit.** Stage 02d, the gate, stage 02e, the back-check and the four + post-passes write the caches only through + `terroir_cache.write_source_cache` / `write_translation_cache` + ([scripts/_lib/terroir_cache.py](scripts/_lib/terroir_cache.py); the + wiring lint in `tests/test_terroir_backup.py` fails on a direct + `cache.write_json`). The first write to a slug in a run copies its + source cache AND its four translation caches to + `raw/terroir-facts-backup//` ([scripts/_lib/terroir_backup.py](scripts/_lib/terroir_backup.py): + one entry file per slug, so parallel per-country processes share a run + id without racing); the run id is `OWM_TERROIR_RUN`, else a + per-process timestamp. `scripts/rollback_terroir_facts.py --list` / + `--run [--only slug] [--dry-run]` restores every file the run + overwrote and deletes every file it created, source and translations + together (a rollback is itself snapshotted, so it can be undone); then + rebuild with stage 04. [scripts/rerun_terroir_facts.py](scripts/rerun_terroir_facts.py) + runs a scoped re-run under one id — it marks the scoped caches stale + (`cahier_source_sha` prefixed `stale:`, written through the backup) and + then chains `02d --batch` per country (parallel) → `02d_verify --batch` + → `02e --batch` per country → `02e_verify --batch` → the audit, logs + under `/tmp/owm-/`. +- **The claim-support gate is part of the pipeline (review 2026-09-12, + R2 / R3 / R7 / R8).** Stage 02d's coverage test only proves the *quote* + exists; [scripts/02d_verify_terroir_facts.py](scripts/02d_verify_terroir_facts.py) + ([scripts/_lib/terroir_gate.py](scripts/_lib/terroir_gate.py)) grades, + per record in ONE request, every bullet's *claim* against the exact + source text 02d graded against (`terroir_sources`), the Wikipedia + hints and the record's feedback sidecar (verified-misleading claims + become explicit checks, record cautions disqualify pasted or mis-bound + text). Verdicts are applied deterministically: `supported` stays; + `rewrite` replaces the bullet with the model's narrower rewrite + (guarded — no number absent from bullet + source, no arrow, sane + length; a refused rewrite keeps the original and is listed by the audit + as `rewrite_rejected`; a rewrite that came back empty keeps the + original as `supported` with `rewrite_missing`, and a *cosmetic* one — + ratio ≥ 95 and no differing word of 4+ letters, so an added hedge is + never cosmetic — keeps the original as `supported` with + `cosmetic_rewrite`, `gate-v2`); `drop` removes it (unsupported, + foreign, tautology, or `restates` another bullet — the semantic + dedupe); `subsection` moves a clearly misfiled bullet. An + `interactions` bullet is supported only when the source sentence + states the causal link, and after the verdicts the deterministic + connective test demotes any that still lacks one. + Each kept fact carries `support` ({verdict, note[, original_bullet][, + moved_from]}); the cache carries a `gate` block (shas it keyed on, + counts, the dropped bullets) and the feedback sidecar a `history` + entry. Translations: index-aligned prune for pure drops; a rewrite + re-keys them `pending:` so 02e re-translates. Incremental — a + record is due (`terroir_gate.needs_gate`, shared with the audit's + `gate_pending`) when any fact carries no gate verdict, or the gate + block predates the record's source sha or `GATE_VERSION`; deliberately + not an exact sha of the bullets, so the normalise / dedupe / + boilerplate post-passes no longer re-fire the gate corpus-wide. The 21 + extraction prompts share the same rules through `STYLE_RULES` in + [scripts/_lib/terroir_prompts.py](scripts/_lib/terroir_prompts.py), + which opens with the claim-support rule — the gate's own over-claim + catalogue (a causal wrapper on a co-occurrence, a narrowed en-bloc + attribution, an invented qualifier, a sibling's statement, a + strengthened hedge) — so the extractor does the gate's job first; then + one full sentence of ~120–220 characters (the 140-character cap that + produced the "Label:" fragments is gone), named entities and figures + first (up to two extra bullets on a long text), an earned + `interactions` sub-section. The audit's `feedback_recurrence` counts a + do-not-claim entry as resolved when the matched fact's + `support.original_bullet` is the claim and the gate's rewrite was not + cosmetic. +- **The translation back-check follows 02e (R6).** + [scripts/02e_verify_terroir_facts.py](scripts/02e_verify_terroir_facts.py) + ([scripts/_lib/terroir_backcheck.py](scripts/_lib/terroir_backcheck.py)) + compares, per (record, locale) in one request, each translated bullet + with its source bullet — changed numbers, dropped or upgraded hedges, + wrong entities and back-formed names, the watch-list false friends + (generoso → generous, tirage → disgorgement, Burgundian climat → + climate, Lehm → clay, Pintes → Pinot, Немски ризлинг ↔ + Welschriesling), untranslated common nouns, missing exonyms + (`exonyms.exonym_hits` feeds the model its hits) — and applies the + corrected translation under the same guards; every checked bullet + carries `check`, the cache a `backcheck` block keyed on both shas. The + glossary ([scripts/_lib/translation_glossary.py](scripts/_lib/translation_glossary.py)) + and the exonym table carry the same terms so 02e gets them right + first time. +- **The LLM audit is the acceptance measure.** + [scripts/audit_terroir_facts_llm.py](scripts/audit_terroir_facts_llm.py) + grades a sample of records' rendered EN bullets against the full source + with an adversarial verifier on a *different* model + (`claude-opus-5`, vs the sonnet-4-6 extractor / gate) and reports the + reader-misled share with a Wilson interval; `--from-backup ` + grades the same records' pre-run state and `--compare A B` prints the + paired before / after. Run it after every scoped re-run; the review's + target is < 1.5 % misleading. Every Batch-API run prices itself: + `batch.run_batch` sums the per-result usage, prices it at the batch + rate (`BATCH_PRICES_USD_PER_M`) and appends one row per batch to + `raw/.batch/costs.jsonl`; the gate and back-check reports carry it + under `batch`, and `rerun_terroir_facts.py` logs the run's spend per + stage at the end. ## Denomination model (sub-denominations) @@ -220,6 +423,18 @@ override URL. Stage 02's cross-bundle rescue then matches the cahier header by name across the corpus, so overrides automatically promote matching stubs to full extracts. Re-run stages 01 → 04 after edits. +The opposite failure — BO Agri serves a PDF that downloads fine but is +verifiably *another* appellation's cahier (Pierrevert carried +Saint-Pourçain's lien; L'Étoile and Grands-Echezeaux carried Bourgogne +Passe-tout-grains') — is pinned in the **checked-in** +[scripts/_lib/fr/register_overrides.json](scripts/_lib/fr/register_overrides.json) +with `prefer_cahier: true`: stage 01 then binds the eAmbrosia register's +cahier attachment ahead of BO Agri (full `eambrosia-register` provenance, +`boagri_url` left empty), bypassing the has-usable-cahier guard that +otherwise protects working resolutions from INAO outages. The audit's FR +name guard (`audit_terroir_facts.py`, strict) flags a lien that never +names its appellation, so a recurrence cannot go unnoticed. + ## eAmbrosia register — second source for the FR cahier BO Agri is the FR pipeline's primary cahier source, and its long tail is @@ -553,7 +768,11 @@ generator against `wiki/_index.json`. ## Scripts contract Each script is independently re-runnable and writes a manifest. Running stage N -twice with no changes upstream must be a no-op (cache hits). +twice with no changes upstream must be a no-op (cache hits). Stage 02's +`--only NAME` (repeatable substring) re-extracts just the matching records and +merges their entries into `_index.json`; it never rewrites the unselected +records (a partial run used to stub the whole corpus). A parser change that +touches many records still wants a full run. | Script | Reads | Writes | |---|---|---| @@ -568,8 +787,18 @@ twice with no changes upstream must be a no-op (cache hits). | 02b_translate_styles.py | raw/wikipedia/styles//*.json | raw/translations/styles//*.json + manifest.json | | 02b_translate_grapes.py | raw/wikipedia/grapes//*.json + grape-corpus dominant-lang | raw/translations/grapes//*.json + manifest.json | | 02c_translate_summaries.py | raw/inao/cahier-extracted/*.json | raw/translations/summaries//*.json | -| 02d_extract_terroir_facts.py | raw/inao/cahier-extracted/*.json + raw/wikipedia/aocs/fr/ | raw/terroir-facts/*.json + manifest.json | +| 02d_extract_terroir_facts.py | raw/inao/cahier-extracted/*.json + raw/wikipedia/aocs/fr/ + raw/terroir-facts-feedback/*.json (optional) | raw/terroir-facts/*.json + manifest.json | | 02e_translate_terroir_facts.py | raw/terroir-facts/*.json | raw/translations/terroir-facts//*.json | +| recompute_terroir_provenance.py | raw/terroir-facts/*.json + each country's 02d source resolver | raw/terroir-facts/*.json (coverage + provenance, in place) + raw/translations/terroir-facts//*.json (provenance) | +| dedupe_terroir_facts.py | raw/terroir-facts/*.json + wiki/_index.json + wiki/data/aocs.en.*.js (sub-denomination roster) | raw/terroir-facts/*.json (facts pruned, `n_deduped`) + raw/translations/terroir-facts//*.json (same indices pruned, `source_facts_sha` updated) | +| normalize_terroir_facts.py | raw/terroir-facts/*.json + raw/translations/terroir-facts//*.json | same files (bullets normalised in place, translation caches re-keyed) | +| filter_terroir_boilerplate.py | raw/terroir-facts/*.json | raw/terroir-facts/*.json (facts pruned, `n_boilerplate`) + raw/translations/terroir-facts//*.json (same indices pruned) | +| build_terroir_feedback.py | a review evidence dir (`confirmed-misleading.json`, `merged.json`) + raw/terroir-facts/*.json (graded-against shas) | raw/terroir-facts-feedback/.json + manifest.json (merged, never overwritten) | +| 02d_verify_terroir_facts.py | raw/terroir-facts/*.json + each country's 02d source resolver + raw/terroir-facts-feedback/*.json | raw/terroir-facts/*.json (facts gated, `support` per fact, `gate` block) + raw/translations/terroir-facts//*.json (pruned / re-keyed `pending:`) + feedback `history` + manifest-gate.json + tmp/terroir-facts-review/gate-.json | +| 02e_verify_terroir_facts.py | raw/translations/terroir-facts//*.json + raw/terroir-facts/*.json + raw/terroir-facts-feedback/*.json | raw/translations/terroir-facts//*.json (fixed bullets, `check` per fact, `backcheck` block) + feedback `history` + manifest-backcheck.json + tmp/terroir-facts-review/backcheck-.json | +| rerun_terroir_facts.py | a scope (slug list) | raw/terroir-facts-backup// (snapshots) → the chain above; logs /tmp/owm-/ | +| rollback_terroir_facts.py | raw/terroir-facts-backup// | raw/terroir-facts/*.json + raw/translations/terroir-facts//*.json (restored / deleted) | +| audit_terroir_facts_llm.py | raw/terroir-facts/ + raw/translations/terroir-facts/en/ (or a backup run) + each country's 02d source resolver | tmp/terroir-facts-review/llm-audit-.json (read-only) | | 02g_fetch_vivc.py | raw/inao/cahier-extracted/*.json + raw/es/pliegos-extracted/ + raw/pt/cadernos-extracted/ + raw/vivc/slug_overrides.json | raw/vivc/{search,passport,by-slug}/*.html\|json + manifest.json + slug_overrides.example.json | | 02i_fetch_wikidata_qids.py | raw/*/*-extracted/*.json (slug + id_eambrosia) + raw/wikipedia/aocs// + raw/wikidata/slug_overrides.json | raw/wikidata/qids-by-slug.json + p9854.json + manifest.json + slug_overrides.example.json | | 03_generate_wiki.py | raw/inao/cahier-extracted/*.json + raw/terroir-facts/ | wiki/*.md, wiki/_index.json | @@ -1179,12 +1408,28 @@ Pipeline per stub: and run `pdftotext -layout`. 4. Carve the layout-text into a `{article_num: body}` dict via `extract_articles` (anchor regex tolerates the form-feed page - breaks pdftotext emits between articles), then parse: + breaks pdftotext emits between articles). A consolidated + disciplinare appends one sub-disciplinare per sottozona (ALLEGATO N — + SOTTOZONA «…», Trentino's TITOLO II), each restarting at Art. 1: + `extract_article_runs` splits the header sequence into runs at every + restart, drops a table of contents (every body a title line) and a + decree preamble bound in front (an Art. 1 that never *reserves* the + name), keeps the parent's own run as `article_bodies` and the later + runs as the sidecar's `annexes` (`{title, article_bodies}`). Before + 2026-09-13 the last occurrence of each article number won, so a parent + with annexes took its summary, roster, area and Art. 9 lien from its + last sottozona (Montepulciano d'Abruzzo → San Martino sulla Marrucina; + 20 parents affected, review R4) — and, as the parent's real Art. 1 now + names them, the stage-04 sottozona detector finds 77 sottozone in 17 + parents (was 38 in 10). Then parse: - **Article 1** → summary (first paragraph, ≤ 600 chars) - **Article 2** → grape varieties via `match_variety` on line/colon/comma-split candidates + `vitigno NAME` regex scan - **Article 3** → geo area / commune list - - **Article 9** → link to terroir + - **Article 9** → link to terroir — `link_to_terroir` is the + 4,000-character panel cut, `link_to_terroir_full` the whole article + (what 02d, the gate and the audits read; `terroir_article` records + which article it came from) 5. Emit a sidecar JSON under [raw/it/masaf-disciplinari-extracted/.json](raw/it/masaf-disciplinari-extracted/) with full provenance (`bundle_key`, `archive_path`, sha256, match @@ -4540,13 +4785,123 @@ entries and never resubmits an already-processed one — and pass 2 matches answers by content, not call order (runs are single-threaded all the same). -Default models (`scripts/_lib/providers.py`): `claude-sonnet-4-6` for -anthropic, `mistral-medium-latest` for mistral — used by `--batch` and by -synchronous `--provider` runs alike; override per run with `--model`. API -keys are read from the environment or a repo-root `.env`. Anthropic +Default models are **per stage** (`providers.STAGE_DEFAULTS`, decided +2026-09-14 after the paired experiments in +[docs/review-terroir-facts-2026-09-12.md](docs/review-terroir-facts-2026-09-12.md)): +02d extraction `claude-sonnet-5` with thinking off; the gate +(`02d_verify`) and the LLM audit `claude-opus-5` with adaptive thinking; +02e, the back-check and 02c `claude-sonnet-4-6`; mistral +`mistral-medium-latest`. Used by `--batch` and by synchronous `--provider` +runs alike (`batch.default_model(provider, stage)` / +`default_thinking()`, `providers.make_provider(..., stage=)`); override +per run with `--model` / `--thinking`, or `OWM_BATCH_THINKING` for an +experiment. `tests/test_stage_defaults.py` pins the configuration. API +keys are read from the environment or a repo-root `.env`. The Claude 5 +family runs adaptive thinking when `thinking` is omitted and every +stage's `max_tokens` is sized for the JSON reply alone, so +`providers.effective_thinking` sends `disabled` for a Claude 5 model on a +stage that sets no mode (2026-09-15: Sonnet 5 on 02e had 64 % of its +replies truncated by thinking and rejected). + +**Prompt caching** ([scripts/_lib/prompt_cache.py](scripts/_lib/prompt_cache.py), +Anthropic only; Mistral / Ollama get the flat text). Text sent more than +once is placed *first* in the system prompt as its own block with +`cache_control`, so every request after the first reads it at 0.1× the +input price: in the 20 non-FR 02d scripts the lien is the leading cached +block (`cached_system(_document_block(lien), instructions)` — the four +sub-section calls of a record each resent the whole lien; the user turn +now carries only the sub-section request; FR slices section X per call +and shares nothing, so it is left alone), and the gate, the back-check, +the LLM audit and the 21 × 02e scripts cache their static system prompt +(`mark_cached`, shared by every record of a batch — 02e's is ≈ 3 K +tokens per locale). A block below the model's minimum (Sonnet 5 / 4.6 +1,024 tokens, Opus 5 512) silently does not cache and costs nothing; the +Batch API processes concurrently, so hits are best-effort (Anthropic +quotes 30–98 %) — the four calls of a record are submitted adjacently and +the ledger's `cache_creation_input_tokens` / `cache_read_input_tokens` +show the achieved rate per batch. `OWM_CACHE_TTL` = `5m` (default; write +1.25×, the four-call pattern breaks even at a 29 % hit rate), `1h` (write +2×, break-even 70 %) or `off`. Inside one large batch a record's four +calls are processed concurrently, so most of them write instead of read +(13–47 % hits on the cfg-2026-09-14 run — break-even, not a saving); +the 20 scripts therefore tag each call with its sub-section +(`cache_phase`) and `batch.run_two_pass` submits **one batch per phase, +in order** (`run_phased`, per-phase sidecars, `OWM_BATCH_PHASED=0` +disables): the first phase writes the lien, the next three read it, and +the lien block carries the 1-hour TTL in that mode (a read refreshes the +timer, so each phase only has to finish within an hour). The trade-off +is wall-clock — four sequential batches per country instead of one. The next +steps — migrating the corpus to this configuration and the remaining +review recommendations — are in +[docs/handoff-terroir-facts-2026-09-14.md](docs/handoff-terroir-facts-2026-09-14.md). Anthropic batches use the Messages Batches SDK; Mistral batches use a file-upload / poll / download REST flow (no `mistralai` SDK dependency). +## Appellation names: traditional term + legal scheme + +Every record carries two naming axes, derived at stage 04 and never +hand-edited ([scripts/_lib/gi_terms.py](scripts/_lib/gi_terms.py)): + +- **`eu_scheme`** — the legally precise scheme: `pdo` / `pgi` (EU wine, + Reg. 1308/2013), `spirit-gi` (the 28 FR eaux-de-vie + Marc d'Alsace: + spirit-drink GIs under Reg. 2019/787, SIQO `signe_ue = IG`), `uk-pdo` / + `uk-pgi` (the six UK wines, GOV.UK register) and `none` (the 75 Swiss + cantonal AOCs, outside the EU scheme). FR comes from `signe_ue` with the + derived `mvt_kind` as fallback; everyone else from the eAmbrosia kind. +- **`national_term`** — the EU-registered *traditional term* (Reg. 1308/2013 + Art. 112(a), Reg. 607/2009 Annex XII) the regulator attaches to the GI as + a whole: AOC, DOCG / DOC / IGT, DOCa / DOQ / DO / Vino de Pago / Vino de + Calidad / Vino de la Tierra, DOC / Vinho Regional, DOC / IG, DAC / + Landwein, DOK / IĠT. The regulator's own string, never gettext-translated + (the region-name rule). Admission needs all three: registered in Annex + XII, GI-wide (lot-level grades — Qualitätswein, Prädikatswein, kakovostno + — are excluded), attached by a public regulator document or a cited pin. + Local abbreviations of PDO/PGI (OEM, ZOP, ΠΟΠ, BOB …) are the scheme, not + a term; the table loader refuses them. +- **`class_key`** (`;pdo;it:docg;`) is the `;`-padded MVT property the + "Appellation type" facet filters on; **`class_label`** is the per-locale + rendered string, composed once in Python (`classification_label`) so the + JS panel, `docTitleFor`, the SSR card, entity `` / meta description, + browse list and children nav all read the same value. + +Rendering is **`TERM (SCHEME)`** with the scheme word in the UI locale — +`DOQ (PDO)` / `DOQ (AOP)` / `DOQ (DOP)` / `DOQ (BOB)`, `AOC (PDO)`, `IGT +(PGI)`, `AOC (spirit-drink GI)`; term only when there is no scheme (Swiss +`AOC`), scheme only when the country has no term (`PDO` for Mosel or +Sussex, `PGI` for a French IGP — never `IGP (IGP)`). Both tokens are +hover/focus targets of the pill tooltip (definition + regulator source from +the term table); the SSR card carries the same text as an `<abbr title>`. +The stored **`kind`** token (AOC/DOP/IGP/EDV) is untouched — it stays the +paint and filter key (map_template.py paint expression + the six `'IGP'` +gates in app.js). + +Sources, all sha-pinned and joined on the EU file number: + +| country | source | result | +|---|---|---| +| IT | MASAF *Elenco alfabetico dei vini DOP* (+ IGP elenco), scraped from IDPagina/4625 by `it/00_fetch_data.py` into `raw/it/masaf-elenchi/`, parsed by [scripts/_lib/it/national_term.py](scripts/_lib/it/national_term.py); IGT constant for IT PGIs | 79 DOCG / 333 DOC / 112 IGT, 524/524; residue pinned in [scripts/_lib/it/national_term_overrides.json](scripts/_lib/it/national_term_overrides.json) (Cirò Classico → DOCG, Reg. 2025/1518; Valtènesi → DOC, Reg. 2026/572; Casauria → DOCG, Reg. 2025/2261). The known static elenco URL serves a 2014 build — the scraper takes the dated `ServeAttachment` link. | +| ES | MAPA *Listado de DOP e IGP de vinos* (Término tradicional column), fetched by `es/00_fetch_data.py` into `raw/es/mapa/`, parsed by [scripts/_lib/es/national_term.py](scripts/_lib/es/national_term.py) | DO 69 / Vino de la Tierra 43 / Vino de Pago 28 / Vino de Calidad 7 / DOQ 1 / DOCa 1, 149/149; [scripts/_lib/es/national_term_overrides.json](scripts/_lib/es/national_term_overrides.json) pins Priorat → `DOQ` (`castilian_form: DOCa`; Llei 2/2020 art. 4(e) — the regional-language legal form wins when the autonomous community's wine law defines it), Tharsys (file-number bridge, VP per its pliego), Urbezo (VP per the MAPA 2024-10-25 release; the listado still prints DO). No PGI ever receives a PDO-only term (asserted). | +| FR / CH / PT / RO / AT / DE / MT / rest | the checked-in ruling table [scripts/_lib/traditional_terms.json](scripts/_lib/traditional_terms.json): per-(country, kind) constants with a cited ruling per country, the 18 Austrian DAC pins by file number (BML DAC-Verordnungen + RIS, `since_vintage`), the `marc-d-alsace-gewurztraminer` slug pin, and the per-scheme / per-term tooltip definitions in en/fr/es/nl with sources | FR AOC (incl. EDV), CH AOC, PT DOC / Vinho Regional, RO DOC / IG, AT DAC / Landwein, DE Landwein (PDO none), MT DOK / IĠT; GB / LU / BE / NL / SI / HR / HU / BG / GR / CZ / SK / CY empty with a recorded reason. | + +Deferred to a curator pin pass (empty renders scheme-only, never wrong): +GR ΟΠΑΠ / ΟΠΕ, CZ VOC (Znojmo), CH Grand Cru / premier cru sub-tiers, NL +Landwijn, SI vino PTP, HU Tájbor, BG Регионално вино, CY ΟΕΟΠ / Τοπικός +Οίνος — see [CURATOR_TODO.md](CURATOR_TODO.md). Wikipedia extracts for the +term tooltips are a follow-up (the 02b style-lexicon pattern). + +Sub-denominations resolve through their own `file_number` / `signe` (ES +subzonas and IT sottozone carry the parent's number) with a parent-slug +fallback in the parents-first record loop; `scripts/audit_gi_terms.py` +asserts every child equals its parent, that every term's registered scheme +matches the record's, that no `class_label` is empty, and reports the +per-country distribution against the rosters. Run it after every stage-04 +build (`--strict` in CI). + +The facet is its own two-level tree ("Appellation type", advanced mode): +scheme rows over flag + term rows, built with `buildTreeFacet` over +`class_key`; counts are parents-only and wines-only, so DOCG = the roster +count, not roster + sottozone. + ## Internationalisation The map UI chrome (sidebar labels, panel headings, link texts, style chip @@ -4586,9 +4941,9 @@ each run; it rebuilds `messages.mo` only when the `.po` is newer (no-op on rerun). ``` -uv run pybabel extract -F locale/babel.cfg -o locale/messages.pot scripts/_lib/ -uv run pybabel update -i locale/messages.pot -d locale # after adding a new msgid -uv run pybabel init -i locale/messages.pot -d locale -l <lang> # to add a new locale +.venv/bin/python -m babel.messages.frontend extract -F locale/babel.cfg -o locale/messages.pot scripts/_lib/ +.venv/bin/python -m babel.messages.frontend update -i locale/messages.pot -d locale --no-fuzzy-matching +.venv/bin/python -m babel.messages.frontend init -i locale/messages.pot -d locale -l <lang> # to add a new locale ``` After editing a `.po`, just rerun `uv run scripts/04_build_maps.py`. @@ -4744,6 +5099,18 @@ static link layer (all in [scripts/_lib/map_template.py](scripts/_lib/map_templa (resolve storage zone by name, set `Custom404FilePath=/404.html`). Apex→www 301 stays a manual dashboard rule (smoke-checked by `check_apex_redirect`). +## Analytics + +Self-hosted Plausible (site id `openwinemap.com`); the snippet is in +`_TEMPLATE`, custom events go through `track()` in +[scripts/_lib/assets/app.js](scripts/_lib/assets/app.js). The event/prop +reference, the goal-configuration recipe (events are stored but invisible +until configured as goals — retroactively), and the known reading artefacts +(replaceState opens are not pageviews; page-load opens are not tracked; +`Appellation Viewed.slug` is the stack focus, split by `via`) live in +[docs/analytics.md](docs/analytics.md). Keep that table in sync when adding +or renaming a `track()` call, and never commit an API key. + ## Code style - Python 3.12, ruff line length 100. diff --git a/CURATOR_TODO.md b/CURATOR_TODO.md index b4b3ee0..3f7fa86 100644 --- a/CURATOR_TODO.md +++ b/CURATOR_TODO.md @@ -81,6 +81,28 @@ DGC cascading unlock realised in this round: **+106 DGCs** (Beaune climats, Chas **To retry the cookie-expired ones:** refresh `cf_clearance` in your browser (open <https://www.legifrance.gouv.fr/loda/id/JORFTEXT000024923948>, copy fresh cookie), update `~/.config/openwinemap/legifrance.json`, then `.venv/bin/python scripts/01b_solve_legifrance.py --refresh --only 71 --only 134 --only 211 --only 230 --only 247`. +### Terroir-fact source contamination — 3 parents bound to the wrong BO Agri PDF — ✅ resolved via the register (2026-09-11) + +Found by the 1,000-bullet terroir-fact review (plan: `docs/plan-terroir-facts-quality.md`, W2b) — **resolved 2026-09-11** without a BO Agri lookup: the eAmbrosia register serves each appellation's own cahier ("cdc Pierrevert BO.pdf", "l-Etoile CDC homologue.pdf", "Grands-Echezeaux CDC publication BO.pdf"). The three are pinned `prefer_cahier: true` in the checked-in `scripts/_lib/fr/register_overrides.json`; stage 01 binds the register attachment ahead of BO Agri for them, stage 02 re-extracted them (liens now name the appellation), and the audit's FR name guard (`audit_terroir_facts.py`) catches any recurrence. + +| id | slug | was bound to | now | +|---:|---|---|---| +| 290 | `pierrevert` | Saint-Pourçain's cahier | register attachment 2952, `eambrosia-register` | +| 187 | `l-etoile` | Bourgogne Passe-tout-grains' cahier | register attachment 2083, `eambrosia-register` | +| 184 | `grands-echezeaux` | Bourgogne Passe-tout-grains' cahier | register attachment 4206, `eambrosia-register` | + + +### Terroir-fact source contamination — Bourgogne Passe-tout-grains + a shared PDF — ✅ resolved (2026-09-13) + +Found by the full-corpus review (`docs/review-terroir-facts-2026-09-12.md`, R4). + +| id | slug | problem | resolution | +|---:|---|---|---| +| 144 | `bourgogne-passe-tout-grains` | bound to PDF `49acff22…` (BO Agri `e89b7ce3…`) whose extracted lien is the **AOC Beaujolais** cahier | `prefer_cahier` pin in `scripts/_lib/fr/register_overrides.json` → register attachment `CDC_Bourgogne_Passe-tout-grains.pdf` (`3ace5ac0…`); re-extracted: lien names Passe-tout-grains 6×, grapes gamay + pinot noir (+ chardonnay / pinot blanc / pinot gris accessory), styles red + rosé; 02d re-run in the R1 batch | +| 870 / 980 | `hautes-alpes` / `haute-vienne` | both manifest entries carry BO Agri `22caf075…` / PDF `1106c71b…` | **not a misattribution**: the PDF is the arrêté of 2 Nov 2011 bundling ~20 IGP cahiers (Agenais, Comté Tolosan, Coteaux de Glanes, …); stage 02's cross-bundle rescue carved each record's own cahier (Hautes-Alpes 3× / Haute-Vienne 16× own name, 0× the other), and the shared `boagri_url` is the document that contains both. No change. | + +The audit's name guard now requires the whole folded name or the stem of its longest token (≥ 6 letters) — "tout" / "grains" no longer pass a Beaujolais cahier — and stays strict for FR (`name_guard`), report-only for the other 20 countries (`name_guard_other`: CZ region-wide texts never name the wine by design). + ### SIQO referentiel — 2 wines missing (eAmbrosia has them, INAO doesn't) — ✅ both RETIRED (2026-08-26) ✅ Web-research pass confirmed both are intentionally absent — no pinning needed; @@ -364,6 +386,18 @@ Shadow-report findings (`raw/inao/register/shadow-report.md`, 466 parents): |---:|---|---:|---:| | 254 | Collioure | 315 | 14 412 | | 217 | Pouilly-Loché | 255 | 8 261 | + + **2026-09-11 (Pouilly-Loché aire)** — the analytics surfaced a 0.4 km² + AOC drawn across all of Burgundy in simple mode: the 2024 PNOCDC writes + `1 - Aire géographique` (no degree sign) and defines the aire as a + sentence ("territoire de la commune de Mâcon"), so stage 02 scanned the + whole section and recorded the *aire de proximité* (366 communes) as the + aire. Fixed in `extract_aire` (degree-less block headers + sentence-form + aires; corpus-wide, 58 single-commune AOCs — Meursault, Pommard, the + Vosne-Romanée and Gevrey grands crus, Barsac, Cornas, Gigondas … — gained + a previously empty aire) and guarded + in stage 04 (`[villages-guard]`). The short `lien` (255 chars) is still + the register re-source candidate above. | 959 | Franche-Comté | 4 062 | 7 583 | - ❌ **103 appellations (22 %) have no register cahier attachment** — a bare @@ -403,6 +437,50 @@ Two pins were needed: | 335 | Calvados Domfontais | `PGI-FR-01837` | SIQO spelling; register + cahier both say *Domfrontais* | | 1091 | Marc d'Alsace Gewurztraminer | `PGI-FR-01836` | SIQO carries no `categorie`, so the product-type partition cannot be picked (same root cause as the `cote-roannaise` / `muscat-du-cap-corse` `is_wine` side-finding above; those two resolve on the all-partition fallback) | +## Cross-country — terroir-facts audit findings after the 2026-09-13 re-run + +Report-only checks added by review R9 (`scripts/audit_terroir_facts.py`, +run `tmp/terroir-facts-review/audit-r1-2026-09-13.json`). Each needs a +human look; none blocks the strict gate. + +### `wiki_binding` — 23 records whose Wikipedia article title shares no token with the name +Pin the right article or `missing` in `raw/wikipedia/aoc_overrides.json`, +then `02b_fetch_aoc_lexicon.py --lang <l> --source <dir> --only <slug> --refresh` +(a changed revision re-triggers 02d for the record). Probably wrong: +`frusinate` → *Provincia di Frosinone* (the province), `lisboa` → *Lista +de vinhos* (a list), `lesvos` → *Λευκό κρασί* (white wine in general), +`regensburger-landwein` → *Baierwein*, `starkenburger-landwein` → +*Hessische Bergstraße*, the 7 HR records bound to *Vinogradarska +područja Republike Hrvatske* (the umbrella article — legitimate as a +hint, like LU / MT, but pin it explicitly so the check stops firing). +Probably right (a naming-only mismatch): `ahr`, `eger` / `mor` +(*borvidék*), `saint-mont`, `var`, `coteaux-de-die`, `english-wine` / +`welsh-wine` (*Wine from the United Kingdom*), `chios` (*Αριούσιος +οίνος* is the Chios wine), `colli-etruschi-viterbesi` (*Tuscia*). + +### `foreign_name` — source text names another appellation ≥ 5× and its own never +`terras-do-dao` (PT: the IGP text names Vinho Verde 21×) and +`malvasia-handakas-candia` (GR: the spec names Κρήτη 20× — check whether +the file is the Cretan umbrella spec); `sobes` / `schwabischer-landwein` +/ `cvicek` are by design (region-wide or parent text). Verified correct +bindings (2026-09-14, from the MASAF sidecars): `terre-del-colleoni` +(its own disciplinare, *bergamasca* is the adjective for the Bergamo +area, 15× in Art. 9) and `pompeiano` (its own IGT disciplinare; Napoli is +the province) — no action. + +### `rewrite_rejected` / `rewrite_missing` — gate rewrites the guards refused +Since gate-v2 (2026-09-14) an empty rewrite keeps the original as +`supported` with `support.rewrite_missing` (audit check +`rewrite_missing`) and a cosmetic one as `supported` with +`support.cosmetic_rewrite`; `rewrite-rejected` is left for the guard +failures (a new number, an arrow, over 420 chars). The corpus migration +re-gates everything (GATE_VERSION bump); hand-check what the audit still +lists afterwards. + +### Records without a resolvable source +`collioure` (already listed under France) — the only record the gate and +the LLM audit skip (`no_source`). + ## Spain ### Pliego URLs — ✅ complete (2026-05-10) @@ -562,6 +640,8 @@ Cross-canonical implication: all six Iberian names for VIVC #12668 (Trousseau No --- ## Code-side follow-ups (not curator data tasks) +- **Terroir-fact quality fixes W1–W8** (2026-09-11): 02e preserve-list split, Alsace shared-cahier slicer, in-record dedupe, ellipsis-aware coverage, style normaliser, boilerplate filter, audit extension — full handoff in `docs/plan-terroir-facts-quality.md`. + These surfaced in the audit but require code changes, not lookups: @@ -814,6 +894,16 @@ or wait out its expected cancellation. (The research prompt formerly at `tmp/it-masaf-disciplinare-research-prompt.md` was cleaned from tmp/; resurface from git history if needed.) +### MASAF article carver takes the last sottozona annex — ✅ fixed (2026-09-13) + +`extract_articles` kept the **last** occurrence of each article number, so a consolidated disciplinare whose sottozona annexes restart at *Art. 1* yielded the last annex's summary / grapes / area / Art. 9 for the whole DOP (review 2026-09-12, R4). `extract_article_runs` in `scripts/_lib/it/masaf.py` now splits the header sequence into runs at every restart, drops a TOC run (all bodies < 200 chars) and a decree preamble bound in front of the disciplinare (Veneto IGT: five transitional articles whose Art. 1 never *reserves* the name), keeps the parent's own run and stores the later runs as the sidecar's `annexes` (`{title, article_bodies}` — "ALLEGATO 3 «MONTEPULCIANO D'ABRUZZO» SOTTOZONA «ALTO TIRINO»"). `parse_grapes_with` also lets a genuine roster phrase vouch for a slug first hit via the DOC name (Trebbiano d'Abruzzo → trebbiano-abruzzese was lost). Parser template bumped to `it-masaf-disciplinare-v2`; 02f re-run for all 522. + +Result: 20 parents structurally corrected (Abruzzo grapes 2→17, Trentino 4→28, Colli Tortonesi 1→24, Langhe 1→14, Romagna 1→15, Terre di Cosenza 9→17, Friuli Colli Orientali 2→19; Montepulciano / Cerasuolo / Trebbiano d'Abruzzo + Abruzzo get the parent's Art. 9 lien) and, because the parent's real Art. 1 now names the sottozone, the stage-04 detector emits **77 sottozone in 17 parents** (was 38 in 10 — Romagna 16, Terre di Cosenza 7, Friuli Colli Orientali 5, Riviera Ligure di Ponente 5, Asti / Barbera d'Asti / Colli Tortonesi 2 each). + +Open follow-ups: +- Sottozone whose parent Art. 1 does not enumerate them but whose annex titles do (Montepulciano d'Abruzzo 9, Abruzzo 4, Trentino 5 + Titolo II Trentino Superiore) — feed the `annexes[].title` roster to `extract_it_sottozone`, and ground each synthesized sottozona on its own annex Art. 9 (the Alsace `terroir_chapters` pattern) instead of inheriting the parent's bullets. +- `friuli-colli-orientali`: the detector merges «Schioppettino di Prepotto» and «Savorgnano» into one slug (`schioppettino-di-prepotto-savorgnano`) — a quote-splitting quirk in `_split_pattern_b_list`. + ### MASAF grape-extraction fix — ✅ landed 2026-05-20 The earlier note here called the 108 `grapes=0` MASAF records a pure @@ -2034,6 +2124,19 @@ now points at the same **Tokaj Wine Road Association** entry as `vinohradnicka-oblast-tokaj` — the two PDOs are the same physical Tokaj oblasť under different brand registrations. +### ΥΠΑΑΤ specs with sections pasted from another PGI — ❌ open (2026-09-12) + +The national technical files reuse text across PGIs; the terroir facts inherit it (`docs/review-terroir-facts-2026-09-12.md`, R4): + +| slug | pasted from | affected | +|---|---|---| +| `fthiotida` | ΠΓΕ Παρνασσός (names it; delimits Gravia / Elateia / Parnassos / Amfikleia above 350 m) | naturels facts #0–#4 | +| `peloponnisos` | ΠΓΕ Αχαΐα / Πλαγιές Αιγιαλείας / Αρκαδία (semi-sparkling section) | #0, #3–#6 | +| `retsina-evias` | Retsina Attikis (Mesogia, Markopoulo, MARKO cooperative) | human-factors facts | +| `ipiros` | Ioannina sparkling section; #5's 1972 recognition is Zitsa's | #5, #7, #8 | + +Route: a foreign-name guard in 02d / the audit (source names another GI of the same country ≥ 3 × and its own 0 ×) so these sections are refused; curator note to ΥΠΑΑΤ optional. + ## Czech Republic Country #14 (added 2026-05-24). 13 wine GIs (11 DOP + 2 PGI), all 13 @@ -2462,6 +2565,15 @@ surfaced. Worth running after each `02b_fetch_aoc_lexicon` sweep. --- +## Cross-country — terroir-fact full-corpus review (2026-09-12) — ❌ open + +Report: `docs/review-terroir-facts-2026-09-12.md`; evidence `tmp/terroir-facts-review/full-review-2026-09-12/`. Data-side items not covered by the country sections above: + +- Wikipedia bindings: `tirol` → de.wikipedia *Toro (Weinbaugebiet)* (the Spanish DO); `montecastelli` → the village article (its wiki-only bullet #4 describes the village hill). Pin both `missing` in `raw/wikipedia/aoc_overrides.json`. +- `sobes` fact #1 (Mikulov bioregion, Pavlov Hills limestone) is the Mikulovská podoblast 50 km east; Šobes sits on Bohemian Massif crystalline rock — the region-wide CHZO grounding plus the podoblast wiki hint. +- `montana` (BG): the IAVV spec has the Danube "to the south" and Stara Planina "to the north"; bullet #0 silently corrects it — a source typo worth a note. +- Re-run scope: `rerun-slugs.txt` (346 records with a verified misleading bullet) and the 730 records with "Label:" bullets (`det_checks.json` → `rows.label_prefix`), all but one extracted before the 2026-09-11 style block. + ## Cross-country — grape pills show another country's spelling ✅ fixed 2026-09-06 Found 2026-09-06 from a GB spot-check: the English PDO's pill reads **"Optima @@ -3164,3 +3276,32 @@ busuioaca-de-bohotin→8248 cristina→21045 korithi→false schiava→false Same applies to the other ~450 pins already in that file; the deployed site is built from the curator's machine, so production is unaffected. + +## Traditional terms — curator pin passes (scripts/_lib/traditional_terms.json) + +Empty renders scheme-only (never wrong); each pin needs the founding act cited. + +- GR — ΟΠΑΠ / ΟΠΕ per PDO (33): pin from the founding ministerial decisions (ΦΕΚ), not the ΥΠΑΑΤ specs (only 1 of 132 cached specs names ΟΠΕ). ΟΠΕ = Samos, Mavrodaphne Patras / Kefallinias, Moschatos Patron / Riou Patron / Kefallinias / Limnou / Rodou; the rest ΟΠΑΠ. +- CZ — VOC (Víno originální certifikace) for `znojmo` only (zákon 321/2004 §23); the other 12 stay empty. +- CH — Grand Cru (12 Valais communal records, roster from Vinum Montis, 2 `to-verify`) and Premier Cru (22 Geneva records, GE règlement) as sub-tier terms; needs the communal / cantonal règlement cited per record before it can enter the table. +- NL — Landwijn (Annex XII PGI term) vs the BGA-labelled provincie PGIs: decide whether the 12 PGIs carry it. +- SI — vino PTP (GI-wide, Uradni list 49/2007); HU — Tájbor; BG — Регионално вино; CY — ΟΕΟΠ / Τοπικός Οίνος: confirm GI-wide use in the regulator specs, then pin. +- IT — re-scrape MASAF IDPagina/4625 when a new DOCG is recognised (the dated `ServeAttachment` elenco; the static URL is the 2014 build). ES — refresh the MAPA listado (dated header) when a new VP is registered; Urbezo is pinned until the listado catches up. +- Tooltip Wikipedia extracts for the terms (02b style-lexicon pattern): en has articles for DOCG, AOC, DOCa, DAC, IGT, Vinho regional, Landwein, PDO; fr/es/nl gaps via 02b-translate. + +## Pipeline — grape canonical ranking depends on the corpus on disk + +**2026-09-11** — `_vivc_canonical_by_id` (scripts/_lib/grape_entity.py) +picks, among VIVC by-slug files sharing a vivc_id, the slug present in the +extracted corpora on disk (then the most frequent). That makes +`raw/inao/cahier-extracted/` an implicit input of every stage-02 / stage-04 +run and the ranking self-reinforcing: a stage-02 run started on a damaged +FR corpus wrote `corvo` for aubun, `rodo` for mondeuse, `araignan` for +picardan, `livornese-bianca` for rolle, `graciano` for morrastel … across +190 FR records, and later runs kept them. Recovered by seeding the FR +records' grape lists from the last good build and re-running (see the +session memory). To do: pin the FR-canonical slugs explicitly (a checked-in +vivc_id → canonical table, or make `GRAPE_ALIAS` the first tiebreaker) so +the choice no longer depends on what happens to be on disk, and add a +stage-04 assertion comparing the principal-slug set against the previous +build's blob. diff --git a/README.md b/README.md index 3d2c600..978670e 100644 --- a/README.md +++ b/README.md @@ -2,24 +2,25 @@ A reference wiki + map of European wine appellations, generated mechanically from public regulator data. France (INAO + JORF) is the canonical pipeline; -Spain (eAmbrosia + EUR-Lex) lives under `scripts/es/`; Portugal (eAmbrosia + -IVV) under `scripts/pt/`; Italy under `scripts/it/`; Austria under -`scripts/at/`; Slovenia under `scripts/si/`. Every per-record fact traces -back to a public-source document — nothing here is hand-written narrative. +the other 20 countries, the United Kingdom included, each have a sibling +pipeline under `scripts/<cc>/` sourced from the eAmbrosia EU register, the +national regulator, the Swiss federal repertoire or the UK GI register. The +per-country list, coverage and mechanics live in `CLAUDE.md`. Every +per-record fact traces back to a public-source document — nothing here is +hand-written narrative. ## Status The FR pipeline runs end-to-end across the full AOC/AOP/IGP corpus, emitting per-denomination markdown pages (one per appellation plus one per DGC — Muscadet sub-crus, Côtes du Rhône Villages, Alsace grands crus, Chablis -premier-cru climats, etc.). The ES pipeline covers the ~149 wine GIs in -eAmbrosia (106 DOP + 43 IGP); coverage is a function of which wines have an -EU-OJ "documento único" — see `CLAUDE.md` for the curator workflow. The PT -pipeline covers the 44 wine GIs (30 DOP + 14 IGP) sourced from eAmbrosia + -the IVV cadernos master indexes. Italy (531 wine GIs), Austria (32) and -Slovenia (17 — 14 DOP + 3 IGP) follow the same eAmbrosia + EU-OJ -single-document pattern. Stage 04 merges all six streams into a single -four-locale interactive map (FR / EN / ES / NL). The site is deployed at +premier-cru climats, etc.). The other 20 country pipelines follow the same +stage layout (00 fetch → 01 fetch documents → 02 extract → 02d/02e terroir +facts → 03 wiki), each with its own document source and geometry chain; +coverage per country is a function of which wines have a fetchable +specification, and `CLAUDE.md` documents every country's sources, coverage +and curator workflow. Stage 04 merges all 21 streams into a single +four-locale interactive map (EN / FR / ES / NL). The site is deployed at <https://www.openwinemap.com>. ## Setup @@ -160,19 +161,22 @@ Same `--emit-todo` / `--import` flags apply to 02d and 02e. ### Internationalisation (map UI chrome) Sidebar labels, panel headings, and style chip names are translated via -gettext. Catalogs live under `locale/<lang>/LC_MESSAGES/messages.po` and are -hand-editable. Stage 04 recompiles `messages.mo` automatically when the `.po` -is newer. +gettext. Catalogs live under `locale/<lang>/LC_MESSAGES/messages.po`, are +committed to the repo, and are hand-editable. Stage 04 recompiles +`messages.mo` automatically when the `.po` is newer. ``` -uv run pybabel extract -F locale/babel.cfg -o locale/messages.pot scripts/_lib/ -uv run pybabel update -i locale/messages.pot -d locale -uv run pybabel init -i locale/messages.pot -d locale -l <lang> # new locale +.venv/bin/python -m babel.messages.frontend extract -F locale/babel.cfg -o locale/messages.pot scripts/_lib/ +.venv/bin/python -m babel.messages.frontend update -i locale/messages.pot -d locale --no-fuzzy-matching +.venv/bin/python -m babel.messages.frontend init -i locale/messages.pot -d locale -l <lang> # new locale ``` Always extract from the directory, not a single file — `style_taxonomy.py` carries msgid anchors that are silently dropped otherwise (and pybabel -update will mark them obsolete). +update will mark them obsolete). Always pass `--no-fuzzy-matching` to +`update`: without it pybabel fills every new msgid with a guess borrowed from +an unrelated existing entry and marks it fuzzy, which is harder to spot than +an empty msgstr. Then set the new msgstrs by hand in each locale. ## Public data sources @@ -195,13 +199,16 @@ for the full rules. | **EUR-Lex — OJ single documents** — `eur-lex.europa.eu` (HTML) | Canonical pliego de condiciones (documento único) for ES wines: zona geográfica, variedades, vínculo, rendimientos. Stage 01 fetches each wine's `publications[0].uri`; stage 01b uses headless Chromium to solve the CloudFront WAF challenge on the blocked subset | EU public sector information | | **MAPA + CCAA national pliegos** — `mapa.gob.es`, JCCM, INCAVI, AGACAL, ITACyL, Aragón, Navarra, GVA, Canarias, Andalucía, Euskadi, Madrid, Extremadura (per-region PDFs) | Secondary/accessory grape varieties not published in the EU-OJ documento único (stage 02f) | Public domain (national/regional gazettes) | | **SIGPAC vineyard parcels** — `fega.es` (per-comarca shapefiles) | Pliego-cited polygon inclusions for fine-grained ES geometry (e.g. Priorat vs Montsant overlap resolution) | Licence-clear under MAPA terms | -| **GISCO LAU 2021** — Eurostat | EU-wide municipio polygons for ES IGP commune-list / province-wide / CCAA-wide geometry fallback | © EuroGeographics for the administrative boundaries (free reuse) | +| **GISCO LAU 2024** — Eurostat | EU-wide municipality polygons for the commune-list / province-wide / region-wide geometry fallbacks (ES IGPs and the other eAmbrosia countries) | © EuroGeographics for the administrative boundaries (free reuse) | | **Bétard 2022 EU PDO geometry** — [Figshare](https://figshare.com/) (`EU_PDO.gpkg`) | Pre-Nov-2021 EU PDO polygons; covers ~99 of 106 ES DOPs and all 30 PT DOPs | CC0 | | **IVV cadernos de especificações** — `ivv.gov.pt` (per-DOP/IGP PDFs) | Canonical legal definition of every Portuguese wine GI — área delimitada, castas, rendimentos, relação com a área geográfica | Public domain (Portuguese state) | | **DGT CAOP 2025** — `geo2.dgterritorio.gov.pt` (Continente + RAA + RAM GPKGs) | Portuguese commune-precision boundaries for future PT IGP commune-list geometry | CC BY 4.0 | | **Wikipedia** — `<lang>.wikipedia.org` REST API | Sidepanel tooltips for grape varieties and distinctive styles (stages 02b/grapes, 02b/styles); per-AOC pages used as a sommelier-vocabulary salience hint for terroir-fact extraction (stage 02b/aocs → 02d) | CC BY-SA 4.0 | | **VIVC** — [Vitis International Variety Catalogue](https://www.vivc.de/), Julius Kühn-Institut Geilweilerhof | Canonical grape-variety names + VIVC variety numbers driving the per-AOC pill's "canonical bracket" (e.g. *Aragonez (Tempranillo Tinto)*), and synonym-aware Wikipedia search (stage 02g + 02b/grapes). Cite: Röckel et al., Vitis International Variety Catalogue — www.vivc.de | Factual citation only — JKI publishes no explicit data licence. We ship VIVC IDs + prime names; verbatim synonym strings are *not* republished pending JKI confirmation. | -| **Anthropic Messages API** — `claude-haiku-4-5` | Cahier-summary translation (02c), terroir-fact extraction from cahier section X + Wikipedia (02d), terroir-fact translation (02e), grape-tooltip translation (02b/grapes-translate); each stage can be swapped to Ollama or to manual round-trip | n/a — outputs are derivatives of the cahier (public domain) and Wikipedia (CC BY-SA 4.0) | +| **MASAF *Elenco alfabetico dei vini DOP*** — `masaf.gov.it` | Source of the Italian traditional term attached to each DOP as a whole (DOCG vs DOC) | Italian public-sector information (MASAF) | +| **MAPA *Listado de DOP e IGP de vinos*** — `mapa.gob.es` | Source of the Spanish traditional term attached to each GI as a whole (DOCa / DOQ, DO, Vino de Pago, Vino de Calidad, Vino de la Tierra) | Spanish public-sector information (MAPA) | +| **Reg. (EC) 607/2009 Annex XII** — [legislation.gov.uk copy](https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht) | Reference list of the traditional terms registered per member state (Reg. 1308/2013 Art. 112(a)); backs the definition and source shown when hovering a term | EU public sector information | +| **Anthropic Messages API** — `claude-sonnet-4-6` (default model) | Terroir-fact extraction from the regulator's terroir text + Wikipedia (02d) and terroir-fact translation (02e); cahier-summary translation (02c) and grape/style-tooltip translation (02b/grapes-translate, 02b/styles-translate) can run here too, but in practice 02c goes through the manual round-trip (a human translator) and the tooltips run mostly on Ollama with Mistral Small 3.2, with a Claude residual; each stage can be swapped to Ollama, Mistral or manual round-trip | n/a — outputs are derivatives of the regulator text (public sector) and Wikipedia (CC BY-SA 4.0) | The map UI displays attribution alongside any Wikipedia extract ("via Wikipedia · CC BY-SA 4.0"), any translated summary ("Machine translated @@ -212,6 +219,25 @@ carry per-bullet provenance (`cahier` / `wiki` / `both`); bullets grounded in Wikipedia render the CC BY-SA 4.0 attribution inline, the rest default to the cahier-PDF footer link. +### Appellation names: traditional term and legal scheme + +Every appellation carries two names, and they mean two different things: +the **traditional term** the country's regulator attaches to the GI as a +whole (DOCG, DOQ, AOC, Vinho Regional, …; Reg. 1308/2013 Art. 112(a), +registered in Reg. (EC) 607/2009 Annex XII) and the **scheme** it is +registered under (the EU's PDO / PGI). The map shows them as +`TERM (SCHEME)`, for example "DOCG (PDO)" or "DOQ (PDO)", with the scheme +word localised per UI language ("DOCG (AOP)" in French, "DOCG (BOB)" in +Dutch). Where a country has no term of its own (Germany's Mosel, the UK), +only the scheme is shown. Swiss AOCs sit outside the EU scheme and carry no +bracket; the United Kingdom registers under its own GI scheme, which keeps +the words PDO and PGI; French eaux-de-vie are spirit-drink GIs, not wine +PDOs. Lot-level quality grades (Qualitätswein, Prädikatswein, kakovostno +vino) are not terms in this sense and are not shown, and neither are scheme +abbreviations (ΠΟΠ, CHOP, ЗНП, BOB). Hovering a term in the panel shows its +definition and source. The stored `kind` token (AOC / DOP / IGP / EDV) is +unchanged; both axes (`eu_scheme`, `national_term`) are derived at stage 04. + ## Licence - **Code** (`scripts/`, `pyproject.toml`, etc.) — MIT, see `LICENSE`. diff --git a/VERIFICATION.md b/VERIFICATION.md index 81389dc..684fcfd 100644 --- a/VERIFICATION.md +++ b/VERIFICATION.md @@ -884,3 +884,32 @@ the **GOV.UK "protected food and drink names" register** run by DEFRA. # Independent check: https://www.gov.uk/protected-food-drink-names # → filter Register = "Wines", Country of origin = "United Kingdom". ``` + +## 2026-09-07 — traditional terms vs the national rosters (IT MASAF elenco, ES MAPA listado) + +Independent cross-check of the `national_term` axis introduced with the two-axis naming +layer (`scripts/_lib/gi_terms.py`, `scripts/_lib/traditional_terms.json`). + +| check | corpus (parents, wines) | reference | result | +|---|---:|---|---| +| IT DOCG | 79 | MASAF *Elenco alfabetico dei vini DOP* agg. 18.03.2026: 79 DOCG rows | ✅ equal | +| IT DOC | 333 | elenco 332 + Valtènesi (registered 2026-03-18, Reg. (EU) 2026/572, pinned) | ✅ equal | +| IT IGT | 111 | eAmbrosia IT PGI records (112 incl. the Salemi stub not on the map) | ✅ | +| ES DOCa/DOQ | 1 + 1 | MAPA listado 2 July 2026: DOCa 2 (Rioja, Priorat; Priorat shown as DOQ) | ✅ | +| ES Vino de Pago | 28 | listado VP 27 + Urbezo (VP per MAPA 2024-10-25, listado still DO; pinned) | ✅ | +| ES Vino de Calidad | 7 | listado VC 7 | ✅ | +| ES DO | 69 | listado DO 70 − Urbezo | ✅ | +| ES Vino de la Tierra | 43 | listado VT 43 (all 43 ES PGIs) | ✅ | +| AT DAC | 18 | BML DAC-Verordnungen page 17 + Wagram (RIS) | ✅ | +| sub == parent | 1,242 subs | every sub-denomination's (eu_scheme, national_term) equals its parent's | ✅ 0 mismatches | +| kind invariant | AOC 1404 / AOP 1 / DOP 944 / EDV 28 / IGP 537 | same counts in `appellations{,-villages}.geojson` before and after | ✅ unchanged | + +Recipe: + +``` +.venv/bin/python scripts/it/00_fetch_data.py # refreshes raw/it/masaf-elenchi/ (dated ServeAttachment link) +.venv/bin/python scripts/es/00_fetch_data.py # refreshes raw/es/mapa/listado-dop-igp-vinos.pdf +.venv/bin/python scripts/04_build_maps.py +.venv/bin/python scripts/audit_gi_terms.py --strict +grep -o '"kind": *"[A-Z]*"' wiki/map-data/appellations.geojson | sort | uniq -c +``` diff --git a/docs/analytics.md b/docs/analytics.md new file mode 100644 index 0000000..d04a20f --- /dev/null +++ b/docs/analytics.md @@ -0,0 +1,83 @@ +# Analytics — Plausible events, goals and how to read the numbers + +The map ships a self-hosted [Plausible](https://plausible.io/) snippet +(`analytics.dev.devloed.com`, injected by `_TEMPLATE` in +[scripts/_lib/map_template.py](../scripts/_lib/map_template.py)); the site id +is **`openwinemap.com`** (Plausible strips the `www.`). Custom events are sent +through the `track(name, props)` helper in +[scripts/_lib/assets/app.js](../scripts/_lib/assets/app.js), which no-ops +when the script is blocked. Every prop is a bounded slug vocabulary — never +raw user text — so breakdowns stay useful and nothing personal can leak. + +## Events sent by app.js + +| Event | Props | Fired when | +|---|---|---| +| `Appellation Viewed` | `slug`, `country`, `kind`, `region`, `stacked`, `stack_size`, `via`, `locale` | the detail panel opens or the stack focus changes. `via` ∈ `map` (click on the map), `cycle` (re-click cycling a stack), `facet`, `omnisearch`, `panel-link` (parent/child link inside a card). **Not** fired for the page-load open of a `/<lang>/<slug>` landing (the pageview already records it) nor for the localStorage restore. | +| `Appellation Opened` | `slug`, `via` (`facet` / `omnisearch`), `locale` | an explicit pick — the cleanest interest signal | +| `Filter Applied` | `facet`, `value`, `locale` | a facet checkbox / chip / country / region / appellation filter changes | +| `Filters Reset` | `locale` | the reset button | +| `Kind Toggled` | `kind` (`igp` / `spirits`), `enabled`, `via` (`reveal-hint`, optional), `locale` | the IGP / spirits switches | +| `View Mode Switched` | `mode` (`simple` / `advanced`), `locale` | the mode toggle | +| `Grape Scope Toggled` | `scope` (`main` / `all`), `locale` | principal-only vs all grapes | +| `Omnisearch Used` | `result_count`, `had_match`, `groups` (a/g/r/s), `query_len`, `locale` | 1 s after typing stops in the omnisearch | +| `Omnisearch Result Picked` | `type` (`grape` / `region` / `style` / `classification`), `locale` | a non-appellation suggestion is picked (appellation picks fire `Appellation Opened`) | +| `Search Used` | `result_count`, `had_match`, `query_len`, `locale` | the legacy sidebar search (superseded by the omnisearch in June 2026) | +| `Theme Changed` | `theme`, `locale` | light / dark toggle | +| `Feedback Clicked` | `channel` (`github` / `email`), `locale` | a link tagged `data-feedback` in the sidebar disclaimer or the About dialog | +| `Outbound Link: Click` | `url` | Plausible's own outbound-link tracking (source PDFs, Wikipedia, interprofessions, GitHub) | + +## Goals — what the dashboard and the Stats API can see + +Plausible stores every custom event but only **shows** the ones configured as +a goal (Site settings → Goals → *Add goal* → *Custom event*, exact name). +Adding a goal is retroactive — history appears immediately. Configured on +2026-09-11: `Appellation Viewed`, `Filter Applied`, `Outbound Link: Click`, +`View Mode Switched`, `Kind Toggled`, `Filters Reset`, `Search Used`. + +**Still to add** (sent since June 2026 but invisible until then): + +- `Appellation Opened` +- `Omnisearch Used` +- `Omnisearch Result Picked` +- `Theme Changed` +- `Grape Scope Toggled` +- `Feedback Clicked` (new, 2026-09) + +Until `Omnisearch Used` is a goal, search usage is unmeasured — `Search Used` +went to zero when the omnisearch replaced the sidebar search on 2026-06-19. + +## Reading the numbers — artefacts to keep in mind + +- **Pageviews on entity paths are entries only.** Opening an appellation + rewrites the URL with `history.replaceState`, which Plausible does not count + as a pageview (by design — panel opens are not page loads). `/en/bourgogne` + showing 85 visitors and 10 pageviews means 10 landings and 85 people whose + panel was on Bourgogne at some point. Use the `Appellation Viewed` slug + breakdown, not the Pages report, for what people look at. +- **Bounce on entity landings was ~0 by construction until 2026-09-11**: the + page-load open fired `Appellation Viewed`, and any second event ends a + bounce. From this build on, a landing that only reads the SSR page counts + as a bounce again, so expect the organic bounce rate to rise from ~8 % to + something honest. +- **Empty entry page + 0 pageviews** (≈ 7 % of visits) are map tabs resumed + after the 30-minute session timeout; they show as *Direct*. Real, engaged + sessions — not a bug. +- **`Appellation Viewed.slug` is the stack focus**, i.e. the smallest + bounding box under the click, so an umbrella appellation whose aire equals + a famous one gets the credit (Coteaux Champenois ≡ Champagne). Filter on + `via` = `facet` / `omnisearch` — or use `Appellation Opened` — for intent. +- A wrongly-huge simple-mode polygon shows up in this breakdown before it + shows up anywhere else (Pouilly-Loché, 2026-09: a 0.4 km² AOC drawn + across all of Burgundy by a stage-02 aire parse miss). Stage 04's + `[villages-guard]` log line now flags the pattern; the analytics is the + second line of defence. + +## Querying + +The Stats API v2 (`POST /api/v2/query`, `Authorization: Bearer <key>`) covers +aggregates, time series and breakdowns by any prop or visit dimension; no DB +tunnel is needed for insights. Keys are created per session in Plausible's +settings and revoked afterwards — never commit or store one. A ClickHouse +tunnel (port 8123) is only worth it for session-sequence questions +("searched, then opened?") that the API cannot express. diff --git a/docs/handoff-terroir-facts-2026-09-14.md b/docs/handoff-terroir-facts-2026-09-14.md new file mode 100644 index 0000000..9541a6b --- /dev/null +++ b/docs/handoff-terroir-facts-2026-09-14.md @@ -0,0 +1,458 @@ +# Terroir facts — hand-off (2026-09-14) + +Where the terroir-fact quality programme stands after the 2026-09-13 runs, +the configuration decided on 2026-09-14, and the remaining work in priority +order, each item with its evidence, the files involved, the acceptance test +and the cost. Written for whoever picks this up next (human or agent); read +[review-terroir-facts-2026-09-12.md](review-terroir-facts-2026-09-12.md) +("Implemented" / "Results") first for what already landed, and the +"Hard rules" bullets on the gate, the back-check and the per-run backups in +[../CLAUDE.md](../CLAUDE.md) for the invariants. + +## 0a. Progress log (2026-09-14, afternoon) + +Landed on `gi-terms-and-analytics` after the hand-off was written — each +a commit, tests green (632), `ruff` clean: + +| item | commit | what changed | +|---|---|---| +| §0 commit | `a1e4920` … `5724847` | the whole programme committed in the four suggested slices | +| 3.1 MASAF cap | `2cfab2e` | 02f emits `link_to_terroir_full` (v3 sidecars, regenerated: 469 of 522 longer, +3.66 M chars); IT 02d / gate / audits read it; panel text byte-identical | +| 3.2 grounding typography | `5fe8c45` | `normalize()` folds quotes / apostrophes / dashes, NFKC, soft hyphens, hyphen-breaks, guillemet spaces. Measured on every kept r1 quote: FR 1,628 / 3,165 higher (1,387 → contiguous), others 484 / 5,679; 0 threshold regressions | +| 3.3 extractor does the gate's job | `0b72dba` | claim-support rule first in `STYLE_RULES`; gate-v2: empty rewrite → supported + `rewrite_missing`, cosmetic rewrite (ratio ≥ 95 **and** no differing word of 4+ letters) → supported + `cosmetic_rewrite`; audit lists `rewrite_missing` | +| 3.4 earned `interactions` | `0b72dba` | `_lib/terroir_interactions.py` (15-language connective table); 02d drops an interactions fact whose *quote* has no connective (cap 2), the gate demotes it. Measured on r1: 69 % of interactions quotes carry one, 8 % bullet-only, 23 % none. The fourth call stays (FR X.3 / EU 8.4 are the earned ones); no promotion (1,388 candidates would triple the share) | +| 3.7 mechanics | `146a6ef` | `needs_gate` shared by 02d_verify + audit: pending = a fact without a verdict, or source / GATE_VERSION changed — no exact sha, so post-passes no longer re-fire the gate; 02d normalises at write time; `feedback_recurrence` resolves on a meaningful `support.original_bullet` rewrite | +| 3.10 cost logging | `86c3af0` | per-result usage from both providers, Batch-API pricing, `raw/.batch/costs.jsonl` ledger, `batch` block in gate / back-check reports, per-stage spend logged by the orchestrator | +| orchestrator | `3133802` | `--scoped-02d` (IT: every record is stale after 3.1 — a scoped run would otherwise re-extract all of Italy) and `--scoped-gate` (gate-v2 made every record ungated — a smoke would otherwise re-gate the corpus) | + +Not done, deliberately: 3.7 `--only-file` across the 42 scripts (the +stale-marking works; `--scoped-02d` covers the case that hurt), the +`multi_sentence` splitter and the Alsace `produit` decision; 3.5, 3.6, +3.8, 3.9 untouched. + +Later the same afternoon: `ff14659` back-check keeps the translation when +a fix comes back empty (`fix_missing`, backcheck-v2 — 3.8's 512); +`f2e5f4d` + `1023576` the audit pinned the MASAF template at v2 (every +regenerated sidecar would have reported stale) and `META_RE` was +English-only — source-language citations ("secondo il disciplinare", +"selon le cahier des charges", "laut Produktspezifikation", "according to +the production specification") are now caught, and the normaliser drops +such a clause when it trails the sentence; `9e0f8f3` FR 02d records +`n_dropped` / `n_deduped` / `n_unearned_interactions`; `43d2d5e` +orchestrator `--scoped-backcheck` (see the smoke). + +**Sequencing.** Items 3.1–3.4 all change what the extractor produces, so +they were landed *before* the corpus migration (2.1) rather than after — +one full pass, not two. GATE_VERSION is `gate-v2` and BACKCHECK_VERSION +`backcheck-v2`, so the migration's corpus-wide gate and back-check steps +redo everything by construction. + +### Smoke of the new chain — `smoke-cfg-2026-09-14` (6 records, $1.12) + +`rerun_terroir_facts.py --scope {barolo, chablis, rioja, mosel, dingac, +santorini} --scoped-02d --scoped-gate` (+ the back-check scoped by hand, +see below), then `normalize_terroir_facts.py --only …`, then the audit: + +| stage | result | +|---|---| +| 02d Sonnet 5 (thinking off) | 58 facts (9.7 / record; the same six had 52 under Sonnet 4.6), 0 grounding drops, 5 unearned `interactions` dropped → 2 kept = **3.4 % share** (was 10.8 %) | +| gate Opus 5 adaptive | 4 / 58 rewritten (**6.9 %**, was 20–27 %), 0 dropped, 1 moved, 0 cosmetic, 0 empty. All four rewrites read as true source-grounded corrections (Rioja Alavesa / Sierra de Cantabria north–south, "von Hand", "milenaria", the Santorini hedge grading) | +| 02e Sonnet 4.6 → back-check | 19 / 210 translated bullets fixed (9.0 %), 2 empty fixes kept as `fix_missing`, 0 rejected | +| audit | strict checks 0; `gate_pending` 1,629 = the rest of the corpus (v2 bump), `feedback_recurrence` 23 — 17 are the do-not-claim entries merged from the two post-r1 LLM audits, still in their (un-re-extracted) records; none in the six | +| ledger | 02d $0.34 · gate $0.37 · 02e $0.26 · back-check $0.14 | + +A third, found by pulling the smoke's raw 02d outputs from the Batch API +(the dropped bullets are not stored anywhere else): of the 5 unearned +`interactions` drops, 3 were **false negatives of the connective table** +("glavni čimbenik", "fördert", "και έτσι"), 1 a real manufactured link +(Rioja Oriental's "explican" — the quote lists the conditions and says +nothing causal) and 1 a non-causal restatement. Sampling the corpus's +unmatched FR / NL / RO quotes showed the same ("déterminent", +"contribuant", "hetgeen zich vertaalt in", "dă vinuri", cedilla ţ/ş, Ambt +Delden's English text under `source_lang: nl`). `a473a47` broadens the +tables (nouns, verb stems, conjunctions, cedilla folding, NL→EN +fallback): interactions quotes matched on r1 go **69 % → 85 %**. So the +smoke's 3.4 % share was partly an artefact; with the corrected table +expect **≈ 8–10 %**, most of it source-stated links — read the +migration's `unsupported-causal-link` tags in the paired audit as the +acceptance signal, not the share. + +Two more things the smoke taught: (1) after a version bump the corpus-wide +gate / back-check steps are the whole corpus — the back-check submitted +5,895 requests before I cancelled it at $0.00 processed; hence +`--scoped-backcheck`, and the rule *a smoke passes all three `--scoped-*` +flags, a migration passes none*. (2) Sonnet 5 still cites the document +once in six records ("secondo il disciplinare") — the normaliser strips a +trailing citation and the audit now flags the source-language forms. + +**Cost projection for 2.1**, from the ledger: $1.12 for the six → ≈ $300 +for 1,640 records as an upper bound (the six include three long sources — +Santorini's four calls alone were 119 K input tokens); the hand-off's +$210 is the lower bound. Expect $210–300 — **before prompt caching**. + +### Prompt caching — `78ef64a` + +`_lib/prompt_cache.py`: in the 20 non-FR 02d scripts the lien is the +leading cached system block (the four sub-section calls of a record each +resent it); the gate (1,365 tokens), the LLM audit (945) and the 21 × 02e +scripts (≈ 3 K tokens per locale, shared by every record of a batch) +cache their static system prompt; the back-check's 860-token prompt is +under Sonnet 4.6's 1,024 minimum (no-op). FR 02d is untouched (its four +calls share nothing). `OWM_CACHE_TTL` = 5m (default) / 1h / off. + +Probe (`gr/02d --only santorini --batch`, 4 requests): **one write of +25,445 tokens, three reads of 25,445** — a 100 % hit rate on the +follow-up calls, the requests having been processed in submission order. +The record's 02d cost fell from $0.14 (smoke) to $0.073: the lien's +4 × 25 K tokens became one 1.25× write plus three 0.1× reads, 52 % of +the uncached input cost. Batch hits stay best-effort corpus-wide (the +ledger's `cache_creation` / `cache_read` columns show the rate per +batch); the four-call pattern breaks even at 29 % on the 5-minute TTL. + +Revised projection for 2.1 with caching: 02d input roughly halves on +the long-lien countries, 02e's shared system prompt (≈ 3 K of a +≈ 4–5 K-token request) reads from cache for all but the first record +per locale — expect **≈ $200–230** for the full corpus. + +## 0b. The migration — `cfg-2026-09-14` (2026-09-14 evening) — DONE + +`rerun_terroir_facts.py --scope scope-all.json --run cfg-2026-09-14 --parallel 7`, +1,640 slugs, 2 h 13 min wall-clock, **$195.49** (02d $51.50 · gate $69.52 · +02e $38.29 · back-check $36.18), plus the `cfg-2026-09-14-fix` follow-up +($0.60: the three FR slicer records + 16 rejected translations) and the +paired audit ($12). Rollback: `rollback_terroir_facts.py --run +cfg-2026-09-14-fix` then `--run cfg-2026-09-14` (newest first). + +| stage | result | +|---|---| +| 02d Sonnet 5 | 1,639 records, **15,945 facts = 9.7 / record (was 6.9)**, **1 grounding drop** corpus-wide (was 965), `interactions` 7.5 % (1,192; 529 unearned dropped at extraction); cache hit rate 37 % overall — 13–25 % in the small first-wave batches, 34–47 % in the large ones (a record's four calls run concurrently inside a batch, see below) | +| gate Opus 5 adaptive | 1,635 records, 0 errors: **8.3 % rewritten** (was 20–27 %), **0.48 % dropped** (was 1.5–2 %), 204 moved, 13 cosmetic, 0 empty, 0 rejected → 15,868 facts; the cached system prompt hit on all but 4 requests | +| 02e Sonnet 4.6 | 5,897 translations; the per-locale system prompt hit 75 % (≈ $14 saved); 16 replies rejected on bullet count and redone in the follow-up | +| back-check | 5,887 caches, 4,685 / 56,600 bullets fixed (8.3 %), 766 empty fixes kept as `fix_missing`, 94 rejected | +| post-passes | normalise: a handful of fixes; dedupe: nothing (02d and the gate dedupe) → **15,893 facts on 1,637 records** | +| strict audit | **0 failures**; report-only: `feedback_recurrence` 4 (was 23), `gate_pending` 1 (Collioure), `translation_stale` 0, `meta_text` 65 source bullets citing the document mid-sentence (0.4 % — the normaliser strips only trailing citations) | +| **paired Opus-5 audit, 120 records** | misleading **2.47 % → 1.31 %** [0.8–2.2] on 849 → **1,147 bullets (+35 %)**; records improved 15 / worse 10 / same 95. Extraction-origin residuals **18 → 3** (0.26 %), `unsupported-causal-link` 5 → 2; **12 of the 15 residuals are translation-origin** (mistranslation, wrong direction, added qualifier in the EN rendering) — 02e + back-check on Sonnet 4.6 is now the dominant lever (item 3.8). The 15 are merged into the feedback sidecars (`llm-audit-2026-09-14-after-cfg`). | + +Found on the way and fixed (`180ae08`): the FR section-X slicer lost the +natural-factors slice on three cahiers (an OCR "l°" for "1°", a lien +opening at "a)" with no "1°", an "a)" heading without its letter) — +Pouilly-Vinzelles had extracted 1 fact from 9.9 K chars; now 11. + +Prompt caching inside a large batch: a record's four 02d calls are +processed concurrently, so most of them write instead of read (hit rates +13–47 %, vs 100 % on a 4-request probe). At the 5-minute TTL that is +break-even to a modest gain, never a loss. To make it a real saving, +either submit the four sub-section calls as four sequential batches +(the first writes, the next three read, 1-hour TTL) or extract all four +sub-sections in one request per record. The static-prompt caching in +02e (75 %) and the gate (≈ 100 %) worked as intended. + +## 0c. 2026-09-15 — after the migration + +- `e1bd757` the 21 × 02e scripts resolved their model through the *generic* + default, not `STAGE_DEFAULTS["02e"]` (they coincided); fixed + wiring + test — a prerequisite for changing the 02e default after the experiment + below. +- `35a7908` phase-sequenced batch submission (`batch.run_phased`, + `cache_phase` on the 20 non-FR 02d scripts, 1-hour TTL on the lien in + that mode, `OWM_BATCH_PHASED=0` to disable): one batch per sub-section, + in order, so the later phases read the lien the first wrote. Trade-off: + four sequential batches per country. Probe + numbers below. +- `bd897cd` **Sonnet 5 runs adaptive thinking when `thinking` is + omitted** (4.x does not); every stage's max_tokens is sized for the JSON + reply, so the first Sonnet-5 02e run spent its 2,000-token budget + thinking and 64 % of the replies were rejected as truncated (the old + 4.6 caches silently stayed). `providers.effective_thinking` now sends + "disabled" for a Claude 5 model on a stage that sets no mode. The + migration's 02e ran on 4.6 and was unaffected. +- Phased runner probe (2 IT records, rolled back): four sequential + batches, **1 write / 3 reads** of the lien in the ledger — the pattern + a single concurrent batch could not give. Provider queueing was slow + that morning (a 2-request phase sat 3 h), so wall-clock is the cost. +- **Experiment 3.8 — Sonnet 5 for 02e + the back-check**, paired on 60 + records (`exp-02e-sonnet5`, kept in place, its own rollback unit; + ≈ $9): misleading **3 → 6 of 575** (0.5 % → 1.0 %, CIs [0.2–1.5] vs + [0.5–2.3]) but the three extra "after" flags are source-bullet defects + the before-grader let pass; **translation-origin errors 3 vs 3** (5's: + a compass swap, *szőlő* → "wine"; 4.6's: a dropped hedge, an inverted + "favourable"); defective 37 → 30 (6.4 % → 5.2 %); the back-check found + 5.5 % to fix vs ≈ 8.5 % on 4.6 translations. **No detectable quality + difference at this size; Sonnet 5 is a third cheaper** ($1 / $5 vs + $1.5 / $7.5 per M batch — ≈ $25 per corpus pass). Recommendation: + switch `STAGE_DEFAULTS["02e"]` / `["backcheck"]` to `claude-sonnet-5` + on cost (Boris's call — he set the 4.6 default on 2026-09-14). The 60 + records (+ 10 substring extras) carry Sonnet-5 translations, + back-checked; `rollback_terroir_facts.py --run exp-02e-sonnet5` + restores 4.6's. + +- **Sub-denomination pages** (2026-09-15 afternoon, from Boris's reading of + Rioja Alavesa): a parent's bullets are inherited by its sub-pages, where + "the appellation" reads as the sub-zone. `1e55136` — for the 133 records + with sub-denominations (`_lib/terroir_roster.py`, shared with the dedupe + post-pass) the 02e user message and the 02d per-record block ask for + the appellation's name where the source is generic; `88e5c43` — the + back-check gets the same note (without it, it reverted the name as an + added entity — caught on the first pass). Re-translated + re-checked + under `ctx-2026-09-15` / `ctx2-2026-09-15` (≈ $12). The source-language + bullets on sub-pages keep the generic wording until the next 02d pass. + `ceacbd4` — the gate's misfiling rule names wine descriptions stated + inside a climate paragraph (Rioja Alavesa's "versatile wines" bullet + was under natural factors; re-gated → produit; the 12 other wine-led + natural-factor bullets in the corpus are genuine natural factors). + `e1fb127` — Dutch river valleys: "Marnevallei / het Marnedal", never the + "Marnedallei" blend Sonnet 4.6 produced (glossary + back-check + watch-list). + +## 0. State you inherit + +- **Corpus**: 1,638 records / 11,255 source bullets / 5,895 translation + caches; strict audit 0 findings (`label_prefix` strict); 1,403 records + extracted 2026-09-13 with Sonnet 4.6 under the current prompt, every + record gated (Sonnet 4.6, `gate-v1`), every translation back-checked. + Measured with the Opus-5 verifier: 3.2 % reader-misleading bullets on a + 120-record sample (from 13.8 % before), 1.4 % on a 60-record sample of + the second run. +- **Runs on disk** (`scripts/rollback_terroir_facts.py --list`): + `r1-2026-09-13` (union scope, 1,634 slugs), `r1b-2026-09-13` (rest + scope, 519), `r1c-2026-09-13` (grounding fix + provenance re-grade, + 1,203), `exp-sonnet5` (rolled back), two `smoke-*`. Roll back newest + first; rebuild stage 04 afterwards. +- **Not committed.** Everything since commit `a857790` on branch + `gi-terms-and-analytics` is uncommitted: the whole review programme + (`_lib/terroir_*`, the 21 × 02d/02e edits, the gate, the back-check, the + orchestrator, the rollback, the audit changes, docs, tests). First act + of the hand-off: commit it — `git status` is the list; a sensible split + is (1) backup + rollback + orchestrator, (2) gate + back-check + prompts + + audit, (3) MASAF slicer + BPTG pin + Wikipedia pins, (4) docs. Raw + caches, backups and feedback sidecars are gitignored and live only on + this machine. +- **Map**: `wiki/` was rebuilt from the final caches (77 IT sottozone, + Montepulciano d'Abruzzo 9 facts); not deployed. + +## 1. Decided configuration (2026-09-14) — already the code default + +| stage | model | thinking | why | +|---|---|---|---| +| 02d extraction (21 scripts) | `claude-sonnet-5` | off | paired test on 120 records: same per-bullet reliability as Sonnet 4.6 (1.7 % vs 1.8 % extraction-origin misleading), **31 % more grounded facts**, 0 grounding drops, 20 % gate-rewrite rate vs 27 %, $2 / $10 per M vs $3 / $15 | +| gate `02d_verify` | `claude-opus-5` | adaptive | the Opus-5 verifier caught residuals a Sonnet gate had passed; `MAX_TOKENS` raised to 8,000 for thinking + reply | +| LLM audit | `claude-opus-5` | adaptive | unchanged | +| 02e translation, back-check | `claude-sonnet-4-6` | — | not re-tested; the back-check already fixes 9 % of translated bullets | + +Where it lives: `providers.STAGE_DEFAULTS` + `stage_default()`, +`batch.default_model(provider, stage)` / `default_thinking()`, the +`thinking` argument threaded through `batch.run_two_pass` → +`_submit_anthropic` and `AnthropicProvider`; every 02d script passes +`stage="02d"`, the gate / back-check / audit their own stage; `--model` and +`--thinking` override per run; `OWM_BATCH_THINKING` overrides for an +experiment. `tests/test_stage_defaults.py` pins it. The orchestrator's +`--model` now applies to 02d only. + +**The corpus is on this configuration since the `cfg-2026-09-14` run** +(§0b); step 2.1 below is what was run. + +**Costs** (measured token usage × Batch-API rates): a full corpus pass is +≈ **$210** in this configuration (02d $47, gate ≈ $79 with adaptive +thinking, 02e $50, back-check $34) vs $157 all-Sonnet-4.6; a 120-record +Opus-5 audit pass is ≈ $6; a scoped re-run costs proportionally. Token +usage per batch is retrievable for 29 days with +`client.messages.batches.results(id)` — there is no cost logging in the +pipeline yet (item 3.10). + +## 2. Runbook + +``` +# one scoped run = one rollback unit; logs under /tmp/owm-<run>/ +.venv/bin/python scripts/rerun_terroir_facts.py --scope SLUGS.json --run <id> [--parallel 7] +# = mark stale → 02d --batch per country → 02d_verify --batch (corpus-wide, ungated) +# → 02e --batch (all countries, stale only) → 02e_verify --batch → audit +.venv/bin/python scripts/normalize_terroir_facts.py && .venv/bin/python scripts/dedupe_terroir_facts.py +.venv/bin/python scripts/audit_terroir_facts.py --strict --strict-labels --quiet --report tmp/terroir-facts-review/audit-<id>.json + +# acceptance: paired, same grader, same records +.venv/bin/python scripts/audit_terroir_facts_llm.py --sample 120 --batch --from-backup <id> --report before.json +.venv/bin/python scripts/audit_terroir_facts_llm.py --slugs-file <the sample> --batch --report after.json +.venv/bin/python scripts/audit_terroir_facts_llm.py --compare before.json after.json +# feed the residue back: --emit-feedback DIR → scripts/build_terroir_feedback.py --evidence DIR --review-id <id> + +# undo +.venv/bin/python scripts/rollback_terroir_facts.py --list +.venv/bin/python scripts/rollback_terroir_facts.py --run <id> [--only slug] [--dry-run] +.venv/bin/python scripts/04_build_maps.py +``` + +Gotchas learned the hard way: the 21 scripts spell their record filter +three ways (`--slug` exact for FR, `--only` *name* substring for stage 02, +`--only` slug substring for the other 02d scripts) — that is why the +orchestrator marks caches stale instead of passing slugs; never run two +chains on overlapping records at once (the gate and 02e both write the +translation caches); a post-pass that changes bullets makes `gate_pending` +fire again (it keys on an exact sha) — re-run `02d_verify` after +normalise / dedupe; the audit's `feedback_recurrence` is lexical and +fuzzy-matches corrected bullets (5 of its 6 residual hits were fixed +bullets), so read its rows before believing the count. + +### 2.1 Migrate the corpus to the decided configuration + +Re-extract everything with Sonnet 5 and re-gate with Opus 5: + +``` +python3 -c "import json,glob; json.dump({'slugs':[p.split('/')[-1][:-5] for p in glob.glob('raw/terroir-facts/*.json') if 'manifest' not in p]}, open('tmp/terroir-facts-review/scope-all.json','w'))" +.venv/bin/python scripts/rerun_terroir_facts.py --scope tmp/terroir-facts-review/scope-all.json --run cfg-2026-xx-xx --parallel 7 +``` + +≈ $200–230 with prompt caching (see §0a), ≈ 45–60 min wall-clock. Then the acceptance pair above; expect the +misleading share at or below 3 % on the 120-record frame, more facts per +record (≈ 8.9 vs 6.8), and check that the Opus gate's rewrite / drop +shares are in the 4.6 gate's range (20–28 % / 1.5–2 %) — a much higher +drop rate means the adaptive-thinking gate is stricter than the prompt +intends and the prompt's "supported" definition needs loosening, not the +model. The 231 records extracted 2026-09-11 (never re-extracted since) +come along in this pass. + +## 3. Remaining recommendations, ranked + +Each item: **evidence** (measured in this programme) → **do** → **accept**. + +### 3.1 Lift the 4,000-character cap on the MASAF terroir text (Italy) +- Evidence: `derive_terroir(body, max_chars=4000)` in + `scripts/_lib/it/masaf.py` truncates the sidecar's `link_to_terroir`; + 312 of 519 IT sidecars have an Art. 9 longer than that — ≈ 2.85 M + characters of regulator terroir text that 02d, the gate and the audit + never see. Italy is the largest and worst-scoring country. +- Do: keep the panel's short text as `link_to_terroir_brief` (stage 04 + reads the short one) and give 02d the full Art. 9 body (or raise the cap + for the 02d path only). Re-run `it/02f_extract_masaf.py --all + --include-nonstub`, then the IT records through the chain (their sha + changes, so the orchestrator picks them up without a scope). +- Accept: IT facts per record up; paired audit on 60 IT records not worse + than before; `audit_it_coverage.py` unchanged. +- Cost: IT-only pass ≈ $70. + +### 3.2 Grounding filter: normalise typography before matching +- Evidence: even after the block-aware coverage (`_lib/terroir_coverage`), + 965 extracted facts were discarded against 11,255 kept (7.9 %); 114 + records lost ≥ 3. The quotes are verbatim; they fail on curly vs straight + apostrophes, hyphen-space artefacts ("gradi- giorno"), soft hyphens, + ligatures. (Sonnet 5 had 0 drops on the 120-record sample, so 2.1 may + make this moot — measure after it.) +- Do: fold apostrophe / quote / dash variants and remove soft hyphens and + hyphen-newline joins in `normalize()` on both sides; keep the 0.6 + threshold; add cases to `tests/test_terroir_audit_checks.py` + (`test_coverage_is_block_aware…`). Then re-run the records with + `n_dropped ≥ 3` (the r1c pattern: `r1c-scope-grounding.json` was built + from that query). +- Accept: `n_dropped` sum < 2 % of kept; no new `eroded_bullets`; the + scattered-phrase and foreign-text probes still score < 0.3. + +### 3.3 Make the extractor do the gate's job +- Evidence: the gate rewrote 27–28 % of Sonnet 4.6 bullets (20 % of + Sonnet 5's); 44 % of rewrites were light edits and 9 % near-cosmetic + (`fuzz.ratio ≥ 95`); 124 rewrites came back empty. +- Do: (a) move the gate's claim-support rules ("assert only what the + quoted sentence states; never turn presence into cause; never narrow an + en-bloc attribution") into `STYLE_RULES` — a first version is there; + make it explicit and first; (b) in `_lib/terroir_gate.apply_verdicts`, + treat a rewrite with `fuzz.ratio(original, rewrite) ≥ 95` as + `supported` (keep the original) so the rewritten cohort is meaning + changes only; (c) in the gate prompt require a non-empty `rewrite` for + the `rewrite` verdict, else downgrade to `supported` with the note. +- Accept: gate rewrite share falls; a 100-rewrite sample graded by the + Opus verifier shows ≥ 90 % meaning-changing. + +### 3.4 Finish R3 — earn the `interactions` sub-section +- Evidence: the `interactions` share is unchanged at 10.8 % of bullets; + the gate rewrites causal wrappers but rarely moves the bullet out of + `interactions`; 7 of the 25 residual misleading bullets on the sample + are still unsupported causal links. +- Do: the review's first option — drop the fourth sub-section call in the + 21 scripts, ask the three remaining calls to mark `causal: true` only + when the source sentence carries an explicit connective, and file those + under `interactions` (cap 1–2). `SUBSECTIONS` is per script; stage 04 + and `terroir_sources` key on the four sub-section names, so keep the + key, change how it is filled. Alternatively a deterministic + connective check per source language in `apply_verdicts` (a bullet + under `interactions` whose quote has no connective → move to + `facteurs_naturels` / `produit` by the verdict's `subsection`). +- Accept: `unsupported-causal-link` tags in the paired audit at or near 0 + (the share itself is not the measure — with the corrected connective + table ≈ 8–10 % of bullets are source-stated links; see §0a). + +### 3.5 Calibrate the measurement +- Evidence: the Opus-5 grader put the pre-programme baseline at 13.8 % + where the review's Sonnet verifier said 6.9 %; nobody has checked the + grader's precision, nor sampled the 3,899 gate rewrites or the 4,483 + back-check fixes for correctness. +- Do: second-opinion 50 of the grader's "misleading" verdicts (a + different model, or a human on the sheet the audit can emit); grade a + 100-rewrite and a 100-fix sample; publish the precision numbers in the + review doc and restate the target on the Opus scale. +- Accept: a precision figure per instrument; the acceptance target + restated ("< X % on the Opus-5 grader"). + +### 3.6 Source-binding residue the new audit checks surfaced +- `wiki_binding` 23: 7 HR records bound to the umbrella *Vinogradarska + područja Republike Hrvatske*, `frusinate` → the province, `lisboa` → a + list article, `lesvos` → "white wine", `regensburger-landwein` → + *Baierwein*, `starkenburger-landwein` → *Hessische Bergstraße*; pin in + `raw/wikipedia/aoc_overrides.json`, refresh with + `02b_fetch_aoc_lexicon.py --lang <l> --source <dir> --only <slug> + --refresh` (a changed revision re-triggers 02d). Full list in + CURATOR_TODO "Cross-country". +- `foreign_name` 3: `terre-del-colleoni` names Bergamasca 8×, `pompeiano` + two provinces 5× — check the MASAF binding. +- CZ register fiche: the `morava` / `slovacka` "terroir" text (31 k + chars) is a per-wine-type description that never names the wine + (`name_guard_other`); check `_lib/register_fiche` slicing of §7. +- MASAF annexes: the sidecars now carry `annexes[].title`; feed those to + `extract_it_sottozone` (Montepulciano d'Abruzzo 9, Abruzzo 4, Trentino + 5 are still undetected) and ground each synthesized sottozona on its + own annex Art. 9 instead of the parent's bullets; fix the detector's + merge of «Schioppettino di Prepotto» + «Savorgnano». +- `rewrite_rejected` 97: originals kept, flagged in `support`; re-gate + after 3.3 or hand-check. + +### 3.7 Smaller mechanics +- `feedback_recurrence`: compare the claim with `support.original_bullet` + too, and count a fact as resolved when its current bullet differs + meaningfully from the matched original. +- Key the gate on a normalised sha so normalise / dedupe do not re-fire + `gate_pending`. +- One `--only-file` across the 21 × 02d and 02e scripts (then the + orchestrator can pass slugs instead of marking caches stale). +- `multi_sentence` 95 source bullets despite the one-sentence rule — + minor; the normaliser could split on the first terminator + capital. +- `cross_record_identical_en` 16: the Alsace grand cru `produit` slice is + one shared paragraph for 51 crus; decide whether to keep it once per + cru or drop it from the sub-record view. + +### 3.8 Translation side +- The back-check fixes 8.9 % of translated bullets; the r1b pass (after + the glossary additions) fixed 8.4 % — the glossary is not the lever, + the back-check is. Consider Sonnet 5 for 02e + back-check (untested; + one paired 60-record run answers it, ≈ $10). +- 512 back-check fixes came back empty (same fix as 3.3c). + +### 3.9 Curator items outside the pipeline +- `collioure`: no source text resolves; the only record the gate and the + audit skip. +- Hautes-Alpes / Haute-Vienne: verified a genuine 20-cahier bundle, no + change needed (closed in CURATOR_TODO). + +### 3.10 Cost logging +- `batch._fetch_anthropic` sees `usage` per result; sum input / output + tokens into the stage manifests and the gate / back-check reports so a + run reports its own cost. Today the numbers in this document came from + re-reading the batches via the API. + +## 4. Things that are settled — do not reopen + +- The coverage threshold stays 0.6; the block-aware measure is the fix + for artefacts, not a lower threshold. +- The name guard is strict for FR only; the other countries' texts often + never name the wine by design (CZ region-wide, Landwein). +- Sonnet 5 as extractor is a coverage / cost choice, not a quality one + (paired test, `exp-sonnet5`); do not expect it to move the misleading + rate — 3.3 / 3.4 / 3.5 are what move it. +- The label-prefix regex was tightened to 1–2-word labels after it + flagged enumerating colons ("Three soil types coexist:"); keep + `label_prefix` strict. diff --git a/docs/plan-eu-scheme-and-national-tier.md b/docs/plan-eu-scheme-and-national-tier.md new file mode 100644 index 0000000..738148b --- /dev/null +++ b/docs/plan-eu-scheme-and-national-tier.md @@ -0,0 +1,448 @@ +# Task brief — separate the EU scheme from the national term, and localise both + +## Objective + +Two related defects in how the corpus names what an appellation *is*: + +1. **`kind` is never localised.** It is stored as a project-internal token (`AOC` / `DOP` / `IGP` / `EDV`) and rendered verbatim in all four + locales. An English visitor sees `DOP` on Chianti — and on the six **British** GIs, where the UK register itself says PDO. +2. **The national term is not modelled at all.** Everything an EU member state protects is flattened to `DOP` / `IGP`, so Barolo **DOCG** + and a generic Piedmont **DOC** are indistinguishable, Priorat **DOQ** looks exactly like Montsant **DO**, Kamptal **DAC** like + Niederösterreich g.U. + +The fix is to stop conflating two orthogonal things: + +| axis | what it is | values | +|---|---|---| +| **EU scheme** (`eu_scheme`, new, derived) | the legal scheme the name is registered under | `pdo` · `pgi` · `spirit-gi` · `uk-pdo` · `uk-pgi` · `none` | +| **National term** (`national_term`, new) | the EU-registered *traditional term* the member state attaches to the GI as a whole — Reg. (EU) 1308/2013 Art. 112(a), Reg. (EC) 607/2009 Annex XII (copy: legislation.gov.uk/eur/2009/607/annex/XII) — inside PDO **or PGI** | DOCG, DOC, IGT, DOQ, DOCa, DO, Vino de Pago, Vino de la Tierra, AOC, DAC, Landwein, Vinho Regional, DOK, … | + +The stored `kind` token stays exactly as it is. Rendering is **`TERM (SCHEME)`** — `DOQ (PDO)`, `DOCG (AOP)` in FR, `AOC (PDO)`, `Vinho +Regional (PGI)` — term only when there is no scheme (Swiss `AOC`), scheme only when the country has no term (UK `PDO`, Mosel `PDO`, French +`PGI`), and never `IGP (IGP)`. + +A third, dependent work item rides along: the **About dialog** and the homepage / browse meta descriptions, which are stale and write the EU +scheme as a language salad (`AOC, AOP, IGP, DOP`). + +## Established facts — verified 2026-09-06/07, do not re-litigate + +### Terminology + +For **wine** the EU defines exactly two schemes; each language abbreviates the same two things differently. + +| | scheme 1 | scheme 2 | +|---|---|---| +| EN | PDO | PGI | +| FR | AOP | IGP | +| ES · IT · PT | DOP | IGP | +| DE | g.U. | g.g.A. | +| NL | BOB | BGA | +| EL | ΠΟΠ | ΠΓΕ | + +Three qualifications, all live in the corpus: + +- **Spirit drinks have one scheme, not two** — the *geographical indication* of Reg. (EU) 2019/787 Art. 3(4). The 28 `EDV` records are + exactly the SIQO rows with `signe_fr = AOC`, `signe_ue = IG` (Cognac, Armagnac, Calvados, but also Whisky breton, Rhum de la Martinique + and three Pommeaux); the eAmbrosia register types them `GI`. "Eau-de-vie" is a product category, not a scheme, and wrong for five of the + 28. +- **The UK registers under its own scheme** (GOV.UK register, since 2021-01-01) using the words PDO / PGI. The six UK wines are *also* in + the EU register (Sussex added 2025-01-31), so a localised scheme word (AOP in the FR locale) is not false; the tooltip names the UK + register. +- **Switzerland is outside any EU scheme.** The 75 CH records carry no `signe_fr` / `signe_ue`; `AOC` is the OFAG répertoire's own + designation. + +**AOC is not a scheme.** It is the French national traditional term for the French PDO (INAO: both words may appear on the bottle), and +separately the Swiss designation. DOCG / DOC / IGT (IT), DOQ / DOCa / DO / Vino de Pago / Vino de Calidad / Vino de la Tierra (ES), DOC / +Vinho Regional (PT), DOC / IG (RO), DAC / Landwein (AT), Landwein (DE), DOK / IĠT (MT) are likewise Annex XII traditional terms bound to PDO +or to PGI. + +**Scheme abbreviations are decoys, not terms.** OEM/OFJ (HU), ZOP/ZGO (SI), ZOI/ZOZP (HR), CHOP/CHZO (SK/CZ), ЗНП/ЗГУ (BG), ΠΟΠ/ΠΓΕ (GR/CY), +g.U./g.g.A., BOB/BGA, PDO/PGI, AOP are the *scheme* in a local alphabet and are never a `national_term`. Lot-level quality grades +(Qualitätswein / Prädikatswein, kakovostno / vrhunsko, kvalitetno / vrhunsko KZP, akostné víno, jakostní víno, minőségi bor) vary inside one +GI and are excluded by construction. + +### Corpus state + +Stored `kind` distribution (2,914 records, from the built startup blob): + +| value | n | where | +|---|---:|---| +| `AOC` | 1,404 | fr 1,329 + ch 75 | +| `DOP` | 944 | every other EU country + gb | +| `IGP` | 537 | all countries | +| `EDV` | 28 | fr spirits (`is_wine: false`) | +| `AOP` | 1 | `marc-d-alsace-gewurztraminer` — stray, see Traps | + +Stage 04 collapses the EU scheme into the stored token (`04_build_maps.py:2320-2331`: `if sfr == "AOC" or sue == "AOP": mvt_kind = "AOC"`), +so the scheme of a stored `AOC` is only recoverable from `country` / `signe_ue` — which is why `eu_scheme` is derived and no label is keyed +on the token. `signe_ue` is not in the startup blob. + +`kind` is **not** translated anywhere; `labels["kind_aoc"]` / `labels["kind_igp"]` (`map_template.py:73-74`) are translatable but every +locale left them at `"AOC / AOP"` / `"IGP"`. The paint is binary (`map_template.py:585-590`: `kind == 'IGP'` green, everything else maroon), +so the maroon swatch covers FR AOC, every DOP, the 75 Swiss AOCs and — toggle on — the 28 spirit GIs. + +**The `classifications` facet is a different axis.** It carries ageing and traditional mentions (riserva 220, reserva 123, crianza 113, +superiore 67, the Prädikat and výběr ladders, Tokaj terms), built by `_aging_tiers_from_text` (`04_build_maps.py:461`) over +`scripts/_lib/aging_taxonomy.py`, which also owns the word `tier`. Folding the national term into that tree as a fourth root was designed +and rejected: one heading for two axes, sub-denomination-inflated counts, and a load-bearing `scan=False` guard so `do` / `doc` / `ig` never +become text-scan keys. + +### Italy — MASAF publishes the roster keyed by EU file number + +The "Elenco alfabetico dei vini DOP" PDF on MASAF IDPagina/4625 carries, per row, the scheme, the *Menzione tradizionale* (DOC / DOCG) and +the eAmbrosia file number: **79 DOCG · 332 DOC** (411 rows; the IGP elenco lists 111 IGT). A file-number join resolves 410 of the 412 IT DOP +records with zero disagreements against the Article-1 text rule. The two misses: `nizza` (elenco `PDO-IT-A1896` vs eAmbrosia `PDO-IT-01896` +— compare by numeric tail) and `valtenesi` (DOC, registered 2026-03-18, newer than the roster). + +Counts that must not be conflated: **522** MASAF sidecars in `raw/it/masaf-disciplinari-extracted/` (plus `_index.json`), **523** +non-sottozone IT parents in the built blob, **524** wines in `raw/it/eambrosia/index.json` = elenco 522 + Valtènesi + Salemi (IGT, pending +cancellation, off the map for lack of geometry). Exactly two eAmbrosia wines have no sidecar, no cached PDF and no bundle entry: +`ciro-classico` (`PDO-IT-03209`, DOCG via Reg. (EU) 2025/1518; a full EUR-Lex extraction that *is* on the map) and `salemi`. Anything +present in eAmbrosia but absent from the elenco is **pinned, never defaulted**. + +The Article-1 regex survives as an **audit**, not a source. Over the 522 sidecars it yields 76 DOCG · 330 DOC · 105 IGT · 11 unresolved; the +79 = those 76 + `sforzato-di-valtellina` (its `article_bodies` has no key `"1"` — keys 2 / 3 / 9 — although `articles_present` is [2..10]; +Article 2 opens with the tier) + `vermentino-di-gallura` (Article 1 reads *"La DOCG «…»"*, an acronym the phrase regex cannot match; the +apostrophe variant does **not** recover it) + `ciro-classico` (no sidecar). The unanchored blob scan gives 77 = anchored + sforzato, a true +positive: unanchored in principle, not a false friend in this corpus. The apostrophe variant (`denominazione d'origine` / `d’origine`) still +matters — requiring `di` alone loses 30 wines: + +```python +ORIG = r"denominazion\w*\s+d(?:i|['’´])\s*origine" +DOCG = re.compile(ORIG + r"\s+controllata\s+e\s+garantita|\bD\.?O\.?C\.?G\b", re.I) +DOC = re.compile(ORIG + r"\s+controllata(?!\s+e\s+garantita)|\bD\.?O\.?C\b(?!\.?G)", re.I) +# head = (article_bodies.get("1") or article_bodies.get("2") or "")[:400] +``` + +Record `kind == IGP` settles IGT with certainty (0 text-vs-kind disagreements over 511 resolved sidecars); an IGT phrase inside a `dop-*` +bundle is a cross-reference to flag. The six regional geoportal layers carry a per-feature tier that `ITZoneIndex` drops — a secondary audit +signal (230 / 233 agreement; Veneto stale on one promotion, Umbria conflates DOC and DOCG), never an authority. **Wikidata P31 was tested +and rejected**: two DOCG-typed items exist in all of Wikidata; stage 02i stays `sameAs`-only. + +### Spain — MAPA publishes the roster keyed by EU file number + +MAPA's "Listado de DOP e IGP de vinos registradas en la UE" (updated 2026-07-02) lists all 149 Spanish wine GIs with a *Término tradicional* +column and the EU file number: **DO 70 · DOCa 2 · VP 27 · VC 7 · VT 43**, joining on `file_number` for 148 / 149 (the miss is `tharsys`, a +Vino de Pago with a file-number mismatch). Display forms are the words on the bottle — `Vino de Pago`, `Vino de Calidad`, `Vino de la +Tierra` — not MAPA's column shorthand. `es/00_fetch_data.py::fetch_mapa_listado` fetches it sha-pinned to +`raw/es/mapa/listado-dop-igp-vinos.pdf` + `manifest.json`; `scripts/_lib/es/national_term.py` parses it. + +A pliego **text scan is NOT safe** and is not used: `Calificada` matches 20 pliegos but only two are real (`cadiz` matches on +*des*calificada); the earlier scan-derived VP (13) and VC (9) lists were materially wrong — `uruena` is a Vino de Pago, +`serra-de-tramuntana-costa-nord` is a PGI (VP / VC / DOCa are PDO-only), `tierra-del-vino-de-zamora` has been a plain DO since 2007, 14 of +the 27 VPs were missing and 12 listado VPs carry no `pago` token at all. The MAPA GIS zone layer is stale (Priorat as plain DO) and is a +cross-check, not a seed. VP / VC therefore **ship in v1**. Priorat's own documento único, its pliego (`PC-Priorat-DOQ-nov-21`), the Consell +Regulador and Catalan wine law (Llei 15/2002) all write **DOQ**; the listado writes DOCa (see Decisions). + +### Austria — DAC is the DOCG-vs-DOC problem with a closed statutory roster + +Exactly **18** of the 27 AT PDOs are DACs — the 27 minus the 9 Bundesland g.U. — per Weingesetz 2009 §10 Abs. 7 and one RIS DAC-Verordnung +each (Wagram: BGBl. II Nr. 30/2022; Thermenregion 2023). The BMLUK DAC page listing 17 is stale (no Wagram); the ÖWM list has 18. eAmbrosia +names carry no "DAC" and only 9 of the 18 single documents mention it (Weinviertel, Wachau, Traisental, Carnuntum, Neusiedlersee, +Südsteiermark, Weststeiermark, Wiener Gemischter Satz, Ruster Ausbruch: 0 hits), so the roster is a pin keyed by file number. The 3 Landwein +PGIs read `Landwein`. Do not copy the two cancelled entries in `scripts/_lib/at/region.py` (`PDO-AT-A0220`, `PDO-AT-A0227`) into the pin. + +### Everywhere else — keyed on (country, kind), a ruling per row + +Constants keyed on country alone would have stamped DO on the 43 Spanish IGPs (MAPA: VT 43 / 43), DOC on the 14 PT IGPs (14 / 14 cadernos +say *Vinho Regional*, 0 say DOC) and DOC on the 12 RO IGPs (0 say DOC; `terasele-dunarii` reads "☐ DOP ☑ IGP ☐ IG") — 69 records. Hence the +(country, kind) table, and a checklist line asserting no IGP record anywhere carries a PDO-only term. + +| country | PDO side | PGI side | source / ruling | +|---|---|---|---| +| fr | `AOC` (`signe_fr`, incl. the 28 spirits; the 4 records with empty signe fields take AOC iff derived `mvt_kind == AOC`) | `""` — the bottle reads IGP; Annex XII's *Vin de pays* is no longer printed | SIQO referentiel | +| ch | `AOC` (all 75) | — | OFAG répertoire; the OFAG tier (cantonale / régionale / locale) and `grand_cru` go in the tooltip, not the term | +| it | `DOCG` / `DOC` (elenco join) | `IGT` (constant) | MASAF | +| es | `DO` / `DOCa` / `DOQ` / `Vino de Pago` / `Vino de Calidad` | `Vino de la Tierra` | MAPA listado | +| pt | `DOC` (29 / 30 cadernos self-declare; `dao` takes the constant) | `Vinho Regional` | IVV cadernos, constant per (country, kind) | +| ro | `DOC` | `IG` | DOCUMENT UNIC / ONVPV, constant | +| at | `DAC` (18, pinned) / `""` (9 Bundesland) | `Landwein` (3) | RIS DAC-Verordnungen + BML DAC page | +| de | `""` — Qualitätswein / Prädikatswein are lot-level | `Landwein` (26 by name + Großräschener See pinned with its BLE Produktspezifikation) | BLE | +| mt | `DOK` | `IĠT` | the MT single documents literally say "The mention DOK" | +| gb | `""` — PDO / PGI are the register's words | `""` | GOV.UK register | +| gr · cy | `""` in v1 (ΟΠΑΠ / ΟΠΕ / ΟΕΟΠ are real Annex XII terms, but only 1 of 132 cached ΥΠΑΑΤ specs carries *Ελεγχόμενη* — no in-build document attaches them per GI) | `""` | empty pin section; curator pass citing the founding FEK decisions | +| cz | `""` in v1 (VOC / Znojmo: producer-association mark, 0 / 13 records mention it) | `""` | empty pin section | +| hu · si · hr · sk · bg · be · nl · lu | `""` — no per-GI traditional term (Marque Nationale abolished; local abbreviations are scheme decoys) | `""` | ruling recorded | + +An empty `national_term` renders as *nothing*, never a placeholder. + +## Decisions taken + +Stated as decided; none is open. + +1. **Rendering is `TERM (SCHEME)`**, the scheme word in the UI locale's own word — EN PDO / PGI, FR AOP / IGP, ES DOP / IGP, NL BOB / BGA. + Term only when the scheme is `none` (CH `AOC`); scheme only when there is no term; never `IGP (IGP)`; both empty renders nothing. Chablis + reads `AOC (PDO)` in EN and `AOC (AOP)` in FR — the two French words the visitor conflates, once each. +2. **UK wines localise the scheme word** (`PDO` / `AOP` / `DOP` / `BOB`), the UK register named in the tooltip; `national_term` is empty. +3. **French spirits read `AOC (spirit-drink GI)` / `AOC (IG spiritueux)`** / `AOC (IG de bebida espirituosa)` / `AOC (GA gedistilleerde + drank)` — the product word never enters the slot. The FR msgid is `IG spiritueux`, not `IG (boisson spiritueuse)`: the render adds the + bracket. +4. **The stored `kind` token is untouched.** Four new derived fields carry everything: `eu_scheme`, `national_term`, `class_key`, + `class_label`. +5. **`national_term` is the regulator's string, never translated** — the region-name rule. Priorat → `DOQ` under the rule *"use the + autonomous community's wine-law form when it has one"* (Llei 15/2002), recorded in `scripts/_lib/es/national_term_overrides.json` with + `castilian_form: DOCa`; the one sanctioned deviation from a roster string. +6. **Admission rule** (all three must hold, else `""` with a recorded ruling): (a) registered in Annex XII / the eAmbrosia traditional-terms + register for that country; (b) applies to the GI as a whole, not per lot; (c) attached to that GI by a public regulator document — a + roster keyed by file number, the GI's own specification in the build, or a checked-in pin citing the founding act. +7. **GR ΟΠΑΠ / ΟΠΕ, CZ VOC and CH Grand Cru / premier cru are scheme-only in v1** — pin sections created empty, filled by a later curator + pass. Unpinned renders scheme-only, never wrong. +8. **Tooltips on both tokens** reuse the pill-tooltip mechanism (definition + regulator source); Wikipedia extracts are a follow-up. +9. **The facet is its own two-level tree "Appellation type"**, not a fourth root of the ageing tree, and ships as the **last** phase. **No + paint change** — the facet, not colour, separates DOCG from DOC. +10. **`locale/*.po` + `messages.pot` are committed**; `.gitignore` narrows to `locale/**/*.mo`. +11. **Stage-03 wiki frontmatter keeps the stored `kind`** — out of scope, noted; the `.md` files are unlinked, octet-stream and absent from + the sitemap. Analytics keep the stored token (`Kind Toggled {kind:'igp'}`, `Appellation Viewed {kind}`) for series continuity. + +## Work item A — data model (`scripts/_lib/gi_terms.py` + `traditional_terms.json`) + +Set at record assembly in `04_build_maps.py` inside `common_props` (2535-2560, next to `"kind": mvt_kind`), computed right after the block +that derives `mvt_kind` from `signe_fr` / `signe_ue` (2312-2331) so every existing signe correction is inherited, and read back at ~3673 +exactly as `kind` is. `common_props` is written verbatim to GeoJSON and tippecanoe, so all four fields **are MVT properties** (negligible +tile cost); all four go into `STARTUP_AOCS_FIELDS` (`map_template.py:1327`) and its contract comment, naming the readers: `docTitleFor` +(runs pre-hydration at `app.js:2441`), the `renderAocCard` meta line, `matchesClient` / `matchesExceptFacets` / `refreshFacetAvailability`, +the omnisearch term index. The lazy panel payload (`wiki/data/d/**`) is untouched and must stay byte-identical. + +1. **`eu_scheme`** (`gi_terms.derive_eu_scheme`) — `ch` → `none`; `gb` → `uk-pdo` / `uk-pgi` (from the register's `protection_type`, or + `mvt_kind`); `fr` → `signe_ue` AOP→`pdo`, IGP→`pgi`, IG→`spirit-gi`, empty `signe_ue` → from `mvt_kind` (AOC→`pdo`, IGP→`pgi`, + EDV→`spirit-gi`; the stray `AOP` → `spirit-gi`, register `PGI-FR-01836` type GI); every other country → DOP→`pdo`, IGP→`pgi`. +2. **`national_term`** (`resolve_national_term`) — in order: FR `signe_fr` (AOC iff derived `mvt_kind ∈ {AOC, EDV}`); CH constant; roster + join on `file_number` (IT elenco, ES listado, AT / DE / MT pins); the (country, kind) constants; else `""`. Rosters, pins, constants, + rulings and the per-scheme / per-term tooltip definitions (4 locales, each with cited sources) live in the checked-in + **`scripts/_lib/traditional_terms.json`**. A scheme abbreviation in a term slot raises at load. +3. **`class_key`** — the facet's filter string, `;{eu_scheme};{cc}:{term-slug};` (`;pdo;it:docg;`, `;pgi;de:landwein;`, `;none;ch:aoc;`, + `;pdo;hu:;`), consumed by the existing `['in', ';v;', ['get', field]]` expression (`app.js:1328`); each key is its own descendant, no + expansion table. +4. **`class_label`** — per locale, precomputed **in Python only** by `gi_terms.classification_label(national_term, eu_scheme, labels)` at + emission (the startup blob and the SSR card are both emitted per locale). JS reads `r.class_label`; SSR reads the same value; nothing + composes the label client-side. The scheme msgids (FR `AOP`, `IGP`, `IG spiritueux`) are folded into `build_labels` so + `RenderCtx.labels` already carries them — no `RenderCtx` signature change; `uk-*` reuses the pdo / pgi msgstrs. + +**Sources fetched, sha-pinned, parsed** (public, licence-clear, joined on file number by numeric tail): + +- IT — `it/00_fetch_data.py` fetches the MASAF *Elenco alfabetico* DOP + IGP PDFs into `raw/it/masaf-elenchi/elenco-{dop,igp}.pdf` (sha256 + + `fetched_at` in the manifest); `scripts/_lib/it/national_term.py` parses `file_number → DOC | DOCG` (79 / 332); `IGT` is a constant for IT + PGIs; `scripts/_lib/it/national_term_overrides.json` pins the residue (Cirò Classico → DOCG citing Reg. (EU) 2025/1518, Valtènesi → DOC + citing its Gazzetta Ufficiale decree, …). The Article-1 regex moves to `audit_gi_terms.py` and must report 0 disagreements. +- ES — `scripts/_lib/es/national_term.py` parses the listado's *Término tradicional* column (DO 70 / VT 43 / VP 27 / VC 7 / DOCa 2); + `scripts/_lib/es/national_term_overrides.json` carries Priorat → DOQ (`castilian_form: DOCa`), the `tharsys` file-number bridge, and any + disagreement the three-way audit (listado ∪ pliego §9 *término tradicional* sentence ∪ zone-layer `tpr_ds_descripcion`) surfaces. +- AT — DAC pinned by file number from the BML DAC-Verordnungen page (18, RIS citations); Landwein constant for the 3 PGIs. PT DOC / Vinho + Regional, RO DOC / IG, DE Landwein (PDO `""`), MT DOK / IĠT — constants or pins in `traditional_terms.json`; everything else `""` with its + ruling row. + +**Sub-denominations** resolve through their own `file_number` / `signe` (ES subzonas carry the parent's number — `rioja-rioja-alavesa` has +`PDO-ES-A0117`; IT sottozone are `dict(record)` copies made at `04_build_maps.py:893`; PT sub-regiões and LU communes share the parent's), +with a **parent-slug fallback pass** in the parents-first record loop (`04_build_maps.py:1022`, the `es_region_by_parent_slug` side-dict +pattern at 2340-2347). Nothing inherits "at the rendering layer" — no code in `app.js` or `content_block.py` looks up a parent at render. A +test asserts every child's `(national_term, eu_scheme)` equals its parent's (`rioja-rioja-alta` == DOCa / pdo, `chianti-rufina` == DOCG / +pdo). + +`audit_gi_terms.py` re-derives every roster and pin on each run and reports CONFIRMED / DRIFT / UNPINNED (the geometry-outlier-override +pattern), lists every corpus record the rosters lack, and diffs the elenco / listado counts against the eAmbrosia corpus, so a silent PDF +layout change shows up as a count mismatch, not as wrong labels. + +## Work item B — render sites + +Every surface renders `class_label`; each is a one-line replacement of `kind`. + +| file:line | surface | note | +|---|---|---| +| `app.js:2328` | client panel meta line | `{flag} {Country} · {class_label} · {Region}`; term and scheme each wrapped `<span class="gi-term has-info">` for the tooltip; bracket muted, non-breaking | +| `content_block.py:858` | SSR panel meta line | same spans, `<abbr title="{definition}">` so crawlers get the definition without JS | +| `app.js:2376` | `docTitleFor` / stack header | reads the startup blob, correct before hydration | +| `map_template.py:1104-1109` | entity page `<title>` | `{name} — {class_label} · {Region}, {Country} · Open Wine Map` — "Barolo DOCG", "Priorat DOQ" are the search phrases | +| `map_template.py:1113-1114` | entity meta description → og:description, Place JSON-LD `description` | `{name} — {class_label}, {Region}, {Country}. {grapes}. {scheme_long}.` clamped to 160; no new schema.org property | +| `map_template.py:1251, 1264-1266` | browse page `<small>` | | +| `map_template.py:1707` → `content_block.py:760` | children nav `sub-kind` | only when the child's label differs from the parent's (never, after inheritance) | +| `map_template.py:73-74`, `2365-2366` | legend swatches | **new msgids** `Protected origin (PDO, AOC, DOC, DO…)` / `Geographical indication (PGI, IGT, Landwein…)` — honest for FR AOC + every DOP + CH + spirits | +| `map_template.py:41-45` | homepage `meta_description` → `<meta>`, og, WebSite JSON-LD (1527), `llms.txt` blockquote (`04_build_maps.py:4269`) | rewritten msgid, no salad | +| `map_template.py:263-266` | `browse_meta_description` | rewritten msgid | +| `map_template.py:82, 93` | `show_igp_label`, `count_hidden_igp_hint` | msgstr-only: EN "Show PGIs", NL "BGA tonen" | +| `wiki/llms.txt` entries | `— DOCG (PDO)` suffix | follows automatically | + +Worked readings (EN unless stated): Priorat `🇪🇸 Spain · DOQ (PDO) · Cataluña` (FR `DOQ (AOP)`, ES `DOQ (DOP)`, NL `DOQ (BOB)`); Montsant `DO +(PDO)`; Barolo `DOCG (PDO)`; Langhe `DOC (PDO)`; Terre Siciliane `IGT (PGI)`; Chablis `AOC (PDO)` / FR `AOC (AOP)`; Pays d'Oc `PGI` / FR +`IGP`; Vully `AOC`; Cognac `AOC (spirit-drink GI)` / FR `AOC (IG spiritueux)`; Sussex `PDO` / FR `AOP`; Landwein Rhein `Landwein (PGI)`; +Mosel `PDO`; Weinviertel `DAC (PDO)` vs Niederösterreich `PDO`; Σάμος `PDO` in v1; Malta `DOK (PDO)`; Rioja Alavesa `DOCa (PDO)` inherited. + +**Tooltips**: the existing pill tooltip (`app.js:2615-2700` — `resolvePillInfo`, `showPillTip`, `aria-describedby`) gains a +`.gi-term.has-info` branch reading a per-locale `TERMS_INFO` (`gi_terms.build_terms_info`) injected like `STYLES_INFO` +(`map_template.py:1598`), keyed `{cc}:{term-slug}` and `scheme:{eu_scheme}`: one definition sentence + the regulator source links (the +`appellation_notes.json` shape; the CH card adds the OFAG tier + cantons from the record). Wikipedia extracts are a follow-up. + +## Work item C — About dialog, meta descriptions, README + +`map_template.py:224-278` is wrong in four ways: `about_roadmap_html` hardcodes "20 pays européens" and enumerates them without Royaume-Uni; +`about_data_html` credits **IGN** as "le fond cartographique" (IGN supplies French commune polygons; the basemap is OpenStreetMap / CARTO); +it names four sources for a pipeline drawing on the EU register plus a dozen national regulators; nothing discloses the LLM layers. + +Replacement, paragraphs in order (as shipped): lead → sources (documentary + cartographic) → the LLM layer → coverage +(interpolated counts) → errors / contributions → browse link → made-by → data-updated. The About dialog is about Open Wine Map, +not about the naming layer: the **"how names work"** explanation was dropped from it on review (owner's call, 2026-09-07) and lives +in the README only; the tooltips on the two tokens carry the per-term explanation in the UI. + +- **Lead** — names the actual registers, in the locale's own scheme words: EN "A reference map of Europe's wine appellations, generated + automatically from public regulator data: the EU register of protected designations of origin (PDO) and protected geographical indications + (PGI), the national regulators behind them, the Swiss federal repertoire of cantonal AOCs and the UK GI register." FR msgid uses « régime + », never « schéma ». +- **How names work** — README subsection "Appellation names: traditional term and legal scheme" only (not an About + paragraph; no `about_names_html` msgid). +- **Homepage `meta_description`** — EN "Interactive map of European wine appellations: PDO and PGI with their label term (AOC, DOCG, DOQ, + DAC…), Swiss AOCs and UK PDOs — grapes, styles and terroir from official registers."; **`browse_meta_description`** likewise. Keep "AOC" + in the string — 1,329 records, the largest search token. +- **Counts** — `{n}` `{c}` `{parents}` `{subs}` computed once in `emit_html` from the same `aocs` dict as the startup blob, **wine-only**, + formatted with `babel.numbers.format_decimal(n, locale=lang)` (2 914 / 2.914 / 2,914), passed to **both** `_build_about_dialog` call sites + (`map_template.py:1640` and `:1676`); no plural msgids; same formatting for `browse_intro_html`. +- **LLM disclosure, per layer, with the models actually recorded**: terroir notes extracted from the regulator text and translated by Claude + Sonnet 4.6; grape / style tooltip extracts translated from Wikipedia mostly by Mistral Small 3.2 via Ollama, with a Claude residual; + cahier summaries mostly human-translated with a machine-translated residual — each item carrying its own attribution line. +- Claims that must **not** be made: ~~"every appellation is on the map"~~ (true today, silently false at the first stub); ~~"the panel shows + each boundary's source"~~ — `_approx_line` (`content_block.py:708`) only flags *approximate* geometries; say "approximations are flagged". +- **README** is a separate work item, not a three-line fold-in: intro still says "six streams" and lists FR/ES/PT/IT/AT/SI, Status counts + are stale, the sources table repeats the IGN-as-basemap error, says GISCO LAU 2021 (now 2024) and names `claude-haiku-4-5` (never used), + the licence list omits OGL v3 / IODL 2.0 / MAPA CC-BY / CARTO + OSM attribution. Either an explicit README checklist or cut Status / + Sources to a pointer at CLAUDE.md plus a build-generated country table. + +## Work item D — facet (last phase) + +Its own `<details data-modes="advanced" data-facet="appellation-type">` under the Appellation facet and above « Classement », heading FR « Type +d'appellation » → EN "Appellation type", ES "Tipo de denominación", NL "Type appellatie". There is no "region facet" to copy — +`facet_regions` (`04_build_maps.py:3807`) only seeds the omnisearch, and the user-facing region control is the country → region tri-state +inside the Appellation tree, which filters by slug expansion. The live model is the classifications tree: `buildTreeFacet` (`app.js:799`) +over the MVT `class_key` with `inField` (`app.js:1325-1330`); the tree comes from `gi_terms.build_term_tree`. + +- Level 1 = scheme rows, gettext-labelled (PDO · PGI · Spirit-drink GI · PDO (UK scheme) · PGI (UK scheme) · AOC (Switzerland)); level 2 = + term rows keyed `{cc}:{term-slug}`, labelled flag + regulator string (`🇮🇹 DOCG 79`, `🇪🇸 Vino de Pago 27`, `🇦🇹 DAC 18`, `🇵🇹 DOC 30`) — the + flag disambiguates the three DOC rows and the two Landwein rows. Countries with no term contribute to the scheme row only. +- Wiring, the ten classifications sites copied: `filters.appellationType` (`app.js:614`), `buildFilterExpr`, `matchesClient` (1345), + `matchesExceptFacets` (1357, new except key), active-filter chips + removal (655-680, 1379), `refreshFacetBadges` (1428), + `refreshFacetAvailability` (1449), reset (1271), `Filter Applied {facet:'appellation-type'}`. No deep-link / localStorage work. +- **Counts are parent-only and wine-only** (skip `is_sub_denomination == "1"` and `is_wine == "0"` in the count loop at `04:3717`), so the + facet numbers equal the acceptance numbers: DOCG = the elenco roster, DOCa = 1, DOQ = 1, DAC = 18 — not roster + 38 inherited sottozone, + not Rioja + 3. +- Omnisearch term index (typing "DOCG" / "DOQ" / "DAC" offers the facet node, the `pickOmni` region pattern, 1646): **deferred** — not + in v1; the facet tree is the entry point. + +## Sequencing + +1. **Copy + legend + catalogs** — commit `locale/*.po` + `.pot`, rewrite the About / meta msgids, new legend msgids, toggle msgstrs, IGN + credit, counts interpolation. Zero runtime risk. +2. **Fields + render sites + tooltips** — `gi_terms.py`, `traditional_terms.json`, the IT / ES parsers and pins, `common_props` + + `STARTUP_AOCS_FIELDS`, the twelve sites above, `TERMS_INFO`, tests. One full rebuild with the golden diff (expected set: `app.<lang>.js`, + `aocs.<lang>.js`, entity / browse / home HTML, `llms.txt`; `wiki/data/d/**` identical). +3. **Facet + omnisearch** — last, its own risk class. + +Deploy: phase 2 rewrites ~11.7k entity pages + 5 homepages + 4 browse pages + 4 `app.js`; the panel JSONs stay byte-identical (every new +field is in the startup blob). Deploy is SHA256-diffed; run `compare_build_output.py` first; take the PUT rate from the last deploy log. + +## i18n mechanics — read before touching any string + +- `locale/*.po` and `messages.pot` are **committed** from this work on; `.gitignore` carries `locale/**/*.mo`. Commit the catalogs *before* + any msgid change so the diff shows the translations. +- **Editing an existing `msgstr` is safe. Changing a `msgid` is not** — pybabel fuzzy-matches new msgids onto unrelated entries, and + `compile_catalogs` → `write_mo(use_fuzzy=False)` silently drops fuzzy entries, so the French msgid renders. Prevent it with + `--no-fuzzy-matching` (below); `.venv/bin/pybabel` has a broken shebang and `uv` is not installed. Extract from the **directory**, never a + single file. Then set every new `msgstr` explicitly in en / es / nl **and** fr; the legend and scheme entries are new msgids, so they + start empty. +- Two pre-existing `#, fuzzy` entries in `locale/fr` (`le registre eAmbrosia de l'UE`, the stale country-count string) — clear them, do not + let them mask new ones. +- `national_term` values are **never** gettext'd; a test asserts the field is byte-identical across the four startup blobs. Stage 04 calls + `compile_catalogs()` at start; it rebuilds `.mo` only when the `.po` is newer. + +``` +.venv/bin/python -m babel.messages.frontend extract -F locale/babel.cfg -o locale/messages.pot scripts/_lib/ +.venv/bin/python -m babel.messages.frontend update --no-fuzzy-matching -i locale/messages.pot -d locale +``` + +## Verification + +- [ ] `.venv/bin/python -m pytest tests/ -q` — 368 before the change; expect edits: `tests/test_content_block.py:227` (EN children list + expects a literal `AOC`) and every meta-line / `kind: "DOP"` fixture. +- [ ] New tests: `classification_label` per locale — IT `DOCG (PDO)` / `(AOP)` / `(DOP)` / `(BOB)`; FR `AOC (PDO)` (never `AOC · AOC`); CH + `AOC` once, no scheme token; GB `PDO` in EN, `AOP` in FR, `national_term == ""`; EDV `AOC (spirit-drink GI)`; French IGP `PGI`, never + `IGP (IGP)`; empty / missing term → no `None`, `—` or dangling ` · `; `_build_entity_meta` description contains `DOCG`; IT audit-regex + fixtures (`«X»` DOCG, straight + typographic apostrophe, `La DOCG «Vermentino di Gallura»`, Article-1-less sidecar whose Art. 2 says + *controllata e garantita*, cross-reference decoy → DOC); ES listado parser fixture (DO 70 / DOCa 2 / VP 27 / VC 7 / VT 43); + sub-denomination == parent; scheme abbreviation in a term slot raises; `national_term` byte-identical across locales; all four fields + in `STARTUP_AOCS_FIELDS`. +- [ ] `.venv/bin/python -m ruff check scripts/ tests/` +- [ ] `npx --yes eslint@9 scripts/_lib/assets/app.js` — 0 errors (6 pre-existing warnings) +- [ ] Full `scripts/04_build_maps.py` run green; asset content-hashes match their filenames. +- [ ] Golden comparator (`scripts/compare_build_output.py`, text files only: html / js / css / xml / txt) against a pre-change snapshot — + expected diff set as under Sequencing; `wiki/data/d/**` identical. +- [ ] **Direct kind-count diff** on `wiki/map-data/appellations.geojson` and `appellations-villages.geojson` (the comparator skips + `.geojson` and `.pmtiles`): `AOC 1404 / AOP 1 / DOP 944 / EDV 28 / IGP 537` unchanged before and after — the mechanical guard for the + six `=== 'IGP'` gates. +- [ ] Counts: IT `national_term` non-empty for every non-sottozone parent (523), DOCG set == the elenco's DOCG rows (79 today; URL + sha256 + + date recorded), audit reports 0 regex disagreements and lists every one-side-only slug; ES DO 70 · DOCa 1 · DOQ 1 · Vino de Pago 27 + · Vino de Calidad 7 · Vino de la Tierra 43; AT DAC 18; no IGP record anywhere carries a PDO-only term (DO / DOC / DOCG / DOCa / DOQ / + VP / VC / DAC / DOK); every `gb` record `eu_scheme` starts `uk-`; every `ch` record `eu_scheme == "none"`. +- [ ] Static grep of `<title>`, `<meta name=description>` and the SSR `.meta` line on one IT / FR / CH / GB / EDV entity page per locale — + server-rendered, so the grep *does* prove them; the whisky-breton / rhum-de-la-martinique titles never contain "Eau-de-vie" as a + classification. +- [ ] Browser check for the client-rendered tree, chips, panel meta, stack header, tooltips and IGP paint: serve with `.venv/bin/python + scripts/serve.py` (Range-capable, port 8765 — plain `http.server` ignores Range and PMTiles never load), then a one-off Playwright + script from the `bootstrap` group (`.venv/bin/python -m playwright install chromium`). Assert SSR meta line == client meta line for + the same slug; nothing enforces it otherwise. +- [ ] IGP toggle still filters and IGP polygons still paint differently; `marc-d-alsace-gewurztraminer` still `is_wine: false`; + `wiki/llms.txt` carries the new copy. + +## Traps + +- **Never change the stored `kind` token.** Six code sites gate on `=== 'IGP'` (`map_template.py:588`, `app.js:747, 1323, 1346, 1358, + 1626`). +- **Never key a label on the stored token.** Stored `AOC` is FR PDO *and* CH no-scheme; stored `EDV` is a spirit GI. Only `eu_scheme` + decides. Do not print an untranslated English "PDO" inside a French line for UK wines either — localise the scheme word and let the 🇬🇧 + chip + tooltip carry the register. +- **`tier` is a taken word twice** — `aging_taxonomy.py` owns it, and every CH extracted record carries an OFAG `tier` plus 12 Valais + `grand_cru` blocks; read neither as the ageing tier, and do not fold either into `national_term` in v1. **`classifications` is ageing + tiers**, not GI terms — different axis, own facet, own MVT property. +- **The ES text scan is a false friend**: 20 "calificada" hits, 2 real. Use the listado. **The IT blob scan is unanchored, not wrong**: in + this corpus it adds only sforzato, a true positive. Neither it nor the anchored regex is the source; the elenco is. +- **Scheme abbreviations are decoys.** A record scan finds OEM (23 / 41 HU), ZOP, ZOI, CHOP, ЗНП, PDO and would produce `ZOP (PDO)`. The + loader raises. +- **Sub-denominations do not inherit at render time.** Resolve in the build loop; a slug-keyed lookup leaves every child empty or wrong. + **Facet counts include subs and spirits by default** (`region_counts` increments per MVT feature) — count parent-only, wine-only. +- **`marc-d-alsace-gewurztraminer`** is the corpus's only `kind: "AOP"`; it is `is_wine: false` only because its SIQO `categorie` is empty, + not because the code knows it is a spirit. `eu_scheme` maps it to `spirit-gi`; leave the token. If fixed at all, fix the SIQO row via the + register pin or stage 02's kind parse — never a stage-04 slug special-case. +- **Stage-03 frontmatter (`type:` / `kind:` / `categories`)** in the 19 per-country `03_generate_wiki.py` keeps the stored token. Out of + scope. + +## Corrections vs. the previous draft + +Facts the review refuted or the owner overruled, folded in above: + +- **`AOC · AOC` contradiction** — the label table keyed on the stored token made decision 1's `PDO · AOC` unreachable and would have stamped + a scheme on Switzerland; replaced by the derived `eu_scheme` (F01). +- **69 IGP records** — "everything else DO", "PT / RO constant DOC" ignored `kind`; the axis now covers PGI-side terms, keyed on (country, + kind) (F02). +- **EDV** — spirits are a single-scheme GI (Reg. 2019/787), nationally AOC; "Eau-de-vie" was a wrong product word for whisky / rum / pommeau + (F03). +- **Austria** — 18 DACs (not "leave empty", not the stale 17); pinned by file number (F04). +- **"All eight" render sites** — missed the homepage `meta_description` → og / WebSite JSON-LD / `llms.txt`, `browse_meta_description` and + the IGP toggle strings (F05). +- **DOCG is 79, not 77** — the checklist hard-coded the blob-scan number B3 itself forbade; the elenco settles it (F06). **522 sidecars, not + 523** (`_index.json` was counted); 523 blob parents; 524 eAmbrosia; `ciro-classico` and `salemi` have no sidecar (F07). +- **Vermentino di Gallura** is recovered by the `DOCG` acronym, not the apostrophe variant; **sforzato** lacks `article_bodies["1"]`, while + `articles_present` is [2..10] (F08, F35). +- **MASAF elenco** joins on file number and is the source; the regex is an audit (F12). **MAPA listado** likewise; the scan-derived VP / VC + lists were wrong and VP / VC ship in v1 (F13). **Priorat → DOQ** is a rule, not a question; no Basque / Galician DO is top-tier (F17). +- **MVT / facet mechanics** — `common_props` is the MVT property path, so "MVT only if paint changes" was false; `facet_regions` is not a + facet; the fields must be in `STARTUP_AOCS_FIELDS` unconditionally (F09, F10). **Inheritance happens at build time**, not "at the + rendering layer" (F11). Facet counts parent-only (F21). +- **`locale/` committed** — hand-carrying msgstrs perpetuated a break of the reproducibility rule (F14); `--no-fuzzy-matching` and + module-form pybabel commands (F30). **Legend** — the maroon swatch is honest only as "Protected origin (PDO, AOC, DOC, DO…)" (F15). +- **GR / CZ / decoys** — ruled explicitly: scheme-only in v1 with empty pin sections; scheme abbreviations and lot-level grades excluded by + rule (F16, F22). FR empty-signe records take AOC from the derived `mvt_kind` (F19); CH `tier` / `grand_cru` acknowledged (F20). +- **Verification method** — the static grep *does* prove SSR title / meta; Playwright needs `scripts/serve.py`, not `http.server` (F18); the + comparator skips `.geojson` / `.pmtiles`, hence the direct kind-count diff (F32); tests named (F24); `RenderCtx` unchanged (F25). +- **Wikidata rejected** as a tier source; geoportal tier a secondary audit signal only (F26, F27). +- **LLM disclosure** per layer with the models actually recorded; `claude-haiku-4-5` was never used (F28); counts locale-formatted, + wine-only, both call sites (F29); README scoped (F31); deploy blast radius ~11.7k pages, "~3 min" uncited (F33); sequencing added (F34); + marc-d-alsace is a stage-02 / SIQO matter (F36). +- **Decisions** — all five former "ask the user" gates are settled (F23); rendering, UK / spirits policy, facet shape and tooltips taken by + the owner from the design panel. + + diff --git a/docs/plan-terroir-facts-quality.md b/docs/plan-terroir-facts-quality.md new file mode 100644 index 0000000..c20790c --- /dev/null +++ b/docs/plan-terroir-facts-quality.md @@ -0,0 +1,548 @@ +# Terroir-fact quality fixes — implementation handoff (2026-09-11) + +> **2026-09-12 — full-corpus review:** every EN bullet was graded and the +> high-severity findings verified; see +> [review-terroir-facts-2026-09-12.md](review-terroir-facts-2026-09-12.md) +> for the measured state after the fixes below and the next ranked work items +> (R1–R10). Its headline: ≈ 7 % of bullets still mislead, three quarters from +> extraction, and 83 % of the corpus predates the style block. + +Self-contained plan for a fresh session. Everything below was established by +reading a stratified sample of 1,000 English terroir-fact bullets (plus an +earlier 100) against their source quotes, and by a set of corpus-wide checks. +Evidence, sampler output and per-bullet tags are preserved under +`tmp/terroir-facts-review/` (gitignored, see "Evidence" at the end). + +Nothing in this plan changes the public-source rules in `CLAUDE.md`: every fix +is either a pipeline-code change, a prompt change, or a curator re-source +through the existing `manual_overrides.json` mechanism. + +## TL;DR + +| finding | scale | root cause | work item | +|---|---|---|---| +| Source-language common nouns and non-Latin script left in EN bullets | 16 % of all bullets; 37–51 % for GR/BG/HU/CZ/HR/LU; 12 % IT | the per-country 02e "Preserve … verbatim" prompt lists contain common nouns (soils, climates, harvest categories, scheme abbreviations) | **W1** | +| Alsace grand-cru pages carry Zotzenberg's geology, slope, Sylvaner rule and 1992 date | 51 records, ~250 of 430 bullets wrong | 02d slicer keys spans by section number over the one shared 339 KB cahier → last chapter wins | **W2a** | +| Pierrevert carries Saint-Pourçain's lien; L'Étoile and Grands-Echezeaux carry Bourgogne Passe-tout-grains' lien | 3 records, 15 bullets, all wrong | stage 01 fetched the wrong BO Agri PDF (L'Étoile and Grands-Echezeaux share one manifest filename) | **W2b** | +| Same fact repeated inside one record | 10 % of bullets (435 identical-quote repeats in 340 records, plus paraphrases) | the four sub-section calls re-extract the same sentence; no dedupe | **W3** | +| "wiki" provenance on facts that are grounded in the cahier | 626 of 982 wiki-labelled facts | quotes containing "[…]" fail the longest-contiguous-match coverage test | **W4** | +| Arrows, INAO colour codes (Pinot Noir N), unexpanded VT/SGN, missing terminal period, meta text ("confirmed by Wikipedia") | 4.3 % + 8 % punctuation | no style rules in the 02d prompt; codes are cahier vocabulary | **W5** | +| Bullets with no information ("uniqueness stems from soil, climate and varieties") | 3.4 % overall, 18 % GR | boilerplate sentence shared by many GR/CY specs | **W6** | +| Hedge dropped: "repose essentiellement sur pinot noir" → "exclusively Pinot Noir" | 14 FR bullets corpus-wide | no "keep hedges" rule | **W5** | +| Fact filed under the wrong sub-section | 1.7 % | sub-section is the slice the call came from, not a per-fact judgement | **W7** | +| Audit covers only fr/es/gb and none of the above checks | — | — | **W8** | + +What is **not** broken: numeric hallucination. Of 1,275 numbers that appear in +a bullet but not in its grounding quote, at most 5 are unsupported by the +source text once number formats are normalised (`13°5`, `22, 5`, `un +centinaio`, `anno mille`). Extraction fidelity is high; the defects are +mechanical and prompt-level. + +Baseline rates (n = 1,000, Wilson 95 % CI) to beat after the fixes: + +| category | rate | +|---|---| +| content error / over-claim / mistranslation | 3.4 % (2.4–4.7) — 1.6 % once W2 is fixed | +| untranslated common noun or non-Latin script | 16.1 % (14.0–18.5) | +| near-duplicate within record | 10.1 % (8.4–12.1) | +| provenance mislabelled as wiki | 5.4 % (4.2–7.0) | +| style (arrow / colour code / abbreviation / meta) | 4.3 % (3.2–5.7) | +| no information | 3.4 % (2.4–4.7) | +| misfiled sub-section | 1.7 % (1.1–2.7) | +| no terminal punctuation | 8.1 % (6.6–10.0) | +| fully clean | 58.1 % | + +## Landed 2026-09-11 — W3 + W4 as cache post-passes (no LLM call) + +- **W4** — `scripts/_lib/terroir_coverage.py` is now the one grounding + function (ellipsis-aware: a multi-span quote is graded span by span, and the + grade is the better of whole-quote and weakest-span, so nothing already + passing is demoted); all 21 `02d` scripts and `audit_terroir_facts.py` + import it. `scripts/recompute_terroir_provenance.py` re-graded the caches + through each country's own 02d source resolver: 467 caches rewritten, + provenance changed on **204 facts (174 wiki→both, 30 cahier→both)**, 481 + translation caches synced; 26 stale records skipped (22 Wikipedia-revision + drift, 3 CH cahier-context drift, `collioure` no longer a 02d target) — + those want a 02d re-run, not a post-pass. The "~600" estimate above counted + every wiki-labelled fact with a cahier quote; only 226 of those quotes carry + an ellipsis, and 174 ground on every span. The residual ~400 are light + paraphrases / typography (coverage 0.4–0.59, no ellipsis) — a normalisation + question for W5/W8, not a coverage-rule bug. Report: + `tmp/terroir-facts-review/provenance-recompute.json`. +- **W3** — `scripts/_lib/terroir_dedupe.py` + `scripts/dedupe_terroir_facts.py`. + The rule as landed, calibrated against the 1,000-bullet review: bullets + near-identical (`token_set_ratio ≥ 85`), or same / contained source quote + (≥ 30 chars) **and** bullet similarity ≥ 60; never when the bullets carry + different number sets; never when they lead with different sub-denomination + names (the roster stage 04's sibling filter uses — commit 24bd059d — so no + sub-zone page loses the bullet about it); transitive. The plan's bare + "identical quote" rule was rejected on evidence: 183 of the 552 + identical-quote pairs are two distinct facts from one source sentence + (alluvial soils / river water supply). Recall against the review's `dup` + tags: 78 %. Applied: **798 of 12,114 facts dropped (6.6 %)** — 698 + same-quote, 100 similar-bullet — in 560 records. FR is nearly untouched (3) + because FR 02d slices the lien into disjoint sub-sections, while every other + country re-reads the whole lien per sub-section call (IT 296, ES 136, BG 69, + GR 50). The 2,132 index-aligned 02e caches were pruned in step and their + `source_facts_sha` updated, so no translation was lost or redone; the sibling + guard never had to fire on the current corpus. Report: + `tmp/terroir-facts-review/dedupe.json`. +- **W2b** — no BO Agri lookup was needed: the eAmbrosia register serves each + of the three appellations' own cahier. Pinned `prefer_cahier: true` in the + checked-in `scripts/_lib/fr/register_overrides.json`; stage 01 gained the + prefer-register path (bypasses its has-usable-cahier guard for a pin), the + three were re-bound and re-extracted (Pierrevert 5.2 KB lien naming + Pierrevert, L'Étoile 9.4 KB / 14 mentions, Grands-Echezeaux 8.6 KB / 17 + mentions; Saint-Pourçain and Passe-tout-grains 0). Name guard → W8 audit. +- **W2a** — `scripts/_lib/terroir_chapters.py` finds the `« Alsace grand cru + X »` chapter headings; FR 02d's `_job_from_record` windows a shared cahier to + the record's own chapter (51 / 51 resolve, 5.9–8.9 KB each, four slices) and + skips a record with no own chapter instead of falling back. The 51 caches + are stale by sha and are in the scoped re-extraction list. +- **W3b** — every 02d script now calls `dedupe_facts` after its sub-section + loop (`n_deduped` in the cache) and carries the shared style block + (`scripts/_lib/terroir_prompts.STYLE_RULES`, spliced before the JSON-only + paragraph of all 21 prompts: full sentences, no arrows / labels / colour + codes, expand VT-SGN-TBA, never mention the document, keep hedges, do not + restate, skip tautologies). +- **W5** — `scripts/_lib/terroir_normalize.py` (colour codes stripped only + after a name the grape matcher's vocabulary recognises — "l'ugni blanc B" → + "l'ugni blanc", "orizzonte B" / "Weinbauzone B" untouched; VT / SGN + expanded; terminal period; and, for the four target locales only, residual + Greek / Cyrillic script Latinised — homoglyphs inside a Latin word mapped + ("Thermoheliоhydric"), whole non-Latin gloss tokens transliterated with + unidecode ("(ξερολιθιές)" → "(xerolithies)"), a Greek-letter chemical + prefix ("α-terpineol") and a predominantly non-Latin bullet left alone). + Applied at stage-04 render after the overlay and sibling filter, and as + `scripts/normalize_terroir_facts.py` over the caches: **1,261 source bullets in 365 records and 4,037 translated bullets + in 1,015 caches** normalised, translation caches re-keyed, nothing + re-translated. Arrows / meta text / dropped hedges go through the scoped + re-extraction (169 arrow records, 9 meta, 5 hedge). +- **W6** — `scripts/_lib/terroir_boilerplate.py` + `filter_terroir_boilerplate.py`. + Grouping by exact quote failed (the sentence embeds the appellation name + and the model quotes spans of varying length), so records are grouped per + (country, tautology pattern): a pattern quoted by ≥ 3 records of one + country is boilerplate for all of them; never a record's only fact, never a + bullet carrying a number. Dry run: 137 facts in 116 records (GR 84, FR 44 — + the Zotzenberg-contaminated Alsace ones, IT 6, DE 3). Applied after the + Alsace re-extraction. + +- **W8** — `scripts/audit_terroir_facts.py` rewritten over + `scripts/_lib/terroir_sources.py` (each country's own 02d resolver, so all + 21 countries are audited, not fr/es/gb). New checks, each counted in + `summary.checks` with its rows in the report: non-Latin script in the four + translation locales (S), colour code (S), missing terminal punctuation (S), + arrow / label prefix / meta text (R), intra-record duplicates via + `duplicate_reason` (S), cross-record shared quotes (R), the FR name guard + (S, whitelist `saone-et-loire`), the shared-cahier own-chapter checks (S), + wiki-provenance-with-cahier-quote (R). `--country`, `--strict` (exit 1 on + any strict finding), 31 unit tests, ~2 min on the corpus. Baseline before + the re-runs: non-Latin 3,167, arrows 819 (216 source), intra-record + duplicates 27, quote-outside-own-chapter 198 (the Alsace crus), colour codes + 0 and missing periods 0 (normaliser already applied). +- **W1** — `translation_rules()` / `translation_system_prompt()` in + `scripts/_lib/terroir_prompts.py`: two buckets (keep names verbatim / + translate the common nouns, with the plan's per-language examples), the + one-time-gloss allowance, the hard Latin-script rule for el/bg, the two W5 + 02e rules and keep-hedges; every 02e script now passes its proper-noun + roster (common nouns dropped per country — see the agent report in the + session) through it and appends the EN/NL glossary (previously FR-only). + `_EN_GLOSSARY` +8 calques. `--only SLUG` on all 21 scripts. + `scripts/detect_untranslated_terroir_facts.py` (non-Latin + the LEAK regex, + minus tokens that are correct target-language words: NL leem / zandleem / + mergel / lege, FR marne + the proper noun Marne) flagged **5,091 bullets = + 1,984 (slug, locale) pairs in 596 slugs** (gr 2,069, bg 1,000, it 974). +- **W1 follow-up (Boris, live check of `/nl/alsace-bergheim`: "Vosges" must be + "Vogezen")** — the keep-verbatim rule was too broad: "region names" kept + mountain ranges, rivers, seas and regions-as-places in the source form. + `translation_rules` now has a geography rule (established target exonym + where one exists — Vogezen / Rijn / Apennijnen / Tuscany / Piedmont / + Burgundy-as-region — while appellation names stay exactly as registered even + when they coincide with a region, and grape or institution names built on a + place stay verbatim). `scripts/_lib/exonyms.py` carries the table and the + detector flags residual source forms (`exonym:` reason; GI-homonym forms + only in place-like context): **418 bullets in 313 (slug, locale) pairs** + (Piemonte 51, Sardegna 44, Tejo 42, Sicilia 33, "Danube Plain" 27, Wien + 25, Toscana 21, Massif Central 20 …), re-translated on Anthropic batch in + two passes. Result: **418 → 120 bullets (75 pairs)**; the second pass + changed nothing, because the model's output is deterministic for an + unchanged prompt — the residue is its settled judgement: Wien / Mosel / + Tejo / Piemonte kept as region names in ES, NL and FR ("la región + vitivinícola de Wien"), named ranges kept as names (Appennino Dauno, Alpi + Apuane), Kärnten as the g.U. English "Mosel" and "Tejo" were removed from + the table — English wine writing uses them for the regions. Vogezen / Rijn + / Apennijnen / Tuscany / Piedmont / Sicily / Sardinia are now in place. + Clearing the last 120 needs either a stronger, example-heavy nudge for + those exact forms or a curated deterministic replacement for the non-GI + ones (Bayern → Bavaria, Appennino → Apennines) — left open; the audit + stays at 0 strict findings. +- **Re-runs (all `--batch --provider anthropic`)**: 02d `--refresh` on 262 + records (51 Alsace, 3 re-sourced, 169 arrow, 9 meta, 5 hedge, 25 stale; + 17 country batches, 1,016 requests, all concurrent) — every record + re-extracted, 0 arrows / 0 meta left in them, Rangen volcanic / + Gloeckelberg granitic / Kitterlé sandstone, no cru mentions Zotzenberg; + then dedupe (26 more) and boilerplate (92 facts: GR 83, IT 6, DE 3); + then 02e `--refresh --only` over the union of flagged and re-extracted + slugs (797 slugs, 19 country batches). Gotcha met on the way: a stale + `raw/.batch/02e-at.json` from May resumed an expired batch — delete + old sidecars before a `--batch` run. + +## Acceptance (2026-09-11, after all of the above) + +200-bullet re-sample (seed 2027, ≥ 5 per country, rest proportional; +`tmp/terroir-facts-review/sample200*.{json,txt}`), tagged with the same +rubric, against the 1,000-bullet baseline: + +| category | baseline (n = 1,000) | after (n = 200) | target | +|---|---|---|---| +| content error / over-claim / mistranslation | 3.4 % | 0.5 % (one inverted "<" sign carried from the source bullet) | < 1.5 % ✓ | +| untranslated common noun or non-Latin script | 16.1 % | 2.0 % (andezity/ryolity, climă temperat-continentală, kontinentális klíma, Kalk/Schiefer) | < 2 % ✓ | +| near-duplicate within record | 10.1 % | 0 % | < 2 % ✓ | +| provenance mislabelled as wiki | 5.4 % | 0 % | ≈ 0 ✓ | +| style (arrow / colour code / abbreviation / meta) | 4.3 % | 0 % | < 1 % ✓ | +| "Label:" prefix (not in the baseline definition; report-only in the audit) | — | 4.5 % | — | +| no information | 3.4 % | 3.5 % (residual: quotes under the 60-char W6 floor, phrasing variants outside the pattern gate) | — | +| no terminal punctuation | 8.1 % | 0 % | — | +| fully clean | 58.1 % | 94 % (89.5 % counting label prefixes) | — | + +Corpus-wide, the extended audit (`audit_terroir_facts.py --strict`, +`tmp/terroir-facts-review/audit-after.json`; the "before" is `audit-w8.json`): + +| check | before | after | +|---|---:|---:| +| non-Latin script in the four translation locales | 3,167 | 0 (45 after the batch re-run; 38 pairs re-translated → 21 residual glosses / homoglyphs, then Latinised deterministically by the normaliser) | +| colour codes / missing terminal period (source + translations) | 0 / 0 (normaliser) | 0 / 0 | +| arrows (source / translated) | 216 / 603 | 0 / 2 | +| meta text (source / translated) | 14 / 48 | 0 / 10 | +| intra-record duplicate pairs | 27 | 0 | +| FR name guard / no own chapter | 0 / 0 | 0 / 0 | +| cahier-grounded quote outside the own Alsace chapter | 198 | 0 (44 Wikipedia-grounded paraphrases were being counted; check narrowed) | +| eroded bullets / cahier drift / wiki drift | 198 / 57 / 22 | 1 (Collioure) / 0 / 0 | +| facts (source) | 11,316 | 11,260 | +| `--strict` exit | 1 (3,392 strict findings) | **0** (`audit-final.json`) | + +~~Still open: **W7** (sub-section per fact), the "Label:" lead prefixes, and +the residual no-information bullets whose quote is shorter than the W6 floor.~~ +**Closed 2026-09-13** by the follow-up programme in +[review-terroir-facts-2026-09-12.md](review-terroir-facts-2026-09-12.md) +("Implemented" / "Results"): W7 is the claim-support gate's `subsection` +verdict (`scripts/02d_verify_terroir_facts.py`), the label prefixes went +with the re-extraction of the whole pre-style-block corpus (1,365 records) +under the 120–220-character rule, and the tautological / no-information +bullets are the gate's `drop` verdict. + +## Recommended order + +1. **W2a** (slicer) and **W3/W4 post-pass scripts** — code only, no LLM cost, + fix the only outright-wrong content. +2. **W2b** curator re-source (three PDFs) — needs a human on the BO Agri UI. +3. **W5** normaliser (deterministic) + **W8** audit — gives the acceptance + checks before any re-run. +4. **W1** prompt split + **W5/W6** prompt rules — then the targeted LLM + re-runs (see "Re-run plan"). +5. **W7** last; lowest impact. + +Follow the memory rule "only fetch what's missing": never re-run 02d/02e +unfiltered across the corpus; every re-run below is scoped to the records a +detector script names. + +--- + +## W1 — Split the 02e "preserve verbatim" lists + +**Files.** `scripts/02e_translate_terroir_facts.py` (`build_system_prompt`, the +FR/ES base — line ~68) and all 20 country scripts +`scripts/<cc>/02e_translate_terroir_facts.py` (at, be, bg, ch, cy, cz, de, es, +gb, gr, hr, hu, it, lu, mt, nl, pt, ro, si, sk). Each has one long +`- Preserve <Language> proper nouns verbatim: …` line (bg:42, gr:42–43, hu:42, +hr:42, it:51, es:47, …). The target-locale glossary lives in +`scripts/_lib/translation_glossary.py` (`_EN_GLOSSARY`, `_NL_GLOSSARY`). + +**Change.** Split every list into two buckets and rewrite the rule text: + +- *Keep verbatim* (proper nouns and registered terms): appellation, region, + commune and vineyard-site **names**; grape **names** (transliterated when the + source script is not Latin); **named** formations and named winds (Marnes à + exogyra virgula, Flysch di Cormons, llicorella, albariza, tuffeau, + Muschelkalk, Rotliegend, Mistral, Bora, Meltemi); registered traditional + terms and Prädikat tiers (Aszú, Szamorodni, Vinsanto, Nychteri, tokajský + výber, Trockenbeerenauslese, Vendanges Tardives, Sélection de Grains Nobles). +- *Translate* (common nouns — this is what currently leaks): generic soil and + rock words (IT argille / calcare / marne / arenaria / scisti / calcareniti / + argilliti; HU lösz / mészkő / homokkő / agyagpala / barna erdőtalaj / + csernozjom / vulkáni talaj; BG льос / чернозем / канелена горска почва / + смолница; HR-SI vapnenac / crvenica / fliš / apnenec / ilovica / laporovec / + lapor; PT xisto; DE Lehm / Quarzit / Gneis / Granit / Urgestein / + Vulkangestein / Steillage / Lagenwein; NL leem / klei / zandleem / mergel; + LU-FR gypse / marnes keupériennes / calcaire conchylien); climate phrases + (умереноконтинентален климат, μεσογειακό κλίμα, ηπειρωτικό κλίμα, + kontinentalna klima, kontinentální podnebí, pannonisches Klima); generic + harvest and wine-law categories (kasna berba, desertno vino, predikatno vino, + pozna trgatev, ledeno vino, suhi jagodni izbor, pozdní sběr, slámové víno, + αφρώδεις οίνοι, λιαστοί οίνοι, pezsgő, gyöngyözőbor, vendemmia, fruttaia); + site words (lege, dűlő, viniční trať, podgorie, borvidék, vinorodni okoliš, + ribera, páramo, gromače, emparrado); scheme abbreviations (ΠΓΕ / ΠΟΠ / ЗНП / + ЗГУ / OEM / OFJ / CHOP / CHZO / ZOI → PGI / PDO). +- Allow a **one-time gloss** of a genuinely technical local term the way + Drenthe's bullet already does: "boulder clay (keileem)", "dry-stone walls + (prizidi)". Never the reverse ("Continental (ηπειρωτικό κλίμα)"). +- Add a **hard script rule** for GR/BG/CY: "The output must be entirely in + Latin script. Transliterate Greek and Cyrillic proper nouns using the + EU-official Latin form (Ξινόμαυρο → Xinomavro, Στара планина → Stara + Planina, Гъмза → Gamza)." The eAmbrosia `transcriptions[0]` field and the + grape lexicon's Latin slugs are the reference spellings. +- Extend `_EN_GLOSSARY` with the calques the sample caught: "minerality NOT + mineralité"; "wine-grape varieties NOT must varieties" (CZ moštové odrůdy); + "style / version NOT typology" (IT tipologia); "actual alcohol NOT developed + alcohol" (IT gradi svolti); "carbonate / calcareous soils NOT carbonated + soils" (FR carbonatés); "vineyard sites NOT lege / dűlők"; "para-barros is a + Portuguese soil class, never invent proto-barros"; Beaujolais "grillage / + pigeage / remontage" = submerged-cap / punch-down / pump-over. +- Optional but recommended: move the two-bucket rule into a shared helper + (`scripts/_lib/terroir_prompts.py`, `preserve_rules(source_lang)`) so 21 + scripts stop drifting. Move-only; keep each country's proper-noun roster. + +**Cache invalidation.** 02e caches key on `source_facts_sha`; a prompt change +does *not* trigger re-translation. Country 02e scripts have `--refresh` but no +`--only` (the 02d scripts have `--only`; FR 02d has `--slug`). Add `--only SLUG` +(repeatable) to every 02e script (mirror the 02d flag), then re-translate only +the slugs a detector names. Detector: scan `raw/translations/terroir-facts/ +<lang>/*.json` for non-Latin script (`[Ѐ-ӿͰ-Ͽ]`) and for the leak regex below; +check **all four target locales**, not only EN. Run with `--batch --provider +anthropic --refresh --only …` (the batch path loads `.env`). + +```python +LEAK = re.compile(r"(?i)\b(lösz|mészkő|homokkő|argille|argilliti|arenari[ae]|calcar[ei]|marn[ae]|scisti|" + r"podgori\w*|mineralité|kasna berba|desertn\w+ vin\w*|predikatn\w+|pozna trgatev|" + r"ledeno vino|okoliš|vapnen\w+|crvenic\w+|fliš|apnen\w+|xisto\w*|leem|zandleem|mergel|" + r"viničn\w+|dűlő\w*|lege\b|Steillage\w*|Lagenwein\w*|Urgestein|typology|must varieties)\b") +``` + +**Acceptance.** Non-Latin characters in Latin-target caches = 0 outside quoted +local words that carry a gloss; leak-regex hits < 20 corpus-wide; a fresh +50-bullet GR/BG/HU/CZ/HR sample reads as English. + +## W2 — Wrong chapter / wrong cahier + +### W2a Shared-cahier slicer (Alsace grand cru, 51 records) + +`scripts/02d_extract_terroir_facts.py`: `_spans_by_top` (line ~183) builds +`out[m.group(1)] = (start, end)` over every `1°- / 2°- / 3°-` anchor in the +lien, so with 51 cru chapters in one 339 KB `lien_au_terroir` the **last** +chapter (Zotzenberg, alphabetically last) wins for every cru. Only the +`produit` slice (generic to all crus) is currently correct. + +Fix in `slice_section_x`: when the lien contains more than one `1°` anchor, +first locate the record's own chapter — the heading `« Alsace grand cru +<Cru> »` (compare accent-folded; the record's `name` is the cru name) — and +restrict the lien to the window from that heading to the next `« Alsace grand +cru` heading before slicing. If no own chapter is found, log loudly and skip the +record rather than fall back. Add a unit test on a synthetic three-chapter lien. + +Then re-run for the 51 slugs only: `02d … --slug alsace-grand-cru-… --refresh` +(FR 02d uses `--slug`), which changes `source_facts_sha` and re-queues 02e for +those records automatically. Expect the new facts to describe each cru's own +geology (Gloeckelberg is granitic sand, Rangen volcanic, Kitterlé sandstone, +etc. — the chapter texts say so). + +The longer-term home for this is stage 02 (CLAUDE.md notes sub-sections of a +shared cahier are not parsed in v1); the 02d fix is sufficient for the terroir +layer. + +### W2b Wrong PDF bound to the record (3 parents) — curator + guard + +| id | slug | lien actually belongs to | evidence | +|---:|---|---|---| +| 290 | `pierrevert` | Saint-Pourçain | lien mentions Saint-Pourçain 6×, Pierrevert 0×; manifest PDF `e2a5794e…` | +| 187 | `l-etoile` | Bourgogne Passe-tout-grains (Mâcon / Jura regional text) | lien byte-identical to `bourgogne-passe-tout-grains`; manifest PDF `49acff22…`, shared with Grands-Echezeaux | +| 184 | `grands-echezeaux` | same as above | same PDF `49acff22…` | + +Route: a curator finds the right cahier on the BO Agri search UI and pins it in +`raw/inao/cahiers/manual_overrides.json` (mechanism in CLAUDE.md, "Manual +override mechanism"), then stage 01 → `02 --only` → `02d --slug` → 02e → 04. +The corresponding entry is in `CURATOR_TODO.md` (France). + +Guard (stage 02 or `audit_terroir_facts.py`): for every FR parent with a lien +≥ 800 chars, tokenise the appellation name (drop stop words such as côtes, de, +saint, grand, cru, village, coteaux) and require at least one token to occur in +the lien; otherwise emit a warning and, in `--strict`, fail. The check that +found these three is in `tmp/terroir-facts-review/` (see Evidence). + +## W3 — Dedupe within a record + +Two layers: + +- **Post-pass now** (no LLM cost): `scripts/dedupe_terroir_facts.py` rewriting + `raw/terroir-facts/*.json` in place. Drop a fact when (a) its normalised + `cahier_quote` or `wiki_quote` (≥ 30 chars) equals an earlier fact's, or + (b) `rapidfuzz.fuzz.token_set_ratio(bullet_a, bullet_b) ≥ 85`. Keep the + earlier fact unless the later one has provenance `both` or a longer quote. + Rewriting changes `source_facts_sha`, so 02e re-translates the ~340 touched + records automatically (`--batch`). +- **In 02d** for future runs: the same routine after the four sub-section calls + (FR: the loop around line ~436 that does `facts.append(classified)`; every + country 02d has the same `f["subsection"] = sub["key"]` pattern, e.g. + `scripts/it/02d_extract_terroir_facts.py:362`). Put it in + `scripts/_lib/terroir_dedupe.py` and call it from all 21 scripts. Add one + prompt line: "Do not restate a fact already covered by another sub-section." + +Sample cases to test against: `skalicky-rubin` facts 3/10, +`moselle-luxembourgeoise` facts 1/2/3/6/7/8/10 (42 km, 129–142 m and the two +cantons appear three times each), `vulkanland-steiermark` 5/7, `sobes` 3/7, +`hrvatsko-podunavlje` 3/6, `colline-lucchesi` 2/7, `halkidiki` 3/7. + +## W4 — Coverage check with ellipses + +`fuzzy_coverage` (FR 02d line ~171; the same function is duplicated in the +country 02d scripts and in `scripts/audit_terroir_facts.py`) uses one +longest-contiguous match. The model legitimately joins two spans with "[…]", +"[...]" or "…", which caps coverage well below 0.6 and flips provenance to +`wiki` (626 cases) or drops the fact. + +Fix: split the quote on `\[…\]|\[\.\.\.\]|…|\.\.\.`, compute coverage per span, +and take the minimum (spans shorter than ~15 chars ignored). Move the function +to `scripts/_lib/terroir_coverage.py` and import it everywhere. Then a post-pass +(`audit_terroir_facts.py --rewrite-provenance`, or a small script) recomputes +`cahier_coverage` / `wiki_coverage` / `provenance` on the existing caches +without an LLM call. Expect roughly 600 facts to move from `wiki` to `cahier` +or `both`; the panel attribution ("via Wikipedia · CC BY-SA 4.0") changes +accordingly at the next stage-04 build. + +## W5 — Style rules and a deterministic normaliser + +**Prompt (02d `EXTRACT_SYSTEM`, FR line ~132, and the country equivalents):** + +- Full sentences ending with a period; no "→"; no label prefixes ("Colour:", + "Climate:", "Soils:"). +- Grape names without INAO colour codes (N / B / G / Rs / Rg). +- Expand VT / SGN / TBA on first use. +- Never refer to "the document", "the cahier", "the disciplinare" or + "Wikipedia" inside the bullet (two bullets leaked "confirmed by Wikipedia" / + "according to the document"). +- Keep the source's hedges: essentiellement, principalement, parfois, souvent, + généralement, "repose sur". Do not strengthen to "exclusively", "100 %", + "only", "always". +- Skip statements that would be true of any appellation. + +**02e:** add "drop INAO colour-code suffixes" and "end every bullet with a +period" to the base rules. + +**Normaliser (`scripts/_lib/terroir_normalize.py`, applied at stage-04 render +and available as a cache post-pass):** append a terminal period when missing; +strip ` (N|B|G|Rs|Rg)` after a token that resolves in the grape lexicon; expand +`VT` / `SGN` in FR/EN/ES/NL. Leave arrows and meta text to a targeted re-run: +the ~150 arrow bullets and the "exclusively / 100 %" bullets are re-extracted +with `--slug … --refresh` once the prompt rules are in. + +## W6 — Boilerplate filter (conservative) + +Detect a `cahier_quote` (normalised, ≥ 60 chars) shared by ≥ 3 records of the +same country **and** matching a tautology pattern (`uniqueness … attributed +to`, `directly linked to the character`, `no uniform description`, `defined by +their typicity`, `favourable soil and climatic conditions`). Drop such facts +unless they are the record's only fact. Do **not** filter on shared quotes +alone — several cahiers are legitimately shared (Anjou family, the Calabrian +IGT text, the RO caiete, the Alsace `produit` slice). Add the pattern list to +the 02d prompt as things to skip. + +## W7 — Sub-section per fact (low priority) + +Either ask the model to return `"subsection"` per fact and use it when it +disagrees with the slice, or add a keyword reclassifier in the post-pass +(yield / density / pruning / vinification / AOC dates → facteurs_humains; soil +/ climate / relief → facteurs_naturels; colour / aroma / palate → produit). +Sample cases: `haspengouwse-wijn` 4/10 (yield cap under naturels), `la-tache` +1/9, `moselle-luxembourgeoise` 6/10, `tarquinia` 5/5 (climate under humains), +`castelli-romani` 3/5, `vin-santo-del-chianti` 8/8. + +**Deferred (Boris, 2026-09-11).** No re-extraction is needed: `subsection` is +a per-fact field in the source cache, copied by index into the four 02e +caches and not part of the translation hash. Preferred routes when picked +up: a keyword reclassifier over the EN rendering (one table, not 21) or a +classification-only LLM pass (one batch request per record, bullets +untouched) — either as a post-pass that rewrites `subsection` and syncs the +translation caches (`_lib/terroir_cache.py` pattern), then one stage-04 +rebuild. Move a fact only on a clear contradiction and never into +`interactions`. + +## W8 — Audit extension + +`scripts/audit_terroir_facts.py` only knows fr / es / gb +(`EXTRACTED_BY_COUNTRY`, line ~45). Extend it to every country by reusing each +country 02d's lien resolver (`_resolve_lien_and_source`, which already handles +the national-spec sidecars), and add checks, each with a count and `--strict` +failure: + +- non-Latin script in Latin-target translation caches; +- INAO colour codes, arrows, label prefixes, missing terminal period; +- intra-record identical quotes and bullet similarity ≥ 85; +- cross-record identical quotes (report only, with the whitelist of shared + cahiers); +- FR parent whose lien never names the appellation (W2b guard); +- shared-cahier records whose quotes are not found in their own chapter (W2a); +- provenance `wiki` with a non-empty `cahier_quote` after W4. + +## Re-run plan and cost + +1. Land W2a, W3 post-pass, W4 post-pass, W5 normaliser, W8 checks. Run the + audit to get the "before" counts. +2. Curator supplies the three PDFs (W2b); run stage 01 → 02 `--only` for them. +3. Land W1 + W5/W6 prompt rules (+ `--only` on the 02e scripts). +4. LLM re-runs, all `--batch --provider anthropic`, all scoped: + - 02d `--refresh`: the 51 Alsace slugs, the 3 re-sourced slugs, the ~150 + arrow records, the 14 "exclusively" records, the GR/AT boilerplate + records — roughly 250 records. + - 02e `--refresh --only …`: every record whose `source_facts_sha` changed + (automatic) plus the W1 detector list in all four target locales — on the + order of 2,000 record-locale pairs. Anthropic batch pricing makes this + cheap; do not use Ollama with `--workers > 1` (memory rule). +5. `scripts/04_build_maps.py`, then the audit in `--strict`. +6. Re-sample 200 bullets (seed 2027) with the sampler below, tag with the same + rubric, and compare against the baseline table. Target: untranslated < 2 %, + duplicates < 2 %, content errors < 1.5 %, style < 1 %, provenance mislabel + ≈ 0. + +## Evidence (`tmp/terroir-facts-review/`, gitignored) + +- `sample1000-review.md` — every flagged bullet of the 1,000 sample with EN, + source bullet, quotes and tags. Rubric: ERR (content error / over-claim), MT + (mistranslation), UNT (untranslated common noun), LOW, SUB, STY, DUP, META; + automatic tags NONLATIN, ARROW, CODE, NOPUNCT, PROVMIS, DUPQ, ALSACE. +- `sample1000_tagged.json` — the same sample as data (`issues` per bullet). +- `tags.txt` — the raw manual tags by sample index. +- `sample.json` — the earlier 100-bullet sample. +- `numcheck.log`, `numcheck_notfound.json` — the corpus-wide beyond-quote + number check. + +Sampler (reproduces `sample1000.json`; seed 2026, ≥ 10 per country, rest +proportional): + +```python +import json, glob, os, random, collections +rows = [] +for p in sorted(glob.glob("raw/terroir-facts/*.json")): + slug = os.path.basename(p)[:-5] + if slug.startswith("manifest"): continue + src = json.load(open(p)); c = src.get("country", "fr") + enp = f"raw/translations/terroir-facts/en/{slug}.json" + en = json.load(open(enp))["facts"] if os.path.exists(enp) else (src["facts"] if c in ("mt", "gb") else None) + if en is None: continue + for i, f in enumerate(src["facts"]): + if i < len(en): + rows.append(dict(slug=slug, country=c, i=i, en=en[i]["bullet"], src=f["bullet"], + cq=f.get("cahier_quote", ""), wq=f.get("wiki_quote", ""), + prov=f.get("provenance"), sub=f.get("subsection"))) +random.seed(2026) +byc = collections.defaultdict(list) +for r in rows: byc[r["country"]].append(r) +sample = [r for c, l in byc.items() for r in random.sample(l, min(10, len(l)))] +rest = [r for r in rows if r not in sample] +sample += random.sample(rest, 1000 - len(sample)); random.shuffle(sample) +``` + +Corpus-wide detectors used in the review (rewrite as audit checks in W8): +non-Latin `re.compile(r"[Ѐ-ӿͰ-Ͽ]")` over EN bullets → 791 hits (GR 503 / 1,170, +BG 264 / 544); INAO code `[a-zé]\s(N|B|G|Rs|Rg)(?=[\s,;:.)]|$)` → 160 FR bullets; +arrows → 152; identical `cahier_quote` within a record → 435; `provenance == +"wiki" and cahier_quote` → 626; lien-never-names-appellation → 4 FR parents +(the 3 above plus `saone-et-loire`, which is fine). diff --git a/docs/review-terroir-facts-2026-09-12.md b/docs/review-terroir-facts-2026-09-12.md new file mode 100644 index 0000000..f0d913c --- /dev/null +++ b/docs/review-terroir-facts-2026-09-12.md @@ -0,0 +1,485 @@ +# Terroir-fact quality review — full EN corpus (2026-09-12) + +Follow-up to [plan-terroir-facts-quality.md](plan-terroir-facts-quality.md) +(W1–W6, W8 landed 2026-09-11; W7 open). That plan was built on a 1,000-bullet +sample; this review graded **every** English terroir fact — 11,260 bullets in +1,634 records across 21 countries — against the source-language bullet, the +grounding quotes and the exact regulator text stage 02d graded against +(`_lib/terroir_sources.py`), then adversarially verified every high-severity +finding plus a stratified sample of the rest. Evidence is under +`tmp/terroir-facts-review/full-review-2026-09-12/` (gitignored; see +"Evidence" at the end). + +## TL;DR + +| finding | scale | root cause | fix | +|---|---|---|---| +| Bullets that would **misinform a reader** (wrong entity, unsupported causal link, invented detail, dropped hedge, wrong number/direction) | ≈ 780 bullets, **6.9 %** (95 % range 5.6–8.5 %); 384 verified one by one, in 346 records | 3 of 4 originate in the **source-language extraction** (02d), not in translation; the pre-2026-09-11 prompt produced them at twice the rate of the current one | R1 broad 02d re-run + R2 claim-support gate | +| `interactions` sub-section asserts terroir → wine causation the source never makes | 73 of the 106 verified "unsupported causal link" bullets; 29 % of all interactions bullets flagged as duplicates, 33 % as content errors | the 4th sub-section call re-reads the whole lien and is asked for a causal link; it manufactures one when the source only lists factors | R2 (gate) + R3 (redesign the call) | +| Records still bound to the **wrong source text** | Bourgogne Passe-tout-grains (Beaujolais cahier); ≥ 15 IT records whose MASAF sidecar carries a sottozona annex (Abruzzo, Montepulciano d'Abruzzo, Trentino, Romagna, Terre di Cosenza, Colli Tortonesi, Riviera Ligure di Ponente, Friuli Colli Orientali, …); 4 GR specs with sections copied from another PGI; 2 FR records sharing one PDF; 2 wrong Wikipedia bindings | name guard defeated by common tokens; MASAF `extract_articles` keeps the *last* occurrence of each article number; ΥΠΑΑΤ files literally paste sections across PGIs | R4 (data fixes + guards) | +| Telegraphic "Label:" bullets | 1,291 EN bullets (11 %) in 730 records — **729 of them extracted before the style block** | never re-extracted; every country prompt still caps bullets at 140 characters, which forces the compression that produces both labels and errors ("sin. Tanaro" → "syn. Tanaro") | R1 (re-run) + R5 (drop the char cap) | +| Translation-stage defects | 225 of the 590 non-refuted verified findings involve 02e: calques (Lehm → clay, generoso → generous, tirage → disgorgement, Burgundian *climat* → climate), a numeric slip (250–290 m → 250–490 m), name back-formation ("Caiata" from *caiatino*), hedges upgraded ("weakly" → "moderately") | no back-check of EN against SRC; glossary and exonym gaps (Bayern, Rodopi/Rhodopes) | R6 | +| Untranslated / source-form words | ≈ 510 bullets (4.5 %, precision-corrected) | residue of W1; German compounds and BG/HU geography dominate | R6 | +| Misfiled sub-section (W7) | ≈ 610 bullets (5.4 %); reviewers supplied `suggested_subsection` for 690 | slice ≠ per-fact judgement | R7 | +| Intra-record restatement after the W3 dedupe | ≈ 540 bullets (4.8 %); 45 % of `interactions` bullets in two-lens batches | dedupe is lexical; the restatements are paraphrases with a causal wrapper | R3 | + +What is **not** broken: numbers. Of 3,875 bullets carrying a number, 42 had a +number found in neither quote nor source after normalisation; all 42 were +refuted by the verifier (decade / century idioms, "ultra-cinquantennale", +"anno mille"). Numeric hallucination stays at ≈ 0. Translation-only numeric +errors exist but are rare (La Grande Rue 490 m, Sithonia 1960s). + +## Method + +1. **Sheets.** One sheet per ~10 records (192 sheets, later re-cut to 151 + single-Read sheets): per fact EN / SRC / cahier quote / wiki quote / + sub-section / provenance, plus the record's full source text and the + per-sub-section Wikipedia hints. Rubric in `REVIEW_GUIDE.md` (14 tags: + ERR HALL MT UNT GRAM STY LOW LEN DUP SUB SIB NAME PROV MISSING). +2. **Review.** 41 sheets by two independent lenses (fidelity, language); + the remaining 151 by one combined lens after the quota made two lenses + unaffordable. Hard-tag rates are stable across the two methods (31 % vs + 33 % of facts); soft-tag rates are not (GRAM 32 % vs 8 %), so every rate + below is either verified or precision-corrected. +3. **Verification.** Every one of the 440 high-severity findings, a + stratified sample of 215 of the 3,238 other hard-tag findings, and the 42 + number residuals were re-judged by an adversarial verifier holding the + record's full source (573 record-level agents, 703 verdicts). A second + judge re-graded 278 soft-tag findings for precision. +4. **Classification.** Each of the 384 verified-misleading bullets was + assigned a failure mode. + +Cost note for the next run: a default subagent carries a 176 k-token base +context (the 300 KB project `CLAUDE.md`); the Explore agent type carries +14 k and produced equivalent findings (28/31 and 30/32 overlap on the same +sheets). The whole verified review ran on ≈ 40 M subagent tokens once +switched. + +## Headline numbers + +| measure | value | basis | +|---|---|---| +| facts reviewed | 11,260 in 1,634 records (100 %) | — | +| facts with any reviewer flag | 6,805 (60 %) | raw; method-sensitive | +| facts with a hard tag (ERR / HALL / MT / SIB / NAME / PROV) | 3,678 (33 %) | raw | +| high-severity findings | 440 | 416 confirmed or partially confirmed on verification (95 %), **355 reader-misleading (81 %)** | +| other hard-tag findings | 3,238 | sample of 215: 78 % confirmed or partial, **13 % reader-misleading** (9–18 %) | +| **misleading bullets, corpus-wide** | **≈ 777 (6.9 %)**, range 634–959 | 355 + 13 % × 3,238 | +| bullets with a verified content-level defect of any weight | ≈ 2,950 (26 %) | 95 % × 440 + 78 % × 3,238; most are "partially confirmed": a dropped qualifier, a narrowed attribution | +| stage of origin (590 non-refuted verdicts) | extraction 365 · translation 127 · both 98 | 79 % involve 02d | +| UNT (precision 94 %) | ≈ 510 (4.5 %) | 543 raw | +| STY (82 %) — "Label:" prefixes, mixed units, stray quotes | ≈ 1,360 (12 %) | 1,646 raw; 1,291 are label prefixes | +| SUB (88 %) | ≈ 610 (5.4 %) | 690 raw | +| DUP (67 %) | ≈ 540 (4.8 %) | 815 raw | +| LEN (40 %) | ≈ 410 (3.6 %) | 1,028 raw | +| GRAM (26 %) | ≈ 390 (3.5 %) | 1,484 raw; the language lens tagged label fragments as GRAM | +| LOW (77 %) | ≈ 300 (2.7 %) | 393 raw | +| records where the source prominently describes something no bullet captures (MISSING) | 1,089 of 1,634 | reviewer notes; not verified | + +The 2026-09-11 acceptance sample reported 0.5 % content errors and 94 % +clean. It was right about the categories it measured (arrows, codes, +punctuation, non-Latin script, duplicates are indeed ≈ 0) and wrong about +content: that sample was graded bullet-by-bullet against the quotes, while +the misleading bullets found here mostly need the *whole source* to be seen +— the quote is genuine, the claim built on it is not. + +## Failure modes (384 verified-misleading bullets) + +| mode | n | share | stage | typical case | +|---|---:|---:|---|---| +| unsupported causal link | 106 | 28 % | extraction 94 | *Taurasi*: "volcanic pyroclastic material … imparts great minerality, structure and an austere character" — the source only records the material's presence | +| mistranslation | 55 | 14 % | translation 40 · both 15 | *Rueda*: "Generous wines" for *vinos generosos* (fortified); *Crémant d'Alsace*: "from the date of disgorgement" for *à compter du tirage* | +| wrong entity | 55 | 14 % | extraction 44 | *Campo de Cartagena*: the Mar Menor "acts as a climatic buffer" — the source credits the open Mediterranean; *Tacoronte-Acentejo*: "volcanic origin distinct from the rest of the Canaries" — the source says the origin is common, the evolution differs | +| invented detail | 54 | 14 % | extraction 37 | *Grignolino d'Asti*: "pale colour typical of thin-skinned red varieties" — nowhere in the sources | +| wrong number / unit | 21 | 5 % | mixed | *La Grande Rue* 250–**490** m (source 290; translation-only); *Barolo* "the Langhe began forming almost 70 million years ago" (the Cenozoic did) | +| wrong direction / sign | 21 | 5 % | mixed | *Volnay*: "shelter afforded by the Morvan to the east" (the Côte lies east of the Morvan); *Nagy-Somló*: varieties "withdrawn" that the source says were brought back | +| wrong grape / name | 17 | 4 % | translation 15 | *Balaton*: Hungarian *Pintes* rendered "Pinot"; *Terre del Volturno*: "the Caiata area" back-formed from *caiatino* (Caiazzo) | +| hedge dropped | 17 | 4 % | extraction 14 | *Canelli*: "pruning exclusively to Guyot" (source: *tipicamente*); *Boliarovo*: "moderately expressed" for *слабо изразено* (weakly) | +| sibling / sub-zone confusion | 11 | 3 % | extraction | *Pouilly-sur-Loire*: the soil → style links belong to Pouilly-Fumé; *Salice Salentino*: the Basso Salento contrast presented as the DOC's soils | +| narrowed attribution | 10 | 3 % | extraction | *Vratsa*: soils "contribute to the softness and long finish" — the source credits climate + relief + soils + human factor en bloc | +| wrong source text | 5 | 1 % | extraction | see below; 5 bullets in the verified set, many more in the affected records | +| added qualifier | 5 | 1 % | extraction | *Dolinata na Struma*: "traditional local varieties" on a bare enumeration | +| other | 7 | 2 % | | | + +By sub-section: `interactions` carries 73 of the 106 unsupported causal +links although it holds only 10 % of the bullets; `facteurs_naturels` +carries the wrong-entity and wrong-direction cases; `facteurs_humains` the +hedge drops and wrong names. By country the profile is uniform except that +Italy contributes 63 of the 106 causal-link cases (the IT prompt re-reads +the whole disciplinare for every sub-section) and Spain / Hungary lean to +mistranslation. + +### Prompt vintage matters more than country + +| records extracted | records | facts | high-severity | verified misleading | label prefix | DUP flag | +|---|---:|---:|---:|---:|---:|---:| +| before the 2026-09-11 style block | 1,365 | 9,090 | 4.4 % | 3.5 % | 14.2 % | 8.1 % | +| after (the scoped W-re-runs) | 269 | 2,170 | 2.0 % | 1.7 % | 0.0 % | 3.6 % | + +The current prompt halves the misleading rate and removes the label +fragments. 83 % of the corpus has never been extracted under it. + +## Systematic source defects (fix the data, not the prompt) + +- **Bourgogne Passe-tout-grains** (`id_appellation` 144) — `raw/inao/cahier-extracted/bourgogne-passe-tout-grains.json` + is bound to PDF `49acff22…` (BO Agri `e89b7ce3…`); the extracted lien is the + **AOC Beaujolais** cahier (23 mentions of Beaujolais, 0 of Passe-tout-grains, + 0 of Bourgogne). All 6 terroir facts describe the Beaujolais; the map blob + carries `styles: ["primeur"]` and no grapes for it. W2b re-bound L'Étoile + and Grands-Echezeaux off this same PDF but left the record it was + attributed to. The `name_guard` passes because "tout" and "grains" (≥ 4 + letters, not stop words) occur in any French cahier. +- **Hautes-Alpes / Haute-Vienne** — both manifest entries point at BO Agri + `22caf075…` and PDF `1106c71b…`; the two extracted liens differ and each + matches its record, so one record's `boagri_url` (and the public PDF link) + is misattributed. +- **IT MASAF sidecars with sottozona annexes** — `scripts/_lib/it/masaf.py` + `extract_articles` keeps the **last** occurrence of each article number + (meant for TOC + body). In a consolidated disciplinare whose sottozona + annexes restart at *Art. 1*, every article — summary, grapes, geo area, + Art. 9 lien — comes from the last annex. 24 PDFs repeat *Art. 1* / *Art. + 9* (`masaf_multi_annex.json`); confirmed wrong on the map or in the + terroir source: `montepulciano-d-abruzzo` (San Martino sulla Marrucina + annex: 1 grape, single-commune area, annex lien), `abruzzo` (Colline + Teramane annex), `trentino` (Valle di Cembra: 4 grapes), `romagna` + (Verucchio summary), `terre-di-cosenza` (Verbicaro), `colli-tortonesi` + (Terre di Libarna: 1 grape), `riviera-ligure-di-ponente` (Taggia), + `friuli-colli-orientali` (Savorgnano), plus `colli-bolognesi`, + `colli-orientali-del-friuli-picolit`, `langhe`, `portofino`, + `sambuca-di-sicilia`, `veneto`, `bardolino`, `riviera-del-garda-classico` + to check. Stubs take the whole sidecar; non-stubs are gap-filled from it. + Fix: per article number keep the first occurrence with a non-trivial + body (or the longest), and treat the annexes as the sottozona records' + own text (the Alsace W2a pattern). +- **GR ΥΠΑΑΤ specs pasted across PGIs** — `fthiotida` (geographic section + names ΠΓΕ Παρνασσός and its municipalities), `peloponnisos` (describes + Αχαΐα / Πλαγιές Αιγιαλείας / Αρκαδία), `retsina-evias` (human-factors + section is Retsina Attikis), `ipiros` (sparkling section from Ioannina). + A source defect at the regulator, but the pipeline should refuse to + ground on it: a foreign-appellation guard (see R4). +- **Wrong Wikipedia binding** — `tirol` → de.wikipedia *Toro + (Weinbaugebiet)* (the Spanish DO); `montecastelli` → the village article + (its one wiki-only bullet describes the village hill). Pin both `missing` + in `raw/wikipedia/aoc_overrides.json`. +- **Šobes** — fact #1 (Mikulov bioregion, Pavlov Hills limestone) is the + Mikulovská podoblast 50 km away; Šobes sits on Bohemian Massif + crystalline rock — a consequence of grounding every CZ record on the + region-wide CHZO text with the podoblast article as hint. +- **Montana (BG)** — the IAVV spec has the Danube "to the south" and Stara + Planina "to the north"; the bullet silently corrects it. Curator note. +- **Collioure** — no source text resolves today and the record carries one + bullet (a 2009 planting grandfather clause). Already in CURATOR_TODO. + +## Country profile + +Hard-flag share is 24–37 % everywhere (SK 11 %); the verified column is +the count of high-severity findings confirmed as misleading, all verified. + +| cc | records | facts | high-severity | verified misleading | UNT | label prefix | wiki-only provenance | +|---|---:|---:|---:|---:|---:|---:|---:| +| fr | 460 | 3,238 | 95 (2.9 %) | 76 | 107 | 385 | 411 | +| it | 519 | 3,045 | 181 (5.9 %) | 139 | 159 | 337 | 296 | +| es | 142 | 1,070 | 27 (2.5 %) | 22 | 58 | 107 | 87 | +| gr | 147 | 1,032 | 35 (3.4 %) | 28 | 9 | 73 | 2 | +| bg | 54 | 456 | 26 (5.7 %) | 22 | 30 | 72 | 0 | +| hu | 41 | 395 | 13 (3.3 %) | 13 | 26 | 47 | 5 | +| ro | 46 | 366 | 12 (3.3 %) | 12 | 20 | 51 | 0 | +| pt | 44 | 345 | 7 (2.0 %) | 6 | 11 | 34 | 12 | +| de | 39 | 288 | 8 (2.8 %) | 6 | 33 | 39 | 9 | +| hr | 18 | 196 | 7 (3.6 %) | 6 | 12 | 27 | 4 | +| at | 29 | 194 | 6 (3.1 %) | 5 | 21 | 21 | 6 | +| nl | 21 | 161 | 2 (1.2 %) | 1 | 28 | 38 | 5 | +| si | 17 | 123 | 3 (2.4 %) | 3 | 2 | 15 | 5 | +| sk | 10 | 85 | 3 (3.5 %) | 3 | 12 | 15 | 1 | +| cz | 13 | 72 | 3 (4.2 %) | 3 | 4 | 22 | 12 | +| cy | 11 | 71 | 3 (4.2 %) | 2 | 1 | 5 | 0 | +| be | 10 | 56 | 1 | 1 | 7 | 2 | 2 | +| gb | 6 | 46 | 5 | 4 | 0 | 1 | 2 | +| ch / lu / mt | 7 | 21 | 3 | 3 | 3 | 0 | 15 | + +Italy's rate is the corpus's worst for a structural reason (whole-document +re-reads per sub-section, 140-character cap, annex mis-slicing); Bulgaria's +comes from the telegraphic IAVV extractions (five of eight bullets in +Vratsa, Lozitsa and Pazardzhik are verbless fragments) and the specs' habit +of crediting all factors en bloc, which the extraction then narrows. + +## Recommended improvements, ranked by verified impact + +**R1 — Re-extract the pre-style-block corpus (02d `--refresh`, batch).** +1,365 records (83 %) predate the current prompt; the post-block cohort has +half the misleading rate, no label fragments and half the duplicates. +Scope first the union of: the 730 label-prefix records, the 346 records +with a verified misleading bullet (`rerun-slugs*.txt`), and the wrong-source +records after R4; then the rest per country as batch budget allows. A full +02d re-run of the corpus is one Anthropic batch per country (the plan's +262-record re-run cost a few dollars); 02e re-translates automatically on +`source_facts_sha` change. This is the single largest lever and needs no +new code. + +**R2 — Add a claim-support gate after extraction (new `02d-verify` step, +batch, one request per bullet).** The coverage test only checks that the +*quote* exists in the source; it never checks that the *bullet's claim* is +what the quote says. Every failure mode above except mistranslation passes +the coverage test. A verifier prompt of the shape used here ("does the +source support each assertion of this bullet; which words go beyond it; +rewrite or drop") confirmed 95 % of the reviewer's high-severity flags and +refuted the number residuals, so it is precise enough to gate on: drop a +bullet whose main claim is unsupported, rewrite one whose qualifier or +causal wrapper is. Run it in 02d after dedupe and before the cache write; +record `support: {verdict, note}` per fact so the audit can count it. + +**R3 — Redesign the `interactions` call.** It produces 10 % of the bullets +and 69 % of the unsupported causal links, and 45 % of its output restates +the naturels / produit bullets. Options, in order of preference: (a) fold +it into the other three calls — ask each sub-section call to mark a bullet +`causal: true` only when the source sentence itself contains a causal +connective, and cap `interactions` at what those yield; (b) keep the call +but require the quote to contain the causal verb and reject otherwise (the +R2 gate does this); (c) for non-FR countries, slice the lien into +sub-section windows as FR does instead of re-reading the whole text four +times (FR has 0.8 % DUP flags, IT 8.5 %). + +**R4 — Fix the source bindings and harden the guards.** +- Re-bind `bourgogne-passe-tout-grains` (register `prefer_cahier` pin or + BO Agri lookup), resolve the Hautes-Alpes / Haute-Vienne shared PDF, pin + `tirol` and `montecastelli` Wikipedia to `missing`. +- `name_guard`: require the *whole* folded name minus stop words (or its + longest token ≥ 6 letters) rather than any ≥ 4-letter token; extend it to + every country through `terroir_sources.py`, and add a **foreign-name + guard**: a source text that names another appellation of the same + country ≥ 3 times and its own 0 times is refused (catches the GR pastes, + the MASAF annexes, Šobes). +- `masaf.extract_articles`: choose the occurrence with the longest body per + article number, then emit the annexes as per-sottozona chapters (the + Alsace `terroir_chapters.py` pattern) so the synthesized sottozona records + get their own text instead of inheriting the parent's. Re-run 02f for the + 24 PDFs in `masaf_multi_annex.json`, then 02d for the affected slugs. + +**R5 — Remove the 140-character cap and the telegraphic register from the +02d prompts.** All 21 prompts still say "≤ 140 caractères chacune" (or its +translation); +the audit already ignores its own soft cap (3,437 bullets over it). The cap +is what produced "Roero (sin. Tanaro)", "Klima: Ø 9,8 °C", the "Label:" +fragments and most hedge drops. Ask for one full sentence of up to ~220 +characters; let the normaliser handle length outliers. + +**R6 — Translation back-check and glossary.** 225 verified defects involve +02e. Add: (a) a cheap batch back-check that compares each EN bullet with +its SRC bullet and flags changed numbers, dropped or upgraded hedges, and +terms in a watch-list (climat, tirage, generoso, Lehm, Spritzigkeit, +capa, tipologia, Urgestein); (b) glossary entries for those terms; (c) +exonyms `Bayern → Bavaria`, `Родопи → Rhodopes`, and a policy for `Stara +Planina` (Balkan Mountains); (d) `Немски ризлинг → Riesling` (rendered +"Welschriesling" in ruse / shumen while the lexicon folds it to Riesling). + +**R7 — W7 sub-section reclassifier.** 690 findings carry a +`suggested_subsection` (precision 88 %): use them as the validation set for +the keyword reclassifier the plan sketched, or run a classification-only +batch pass. Move only on a clear contradiction; the `interactions` label +should be *earned* by a causal sentence (R3), not assigned by slice. + +**R8 — Semantic dedupe.** The lexical W3 rule leaves ≈ 540 restatements, +almost all "naturels fact + produit fact rewritten as a causal sentence". +After R3 most disappear; for the rest, an embedding or LLM pairwise pass +over each record's ≤ 15 bullets is cheap. + +**R9 — Audit additions (`audit_terroir_facts.py`).** Make `label_prefix` +strict once R1 lands; add `multi_sentence` (100 bullets), `en_equals_src` +(2), `cross_record_identical_en` (26 bullets in 55 records — the Alsace +produit slice and the GR retsina cluster), the foreign-name guard, the +MASAF repeated-article detector, and a Wikipedia-binding sanity check +(article title must share a token with the record name). Keep the LLM +review harness (`workflows/` in the evidence directory) as a repeatable +`audit_terroir_facts_llm.py --sample N` for regression: with Explore-type +agents a 60-fact sheet costs ≈ 60 k tokens. + +**R10 — Coverage (lower priority).** Reviewers noted a prominent +uncaptured element in 1,089 records (lakes as climate regulators, named +winds, headline hectares, defining practices). The 5 / 2 / 2 / 2 caps are +tight for long liens; consider scaling the naturels cap with lien length, +and asking for named entities first. + +## Implemented after the review (2026-09-13): per-record feedback layer + +The review's verified findings now live as constraints the next +extraction reads: `raw/terroir-facts-feedback/<slug>.json` (1,238 records: +384 do-not-claim entries in 346 records, 1,133 capture hints, 96 +cautions), built by `scripts/build_terroir_feedback.py` from the evidence +directory and read by every 02d script through +`_lib/terroir_feedback.with_feedback` (live call and `--emit-todo`). +The prompt block lists each misleading claim as "do not assert … unless +the source states it explicitly" with the verifier's source-quoting +reason, the uncaptured elements as "if the text describes it, capture", +and the record cautions; it never carries the corrected English text, and +only `extraction` / `both` entries reach 02d. `audit_terroir_facts.py` +gained `feedback_recurrence` (report-only): known-bad claims still +matching a current bullet — the regression measure for R1. Baseline +before any re-run: **311 known-bad claims in 285 records** (the 384 +misleading bullets minus the 73 translation-only ones, all still present +by construction); the number to watch is that count after the re-run. + +## Implemented (2026-09-13): R1–R10 landed — see "Results" below + +Everything runs as one rollback unit (`scripts/rerun_terroir_facts.py`, +run id `r1-2026-09-13`; undo with +`scripts/rollback_terroir_facts.py --run r1-2026-09-13`): + +| rec. | what landed | where | +|---|---|---| +| backup | every 02d / gate / 02e / back-check / post-pass write snapshots the slug's source + 4 translation caches per run; per-slug entry files; rollback restores overwritten files and deletes created ones | `_lib/terroir_backup.py`, `_lib/terroir_cache.write_*_cache`, `rollback_terroir_facts.py`, wiring lint `tests/test_terroir_backup.py` | +| R1 | 918 records re-extracted (the union: 730 label-prefix + 346 verified-misleading + 23 wrong-source, all countries) under the new prompt, with the feedback sidecars in the prompt; the remaining 486 pre-block records are `tmp/terroir-facts-review/r1-scope-rest.json` | `rerun_terroir_facts.py --scope …` | +| R2 | claim-support gate, one request per record against the exact 02d source + hints + feedback; supported / rewrite (guarded) / drop; `support` per fact, `gate` block, feedback `history` | `02d_verify_terroir_facts.py`, `_lib/terroir_gate.py` | +| R3 | `interactions` earned: the shared prompt block admits a causal bullet only on an explicit connective in the source sentence; the gate rewrites or drops the rest | `_lib/terroir_prompts.STYLE_RULES`, gate prompt | +| R4 | Bourgogne Passe-tout-grains re-bound to its register cahier (`prefer_cahier` pin); Hautes-Alpes / Haute-Vienne verified as a genuine 20-cahier bundle (no change); `tirol` / `montecastelli` Wikipedia pinned `missing` (02b gained `--only`); MASAF `extract_article_runs` (annexes as chapters, 20 parents corrected, 77 sottozone detected vs 38); name guard on whole name / long-token stem, all countries (strict FR, report elsewhere); foreign-name guard; Wikipedia-binding check | `register_overrides.json`, `aoc_overrides.json`, `_lib/it/masaf.py`, `audit_terroir_facts.py` | +| R5 | the 140-character cap replaced in all 21 prompts by "one full sentence of ~120–220 characters, never a telegraphic fragment"; audit soft cap 240 | 21 × `02d_extract_terroir_facts.py` | +| R6 | translation back-check per (record, locale) with fixes under guards; glossary entries (climat, tirage, generoso, Lehm, Spritzigkeit, capa, Urgestein, Pintes, Немски ризлинг, caiatino); exonyms (Rodopi → Rhodopes, Bayern → Bavaria, Carpathians, Danube forms, Peloponnese, Crete, …) | `02e_verify_terroir_facts.py`, `_lib/terroir_backcheck.py`, `_lib/translation_glossary.py`, `_lib/exonyms.py` | +| R7 | the gate returns `subsection` on a clear contradiction only; `support.moved_from` records the move; translations re-synced by index | gate | +| R8 | the gate's `restates` + the lexical dedupe after rewrites | gate | +| R9 | audit: `multi_sentence`, `en_equals_src`, `cross_record_identical_en`, `foreign_name`, `wiki_binding`, `masaf_sidecar_stale`, `gate_pending`, `rewrite_rejected`, `name_guard_other`, `--strict-labels`; the LLM audit as a script with an independent grader (`claude-opus-5`) and a paired before / after | `audit_terroir_facts.py`, `audit_terroir_facts_llm.py` | +| R10 | named entities and figures first; up to two extra bullets on a long text | `STYLE_RULES` | + +## Results (2026-09-13, runs `r1-2026-09-13` + `r1b-2026-09-13`) + +Everything below is undoable: `scripts/rollback_terroir_facts.py --run +r1c-2026-09-13`, then `--run r1b-2026-09-13`, then `--run r1-2026-09-13` +(newest first; 2,153 + 28 snapshot entries). + +**What ran.** 1,403 of 1,634 records re-extracted under the new prompt +(the 918-record union first, then the 486 remaining pre-block records; +the 231 untouched records are the post-block cohort of 2026-09-11 that +the review already measured as clean), every record through the gate, +every changed record re-translated, every translation cache +back-checked, then the deterministic post-passes (normalise, dedupe) and +the strict audit. + +| step | scale | outcome | +|---|---:|---| +| 02d re-extraction | 1,403 records, 21 countries | 0 records without facts; median bullet 195 chars (was ≤ 140); **0 label-prefix bullets** (was 1,291) | +| gate, pass 1 (corpus) | 1,633 records, 11,202 bullets | 27 % rewritten, 1.5 % dropped, 498 moved sub-section, 283 rewrites refused by the guards | +| gate, re-gate of refused | 251 records | cap raised 320 → 420 chars; 52 still refused | +| gate, pass 2 (rest scope) | 492 records, 2,978 bullets | 28 % rewritten, 1.6 % dropped | +| 02e re-translation | ≈ 6,900 (record, locale) caches | 5,891 / 5,891 aligned at the end | +| back-check | 7,731 caches, 50,555 translated bullets | **4,483 fixed (8.9 %)**, 802 fixes refused by the guards | +| strict audit | 1,638 caches, 11,140 bullets, 39,997 translated | **0 strict findings**, with `label_prefix` promoted to strict | +| `feedback_recurrence` | 311 known-bad claims before | **6** after the runs (5 of them fuzzy matches on bullets the gate had already corrected; 1 real — the Rosé d'Anjou sibling text); 19 after merging the new audit findings below | + +**Validation — paired, same records, same independent grader +(`claude-opus-5`, a different model from the sonnet-4-6 extractor / gate), +grading the EN bullets a reader sees against the full source +(`scripts/audit_terroir_facts_llm.py`).** + +| sample | state | bullets | misleading | share (95 % CI) | records with ≥ 1 | +|---|---|---:|---:|---|---:| +| 120 records, whole corpus, seed 0 | before the programme (backup `r1`) | 804 | 111 | **13.8 %** (11.6–16.4) | 76 | +| same 120 | after `r1` (re-extraction + gate + 02e + back-check) | 793 | 25 | **3.2 %** (2.1–4.6) | 21 | +| 60 records, rest scope, seed 7 | after the gate, before their re-extraction (backup `r1b`) | 349 | 13 | 3.7 % (2.2–6.3) | 12 | +| same 60 | after `r1b` | 348 | 5 | **1.4 %** (0.6–3.3) | 5 | + +Paired: 64 of 120 records improved, 6 worse, 50 unchanged (first sample); +10 / 3 / 47 (second). The six "worse" records are re-extracted records +whose new bullets carry a small residual defect (a dropped *plutôt*, a +soil subtype in a parenthesis the source lists elsewhere) — none is a +gate rewrite that went wrong. The independent grader is stricter than +the review's verifier (it put the pre-programme state at 13.8 %, the +review at 6.9 %), so the absolute target of < 1.5 % is met on the second +sample and missed on the first (3.2 %); relative to its own baseline the +programme removes **77 %** of the misleading bullets on the first sample +and 62 % of what the gate alone had left on the second. Residual failure +modes after: narrowed attribution 8, unsupported causal link 7, hedge 6, +wrong entity 6, invented detail 5 — the same family as before, at a +fifth of the rate; 16 of the 25 are extraction, 5 translation, 4 both. +The 30 residual bullets were merged into the feedback sidecars +(`llm-audit-2026-09-13-after-r1` / `-r1b`), so the next re-run of those +records reads them as constraints. + +**Run `r1c` — the grounding filter.** Montepulciano d'Abruzzo came out +of `r1` with 1 fact: 8 verbatim quotes had been dropped because the +MASAF text carries pdftotext artefacts ("gradi- giorno") and one wrong +character halves a single-longest-contiguous-match coverage. The +coverage test in `_lib/terroir_coverage.py` is now also graded by +verbatim blocks (≥ 12 chars, summed; threshold unchanged at 0.6 — +scattered phrases and foreign text still score < 0.3). The 28 records +that had lost ≥ 4 facts to grounding were re-run through the chain (91 → +207 facts; Montepulciano 1 → 9, Montefalco 6 → 13) and every cache +re-graded (`recompute_terroir_provenance.py`): wiki-only provenance fell +from 830 to 394 bullets and `wiki_with_cahier_quote` from 446 to 5 — the +cahier quote *was* there; the old measure could not see past one +artefact. Final corpus: 1,638 records, **11,255 bullets**, strict audit +0 findings (labels strict). + +**Experiment (2026-09-14, run `exp-sonnet5`, rolled back): Sonnet 5 as +extractor.** The 120-record sample was re-extracted with +`claude-sonnet-5` (thinking off, `OWM_BATCH_THINKING=disabled`), then +gated / translated / back-checked by the unchanged Sonnet 4.6 stages, +and graded paired by the Opus-5 verifier against its pre-experiment +state. Per bullet, reader-misleading extraction defects were the same — +1.8 % (15 / 812) on 4.6 vs 1.7 % (18 / 1,064) on Sonnet 5 — with +overlapping intervals; the gate rewrote 20 % of Sonnet 5's bullets vs +27 % of 4.6's, and Sonnet 5 produced 31 % more bullets per record (8.9 +vs 6.8), every quote grounding (0 grounding drops vs 73). Translation- +origin defects rose with the volume (4 → 10). Conclusion: the extractor +model is not the lever for the misleading rate; Sonnet 5 buys coverage +and a third off the extraction price at equal per-bullet reliability. A +switch is a coverage / cost decision, not a quality one; the run was +rolled back so the corpus stays on one extractor. + +**Configuration decided 2026-09-14** (Boris): Sonnet 5 as the extractor +(coverage + cost), Opus 5 with adaptive thinking as the gate; the code +defaults are set (`providers.STAGE_DEFAULTS`), the corpus is not yet +migrated. The hand-off with the remaining recommendations is +[handoff-terroir-facts-2026-09-14.md](handoff-terroir-facts-2026-09-14.md). + +**Side effects worth knowing.** The MASAF annex fix corrected 20 IT +parents' sidecars (Trentino 4 → 28 grapes, Colli Tortonesi 1 → 24, +Langhe 1 → 14, Romagna 1 → 15) and the sottozona detector now finds 77 +sottozone in 17 parents (was 38 in 10) — visible on the map after stage +04. The gate's `moved` verdicts (≈ 5 % of bullets) closed W7 without a +separate reclassifier. Cost: the whole programme ran as Anthropic +batches (≈ 12,000 extraction, 2,400 gate, 7,000 translation, 7,700 +back-check, 360 opus-5 grading requests) in about two hours of +wall-clock. + +**Still open.** `rewrite_rejected` 97 (an empty or over-long rewrite; the +original bullet stays, flagged), `wiki_binding` 23 and `foreign_name` 3 +for a curator (see CURATOR_TODO "Cross-country"), `collioure` (no +source), and the 231 post-block records of 2026-09-11: gated like every +record, but not re-extracted (their prompt already carried the style +block); they are the cleanest cohort and can wait for the next scoped +run. + +## Re-run scope + +`tmp/terroir-facts-review/full-review-2026-09-12/`: + +- `rerun-slugs.txt` — 346 records with a verified misleading bullet + (`rerun-slugs-<cc>.txt` per country; IT 146 records, FR 81, GR 34, ES 26, + BG 23, HU 16, RO 12, …). +- `confirmed-misleading.json` — the 384 bullets with mode, stage, + verifier reason and a corrected English rendering (for spot-checking the + re-run, not for hand-editing caches). +- `masaf_multi_annex.json` — the 24 MASAF PDFs to re-slice. +- `det_checks.json` → `rows.label_prefix` — the 1,291 label bullets / 730 + records. + +Suggested order: R4 data fixes → R5 prompt edit + R2/R3 code → 02d +`--refresh` on the union above (then the rest of the pre-block corpus) → +02e → 04 → `audit_terroir_facts.py --strict` → re-sample 200 with the +verifier prompt, target misleading < 1.5 %. + +## Evidence + +`tmp/terroir-facts-review/full-review-2026-09-12/` (15 MB, gitignored): + +- `REVIEW_GUIDE.md` — rubric and pipeline context given to every reviewer. +- `findings/` — per-sheet reviewer output (`<sheet>-fidelity|language|explore.json`). +- `merged.json` — per-fact merged findings + summary; `record_notes` holds + the 1,263 MISSING / 96 OTHER / 10 WRONG_SOURCE / 4 SIB record notes. +- `verdicts/`, `verdicts_merged.json` — 703 adversarial verdicts + (`verdict`, `reader_misled`, `where`, `confirmed_tags`, `reason`, + `corrected_en`). +- `precision/` — 278 soft-tag second opinions; `modes/`, `modes_merged.json` + — failure mode per misleading bullet. +- `det_checks.json`, `numcheck_notfound.json` — deterministic checks. +- `report_data.json` — every number in this document. +- `workflows/` + `*.py` — the sheet builders, merge, harvest and the + Workflow scripts (review, verify, precision, modes) to re-run. diff --git a/locale/babel.cfg b/locale/babel.cfg new file mode 100644 index 0000000..efceab8 --- /dev/null +++ b/locale/babel.cfg @@ -0,0 +1 @@ +[python: **.py] diff --git a/locale/en/LC_MESSAGES/messages.po b/locale/en/LC_MESSAGES/messages.po new file mode 100644 index 0000000..3269313 --- /dev/null +++ b/locale/en/LC_MESSAGES/messages.po @@ -0,0 +1,1346 @@ +# English translations for Open Wine Map. Hand-seeded; review and amend in +# place. +msgid "" +msgstr "" +"Project-Id-Version: open-wine-map\n" +"Report-Msgid-Bugs-To: EMAIL@ADDRESS\n" +"POT-Creation-Date: 2026-09-08 21:10+0200\n" +"PO-Revision-Date: YEAR-MO-DA HO:MI+ZONE\n" +"Last-Translator: FULL NAME <EMAIL@ADDRESS>\n" +"Language: en\n" +"Language-Team: en <LL@li.org>\n" +"Plural-Forms: nplurals=2; plural=(n != 1);\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.18.0\n" + +#: scripts/_lib/aging_taxonomy.py:184 +msgid "Vieillissement" +msgstr "Aging" + +#: scripts/_lib/aging_taxonomy.py:185 +msgid "Prädikat" +msgstr "Prädikat" + +#: scripts/_lib/aging_taxonomy.py:186 +msgid "Sélection" +msgstr "Selection" + +#: scripts/_lib/map_template.py:46 +msgid "Open Wine Map — appellations viticoles d'Europe" +msgstr "Open Wine Map — European wine appellations" +#: scripts/_lib/map_template.py:47 +msgid "carte des appellations viticoles" +msgstr "Map of wine appellations" + +#: scripts/_lib/map_template.py:49 +msgid "" +"Carte interactive des appellations viticoles d'Europe : cépages, styles " +"et terroir, d'après les registres officiels." +msgstr "Interactive map of Europe's wine appellations: grape varieties, styles and terroir, from the official registers." +#: scripts/_lib/map_template.py:52 +msgid "Chargement…" +msgstr "Loading…" + +#: scripts/_lib/map_template.py:53 +msgid "Recherche" +msgstr "Search" + +#: scripts/_lib/map_template.py:54 +msgid "nom d'appellation…" +msgstr "appellation name…" + +#: scripts/_lib/map_template.py:55 +msgid "Recherche d'appellation…" +msgstr "Search appellations…" + +#: scripts/_lib/map_template.py:56 +msgid "Recherche de cépage…" +msgstr "Search grapes…" + +#: scripts/_lib/map_template.py:58 +msgid "Rechercher une appellation, un cépage, une région…" +msgstr "Search an appellation, grape or region…" + +#: scripts/_lib/map_template.py:59 +msgid "Cépage principal uniquement" +msgstr "Main grape only" + +#: scripts/_lib/map_template.py:60 +#, python-brace-format +msgid "Aucun résultat pour « {q} »" +msgstr "No results for “{q}”" + +#: scripts/_lib/map_template.py:61 +msgid "Options" +msgstr "Options" + +#: scripts/_lib/map_template.py:62 +msgid "Filtres actifs" +msgstr "Active filters" + +#: scripts/_lib/map_template.py:63 +msgid "Tout sélectionner" +msgstr "Select all" + +#: scripts/_lib/map_template.py:65 +#, python-brace-format +msgid "{p} appellations · {s} dénominations géographiques complémentaires" +msgstr "{p} appellations · {s} complementary geographic designations" + +#: scripts/_lib/map_template.py:67 +#, python-brace-format +msgid "{p} appellations" +msgstr "{p} appellations" + +#: scripts/_lib/map_template.py:68 +#, python-brace-format +msgid "Ouvrir la fiche de {name}" +msgstr "Open the {name} details" + +#: scripts/_lib/map_template.py:69 +msgid "Ouvrir la fiche" +msgstr "Open details" + +#: scripts/_lib/map_template.py:70 +msgid "Inclure les spiritueux" +msgstr "Include spirits" + +#: scripts/_lib/map_template.py:71 +msgid "Style de vin" +msgstr "Wine style" + +#: scripts/_lib/map_template.py:72 +msgid "Classement" +msgstr "Classification" + +#: scripts/_lib/map_template.py:73 +msgid "Cépages principaux" +msgstr "Principal grape varieties" + +#: scripts/_lib/map_template.py:74 +msgid "Cépages accessoires" +msgstr "Accessory grape varieties" + +#: scripts/_lib/map_template.py:75 scripts/_lib/map_template.py:189 +msgid "Cépages" +msgstr "Grape varieties" + +#: scripts/_lib/map_template.py:76 +msgid "Région" +msgstr "Region" + +#: scripts/_lib/map_template.py:77 +msgid "Appellation" +msgstr "Appellation" + +#: scripts/_lib/map_template.py:78 +msgid "Type" +msgstr "Type" + +#: scripts/_lib/map_template.py:79 +msgid "Origine protégée (AOP, AOC, DOC, DO…)" +msgstr "Protected origin (PDO, AOC, DOC, DO…)" + +#: scripts/_lib/map_template.py:80 +msgid "Indication géographique (IGP, IGT, Landwein…)" +msgstr "Geographical indication (PGI, IGT, Landwein…)" + +#: scripts/_lib/map_template.py:81 +msgid "AOP" +msgstr "PDO" + +#: scripts/_lib/map_template.py:82 +msgid "IGP" +msgstr "PGI" + +#: scripts/_lib/map_template.py:83 +msgid "IG spiritueux" +msgstr "spirit-drink GI" + +#: scripts/_lib/map_template.py:84 +msgid "Type d'appellation" +msgstr "Appellation type" + +#: scripts/_lib/map_template.py:85 +msgid "AOP (régime britannique)" +msgstr "PDO (UK scheme)" + +#: scripts/_lib/map_template.py:86 +msgid "IGP (régime britannique)" +msgstr "PGI (UK scheme)" + +#: scripts/_lib/map_template.py:87 +msgid "AOC (Suisse)" +msgstr "AOC (Switzerland)" + +#: scripts/_lib/map_template.py:88 +msgid "Source" +msgstr "Source" + +#: scripts/_lib/map_template.py:89 +msgid "Vue" +msgstr "View" + +#: scripts/_lib/map_template.py:90 +msgid "Simple" +msgstr "Simple" + +#: scripts/_lib/map_template.py:91 +msgid "Avancée" +msgstr "Advanced" + +#: scripts/_lib/map_template.py:92 +msgid "Thème" +msgstr "Theme" + +#: scripts/_lib/map_template.py:93 +msgid "Clair" +msgstr "Light" + +#: scripts/_lib/map_template.py:94 +msgid "Sombre" +msgstr "Dark" + +#: scripts/_lib/map_template.py:95 +msgid "Système" +msgstr "System" + +#: scripts/_lib/map_template.py:96 +msgid "Afficher les IGP" +msgstr "Show PGIs" + +#: scripts/_lib/map_template.py:97 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:361 +msgid "rouge" +msgstr "red" + +#: scripts/_lib/map_template.py:98 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:362 +msgid "blanc" +msgstr "white" + +#: scripts/_lib/map_template.py:99 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:363 +msgid "rosé" +msgstr "rosé" + +#: scripts/_lib/map_template.py:100 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:364 +msgid "mousseux" +msgstr "sparkling" + +#: scripts/_lib/map_template.py:101 +msgid "moelleux / liquoreux" +msgstr "sweet / dessert" + +#: scripts/_lib/map_template.py:102 scripts/_lib/style_taxonomy.py:345 +msgid "oxydatif" +msgstr "oxidative" + +#: scripts/_lib/map_template.py:103 +msgid "autre" +msgstr "other" + +#: scripts/_lib/map_template.py:104 +msgid "Réinitialiser" +msgstr "Reset" + +#: scripts/_lib/map_template.py:105 +#, python-brace-format +msgid "{n} appellations" +msgstr "{n} appellations" + +#: scripts/_lib/map_template.py:106 +#, python-brace-format +msgid "{n} / {total} appellations" +msgstr "{n} / {total} appellations" + +#: scripts/_lib/map_template.py:107 +#, python-brace-format +msgid "{n} masquées dans les IGP · afficher" +msgstr "{n} hidden in PGIs · show" + +#: scripts/_lib/map_template.py:108 +msgid "Fermer" +msgstr "Close" + +#: scripts/_lib/map_template.py:109 +msgid "Détails de l'appellation" +msgstr "Appellation details" + +#: scripts/_lib/map_template.py:110 +#, python-brace-format +msgid "Retirer le filtre {label}" +msgstr "Remove filter {label}" + +#: scripts/_lib/map_template.py:111 +msgid "Filtres et options de la carte" +msgstr "Filters and map options" + +#: scripts/_lib/map_template.py:112 +msgid "Langue" +msgstr "Language" + +#: scripts/_lib/map_template.py:113 +msgid "Carte des appellations viticoles" +msgstr "Map of wine appellations" + +#: scripts/_lib/map_template.py:114 +msgid "Aller à la carte" +msgstr "Skip to the map" + +#: scripts/_lib/map_template.py:115 +msgid "Styles" +msgstr "Styles" + +#: scripts/_lib/map_template.py:116 +msgid "Variétés d'intérêt" +msgstr "Varieties of interest" + +#: scripts/_lib/map_template.py:117 +msgid "Sources" +msgstr "Sources" + +#: scripts/_lib/map_template.py:118 +msgid "Terroir" +msgstr "Terroir" + +#: scripts/_lib/map_template.py:119 +#, python-brace-format +msgid "Dűlők (lieux-dits) : {n}" +msgstr "Dűlők (named vineyards): {n}" + +#: scripts/_lib/map_template.py:120 +#, python-brace-format +msgid "Menzioni geografiche aggiuntive (crus) : {n}" +msgstr "Additional geographical mentions (crus): {n}" + +#: scripts/_lib/map_template.py:121 +msgid "Facteurs naturels" +msgstr "Natural factors" + +#: scripts/_lib/map_template.py:122 +msgid "Facteurs humains" +msgstr "Human factors" + +#: scripts/_lib/map_template.py:123 +msgid "Caractéristiques du produit" +msgstr "Product characteristics" + +#: scripts/_lib/map_template.py:124 +msgid "Lien terroir / vin" +msgstr "Terroir / wine link" + +#: scripts/_lib/map_template.py:126 +#, python-brace-format +msgid "" +"Faits dégagés du Lien au terroir par interprétation automatique — voir la" +" {source}." +msgstr "" +"Facts drawn from the cahier's terroir-link section (Lien au terroir) by " +"automatic interpretation — see the {source}." + +#: scripts/_lib/map_template.py:128 +msgid "source" +msgstr "source" + +#: scripts/_lib/map_template.py:129 +msgid "via Wikipedia · CC BY-SA 4.0" +msgstr "via Wikipedia · CC BY-SA 4.0" + +#: scripts/_lib/map_template.py:131 +#, python-brace-format +msgid "Citation textuelle du Lien au terroir — voir la {source}." +msgstr "" +"Verbatim quote from the cahier's terroir-link section (Lien au terroir) —" +" see the {source}." + +#: scripts/_lib/map_template.py:133 +msgid "à vérifier — texte source court" +msgstr "to verify — short source text" + +#: scripts/_lib/map_template.py:135 +#, python-brace-format +msgid "" +"Ces repères décrivent l'appellation englobante {parent} — pas " +"spécifiquement cette dénomination." +msgstr "" +"These highlights describe the wider {parent} appellation — not this " +"denomination specifically." + +#: scripts/_lib/map_template.py:138 +msgid "sans région" +msgstr "no region" + +#: scripts/_lib/map_template.py:139 +#, python-brace-format +msgid "{n} commune(s) INAO" +msgstr "{n} INAO commune(s)" + +#: scripts/_lib/map_template.py:140 +#, python-brace-format +msgid "{n} commune(s)" +msgstr "{n} commune(s)" + +#: scripts/_lib/map_template.py:141 +msgid "aire approchée" +msgstr "approximate area" + +#: scripts/_lib/map_template.py:142 +msgid "aire approchée (à l'échelle communale)" +msgstr "approximate area (commune level)" + +#: scripts/_lib/map_template.py:144 +#, python-brace-format +msgid "" +"Aire approchée — pas de données parcellaires précises pour cette " +"dénomination ; polygone hérité de {umbrella}." +msgstr "" +"Approximate area — no parcel-precise data published for this " +"denomination; polygon inherited from {umbrella}." + +#: scripts/_lib/map_template.py:148 +msgid "" +"Aire approchée — pas de données parcellaires précises pour cette " +"dénomination ; polygone hérité de l'appellation parente." +msgstr "" +"Approximate area — no parcel-precise data published for this " +"denomination; polygon inherited from the parent appellation." + +#: scripts/_lib/map_template.py:152 +msgid "" +"Aire approchée — pas de données parcellaires disponibles ; affichée comme" +" l'emprise de la commune où se situe la dénomination." +msgstr "" +"Approximate area — no parcel data available; shown as the outline of the " +"commune containing the denomination." + +#: scripts/_lib/map_template.py:156 +#, python-brace-format +msgid "" +"Aire issue du lieu-dit cadastral « {lieu_dit} » (commune de {commune}, " +"{source})." +msgstr "" +"Boundary derived from the cadastral lieu-dit “{lieu_dit}” (commune of " +"{commune}, {source})." + +#: scripts/_lib/map_template.py:159 +msgid "cadastre.data.gouv.fr" +msgstr "cadastre.data.gouv.fr" + +#: scripts/_lib/map_template.py:161 +msgid "" +"Aire approchée — reconstituée à partir des références parcellaires du " +"plan de l'aire délimitée annexé au cahier des charges ; ce n'est pas une " +"limite officielle." +msgstr "" +"Approximate area — reconstructed from the land-parcel references on the " +"plan of the demarcated area annexed to the product specification; this is" +" not an official boundary." + +#: scripts/_lib/map_template.py:165 +#, python-brace-format +msgid "{n} appellations à ce point" +msgstr "{n} appellations at this point" + +#: scripts/_lib/map_template.py:166 +msgid "Cliquer à nouveau pour parcourir les autres" +msgstr "Click again to cycle through the others" + +#: scripts/_lib/map_template.py:167 +msgid "Cahier des charges (BO Agri, PDF)" +msgstr "Product specification (BO Agri, PDF)" + +#: scripts/_lib/map_template.py:168 +msgid "Cahier des charges (registre GI de l'UE, PDF)" +msgstr "Product specification (EU GI register, PDF)" + +#: scripts/_lib/map_template.py:169 +msgid "homologué" +msgstr "approved" + +#: scripts/_lib/map_template.py:170 +msgid "JORF" +msgstr "JORF" + +#: scripts/_lib/map_template.py:171 +msgid "Texte officiel INAO (show_texte)" +msgstr "Official INAO text (show_texte)" + +#: scripts/_lib/map_template.py:172 +msgid "Fiche produit INAO" +msgstr "INAO product entry" + +#: scripts/_lib/map_template.py:173 +msgid "Site officiel de l'interprofession" +msgstr "Official trade body site" + +#: scripts/_lib/map_template.py:174 +msgid "Cahier des charges (EUR-Lex, document unique)" +msgstr "Specification (EUR-Lex, single document)" + +#: scripts/_lib/map_template.py:175 +msgid "Pliego de condiciones (national, PDF)" +msgstr "National pliego de condiciones (PDF)" + +#: scripts/_lib/map_template.py:176 +msgid "variétés ajoutées" +msgstr "varieties added" + +#: scripts/_lib/map_template.py:177 +msgid "Cahier des charges national (PDF)" +msgstr "National product specification (PDF)" + +#: scripts/_lib/map_template.py:178 +msgid "Spécification du produit (IGP, PDF)" +msgstr "Product specification (PGI, PDF)" + +#: scripts/_lib/map_template.py:179 +msgid "Registre régional des cépages (PDF)" +msgstr "Regional grape register (PDF)" + +#: scripts/_lib/map_template.py:180 +msgid "Registre eAmbrosia (UE)" +msgstr "eAmbrosia register (EU)" + +#: scripts/_lib/map_template.py:181 +msgid "Numéro de dossier" +msgstr "File number" + +#: scripts/_lib/map_template.py:182 +msgid "Règlement cantonal sur la vigne et le vin" +msgstr "Cantonal wine regulation" + +#: scripts/_lib/map_template.py:183 +msgid "Répertoire suisse des AOC (OFAG/BLW)" +msgstr "Swiss AOC register (FOAG/BLW)" + +#: scripts/_lib/map_template.py:184 +msgid "Cahier des charges (GOV.UK, PDF)" +msgstr "Product specification (GOV.UK, PDF)" + +#: scripts/_lib/map_template.py:185 +msgid "Registre des IG du Royaume-Uni" +msgstr "UK GI register" + +#: scripts/_lib/map_template.py:186 +msgid "Légende couleurs" +msgstr "Colour legend" + +#: scripts/_lib/map_template.py:187 +msgid "Bassin viticole" +msgstr "Wine basin" + +#: scripts/_lib/map_template.py:188 +msgid "Plus l'aire est petite, plus la teinte est dense." +msgstr "Smaller area = denser fill." + +#: scripts/_lib/map_template.py:190 +msgid "principal — variété de la cuvée" +msgstr "principal — main variety" + +#: scripts/_lib/map_template.py:191 +msgid "accessoire — assemblage limité" +msgstr "accessory — limited blending" + +#: scripts/_lib/map_template.py:192 +msgid "intérêt — observation/conservation" +msgstr "of interest — observation / conservation" + +#: scripts/_lib/map_template.py:194 +msgid "" +"Le régulateur portugais (IVV) n'établit pas de distinction " +"principal/accessoire — toutes les castas autorisées sont listées ensemble" +" dans le caderno de especificações." +msgstr "" +"The Portuguese regulator (IVV) does not distinguish principal vs " +"accessory varieties — every authorised casta is listed together in the " +"caderno de especificações." + +#: scripts/_lib/map_template.py:198 +msgid "" +"Bianchello (ou Biancame) est traité ici comme un cépage distinct, " +"conformément au disciplinare de la DOP Bianchello del Metauro ; le " +"catalogue VIVC le recense comme synonyme du Trebbiano Toscano." +msgstr "" +"Bianchello (also called Biancame) is treated here as a distinct variety, " +"in line with the disciplinare of the Bianchello del Metauro DOP; the VIVC" +" catalogue records it as a synonym of Trebbiano Toscano." + +#: scripts/_lib/map_template.py:203 +#, python-brace-format +msgid "Open Wine Map n'a pas encore trouvé de {doc} pour cette appellation." +msgstr "Open Wine Map has not yet located a {doc} for this appellation." + +#: scripts/_lib/map_template.py:205 +msgid "aidez-nous à le trouver" +msgstr "help us find it" + +#: scripts/_lib/map_template.py:211 +#, python-brace-format +msgid "Délimitée par {regulator} dans son {doc}, qui autorise {grapes}." +msgstr "Delimited by {regulator} in its {doc}, which authorises {grapes}." + +#: scripts/_lib/map_template.py:213 +#, python-brace-format +msgid "Délimitée par {regulator} dans son {doc}{extra}." +msgstr "Delimited by {regulator} in its {doc}{extra}." + +#: scripts/_lib/map_template.py:214 +#, python-brace-format +msgid "{names} et {n} autres cépages" +msgstr "{names} and {n} other grape varieties" + +#: scripts/_lib/map_template.py:215 +msgid "le registre eAmbrosia de l'UE" +msgstr "the EU eAmbrosia register" + +#: scripts/_lib/map_template.py:216 +#, python-brace-format +msgid "le canton de {canton}" +msgstr "the canton of {canton}" + +#: scripts/_lib/map_template.py:217 +msgid "(français)" +msgstr "(source: French)" + +#: scripts/_lib/map_template.py:218 +msgid "Texte source en français" +msgstr "Source text in French" + +#: scripts/_lib/map_template.py:219 +msgid "(español)" +msgstr "(source: Spanish)" + +#: scripts/_lib/map_template.py:220 +msgid "Texte source en espagnol" +msgstr "Source text in Spanish" + +#: scripts/_lib/map_template.py:221 +msgid "(português)" +msgstr "(source: Portuguese)" + +#: scripts/_lib/map_template.py:222 +msgid "Texte source en portugais" +msgstr "Source text in Portuguese" + +#: scripts/_lib/map_template.py:223 +msgid "Filtres" +msgstr "Filters" + +#: scripts/_lib/map_template.py:224 +#, python-brace-format +msgid "Traduit de {wiki} · CC BY-SA 4.0" +msgstr "Translated from {wiki} · CC BY-SA 4.0" + +#: scripts/_lib/map_template.py:225 +msgid "Wikipédia en anglais" +msgstr "English Wikipedia" + +#: scripts/_lib/map_template.py:226 +msgid "Wikipédia en français" +msgstr "French Wikipedia" + +#: scripts/_lib/map_template.py:227 +msgid "Wikipédia en espagnol" +msgstr "Spanish Wikipedia" + +#: scripts/_lib/map_template.py:228 +msgid "Wikipédia en néerlandais" +msgstr "Dutch Wikipedia" + +#: scripts/_lib/map_template.py:229 +msgid "Wikipédia en portugais" +msgstr "Portuguese Wikipedia" + +#: scripts/_lib/map_template.py:230 +msgid "Wikipédia en croate" +msgstr "Croatian Wikipedia" + +#: scripts/_lib/map_template.py:231 +msgid "Vitis International Variety Catalogue (Julius Kühn-Institut)" +msgstr "Vitis International Variety Catalogue (Julius Kühn-Institut)" + +#: scripts/_lib/map_template.py:232 +#, python-brace-format +msgid "VIVC #{id}" +msgstr "VIVC #{id}" + +#: scripts/_lib/map_template.py:233 +#, python-brace-format +msgid "Traduction automatique depuis {source}" +msgstr "Machine translated from {source}" + +#: scripts/_lib/map_template.py:234 +msgid "le cahier des charges" +msgstr "the cahier des charges" + +#: scripts/_lib/map_template.py:235 scripts/_lib/map_template.py:237 +msgid "le pliego de condiciones" +msgstr "the specification" + +#: scripts/_lib/map_template.py:236 scripts/_lib/map_template.py:238 +msgid "le caderno de especificações" +msgstr "the caderno de especificações" + +#: scripts/_lib/map_template.py:239 +msgid "Dénomination géographique complémentaire de" +msgstr "Sub-region of" + +#: scripts/_lib/map_template.py:240 +msgid "À propos" +msgstr "About" + +#: scripts/_lib/map_template.py:241 +msgid "À propos d'Open Wine Map" +msgstr "About Open Wine Map" + +#: scripts/_lib/map_template.py:243 +msgid "" +"Carte de référence des appellations viticoles, générée automatiquement à " +"partir des registres publics : le registre de l'Union européenne des AOP " +"et IGP, les régulateurs nationaux, le répertoire fédéral suisse des AOC " +"cantonales et le registre britannique des indications géographiques." +msgstr "" +"Reference map of wine appellations, generated automatically from public " +"registers: the EU register of PDOs and PGIs, the national regulators, the" +" Swiss federal repertoire of cantonal AOCs and the UK register of " +"geographical indications." + +#: scripts/_lib/map_template.py:249 +msgid "" +"Une partie du texte est produite par des modèles de langage : les repères" +" de terroir sont dégagés du texte du régulateur et traduits par Claude " +"(Sonnet 4.6) ; les extraits Wikipedia des infobulles de cépages et de " +"styles sont traduits pour l'essentiel par Mistral Small 3.2, exécuté " +"localement, et pour quelques-uns par Claude ; les résumés des cahiers des" +" charges ont été traduits par un traducteur humain, à l'exception d'un " +"petit reliquat traduit automatiquement. Chaque élément porte sa propre " +"ligne d'attribution dans le panneau." +msgstr "" +"Part of the text is produced by language models: the terroir notes are " +"extracted from the regulator's text and translated by Claude (Sonnet " +"4.6); the Wikipedia extracts in the grape and style tooltips are " +"translated mostly by Mistral Small 3.2 running locally, and some by " +"Claude; the cahier summaries were translated by a human translator, with " +"a small machine-translated remainder. Each item carries its own " +"attribution line in the panel." + +#: scripts/_lib/map_template.py:258 +#, python-brace-format +msgid "Réalisé avec ♡ par {devloed}." +msgstr "Made with ♡ by {devloed}." + +#: scripts/_lib/map_template.py:260 +#, python-brace-format +msgid "" +"Sources : INAO ({inao}) pour les cahiers des charges et les aires " +"parcellaires françaises, IGN ({ign}) pour les contours des communes " +"françaises, le registre des indications géographiques de l'UE, les " +"régulateurs nationaux, Eurostat GISCO et Bétard 2022 pour les autres " +"pays, OpenStreetMap et CARTO pour le fond de carte, Wikipedia " +"({wikipedia}) pour quelques compléments narratifs (CC BY-SA 4.0), et VIVC" +" ({vivc}), le Vitis International Variety Catalogue du Julius Kühn-" +"Institut, pour les noms canoniques et numéros de cépage (citation Röckel " +"et al.). Tout extrait Wikipedia est signalé sur place. Détails et " +"licences dans le {readme}." +msgstr "" +"Sources: INAO ({inao}) for the French cahiers des charges and parcel " +"boundaries, IGN ({ign}) for French commune polygons, the EU geographical-" +"indications register, the national regulators, Eurostat GISCO and Bétard " +"2022 for the other countries, OpenStreetMap and CARTO for the basemap, " +"Wikipedia ({wikipedia}) for a handful of narrative extracts (CC BY-SA " +"4.0), and VIVC ({vivc}), the Julius Kühn-Institut's Vitis International " +"Variety Catalogue, for canonical grape names and variety numbers (citing " +"Röckel et al.). Every Wikipedia excerpt is flagged where it appears. Full" +" source list and licences in the {readme}." + +#: scripts/_lib/map_template.py:270 +#, python-brace-format +msgid "Suggestions et pull requests bienvenues sur {github}." +msgstr "Issues and pull requests welcome on {github}." + +#: scripts/_lib/map_template.py:271 +msgid "ticket GitHub" +msgstr "GitHub issue" + +#: scripts/_lib/map_template.py:272 +msgid "e-mail" +msgstr "email" + +#: scripts/_lib/map_template.py:273 +msgid "E-mail copié dans le presse-papiers" +msgstr "Email copied to clipboard" + +#: scripts/_lib/map_template.py:275 +#, python-brace-format +msgid "" +"Carte générée automatiquement — des erreurs sont possibles. Signalez-les " +"via {issue} ou {email}." +msgstr "" +"Auto-generated map — errors are possible. Report them via {issue} or " +"{email}." + +#: scripts/_lib/map_template.py:279 +#, python-brace-format +msgid "" +"{c} pays européens cartographiés ({n} appellations : {parents} " +"appellations et {subs} dénominations rattachées). Des itérations " +"supplémentaires affineront la qualité des données et étendront la " +"couverture au-delà de l'UE, de la Suisse et du Royaume-Uni." +msgstr "" +"{c} European countries mapped ({n} entries: {parents} appellations and " +"{subs} related denominations). Further iterations will refine data " +"quality and extend coverage beyond the EU, Switzerland and the United " +"Kingdom." + +#: scripts/_lib/map_template.py:284 +msgid "Toutes les appellations" +msgstr "All appellations" + +#: scripts/_lib/map_template.py:285 +msgid "Toutes les appellations viticoles — Open Wine Map" +msgstr "All wine appellations — Open Wine Map" + +#: scripts/_lib/map_template.py:287 +#, python-brace-format +msgid "" +"Liste des {n} appellations viticoles cartographiées sur Open Wine Map, " +"classées par pays : AOP et IGP de l'UE avec leurs termes traditionnels " +"(AOC, DOCG, DOQ…), AOC suisses et IG britanniques." +msgstr "" +"List of the {n} wine appellations mapped on Open Wine Map, by country: EU" +" PDOs and PGIs with their traditional terms (AOC, DOCG, DOQ…), Swiss AOCs" +" and UK GIs." + +#: scripts/_lib/map_template.py:292 +#, python-brace-format +msgid "" +"Les {n} appellations ci-dessous sont classées par pays. Retour à la " +"{map_link}." +msgstr "The {n} appellations below are grouped by country. Back to the {map_link}." + +#: scripts/_lib/map_template.py:295 +msgid "carte interactive" +msgstr "interactive map" + +#: scripts/_lib/map_template.py:296 +msgid "Liste des appellations par pays" +msgstr "List of appellations by country" + +#: scripts/_lib/map_template.py:297 +msgid "Appellation parente" +msgstr "Parent appellation" + +#: scripts/_lib/map_template.py:298 +msgid "Dénominations rattachées" +msgstr "Related denominations" + +#: scripts/_lib/map_template.py:300 +#, python-brace-format +msgid "Parcourir la liste complète des appellations : {browse_link}." +msgstr "Browse the full list of appellations: {browse_link}." + +#: scripts/_lib/map_template.py:302 +#, python-brace-format +msgid "Données mises à jour le {date}." +msgstr "Data updated on {date}." + +#: scripts/_lib/map_template.py:317 +msgid "BOURGOGNE" +msgstr "Burgundy" + +#: scripts/_lib/map_template.py:318 +msgid "BEAUJOLAIS" +msgstr "Beaujolais" + +#: scripts/_lib/map_template.py:319 +msgid "JURA" +msgstr "Jura" + +#: scripts/_lib/map_template.py:320 +msgid "SAVOIE" +msgstr "Savoie" + +#: scripts/_lib/map_template.py:321 +msgid "BUGEY" +msgstr "Bugey" + +#: scripts/_lib/map_template.py:322 +msgid "ALSACE ET EST" +msgstr "Alsace and East" + +#: scripts/_lib/map_template.py:323 +msgid "VAL DE LOIRE" +msgstr "Loire Valley" + +#: scripts/_lib/map_template.py:324 +msgid "SUD-OUEST" +msgstr "South-West" + +#: scripts/_lib/map_template.py:325 +msgid "VALLEE DU RHÔNE" +msgstr "Rhône Valley" + +#: scripts/_lib/map_template.py:326 +msgid "LANGUEDOC-ROUSSILLON" +msgstr "Languedoc-Roussillon" + +#: scripts/_lib/map_template.py:327 +msgid "TOULOUSE-PYRENEES" +msgstr "Toulouse-Pyrenees" + +#: scripts/_lib/map_template.py:328 +msgid "PROVENCE-CORSE" +msgstr "Provence-Corsica" + +#: scripts/_lib/map_template.py:329 +msgid "CHAMPAGNE" +msgstr "Champagne" + +#: scripts/_lib/map_template.py:330 +msgid "EAUX-DE-VIE DE CIDRE" +msgstr "Cider eaux-de-vie" + +#: scripts/_lib/map_template.py:331 +msgid "VIN DOUX NATURELS" +msgstr "Natural sweet wines" + +#: scripts/_lib/map_template.py:332 +msgid "COGNAC" +msgstr "Cognac" + +#: scripts/_lib/map_template.py:333 +msgid "ARMAGNAC" +msgstr "Armagnac" + +#: scripts/_lib/map_template.py:334 +msgid "RHUM" +msgstr "Rum" + +#: scripts/_lib/map_template.py:374 +msgid "France" +msgstr "France" + +#: scripts/_lib/map_template.py:375 +msgid "Espagne" +msgstr "Spain" + +#: scripts/_lib/map_template.py:376 +msgid "Portugal" +msgstr "Portugal" + +#: scripts/_lib/map_template.py:377 +msgid "Italie" +msgstr "Italy" + +#: scripts/_lib/map_template.py:378 +msgid "Autriche" +msgstr "Austria" + +#: scripts/_lib/map_template.py:379 +msgid "Slovénie" +msgstr "Slovenia" + +#: scripts/_lib/map_template.py:380 +msgid "Croatie" +msgstr "Croatia" + +#: scripts/_lib/map_template.py:381 +msgid "Hongrie" +msgstr "Hungary" + +#: scripts/_lib/map_template.py:382 +msgid "Roumanie" +msgstr "Romania" + +#: scripts/_lib/map_template.py:383 +msgid "Bulgarie" +msgstr "Bulgaria" + +#: scripts/_lib/map_template.py:384 +msgid "Grèce" +msgstr "Greece" + +#: scripts/_lib/map_template.py:385 +msgid "Allemagne" +msgstr "Germany" + +#: scripts/_lib/map_template.py:386 +msgid "Slovaquie" +msgstr "Slovakia" + +#: scripts/_lib/map_template.py:387 +msgid "Suisse" +msgstr "Switzerland" + +#: scripts/_lib/map_template.py:388 +msgid "Tchéquie" +msgstr "Czechia" + +#: scripts/_lib/map_template.py:389 +msgid "Luxembourg" +msgstr "Luxembourg" + +#: scripts/_lib/map_template.py:390 +msgid "Belgique" +msgstr "Belgium" + +#: scripts/_lib/map_template.py:391 +msgid "Pays-Bas" +msgstr "Netherlands" + +#: scripts/_lib/map_template.py:392 +msgid "Malte" +msgstr "Malta" + +#: scripts/_lib/map_template.py:393 +msgid "Chypre" +msgstr "Cyprus" + +#: scripts/_lib/map_template.py:394 +msgid "Royaume-Uni" +msgstr "United Kingdom" + +#: scripts/_lib/style_taxonomy.py:332 scripts/_lib/style_taxonomy.py:365 +msgid "doux" +msgstr "sweet" + +#: scripts/_lib/style_taxonomy.py:333 scripts/_lib/style_taxonomy.py:366 +msgid "autres" +msgstr "other" + +#: scripts/_lib/style_taxonomy.py:334 +msgid "clairet" +msgstr "clairet" + +#: scripts/_lib/style_taxonomy.py:334 +msgid "primeur" +msgstr "primeur" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "mousseux de qualité" +msgstr "quality sparkling" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "crémant" +msgstr "crémant" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "méthode ancestrale" +msgstr "ancestral method" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode traditionnelle" +msgstr "traditional method" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode Charmat" +msgstr "Charmat method" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode dioise" +msgstr "méthode dioise" + +#: scripts/_lib/style_taxonomy.py:337 +msgid "pétillant" +msgstr "semi-sparkling" + +#: scripts/_lib/style_taxonomy.py:338 +msgid "vin muté" +msgstr "fortified" + +#: scripts/_lib/style_taxonomy.py:338 +msgid "vendanges tardives (catégorie)" +msgstr "late-harvest (category)" + +#: scripts/_lib/style_taxonomy.py:339 +msgid "vin de raisins passerillés" +msgstr "raisin wine" + +#: scripts/_lib/style_taxonomy.py:340 +msgid "demi-doux" +msgstr "semi-sweet" + +#: scripts/_lib/style_taxonomy.py:340 +msgid "vin de glace" +msgstr "ice wine" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "vin doux naturel" +msgstr "vin doux naturel" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "vin de liqueur" +msgstr "vin de liqueur" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "mistelle" +msgstr "mistelle" + +#: scripts/_lib/style_taxonomy.py:342 +msgid "vendanges tardives" +msgstr "late harvest" + +#: scripts/_lib/style_taxonomy.py:342 +msgid "uvas sobremaduradas" +msgstr "overripe grapes" + +#: scripts/_lib/style_taxonomy.py:343 +msgid "vin naturellement doux" +msgstr "naturally sweet" + +#: scripts/_lib/style_taxonomy.py:343 +msgid "grains nobles" +msgstr "grains nobles" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "vin de paille" +msgstr "vin de paille" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "uvas pasificadas" +msgstr "dried grapes" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "vin santo" +msgstr "vin santo" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "tranquille" +msgstr "still" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "sur lie" +msgstr "sur lie" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "sec" +msgstr "dry" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "demi-sec" +msgstr "semi-dry" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "rancio" +msgstr "rancio" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "vin jaune" +msgstr "vin jaune" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "generoso" +msgstr "generoso" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "fino" +msgstr "fino" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "manzanilla" +msgstr "manzanilla" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "amontillado" +msgstr "amontillado" + +#: scripts/_lib/style_taxonomy.py:348 +msgid "oloroso" +msgstr "oloroso" + +#: scripts/_lib/style_taxonomy.py:348 +msgid "palo cortado" +msgstr "palo cortado" + +#~ msgid "carte des appellations françaises" +#~ msgstr "map of french appellations" + +#~ msgid "" +#~ "Priorité actuelle : affiner la " +#~ "couverture française — précision des " +#~ "aires, qualité des extraits, climats et" +#~ " lieux-dits. L'extension à d'autres " +#~ "pays viticoles viendra ensuite." +#~ msgstr "" +#~ "Current priority: tightening the French " +#~ "coverage — boundary accuracy, extract " +#~ "quality, climats and lieux-dits. Other" +#~ " wine-producing countries will follow." + +#~ msgid "anglais" +#~ msgstr "" + +#~ msgid "français" +#~ msgstr "(French)" + +#~ msgid "espagnol" +#~ msgstr "(Spanish)" + +#~ msgid "néerlandais" +#~ msgstr "" + +#~ msgid "" +#~ "Couverture actuelle : France et une " +#~ "première version de l'Espagne. Prochaines " +#~ "étapes : amélioration continue de la " +#~ "qualité des données, puis ajout du " +#~ "Portugal." +#~ msgstr "" +#~ "Currently covered: France and an initial" +#~ " version of Spain. Next up: ongoing" +#~ " data-quality improvements, then Portugal." + +#~ msgid "moelleux" +#~ msgstr "medium-sweet" + +#~ msgid "" +#~ "Carte interactive des appellations viticoles" +#~ " européennes (AOC, AOP, IGP, DOP) :" +#~ " cépages, styles et terroir, d'après " +#~ "les registres officiels (INAO, EUR-Lex)." +#~ msgstr "" +#~ "Interactive map of European wine " +#~ "appellations (AOC, AOP, IGP, DOP): grape" +#~ " varieties, styles and terroir, from " +#~ "official public registry data (INAO, " +#~ "EUR-Lex)." + +#~ msgid "AOC / AOP" +#~ msgstr "AOC / AOP" + +#~ msgid "{n} dans IGP masquées — afficher" +#~ msgstr "{n} hidden in IGPs — show" + +#~ msgid "" +#~ "Carte de référence des appellations " +#~ "viticoles (AOC, AOP, IGP, DOP), générée" +#~ " automatiquement à partir des données " +#~ "publiques." +#~ msgstr "" +#~ "An open reference map of wine " +#~ "appellations (AOC, AOP, IGP, DOP), " +#~ "generated automatically from public data." + +#~ msgid "" +#~ "Sources : INAO ({inao}) pour les " +#~ "cahiers des charges et les aires " +#~ "parcellaires, IGN ({ign}) pour le fond" +#~ " cartographique, Wikipedia ({wikipedia}) pour " +#~ "quelques compléments narratifs (CC BY-SA" +#~ " 4.0), VIVC ({vivc}) — Vitis " +#~ "International Variety Catalogue, Julius " +#~ "Kühn-Institut — pour les noms " +#~ "canoniques et numéros de cépage " +#~ "(citation Röckel et al.). Tout extrait" +#~ " Wikipedia est signalé sur place. " +#~ "Détails et licences dans le {readme}." +#~ msgstr "" +#~ "Sources: INAO ({inao}) for the cahiers" +#~ " des charges and parcel boundaries, " +#~ "IGN ({ign}) for the base map, " +#~ "Wikipedia ({wikipedia}) for a handful of" +#~ " narrative extracts (CC BY-SA 4.0)," +#~ " VIVC ({vivc}) — Vitis International " +#~ "Variety Catalogue, Julius Kühn-Institut " +#~ "— for canonical variety names and " +#~ "numbers (citing Röckel et al.). Every" +#~ " Wikipedia excerpt is flagged where " +#~ "it appears. Full source list and " +#~ "licences in the {readme}." + +#~ msgid "" +#~ "20 pays européens cartographiés : " +#~ "France, Espagne, Portugal, Italie, Autriche," +#~ " Allemagne, Suisse, Slovénie, Croatie, " +#~ "Hongrie, Roumanie, Bulgarie, Grèce, Slovaquie," +#~ " Tchéquie, Luxembourg, Belgique, Pays-Bas," +#~ " Malte et Chypre. Des itérations " +#~ "supplémentaires viendront affiner la qualité" +#~ " des données. La couverture sera " +#~ "étendue au-delà de l'UE et de " +#~ "la Suisse, ainsi qu'aux classifications " +#~ "hors AOP." +#~ msgstr "" +#~ "20 European countries now mapped: " +#~ "France, Spain, Portugal, Italy, Austria, " +#~ "Germany, Switzerland, Slovenia, Croatia, " +#~ "Hungary, Romania, Bulgaria, Greece, Slovakia," +#~ " Czech Republic, Luxembourg, Belgium, " +#~ "Netherlands, Malta and Cyprus. Further " +#~ "iterations will continue to refine data" +#~ " quality. Coverage will expand beyond " +#~ "the EU and Switzerland, as well as" +#~ " to non-PDO classifications." + +#~ msgid "" +#~ "Liste des {n} appellations viticoles " +#~ "européennes cartographiées sur Open Wine " +#~ "Map, classées par pays — AOC, AOP," +#~ " IGP, DOP." +#~ msgstr "" +#~ "List of {n} European wine appellations" +#~ " mapped on Open Wine Map, organised" +#~ " by country — AOC, AOP, IGP, " +#~ "DOP." + +#~ msgid "" +#~ "Chaque appellation porte deux noms, qui" +#~ " désignent deux choses : le terme " +#~ "traditionnel que le régulateur du pays" +#~ " lui rattache, et le régime sous " +#~ "lequel elle est enregistrée. La carte" +#~ " les affiche sous la forme <em>TERME" +#~ " (RÉGIME)</em>, par exemple « DOCG " +#~ "(AOP) » ou « DOQ (AOP) ». " +#~ "Lorsqu'un pays n'a pas de terme " +#~ "propre, seul le régime apparaît. Les " +#~ "AOC suisses sont hors du régime de" +#~ " l'UE et ne portent aucune parenthèse" +#~ " ; le Royaume-Uni enregistre ses " +#~ "appellations dans son propre régime, qui" +#~ " conserve les mots PDO et PGI ;" +#~ " les eaux-de-vie françaises sont " +#~ "des indications géographiques de boissons " +#~ "spiritueuses, pas des AOP viticoles. " +#~ "Survolez un terme pour en lire la" +#~ " définition et la source." +#~ msgstr "" +#~ "Every appellation carries two names, and" +#~ " they mean two different things: the" +#~ " traditional term the country's regulator" +#~ " attaches to it, and the scheme " +#~ "it is registered under. The map " +#~ "shows them as <em>TERM (SCHEME)</em>, " +#~ "for example “DOCG (PDO)” or “DOQ " +#~ "(PDO)”. Where a country has no " +#~ "term of its own, only the scheme" +#~ " is shown. Swiss AOCs sit outside " +#~ "the EU scheme and carry no " +#~ "bracket; the United Kingdom registers " +#~ "under its own scheme, which keeps " +#~ "the words PDO and PGI; French " +#~ "eaux-de-vie are spirit-drink " +#~ "geographical indications, not wine PDOs. " +#~ "Hover over a term for its " +#~ "definition and source." + +#~ msgid "Open Wine Map — carte des appellations" +#~ msgstr "Open Wine Map — appellation map" + +#~ msgid "" +#~ "Carte des AOP et IGP viticoles " +#~ "d'Europe (AOC, DOCG, DOQ, DAC…), AOC " +#~ "suisses et IG britanniques : cépages," +#~ " styles et terroir, d'après les " +#~ "registres officiels." +#~ msgstr "" +#~ "Interactive map of Europe's wine PDOs" +#~ " and PGIs (AOC, DOCG, DOQ, DAC…), " +#~ "Swiss AOCs and UK GIs: grape " +#~ "varieties, styles and terroir from the" +#~ " official registers." + diff --git a/locale/es/LC_MESSAGES/messages.po b/locale/es/LC_MESSAGES/messages.po new file mode 100644 index 0000000..3a81c0f --- /dev/null +++ b/locale/es/LC_MESSAGES/messages.po @@ -0,0 +1,1355 @@ +# Spanish translations for Open Wine Map. Hand-seeded; review and amend in +# place. +msgid "" +msgstr "" +"Project-Id-Version: open-wine-map\n" +"Report-Msgid-Bugs-To: EMAIL@ADDRESS\n" +"POT-Creation-Date: 2026-09-08 21:10+0200\n" +"PO-Revision-Date: YEAR-MO-DA HO:MI+ZONE\n" +"Last-Translator: FULL NAME <EMAIL@ADDRESS>\n" +"Language: es\n" +"Language-Team: es <LL@li.org>\n" +"Plural-Forms: nplurals=2; plural=(n != 1);\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.18.0\n" + +#: scripts/_lib/aging_taxonomy.py:184 +msgid "Vieillissement" +msgstr "Envejecimiento" + +#: scripts/_lib/aging_taxonomy.py:185 +msgid "Prädikat" +msgstr "Prädikat" + +#: scripts/_lib/aging_taxonomy.py:186 +msgid "Sélection" +msgstr "Selección" + +#: scripts/_lib/map_template.py:46 +msgid "Open Wine Map — appellations viticoles d'Europe" +msgstr "Open Wine Map — denominaciones vitivinícolas de Europa" +#: scripts/_lib/map_template.py:47 +msgid "carte des appellations viticoles" +msgstr "Mapa de denominaciones vitivinícolas" + +#: scripts/_lib/map_template.py:49 +msgid "" +"Carte interactive des appellations viticoles d'Europe : cépages, styles " +"et terroir, d'après les registres officiels." +msgstr "Mapa interactivo de las denominaciones vitivinícolas de Europa: variedades, estilos y terruño, según los registros oficiales." +#: scripts/_lib/map_template.py:52 +msgid "Chargement…" +msgstr "Cargando…" + +#: scripts/_lib/map_template.py:53 +msgid "Recherche" +msgstr "Búsqueda" + +#: scripts/_lib/map_template.py:54 +msgid "nom d'appellation…" +msgstr "nombre de denominación…" + +#: scripts/_lib/map_template.py:55 +msgid "Recherche d'appellation…" +msgstr "Buscar denominación…" + +#: scripts/_lib/map_template.py:56 +msgid "Recherche de cépage…" +msgstr "Buscar variedad…" + +#: scripts/_lib/map_template.py:58 +msgid "Rechercher une appellation, un cépage, une région…" +msgstr "Buscar una denominación, una uva o una región…" + +#: scripts/_lib/map_template.py:59 +msgid "Cépage principal uniquement" +msgstr "Solo uva principal" + +#: scripts/_lib/map_template.py:60 +#, python-brace-format +msgid "Aucun résultat pour « {q} »" +msgstr "Sin resultados para «{q}»" + +#: scripts/_lib/map_template.py:61 +msgid "Options" +msgstr "Opciones" + +#: scripts/_lib/map_template.py:62 +msgid "Filtres actifs" +msgstr "Filtros activos" + +#: scripts/_lib/map_template.py:63 +msgid "Tout sélectionner" +msgstr "Seleccionar todo" + +#: scripts/_lib/map_template.py:65 +#, python-brace-format +msgid "{p} appellations · {s} dénominations géographiques complémentaires" +msgstr "{p} denominaciones · {s} denominaciones geográficas complementarias" + +#: scripts/_lib/map_template.py:67 +#, python-brace-format +msgid "{p} appellations" +msgstr "{p} denominaciones" + +#: scripts/_lib/map_template.py:68 +#, python-brace-format +msgid "Ouvrir la fiche de {name}" +msgstr "Abrir la ficha de {name}" + +#: scripts/_lib/map_template.py:69 +msgid "Ouvrir la fiche" +msgstr "Abrir la ficha" + +#: scripts/_lib/map_template.py:70 +msgid "Inclure les spiritueux" +msgstr "Incluir destilados" + +#: scripts/_lib/map_template.py:71 +msgid "Style de vin" +msgstr "Estilo de vino" + +#: scripts/_lib/map_template.py:72 +msgid "Classement" +msgstr "Clasificación" + +#: scripts/_lib/map_template.py:73 +msgid "Cépages principaux" +msgstr "Variedades principales" + +#: scripts/_lib/map_template.py:74 +msgid "Cépages accessoires" +msgstr "Variedades complementarias" + +#: scripts/_lib/map_template.py:75 scripts/_lib/map_template.py:189 +msgid "Cépages" +msgstr "Variedades de uva" + +#: scripts/_lib/map_template.py:76 +msgid "Région" +msgstr "Región" + +#: scripts/_lib/map_template.py:77 +msgid "Appellation" +msgstr "Denominación" + +#: scripts/_lib/map_template.py:78 +msgid "Type" +msgstr "Tipo" + +#: scripts/_lib/map_template.py:79 +msgid "Origine protégée (AOP, AOC, DOC, DO…)" +msgstr "Origen protegido (DOP, AOC, DOC, DO…)" + +#: scripts/_lib/map_template.py:80 +msgid "Indication géographique (IGP, IGT, Landwein…)" +msgstr "Indicación geográfica (IGP, IGT, Landwein…)" + +#: scripts/_lib/map_template.py:81 +msgid "AOP" +msgstr "DOP" + +#: scripts/_lib/map_template.py:82 +msgid "IGP" +msgstr "IGP" + +#: scripts/_lib/map_template.py:83 +msgid "IG spiritueux" +msgstr "IG de bebida espirituosa" + +#: scripts/_lib/map_template.py:84 +msgid "Type d'appellation" +msgstr "Tipo de denominación" + +#: scripts/_lib/map_template.py:85 +msgid "AOP (régime britannique)" +msgstr "DOP (régimen británico)" + +#: scripts/_lib/map_template.py:86 +msgid "IGP (régime britannique)" +msgstr "IGP (régimen británico)" + +#: scripts/_lib/map_template.py:87 +msgid "AOC (Suisse)" +msgstr "AOC (Suiza)" + +#: scripts/_lib/map_template.py:88 +msgid "Source" +msgstr "Fuente" + +#: scripts/_lib/map_template.py:89 +msgid "Vue" +msgstr "Vista" + +#: scripts/_lib/map_template.py:90 +msgid "Simple" +msgstr "Simple" + +#: scripts/_lib/map_template.py:91 +msgid "Avancée" +msgstr "Avanzada" + +#: scripts/_lib/map_template.py:92 +msgid "Thème" +msgstr "Tema" + +#: scripts/_lib/map_template.py:93 +msgid "Clair" +msgstr "Claro" + +#: scripts/_lib/map_template.py:94 +msgid "Sombre" +msgstr "Oscuro" + +#: scripts/_lib/map_template.py:95 +msgid "Système" +msgstr "Sistema" + +#: scripts/_lib/map_template.py:96 +msgid "Afficher les IGP" +msgstr "Mostrar las IGP" + +#: scripts/_lib/map_template.py:97 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:361 +msgid "rouge" +msgstr "tinto" + +#: scripts/_lib/map_template.py:98 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:362 +msgid "blanc" +msgstr "blanco" + +#: scripts/_lib/map_template.py:99 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:363 +msgid "rosé" +msgstr "rosado" + +#: scripts/_lib/map_template.py:100 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:364 +msgid "mousseux" +msgstr "espumoso" + +#: scripts/_lib/map_template.py:101 +msgid "moelleux / liquoreux" +msgstr "dulce / licoroso" + +#: scripts/_lib/map_template.py:102 scripts/_lib/style_taxonomy.py:345 +msgid "oxydatif" +msgstr "oxidativo" + +#: scripts/_lib/map_template.py:103 +msgid "autre" +msgstr "otro" + +#: scripts/_lib/map_template.py:104 +msgid "Réinitialiser" +msgstr "Reiniciar" + +#: scripts/_lib/map_template.py:105 +#, python-brace-format +msgid "{n} appellations" +msgstr "{n} denominaciones" + +#: scripts/_lib/map_template.py:106 +#, python-brace-format +msgid "{n} / {total} appellations" +msgstr "{n} / {total} denominaciones" + +#: scripts/_lib/map_template.py:107 +#, python-brace-format +msgid "{n} masquées dans les IGP · afficher" +msgstr "{n} ocultas en las IGP · mostrar" + +#: scripts/_lib/map_template.py:108 +msgid "Fermer" +msgstr "Cerrar" + +#: scripts/_lib/map_template.py:109 +msgid "Détails de l'appellation" +msgstr "Detalles de la denominación" + +#: scripts/_lib/map_template.py:110 +#, python-brace-format +msgid "Retirer le filtre {label}" +msgstr "Quitar filtro {label}" + +#: scripts/_lib/map_template.py:111 +msgid "Filtres et options de la carte" +msgstr "Filtros y opciones del mapa" + +#: scripts/_lib/map_template.py:112 +msgid "Langue" +msgstr "Idioma" + +#: scripts/_lib/map_template.py:113 +msgid "Carte des appellations viticoles" +msgstr "Mapa de denominaciones vitivinícolas" + +#: scripts/_lib/map_template.py:114 +msgid "Aller à la carte" +msgstr "Ir al mapa" + +#: scripts/_lib/map_template.py:115 +msgid "Styles" +msgstr "Estilos" + +#: scripts/_lib/map_template.py:116 +msgid "Variétés d'intérêt" +msgstr "Variedades de interés" + +#: scripts/_lib/map_template.py:117 +msgid "Sources" +msgstr "Fuentes" + +#: scripts/_lib/map_template.py:118 +msgid "Terroir" +msgstr "Terroir" + +#: scripts/_lib/map_template.py:119 +#, python-brace-format +msgid "Dűlők (lieux-dits) : {n}" +msgstr "Dűlők (pagos): {n}" + +#: scripts/_lib/map_template.py:120 +#, python-brace-format +msgid "Menzioni geografiche aggiuntive (crus) : {n}" +msgstr "Menciones geográficas adicionales (crus): {n}" + +#: scripts/_lib/map_template.py:121 +msgid "Facteurs naturels" +msgstr "Factores naturales" + +#: scripts/_lib/map_template.py:122 +msgid "Facteurs humains" +msgstr "Factores humanos" + +#: scripts/_lib/map_template.py:123 +msgid "Caractéristiques du produit" +msgstr "Características del producto" + +#: scripts/_lib/map_template.py:124 +msgid "Lien terroir / vin" +msgstr "Vínculo terroir / vino" + +#: scripts/_lib/map_template.py:126 +#, python-brace-format +msgid "" +"Faits dégagés du Lien au terroir par interprétation automatique — voir la" +" {source}." +msgstr "" +"Hechos extraídos de la sección de vínculo con el terruño (Lien au " +"terroir) por interpretación automática — véase el {source}." + +#: scripts/_lib/map_template.py:128 +msgid "source" +msgstr "fuente" + +#: scripts/_lib/map_template.py:129 +msgid "via Wikipedia · CC BY-SA 4.0" +msgstr "vía Wikipedia · CC BY-SA 4.0" + +#: scripts/_lib/map_template.py:131 +#, python-brace-format +msgid "Citation textuelle du Lien au terroir — voir la {source}." +msgstr "" +"Cita textual de la sección de vínculo con el terruño (Lien au terroir) — " +"véase el {source}." + +#: scripts/_lib/map_template.py:133 +msgid "à vérifier — texte source court" +msgstr "por verificar — texto fuente breve" + +#: scripts/_lib/map_template.py:135 +#, python-brace-format +msgid "" +"Ces repères décrivent l'appellation englobante {parent} — pas " +"spécifiquement cette dénomination." +msgstr "" +"Estas notas describen la denominación matriz {parent} — no " +"específicamente esta denominación." + +#: scripts/_lib/map_template.py:138 +msgid "sans région" +msgstr "sin región" + +#: scripts/_lib/map_template.py:139 +#, python-brace-format +msgid "{n} commune(s) INAO" +msgstr "{n} comuna(s) INAO" + +#: scripts/_lib/map_template.py:140 +#, python-brace-format +msgid "{n} commune(s)" +msgstr "{n} comuna(s)" + +#: scripts/_lib/map_template.py:141 +msgid "aire approchée" +msgstr "área aproximada" + +#: scripts/_lib/map_template.py:142 +msgid "aire approchée (à l'échelle communale)" +msgstr "área aproximada (nivel municipal)" + +#: scripts/_lib/map_template.py:144 +#, python-brace-format +msgid "" +"Aire approchée — pas de données parcellaires précises pour cette " +"dénomination ; polygone hérité de {umbrella}." +msgstr "" +"Área aproximada — no hay datos parcelarios precisos para esta " +"denominación; polígono heredado de {umbrella}." + +#: scripts/_lib/map_template.py:148 +msgid "" +"Aire approchée — pas de données parcellaires précises pour cette " +"dénomination ; polygone hérité de l'appellation parente." +msgstr "" +"Área aproximada — no hay datos parcelarios precisos para esta " +"denominación; polígono heredado de la denominación de origen principal." + +#: scripts/_lib/map_template.py:152 +msgid "" +"Aire approchée — pas de données parcellaires disponibles ; affichée comme" +" l'emprise de la commune où se situe la dénomination." +msgstr "" +"Área aproximada — sin datos parcelarios disponibles; se muestra como el " +"contorno del municipio que contiene la denominación." + +#: scripts/_lib/map_template.py:156 +#, python-brace-format +msgid "" +"Aire issue du lieu-dit cadastral « {lieu_dit} » (commune de {commune}, " +"{source})." +msgstr "" +"Superficie derivada del paraje catastral «{lieu_dit}» (municipio de " +"{commune}, {source})." + +#: scripts/_lib/map_template.py:159 +msgid "cadastre.data.gouv.fr" +msgstr "cadastre.data.gouv.fr" + +#: scripts/_lib/map_template.py:161 +msgid "" +"Aire approchée — reconstituée à partir des références parcellaires du " +"plan de l'aire délimitée annexé au cahier des charges ; ce n'est pas une " +"limite officielle." +msgstr "" +"Área aproximada — reconstruida a partir de las referencias de parcelas " +"del plano de la zona delimitada anexo al pliego de condiciones; no es un " +"límite oficial." + +#: scripts/_lib/map_template.py:165 +#, python-brace-format +msgid "{n} appellations à ce point" +msgstr "{n} denominaciones en este punto" + +#: scripts/_lib/map_template.py:166 +msgid "Cliquer à nouveau pour parcourir les autres" +msgstr "Vuelve a hacer clic para recorrer las demás" + +#: scripts/_lib/map_template.py:167 +msgid "Cahier des charges (BO Agri, PDF)" +msgstr "Pliego de condiciones (BO Agri, PDF)" + +#: scripts/_lib/map_template.py:168 +msgid "Cahier des charges (registre GI de l'UE, PDF)" +msgstr "Pliego de condiciones (registro de IG de la UE, PDF)" + +#: scripts/_lib/map_template.py:169 +msgid "homologué" +msgstr "homologado" + +#: scripts/_lib/map_template.py:170 +msgid "JORF" +msgstr "JORF" + +#: scripts/_lib/map_template.py:171 +msgid "Texte officiel INAO (show_texte)" +msgstr "Texto oficial INAO (show_texte)" + +#: scripts/_lib/map_template.py:172 +msgid "Fiche produit INAO" +msgstr "Ficha de producto INAO" + +#: scripts/_lib/map_template.py:173 +msgid "Site officiel de l'interprofession" +msgstr "Sitio oficial de la interprofesión" + +#: scripts/_lib/map_template.py:174 +msgid "Cahier des charges (EUR-Lex, document unique)" +msgstr "Pliego de condiciones (EUR-Lex, documento único)" + +#: scripts/_lib/map_template.py:175 +msgid "Pliego de condiciones (national, PDF)" +msgstr "Pliego de condiciones nacional (PDF)" + +#: scripts/_lib/map_template.py:176 +msgid "variétés ajoutées" +msgstr "variedades añadidas" + +#: scripts/_lib/map_template.py:177 +msgid "Cahier des charges national (PDF)" +msgstr "Pliego de condiciones nacional (PDF)" + +#: scripts/_lib/map_template.py:178 +msgid "Spécification du produit (IGP, PDF)" +msgstr "Pliego de condiciones (IGP, PDF)" + +#: scripts/_lib/map_template.py:179 +msgid "Registre régional des cépages (PDF)" +msgstr "Registro regional de variedades (PDF)" + +#: scripts/_lib/map_template.py:180 +msgid "Registre eAmbrosia (UE)" +msgstr "Registro eAmbrosia (UE)" + +#: scripts/_lib/map_template.py:181 +msgid "Numéro de dossier" +msgstr "Número de expediente" + +#: scripts/_lib/map_template.py:182 +msgid "Règlement cantonal sur la vigne et le vin" +msgstr "Reglamento cantonal de la viña y el vino" + +#: scripts/_lib/map_template.py:183 +msgid "Répertoire suisse des AOC (OFAG/BLW)" +msgstr "Registro suizo de AOC (OFAG/BLW)" + +#: scripts/_lib/map_template.py:184 +msgid "Cahier des charges (GOV.UK, PDF)" +msgstr "Pliego de condiciones (GOV.UK, PDF)" + +#: scripts/_lib/map_template.py:185 +msgid "Registre des IG du Royaume-Uni" +msgstr "Registro de IG del Reino Unido" + +#: scripts/_lib/map_template.py:186 +msgid "Légende couleurs" +msgstr "Leyenda de colores" + +#: scripts/_lib/map_template.py:187 +msgid "Bassin viticole" +msgstr "Cuenca vitícola" + +#: scripts/_lib/map_template.py:188 +msgid "Plus l'aire est petite, plus la teinte est dense." +msgstr "Cuanto menor es el área, más densa es la tonalidad." + +#: scripts/_lib/map_template.py:190 +msgid "principal — variété de la cuvée" +msgstr "principal — variedad de la mezcla" + +#: scripts/_lib/map_template.py:191 +msgid "accessoire — assemblage limité" +msgstr "accesoria — mezcla limitada" + +#: scripts/_lib/map_template.py:192 +msgid "intérêt — observation/conservation" +msgstr "de interés — observación/conservación" + +#: scripts/_lib/map_template.py:194 +msgid "" +"Le régulateur portugais (IVV) n'établit pas de distinction " +"principal/accessoire — toutes les castas autorisées sont listées ensemble" +" dans le caderno de especificações." +msgstr "" +"El regulador portugués (IVV) no distingue entre castas principales y " +"accesorias — todas las castas autorizadas se enumeran juntas en el " +"caderno de especificações." + +#: scripts/_lib/map_template.py:198 +msgid "" +"Bianchello (ou Biancame) est traité ici comme un cépage distinct, " +"conformément au disciplinare de la DOP Bianchello del Metauro ; le " +"catalogue VIVC le recense comme synonyme du Trebbiano Toscano." +msgstr "" +"Bianchello (o Biancame) se trata aquí como una variedad distinta, " +"conforme al disciplinare de la DOP Bianchello del Metauro; el catálogo " +"VIVC lo registra como sinónimo de Trebbiano Toscano." + +#: scripts/_lib/map_template.py:203 +#, python-brace-format +msgid "Open Wine Map n'a pas encore trouvé de {doc} pour cette appellation." +msgstr "Open Wine Map todavía no ha localizado un {doc} para esta denominación." + +#: scripts/_lib/map_template.py:205 +msgid "aidez-nous à le trouver" +msgstr "ayúdanos a encontrarlo" + +#: scripts/_lib/map_template.py:211 +#, python-brace-format +msgid "Délimitée par {regulator} dans son {doc}, qui autorise {grapes}." +msgstr "Delimitada por {regulator} en su {doc}, que autoriza {grapes}." + +#: scripts/_lib/map_template.py:213 +#, python-brace-format +msgid "Délimitée par {regulator} dans son {doc}{extra}." +msgstr "Delimitada por {regulator} en su {doc}{extra}." + +#: scripts/_lib/map_template.py:214 +#, python-brace-format +msgid "{names} et {n} autres cépages" +msgstr "{names} y otras {n} variedades de uva" + +#: scripts/_lib/map_template.py:215 +msgid "le registre eAmbrosia de l'UE" +msgstr "el registro eAmbrosia de la UE" + +#: scripts/_lib/map_template.py:216 +#, python-brace-format +msgid "le canton de {canton}" +msgstr "el cantón de {canton}" + +#: scripts/_lib/map_template.py:217 +msgid "(français)" +msgstr "(fuente: francés)" + +#: scripts/_lib/map_template.py:218 +msgid "Texte source en français" +msgstr "Texto fuente en francés" + +#: scripts/_lib/map_template.py:219 +msgid "(español)" +msgstr "(fuente: español)" + +#: scripts/_lib/map_template.py:220 +msgid "Texte source en espagnol" +msgstr "Texto fuente en español" + +#: scripts/_lib/map_template.py:221 +msgid "(português)" +msgstr "(fuente: portugués)" + +#: scripts/_lib/map_template.py:222 +msgid "Texte source en portugais" +msgstr "Texto fuente en portugués" + +#: scripts/_lib/map_template.py:223 +msgid "Filtres" +msgstr "Filtros" + +#: scripts/_lib/map_template.py:224 +#, python-brace-format +msgid "Traduit de {wiki} · CC BY-SA 4.0" +msgstr "Traducido de {wiki} · CC BY-SA 4.0" + +#: scripts/_lib/map_template.py:225 +msgid "Wikipédia en anglais" +msgstr "Wikipedia en inglés" + +#: scripts/_lib/map_template.py:226 +msgid "Wikipédia en français" +msgstr "Wikipedia en francés" + +#: scripts/_lib/map_template.py:227 +msgid "Wikipédia en espagnol" +msgstr "Wikipedia en español" + +#: scripts/_lib/map_template.py:228 +msgid "Wikipédia en néerlandais" +msgstr "Wikipedia en neerlandés" + +#: scripts/_lib/map_template.py:229 +msgid "Wikipédia en portugais" +msgstr "Wikipedia en portugués" + +#: scripts/_lib/map_template.py:230 +msgid "Wikipédia en croate" +msgstr "Wikipedia en croata" + +#: scripts/_lib/map_template.py:231 +msgid "Vitis International Variety Catalogue (Julius Kühn-Institut)" +msgstr "Vitis International Variety Catalogue (Julius Kühn-Institut)" + +#: scripts/_lib/map_template.py:232 +#, python-brace-format +msgid "VIVC #{id}" +msgstr "VIVC #{id}" + +#: scripts/_lib/map_template.py:233 +#, python-brace-format +msgid "Traduction automatique depuis {source}" +msgstr "Traducción automática desde {source}" + +#: scripts/_lib/map_template.py:234 +msgid "le cahier des charges" +msgstr "el pliego de condiciones" + +#: scripts/_lib/map_template.py:235 scripts/_lib/map_template.py:237 +msgid "le pliego de condiciones" +msgstr "el pliego de condiciones" + +#: scripts/_lib/map_template.py:236 scripts/_lib/map_template.py:238 +msgid "le caderno de especificações" +msgstr "el caderno de especificações" + +#: scripts/_lib/map_template.py:239 +msgid "Dénomination géographique complémentaire de" +msgstr "Subzona de" + +#: scripts/_lib/map_template.py:240 +msgid "À propos" +msgstr "Acerca de" + +#: scripts/_lib/map_template.py:241 +msgid "À propos d'Open Wine Map" +msgstr "Acerca de Open Wine Map" + +#: scripts/_lib/map_template.py:243 +msgid "" +"Carte de référence des appellations viticoles, générée automatiquement à " +"partir des registres publics : le registre de l'Union européenne des AOP " +"et IGP, les régulateurs nationaux, le répertoire fédéral suisse des AOC " +"cantonales et le registre britannique des indications géographiques." +msgstr "" +"Mapa de referencia de las denominaciones vitivinícolas, generado " +"automáticamente a partir de registros públicos: el registro de DOP e IGP " +"de la Unión Europea, los reguladores nacionales, el repertorio federal " +"suizo de AOC cantonales y el registro británico de indicaciones " +"geográficas." + +#: scripts/_lib/map_template.py:249 +msgid "" +"Une partie du texte est produite par des modèles de langage : les repères" +" de terroir sont dégagés du texte du régulateur et traduits par Claude " +"(Sonnet 4.6) ; les extraits Wikipedia des infobulles de cépages et de " +"styles sont traduits pour l'essentiel par Mistral Small 3.2, exécuté " +"localement, et pour quelques-uns par Claude ; les résumés des cahiers des" +" charges ont été traduits par un traducteur humain, à l'exception d'un " +"petit reliquat traduit automatiquement. Chaque élément porte sa propre " +"ligne d'attribution dans le panneau." +msgstr "" +"Parte del texto lo generan modelos de lenguaje: las notas de terruño se " +"extraen del texto del regulador y las traduce Claude (Sonnet 4.6); los " +"extractos de Wikipedia de las etiquetas emergentes de variedades y " +"estilos los traduce en su mayoría Mistral Small 3.2, ejecutado en local, " +"y algunos Claude; los resúmenes de los pliegos de condiciones los tradujo" +" un traductor humano, salvo un pequeño resto traducido automáticamente. " +"Cada elemento lleva su propia línea de atribución en el panel." + +#: scripts/_lib/map_template.py:258 +#, python-brace-format +msgid "Réalisé avec ♡ par {devloed}." +msgstr "Hecho con ♡ por {devloed}." + +#: scripts/_lib/map_template.py:260 +#, python-brace-format +msgid "" +"Sources : INAO ({inao}) pour les cahiers des charges et les aires " +"parcellaires françaises, IGN ({ign}) pour les contours des communes " +"françaises, le registre des indications géographiques de l'UE, les " +"régulateurs nationaux, Eurostat GISCO et Bétard 2022 pour les autres " +"pays, OpenStreetMap et CARTO pour le fond de carte, Wikipedia " +"({wikipedia}) pour quelques compléments narratifs (CC BY-SA 4.0), et VIVC" +" ({vivc}), le Vitis International Variety Catalogue du Julius Kühn-" +"Institut, pour les noms canoniques et numéros de cépage (citation Röckel " +"et al.). Tout extrait Wikipedia est signalé sur place. Détails et " +"licences dans le {readme}." +msgstr "" +"Fuentes: INAO ({inao}) para los pliegos de condiciones y los límites " +"parcelarios franceses, IGN ({ign}) para los contornos de los municipios " +"franceses, el registro de indicaciones geográficas de la UE, los " +"reguladores nacionales, Eurostat GISCO y Bétard 2022 para los demás " +"países, OpenStreetMap y CARTO para el mapa base, Wikipedia ({wikipedia}) " +"para algunos complementos narrativos (CC BY-SA 4.0) y VIVC ({vivc}), el " +"Vitis International Variety Catalogue del Julius Kühn-Institut, para los " +"nombres canónicos y los números de variedad (cita: Röckel et al.). Cada " +"extracto de Wikipedia se indica allí donde aparece. Detalles y licencias " +"en el {readme}." + +#: scripts/_lib/map_template.py:270 +#, python-brace-format +msgid "Suggestions et pull requests bienvenues sur {github}." +msgstr "Sugerencias y pull requests bienvenidos en {github}." + +#: scripts/_lib/map_template.py:271 +msgid "ticket GitHub" +msgstr "ticket en GitHub" + +#: scripts/_lib/map_template.py:272 +msgid "e-mail" +msgstr "correo" + +#: scripts/_lib/map_template.py:273 +msgid "E-mail copié dans le presse-papiers" +msgstr "Correo copiado al portapapeles" + +#: scripts/_lib/map_template.py:275 +#, python-brace-format +msgid "" +"Carte générée automatiquement — des erreurs sont possibles. Signalez-les " +"via {issue} ou {email}." +msgstr "" +"Mapa generado automáticamente — puede contener errores. Comunícalos por " +"{issue} o {email}." + +#: scripts/_lib/map_template.py:279 +#, python-brace-format +msgid "" +"{c} pays européens cartographiés ({n} appellations : {parents} " +"appellations et {subs} dénominations rattachées). Des itérations " +"supplémentaires affineront la qualité des données et étendront la " +"couverture au-delà de l'UE, de la Suisse et du Royaume-Uni." +msgstr "" +"{c} países europeos cartografiados ({n} entradas: {parents} " +"denominaciones y {subs} denominaciones vinculadas). Las próximas " +"iteraciones afinarán la calidad de los datos y ampliarán la cobertura más" +" allá de la UE, Suiza y el Reino Unido." + +#: scripts/_lib/map_template.py:284 +msgid "Toutes les appellations" +msgstr "Todas las denominaciones" + +#: scripts/_lib/map_template.py:285 +msgid "Toutes les appellations viticoles — Open Wine Map" +msgstr "Todas las denominaciones vitivinícolas — Open Wine Map" + +#: scripts/_lib/map_template.py:287 +#, python-brace-format +msgid "" +"Liste des {n} appellations viticoles cartographiées sur Open Wine Map, " +"classées par pays : AOP et IGP de l'UE avec leurs termes traditionnels " +"(AOC, DOCG, DOQ…), AOC suisses et IG britanniques." +msgstr "" +"Lista de las {n} denominaciones vitivinícolas cartografiadas en Open Wine" +" Map, por país: DOP e IGP de la UE con sus términos tradicionales (DOCa, " +"DOQ, DOCG…), AOC suizas e IG británicas." + +#: scripts/_lib/map_template.py:292 +#, python-brace-format +msgid "" +"Les {n} appellations ci-dessous sont classées par pays. Retour à la " +"{map_link}." +msgstr "" +"Las {n} denominaciones siguientes están agrupadas por país. Volver al " +"{map_link}." + +#: scripts/_lib/map_template.py:295 +msgid "carte interactive" +msgstr "mapa interactivo" + +#: scripts/_lib/map_template.py:296 +msgid "Liste des appellations par pays" +msgstr "Lista de denominaciones por país" + +#: scripts/_lib/map_template.py:297 +msgid "Appellation parente" +msgstr "Denominación matriz" + +#: scripts/_lib/map_template.py:298 +msgid "Dénominations rattachées" +msgstr "Denominaciones vinculadas" + +#: scripts/_lib/map_template.py:300 +#, python-brace-format +msgid "Parcourir la liste complète des appellations : {browse_link}." +msgstr "Explore la lista completa de denominaciones: {browse_link}." + +#: scripts/_lib/map_template.py:302 +#, python-brace-format +msgid "Données mises à jour le {date}." +msgstr "Datos actualizados el {date}." + +#: scripts/_lib/map_template.py:317 +msgid "BOURGOGNE" +msgstr "Borgoña" + +#: scripts/_lib/map_template.py:318 +msgid "BEAUJOLAIS" +msgstr "Beaujolais" + +#: scripts/_lib/map_template.py:319 +msgid "JURA" +msgstr "Jura" + +#: scripts/_lib/map_template.py:320 +msgid "SAVOIE" +msgstr "Saboya" + +#: scripts/_lib/map_template.py:321 +msgid "BUGEY" +msgstr "Bugey" + +#: scripts/_lib/map_template.py:322 +msgid "ALSACE ET EST" +msgstr "Alsacia y Este" + +#: scripts/_lib/map_template.py:323 +msgid "VAL DE LOIRE" +msgstr "Valle del Loira" + +#: scripts/_lib/map_template.py:324 +msgid "SUD-OUEST" +msgstr "Suroeste" + +#: scripts/_lib/map_template.py:325 +msgid "VALLEE DU RHÔNE" +msgstr "Valle del Ródano" + +#: scripts/_lib/map_template.py:326 +msgid "LANGUEDOC-ROUSSILLON" +msgstr "Languedoc-Rosellón" + +#: scripts/_lib/map_template.py:327 +msgid "TOULOUSE-PYRENEES" +msgstr "Tolosa-Pirineos" + +#: scripts/_lib/map_template.py:328 +msgid "PROVENCE-CORSE" +msgstr "Provenza-Córcega" + +#: scripts/_lib/map_template.py:329 +msgid "CHAMPAGNE" +msgstr "Champaña" + +#: scripts/_lib/map_template.py:330 +msgid "EAUX-DE-VIE DE CIDRE" +msgstr "Aguardientes de sidra" + +#: scripts/_lib/map_template.py:331 +msgid "VIN DOUX NATURELS" +msgstr "Vinos dulces naturales" + +#: scripts/_lib/map_template.py:332 +msgid "COGNAC" +msgstr "Coñac" + +#: scripts/_lib/map_template.py:333 +msgid "ARMAGNAC" +msgstr "Armañac" + +#: scripts/_lib/map_template.py:334 +msgid "RHUM" +msgstr "Ron" + +#: scripts/_lib/map_template.py:374 +msgid "France" +msgstr "Francia" + +#: scripts/_lib/map_template.py:375 +msgid "Espagne" +msgstr "España" + +#: scripts/_lib/map_template.py:376 +msgid "Portugal" +msgstr "Portugal" + +#: scripts/_lib/map_template.py:377 +msgid "Italie" +msgstr "Italia" + +#: scripts/_lib/map_template.py:378 +msgid "Autriche" +msgstr "Austria" + +#: scripts/_lib/map_template.py:379 +msgid "Slovénie" +msgstr "Eslovenia" + +#: scripts/_lib/map_template.py:380 +msgid "Croatie" +msgstr "Croacia" + +#: scripts/_lib/map_template.py:381 +msgid "Hongrie" +msgstr "Hungría" + +#: scripts/_lib/map_template.py:382 +msgid "Roumanie" +msgstr "Rumanía" + +#: scripts/_lib/map_template.py:383 +msgid "Bulgarie" +msgstr "Bulgaria" + +#: scripts/_lib/map_template.py:384 +msgid "Grèce" +msgstr "Grecia" + +#: scripts/_lib/map_template.py:385 +msgid "Allemagne" +msgstr "Alemania" + +#: scripts/_lib/map_template.py:386 +msgid "Slovaquie" +msgstr "Eslovaquia" + +#: scripts/_lib/map_template.py:387 +msgid "Suisse" +msgstr "Suiza" + +#: scripts/_lib/map_template.py:388 +msgid "Tchéquie" +msgstr "Chequia" + +#: scripts/_lib/map_template.py:389 +msgid "Luxembourg" +msgstr "Luxemburgo" + +#: scripts/_lib/map_template.py:390 +msgid "Belgique" +msgstr "Bélgica" + +#: scripts/_lib/map_template.py:391 +msgid "Pays-Bas" +msgstr "Países Bajos" + +#: scripts/_lib/map_template.py:392 +msgid "Malte" +msgstr "Malta" + +#: scripts/_lib/map_template.py:393 +msgid "Chypre" +msgstr "Chipre" + +#: scripts/_lib/map_template.py:394 +msgid "Royaume-Uni" +msgstr "Reino Unido" + +#: scripts/_lib/style_taxonomy.py:332 scripts/_lib/style_taxonomy.py:365 +msgid "doux" +msgstr "dulce" + +#: scripts/_lib/style_taxonomy.py:333 scripts/_lib/style_taxonomy.py:366 +msgid "autres" +msgstr "otros" + +#: scripts/_lib/style_taxonomy.py:334 +msgid "clairet" +msgstr "clarete" + +#: scripts/_lib/style_taxonomy.py:334 +msgid "primeur" +msgstr "primeur" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "mousseux de qualité" +msgstr "espumoso de calidad" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "crémant" +msgstr "crémant" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "méthode ancestrale" +msgstr "método ancestral" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode traditionnelle" +msgstr "método tradicional" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode Charmat" +msgstr "método Charmat" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode dioise" +msgstr "méthode dioise" + +#: scripts/_lib/style_taxonomy.py:337 +msgid "pétillant" +msgstr "de aguja" + +#: scripts/_lib/style_taxonomy.py:338 +msgid "vin muté" +msgstr "vino encabezado" + +#: scripts/_lib/style_taxonomy.py:338 +msgid "vendanges tardives (catégorie)" +msgstr "vendimia tardía (categoría)" + +#: scripts/_lib/style_taxonomy.py:339 +msgid "vin de raisins passerillés" +msgstr "vino de uvas pasificadas" + +#: scripts/_lib/style_taxonomy.py:340 +msgid "demi-doux" +msgstr "semidulce" + +#: scripts/_lib/style_taxonomy.py:340 +msgid "vin de glace" +msgstr "vino de hielo" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "vin doux naturel" +msgstr "vino dulce natural" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "vin de liqueur" +msgstr "vino de licor" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "mistelle" +msgstr "mistela" + +#: scripts/_lib/style_taxonomy.py:342 +msgid "vendanges tardives" +msgstr "vendimia tardía" + +#: scripts/_lib/style_taxonomy.py:342 +msgid "uvas sobremaduradas" +msgstr "uvas sobremaduradas" + +#: scripts/_lib/style_taxonomy.py:343 +msgid "vin naturellement doux" +msgstr "naturalmente dulce" + +#: scripts/_lib/style_taxonomy.py:343 +msgid "grains nobles" +msgstr "granos nobles" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "vin de paille" +msgstr "vino de paja" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "uvas pasificadas" +msgstr "uvas pasificadas" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "vin santo" +msgstr "vin santo" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "tranquille" +msgstr "tranquilo" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "sur lie" +msgstr "sur lie" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "sec" +msgstr "seco" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "demi-sec" +msgstr "semiseco" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "rancio" +msgstr "rancio" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "vin jaune" +msgstr "vino amarillo" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "generoso" +msgstr "generoso" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "fino" +msgstr "fino" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "manzanilla" +msgstr "manzanilla" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "amontillado" +msgstr "amontillado" + +#: scripts/_lib/style_taxonomy.py:348 +msgid "oloroso" +msgstr "oloroso" + +#: scripts/_lib/style_taxonomy.py:348 +msgid "palo cortado" +msgstr "palo cortado" + +#~ msgid "carte des appellations françaises" +#~ msgstr "mapa de denominaciones francesas" + +#~ msgid "" +#~ "Priorité actuelle : affiner la " +#~ "couverture française — précision des " +#~ "aires, qualité des extraits, climats et" +#~ " lieux-dits. L'extension à d'autres " +#~ "pays viticoles viendra ensuite." +#~ msgstr "" +#~ "Prioridad actual: afinar la cobertura " +#~ "francesa — precisión de las áreas, " +#~ "calidad de los extractos, climats y " +#~ "lieux-dits. La extensión a otros " +#~ "países vitivinícolas vendrá después." + +#~ msgid "anglais" +#~ msgstr "" + +#~ msgid "français" +#~ msgstr "(francés)" + +#~ msgid "espagnol" +#~ msgstr "(español)" + +#~ msgid "néerlandais" +#~ msgstr "" + +#~ msgid "" +#~ "Couverture actuelle : France et une " +#~ "première version de l'Espagne. Prochaines " +#~ "étapes : amélioration continue de la " +#~ "qualité des données, puis ajout du " +#~ "Portugal." +#~ msgstr "" +#~ "Cobertura actual: Francia y una primera" +#~ " versión de España. Próximos pasos: " +#~ "mejora continua de la calidad de " +#~ "los datos y, después, Portugal." + +#~ msgid "moelleux" +#~ msgstr "semidulce" + +#~ msgid "" +#~ "Carte interactive des appellations viticoles" +#~ " européennes (AOC, AOP, IGP, DOP) :" +#~ " cépages, styles et terroir, d'après " +#~ "les registres officiels (INAO, EUR-Lex)." +#~ msgstr "" +#~ "Mapa interactivo de las denominaciones " +#~ "vitivinícolas europeas (DOP, IGP): variedades," +#~ " estilos y terruño, según los " +#~ "registros oficiales (INAO, EUR-Lex)." + +#~ msgid "AOC / AOP" +#~ msgstr "AOC / AOP" + +#~ msgid "{n} dans IGP masquées — afficher" +#~ msgstr "{n} ocultos en IGP — mostrar" + +#~ msgid "" +#~ "Carte de référence des appellations " +#~ "viticoles (AOC, AOP, IGP, DOP), générée" +#~ " automatiquement à partir des données " +#~ "publiques." +#~ msgstr "" +#~ "Mapa de referencia abierto de las " +#~ "denominaciones vitivinícolas (AOC, AOP, IGP," +#~ " DOP), generado automáticamente a partir" +#~ " de datos públicos." + +#~ msgid "" +#~ "Sources : INAO ({inao}) pour les " +#~ "cahiers des charges et les aires " +#~ "parcellaires, IGN ({ign}) pour le fond" +#~ " cartographique, Wikipedia ({wikipedia}) pour " +#~ "quelques compléments narratifs (CC BY-SA" +#~ " 4.0), VIVC ({vivc}) — Vitis " +#~ "International Variety Catalogue, Julius " +#~ "Kühn-Institut — pour les noms " +#~ "canoniques et numéros de cépage " +#~ "(citation Röckel et al.). Tout extrait" +#~ " Wikipedia est signalé sur place. " +#~ "Détails et licences dans le {readme}." +#~ msgstr "" +#~ "Fuentes: INAO ({inao}) para los pliegos" +#~ " de condiciones y los límites " +#~ "parcelarios, IGN ({ign}) para la " +#~ "cartografía de base, Wikipedia ({wikipedia})" +#~ " para algunos complementos narrativos (CC" +#~ " BY-SA 4.0), VIVC ({vivc}) — " +#~ "Vitis International Variety Catalogue, Julius" +#~ " Kühn-Institut — para los nombres " +#~ "canónicos y números de variedad (cita" +#~ " Röckel et al.). Cada extracto de " +#~ "Wikipedia se indica en el sitio " +#~ "donde aparece. Detalles y licencias en" +#~ " el {readme}." + +#~ msgid "" +#~ "20 pays européens cartographiés : " +#~ "France, Espagne, Portugal, Italie, Autriche," +#~ " Allemagne, Suisse, Slovénie, Croatie, " +#~ "Hongrie, Roumanie, Bulgarie, Grèce, Slovaquie," +#~ " Tchéquie, Luxembourg, Belgique, Pays-Bas," +#~ " Malte et Chypre. Des itérations " +#~ "supplémentaires viendront affiner la qualité" +#~ " des données. La couverture sera " +#~ "étendue au-delà de l'UE et de " +#~ "la Suisse, ainsi qu'aux classifications " +#~ "hors AOP." +#~ msgstr "" +#~ "20 países europeos ya cartografiados: " +#~ "Francia, España, Portugal, Italia, Austria," +#~ " Alemania, Suiza, Eslovenia, Croacia, " +#~ "Hungría, Rumanía, Bulgaria, Grecia, " +#~ "Eslovaquia, República Checa, Luxemburgo, " +#~ "Bélgica, Países Bajos, Malta y Chipre." +#~ " Se seguirá mejorando la calidad de" +#~ " los datos. La cobertura se ampliará" +#~ " más allá de la UE y Suiza, " +#~ "así como a clasificaciones no DOP." + +#~ msgid "" +#~ "Liste des {n} appellations viticoles " +#~ "européennes cartographiées sur Open Wine " +#~ "Map, classées par pays — AOC, AOP," +#~ " IGP, DOP." +#~ msgstr "" +#~ "Lista de las {n} denominaciones " +#~ "vitivinícolas europeas cartografiadas en Open" +#~ " Wine Map, organizadas por país — " +#~ "DOP, IGP." + +#~ msgid "" +#~ "Chaque appellation porte deux noms, qui" +#~ " désignent deux choses : le terme " +#~ "traditionnel que le régulateur du pays" +#~ " lui rattache, et le régime sous " +#~ "lequel elle est enregistrée. La carte" +#~ " les affiche sous la forme <em>TERME" +#~ " (RÉGIME)</em>, par exemple « DOCG " +#~ "(AOP) » ou « DOQ (AOP) ». " +#~ "Lorsqu'un pays n'a pas de terme " +#~ "propre, seul le régime apparaît. Les " +#~ "AOC suisses sont hors du régime de" +#~ " l'UE et ne portent aucune parenthèse" +#~ " ; le Royaume-Uni enregistre ses " +#~ "appellations dans son propre régime, qui" +#~ " conserve les mots PDO et PGI ;" +#~ " les eaux-de-vie françaises sont " +#~ "des indications géographiques de boissons " +#~ "spiritueuses, pas des AOP viticoles. " +#~ "Survolez un terme pour en lire la" +#~ " définition et la source." +#~ msgstr "" +#~ "Cada denominación lleva dos nombres, y" +#~ " designan dos cosas distintas: el " +#~ "término tradicional que el regulador del" +#~ " país le asocia y el régimen " +#~ "bajo el que está registrada. El " +#~ "mapa los muestra como <em>TÉRMINO " +#~ "(RÉGIMEN)</em>, por ejemplo «DOCG (DOP)» " +#~ "o «DOQ (DOP)». Cuando un país no" +#~ " tiene un término propio, solo " +#~ "aparece el régimen. Las AOC suizas " +#~ "quedan fuera del régimen de la UE" +#~ " y no llevan paréntesis; el Reino " +#~ "Unido registra sus denominaciones en un" +#~ " régimen propio, que conserva las " +#~ "palabras PDO y PGI; los aguardientes " +#~ "franceses (eaux-de-vie) son indicaciones" +#~ " geográficas de bebidas espirituosas, no" +#~ " DOP vinícolas. Pase el cursor sobre" +#~ " un término para ver su definición" +#~ " y su fuente." + +#~ msgid "Open Wine Map — carte des appellations" +#~ msgstr "Open Wine Map — mapa de denominaciones" + +#~ msgid "" +#~ "Carte des AOP et IGP viticoles " +#~ "d'Europe (AOC, DOCG, DOQ, DAC…), AOC " +#~ "suisses et IG britanniques : cépages," +#~ " styles et terroir, d'après les " +#~ "registres officiels." +#~ msgstr "" +#~ "Mapa de las DOP e IGP vinícolas" +#~ " de Europa (DOCa, DOQ, DOCG, AOC…)," +#~ " AOC suizas e IG británicas: " +#~ "variedades, estilos y terruño según los" +#~ " registros oficiales." + diff --git a/locale/fr/LC_MESSAGES/messages.po b/locale/fr/LC_MESSAGES/messages.po new file mode 100644 index 0000000..946c177 --- /dev/null +++ b/locale/fr/LC_MESSAGES/messages.po @@ -0,0 +1,1345 @@ +# French translations for Open Wine Map (source language; identity catalog). +msgid "" +msgstr "" +"Project-Id-Version: open-wine-map\n" +"Report-Msgid-Bugs-To: EMAIL@ADDRESS\n" +"POT-Creation-Date: 2026-09-08 21:10+0200\n" +"PO-Revision-Date: YEAR-MO-DA HO:MI+ZONE\n" +"Last-Translator: FULL NAME <EMAIL@ADDRESS>\n" +"Language: fr\n" +"Language-Team: fr <LL@li.org>\n" +"Plural-Forms: nplurals=2; plural=(n > 1);\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.18.0\n" + +#: scripts/_lib/aging_taxonomy.py:184 +msgid "Vieillissement" +msgstr "Vieillissement" + +#: scripts/_lib/aging_taxonomy.py:185 +msgid "Prädikat" +msgstr "Prädikat" + +#: scripts/_lib/aging_taxonomy.py:186 +msgid "Sélection" +msgstr "Sélection" + +#: scripts/_lib/map_template.py:46 +msgid "Open Wine Map — appellations viticoles d'Europe" +msgstr "Open Wine Map — appellations viticoles d'Europe" +#: scripts/_lib/map_template.py:47 +msgid "carte des appellations viticoles" +msgstr "carte des appellations viticoles" + +#: scripts/_lib/map_template.py:49 +msgid "" +"Carte interactive des appellations viticoles d'Europe : cépages, styles " +"et terroir, d'après les registres officiels." +msgstr "Carte interactive des appellations viticoles d'Europe : cépages, styles et terroir, d'après les registres officiels." +#: scripts/_lib/map_template.py:52 +msgid "Chargement…" +msgstr "Chargement…" + +#: scripts/_lib/map_template.py:53 +msgid "Recherche" +msgstr "Recherche" + +#: scripts/_lib/map_template.py:54 +msgid "nom d'appellation…" +msgstr "nom d'appellation…" + +#: scripts/_lib/map_template.py:55 +msgid "Recherche d'appellation…" +msgstr "Recherche d'appellation…" + +#: scripts/_lib/map_template.py:56 +msgid "Recherche de cépage…" +msgstr "Recherche de cépage…" + +#: scripts/_lib/map_template.py:58 +msgid "Rechercher une appellation, un cépage, une région…" +msgstr "Rechercher une appellation, un cépage, une région…" + +#: scripts/_lib/map_template.py:59 +msgid "Cépage principal uniquement" +msgstr "Cépage principal uniquement" + +#: scripts/_lib/map_template.py:60 +#, python-brace-format +msgid "Aucun résultat pour « {q} »" +msgstr "" + +#: scripts/_lib/map_template.py:61 +msgid "Options" +msgstr "Options" + +#: scripts/_lib/map_template.py:62 +msgid "Filtres actifs" +msgstr "Filtres actifs" + +#: scripts/_lib/map_template.py:63 +msgid "Tout sélectionner" +msgstr "Tout sélectionner" + +#: scripts/_lib/map_template.py:65 +#, python-brace-format +msgid "{p} appellations · {s} dénominations géographiques complémentaires" +msgstr "{p} appellations · {s} dénominations géographiques complémentaires" + +#: scripts/_lib/map_template.py:67 +#, python-brace-format +msgid "{p} appellations" +msgstr "{p} appellations" + +#: scripts/_lib/map_template.py:68 +#, python-brace-format +msgid "Ouvrir la fiche de {name}" +msgstr "Ouvrir la fiche de {name}" + +#: scripts/_lib/map_template.py:69 +msgid "Ouvrir la fiche" +msgstr "Ouvrir la fiche" + +#: scripts/_lib/map_template.py:70 +msgid "Inclure les spiritueux" +msgstr "Inclure les spiritueux" + +#: scripts/_lib/map_template.py:71 +msgid "Style de vin" +msgstr "Style de vin" + +#: scripts/_lib/map_template.py:72 +msgid "Classement" +msgstr "Classement" + +#: scripts/_lib/map_template.py:73 +msgid "Cépages principaux" +msgstr "Cépages principaux" + +#: scripts/_lib/map_template.py:74 +msgid "Cépages accessoires" +msgstr "Cépages accessoires" + +#: scripts/_lib/map_template.py:75 scripts/_lib/map_template.py:189 +msgid "Cépages" +msgstr "Cépages" + +#: scripts/_lib/map_template.py:76 +msgid "Région" +msgstr "Région" + +#: scripts/_lib/map_template.py:77 +msgid "Appellation" +msgstr "Appellation" + +#: scripts/_lib/map_template.py:78 +msgid "Type" +msgstr "Type" + +#: scripts/_lib/map_template.py:79 +msgid "Origine protégée (AOP, AOC, DOC, DO…)" +msgstr "Origine protégée (AOP, AOC, DOC, DO…)" + +#: scripts/_lib/map_template.py:80 +msgid "Indication géographique (IGP, IGT, Landwein…)" +msgstr "Indication géographique (IGP, IGT, Landwein…)" + +#: scripts/_lib/map_template.py:81 +msgid "AOP" +msgstr "AOP" + +#: scripts/_lib/map_template.py:82 +msgid "IGP" +msgstr "IGP" + +#: scripts/_lib/map_template.py:83 +msgid "IG spiritueux" +msgstr "IG spiritueux" + +#: scripts/_lib/map_template.py:84 +msgid "Type d'appellation" +msgstr "Type d'appellation" + +#: scripts/_lib/map_template.py:85 +msgid "AOP (régime britannique)" +msgstr "" + +#: scripts/_lib/map_template.py:86 +msgid "IGP (régime britannique)" +msgstr "" + +#: scripts/_lib/map_template.py:87 +msgid "AOC (Suisse)" +msgstr "" + +#: scripts/_lib/map_template.py:88 +msgid "Source" +msgstr "Source" + +#: scripts/_lib/map_template.py:89 +msgid "Vue" +msgstr "Vue" + +#: scripts/_lib/map_template.py:90 +msgid "Simple" +msgstr "Simple" + +#: scripts/_lib/map_template.py:91 +msgid "Avancée" +msgstr "Avancée" + +#: scripts/_lib/map_template.py:92 +msgid "Thème" +msgstr "Thème" + +#: scripts/_lib/map_template.py:93 +msgid "Clair" +msgstr "Clair" + +#: scripts/_lib/map_template.py:94 +msgid "Sombre" +msgstr "Sombre" + +#: scripts/_lib/map_template.py:95 +msgid "Système" +msgstr "Système" + +#: scripts/_lib/map_template.py:96 +msgid "Afficher les IGP" +msgstr "Afficher les IGP" + +#: scripts/_lib/map_template.py:97 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:361 +msgid "rouge" +msgstr "rouge" + +#: scripts/_lib/map_template.py:98 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:362 +msgid "blanc" +msgstr "blanc" + +#: scripts/_lib/map_template.py:99 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:363 +msgid "rosé" +msgstr "rosé" + +#: scripts/_lib/map_template.py:100 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:364 +msgid "mousseux" +msgstr "mousseux" + +#: scripts/_lib/map_template.py:101 +msgid "moelleux / liquoreux" +msgstr "moelleux / liquoreux" + +#: scripts/_lib/map_template.py:102 scripts/_lib/style_taxonomy.py:345 +msgid "oxydatif" +msgstr "oxydatif" + +#: scripts/_lib/map_template.py:103 +msgid "autre" +msgstr "autre" + +#: scripts/_lib/map_template.py:104 +msgid "Réinitialiser" +msgstr "Réinitialiser" + +#: scripts/_lib/map_template.py:105 +#, python-brace-format +msgid "{n} appellations" +msgstr "{n} appellations" + +#: scripts/_lib/map_template.py:106 +#, python-brace-format +msgid "{n} / {total} appellations" +msgstr "{n} / {total} appellations" + +#: scripts/_lib/map_template.py:107 +#, python-brace-format +msgid "{n} masquées dans les IGP · afficher" +msgstr "{n} masquées dans les IGP · afficher" + +#: scripts/_lib/map_template.py:108 +msgid "Fermer" +msgstr "Fermer" + +#: scripts/_lib/map_template.py:109 +msgid "Détails de l'appellation" +msgstr "Détails de l'appellation" + +#: scripts/_lib/map_template.py:110 +#, python-brace-format +msgid "Retirer le filtre {label}" +msgstr "Retirer le filtre {label}" + +#: scripts/_lib/map_template.py:111 +msgid "Filtres et options de la carte" +msgstr "Filtres et options de la carte" + +#: scripts/_lib/map_template.py:112 +msgid "Langue" +msgstr "Langue" + +#: scripts/_lib/map_template.py:113 +msgid "Carte des appellations viticoles" +msgstr "Carte des appellations viticoles" + +#: scripts/_lib/map_template.py:114 +msgid "Aller à la carte" +msgstr "Aller à la carte" + +#: scripts/_lib/map_template.py:115 +msgid "Styles" +msgstr "Styles" + +#: scripts/_lib/map_template.py:116 +msgid "Variétés d'intérêt" +msgstr "Variétés d'intérêt" + +#: scripts/_lib/map_template.py:117 +msgid "Sources" +msgstr "Sources" + +#: scripts/_lib/map_template.py:118 +msgid "Terroir" +msgstr "Terroir" + +#: scripts/_lib/map_template.py:119 +#, python-brace-format +msgid "Dűlők (lieux-dits) : {n}" +msgstr "Dűlők (lieux-dits) : {n}" + +#: scripts/_lib/map_template.py:120 +#, python-brace-format +msgid "Menzioni geografiche aggiuntive (crus) : {n}" +msgstr "Mentions géographiques complémentaires (crus) : {n}" + +#: scripts/_lib/map_template.py:121 +msgid "Facteurs naturels" +msgstr "Facteurs naturels" + +#: scripts/_lib/map_template.py:122 +msgid "Facteurs humains" +msgstr "Facteurs humains" + +#: scripts/_lib/map_template.py:123 +msgid "Caractéristiques du produit" +msgstr "Caractéristiques du produit" + +#: scripts/_lib/map_template.py:124 +msgid "Lien terroir / vin" +msgstr "Lien terroir / vin" + +#: scripts/_lib/map_template.py:126 +#, python-brace-format +msgid "" +"Faits dégagés du Lien au terroir par interprétation automatique — voir la" +" {source}." +msgstr "" +"Faits dégagés du Lien au terroir par interprétation automatique — voir la" +" {source}." + +#: scripts/_lib/map_template.py:128 +msgid "source" +msgstr "source" + +#: scripts/_lib/map_template.py:129 +msgid "via Wikipedia · CC BY-SA 4.0" +msgstr "via Wikipedia · CC BY-SA 4.0" + +#: scripts/_lib/map_template.py:131 +#, python-brace-format +msgid "Citation textuelle du Lien au terroir — voir la {source}." +msgstr "Citation textuelle du Lien au terroir — voir la {source}." + +#: scripts/_lib/map_template.py:133 +msgid "à vérifier — texte source court" +msgstr "" + +#: scripts/_lib/map_template.py:135 +#, python-brace-format +msgid "" +"Ces repères décrivent l'appellation englobante {parent} — pas " +"spécifiquement cette dénomination." +msgstr "" + +#: scripts/_lib/map_template.py:138 +msgid "sans région" +msgstr "sans région" + +#: scripts/_lib/map_template.py:139 +#, python-brace-format +msgid "{n} commune(s) INAO" +msgstr "{n} commune(s) INAO" + +#: scripts/_lib/map_template.py:140 +#, python-brace-format +msgid "{n} commune(s)" +msgstr "{n} commune(s)" + +#: scripts/_lib/map_template.py:141 +msgid "aire approchée" +msgstr "aire approchée" + +#: scripts/_lib/map_template.py:142 +msgid "aire approchée (à l'échelle communale)" +msgstr "aire approchée (à l'échelle communale)" + +#: scripts/_lib/map_template.py:144 +#, python-brace-format +msgid "" +"Aire approchée — pas de données parcellaires précises pour cette " +"dénomination ; polygone hérité de {umbrella}." +msgstr "" +"Aire approchée — pas de données parcellaires précises pour cette " +"dénomination ; polygone hérité de {umbrella}." + +#: scripts/_lib/map_template.py:148 +msgid "" +"Aire approchée — pas de données parcellaires précises pour cette " +"dénomination ; polygone hérité de l'appellation parente." +msgstr "" +"Aire approchée — pas de données parcellaires précises pour cette " +"dénomination ; polygone hérité de l'appellation parente." + +#: scripts/_lib/map_template.py:152 +msgid "" +"Aire approchée — pas de données parcellaires disponibles ; affichée comme" +" l'emprise de la commune où se situe la dénomination." +msgstr "" +"Aire approchée — pas de données parcellaires disponibles ; affichée comme" +" l'emprise de la commune où se situe la dénomination." + +#: scripts/_lib/map_template.py:156 +#, python-brace-format +msgid "" +"Aire issue du lieu-dit cadastral « {lieu_dit} » (commune de {commune}, " +"{source})." +msgstr "" + +#: scripts/_lib/map_template.py:159 +msgid "cadastre.data.gouv.fr" +msgstr "" + +#: scripts/_lib/map_template.py:161 +msgid "" +"Aire approchée — reconstituée à partir des références parcellaires du " +"plan de l'aire délimitée annexé au cahier des charges ; ce n'est pas une " +"limite officielle." +msgstr "" +"Aire approchée — reconstituée à partir des références parcellaires du " +"plan de l'aire délimitée annexé au cahier des charges ; ce n'est pas une " +"limite officielle." + +#: scripts/_lib/map_template.py:165 +#, python-brace-format +msgid "{n} appellations à ce point" +msgstr "{n} appellations à ce point" + +#: scripts/_lib/map_template.py:166 +msgid "Cliquer à nouveau pour parcourir les autres" +msgstr "Cliquer à nouveau pour parcourir les autres" + +#: scripts/_lib/map_template.py:167 +msgid "Cahier des charges (BO Agri, PDF)" +msgstr "Cahier des charges (BO Agri, PDF)" + +#: scripts/_lib/map_template.py:168 +msgid "Cahier des charges (registre GI de l'UE, PDF)" +msgstr "" + +#: scripts/_lib/map_template.py:169 +msgid "homologué" +msgstr "homologué" + +#: scripts/_lib/map_template.py:170 +msgid "JORF" +msgstr "JORF" + +#: scripts/_lib/map_template.py:171 +msgid "Texte officiel INAO (show_texte)" +msgstr "Texte officiel INAO (show_texte)" + +#: scripts/_lib/map_template.py:172 +msgid "Fiche produit INAO" +msgstr "Fiche produit INAO" + +#: scripts/_lib/map_template.py:173 +msgid "Site officiel de l'interprofession" +msgstr "Site officiel de l'interprofession" + +#: scripts/_lib/map_template.py:174 +msgid "Cahier des charges (EUR-Lex, document unique)" +msgstr "Cahier des charges (EUR-Lex, document unique)" + +#: scripts/_lib/map_template.py:175 +msgid "Pliego de condiciones (national, PDF)" +msgstr "Pliego de condiciones (national, PDF)" + +#: scripts/_lib/map_template.py:176 +msgid "variétés ajoutées" +msgstr "variétés ajoutées" + +#: scripts/_lib/map_template.py:177 +msgid "Cahier des charges national (PDF)" +msgstr "Cahier des charges national (PDF)" + +#: scripts/_lib/map_template.py:178 +msgid "Spécification du produit (IGP, PDF)" +msgstr "Spécification du produit (IGP, PDF)" + +#: scripts/_lib/map_template.py:179 +msgid "Registre régional des cépages (PDF)" +msgstr "Registre régional des cépages (PDF)" + +#: scripts/_lib/map_template.py:180 +msgid "Registre eAmbrosia (UE)" +msgstr "Registre eAmbrosia (UE)" + +#: scripts/_lib/map_template.py:181 +msgid "Numéro de dossier" +msgstr "Numéro de dossier" + +#: scripts/_lib/map_template.py:182 +msgid "Règlement cantonal sur la vigne et le vin" +msgstr "Règlement cantonal sur la vigne et le vin" + +#: scripts/_lib/map_template.py:183 +msgid "Répertoire suisse des AOC (OFAG/BLW)" +msgstr "Répertoire suisse des AOC (OFAG/BLW)" + +#: scripts/_lib/map_template.py:184 +msgid "Cahier des charges (GOV.UK, PDF)" +msgstr "Cahier des charges (GOV.UK, PDF)" + +#: scripts/_lib/map_template.py:185 +msgid "Registre des IG du Royaume-Uni" +msgstr "Registre des IG du Royaume-Uni" + +#: scripts/_lib/map_template.py:186 +msgid "Légende couleurs" +msgstr "Légende couleurs" + +#: scripts/_lib/map_template.py:187 +msgid "Bassin viticole" +msgstr "Bassin viticole" + +#: scripts/_lib/map_template.py:188 +msgid "Plus l'aire est petite, plus la teinte est dense." +msgstr "Plus l'aire est petite, plus la teinte est dense." + +#: scripts/_lib/map_template.py:190 +msgid "principal — variété de la cuvée" +msgstr "principal — variété de la cuvée" + +#: scripts/_lib/map_template.py:191 +msgid "accessoire — assemblage limité" +msgstr "accessoire — assemblage limité" + +#: scripts/_lib/map_template.py:192 +msgid "intérêt — observation/conservation" +msgstr "intérêt — observation/conservation" + +#: scripts/_lib/map_template.py:194 +msgid "" +"Le régulateur portugais (IVV) n'établit pas de distinction " +"principal/accessoire — toutes les castas autorisées sont listées ensemble" +" dans le caderno de especificações." +msgstr "" + +#: scripts/_lib/map_template.py:198 +msgid "" +"Bianchello (ou Biancame) est traité ici comme un cépage distinct, " +"conformément au disciplinare de la DOP Bianchello del Metauro ; le " +"catalogue VIVC le recense comme synonyme du Trebbiano Toscano." +msgstr "" + +#: scripts/_lib/map_template.py:203 +#, python-brace-format +msgid "Open Wine Map n'a pas encore trouvé de {doc} pour cette appellation." +msgstr "" + +#: scripts/_lib/map_template.py:205 +msgid "aidez-nous à le trouver" +msgstr "aidez-nous à le trouver" + +#: scripts/_lib/map_template.py:211 +#, python-brace-format +msgid "Délimitée par {regulator} dans son {doc}, qui autorise {grapes}." +msgstr "" + +#: scripts/_lib/map_template.py:213 +#, python-brace-format +msgid "Délimitée par {regulator} dans son {doc}{extra}." +msgstr "" + +#: scripts/_lib/map_template.py:214 +#, python-brace-format +msgid "{names} et {n} autres cépages" +msgstr "" + +#: scripts/_lib/map_template.py:215 +#, fuzzy +msgid "le registre eAmbrosia de l'UE" +msgstr "Registre eAmbrosia (UE)" + +#: scripts/_lib/map_template.py:216 +#, python-brace-format +msgid "le canton de {canton}" +msgstr "" + +#: scripts/_lib/map_template.py:217 +msgid "(français)" +msgstr "(français)" + +#: scripts/_lib/map_template.py:218 +msgid "Texte source en français" +msgstr "Texte source en français" + +#: scripts/_lib/map_template.py:219 +msgid "(español)" +msgstr "(español)" + +#: scripts/_lib/map_template.py:220 +msgid "Texte source en espagnol" +msgstr "Texte source en espagnol" + +#: scripts/_lib/map_template.py:221 +msgid "(português)" +msgstr "" + +#: scripts/_lib/map_template.py:222 +msgid "Texte source en portugais" +msgstr "Texte source en portugais" + +#: scripts/_lib/map_template.py:223 +msgid "Filtres" +msgstr "Filtres" + +#: scripts/_lib/map_template.py:224 +#, python-brace-format +msgid "Traduit de {wiki} · CC BY-SA 4.0" +msgstr "Traduit de {wiki} · CC BY-SA 4.0" + +#: scripts/_lib/map_template.py:225 +msgid "Wikipédia en anglais" +msgstr "Wikipédia en anglais" + +#: scripts/_lib/map_template.py:226 +msgid "Wikipédia en français" +msgstr "Wikipédia en français" + +#: scripts/_lib/map_template.py:227 +msgid "Wikipédia en espagnol" +msgstr "Wikipédia en espagnol" + +#: scripts/_lib/map_template.py:228 +msgid "Wikipédia en néerlandais" +msgstr "Wikipédia en néerlandais" + +#: scripts/_lib/map_template.py:229 +msgid "Wikipédia en portugais" +msgstr "Wikipédia en portugais" + +#: scripts/_lib/map_template.py:230 +msgid "Wikipédia en croate" +msgstr "Wikipédia en croate" + +#: scripts/_lib/map_template.py:231 +msgid "Vitis International Variety Catalogue (Julius Kühn-Institut)" +msgstr "" + +#: scripts/_lib/map_template.py:232 +#, python-brace-format +msgid "VIVC #{id}" +msgstr "" + +#: scripts/_lib/map_template.py:233 +#, python-brace-format +msgid "Traduction automatique depuis {source}" +msgstr "Traduction automatique depuis {source}" + +#: scripts/_lib/map_template.py:234 +msgid "le cahier des charges" +msgstr "le cahier des charges" + +#: scripts/_lib/map_template.py:235 scripts/_lib/map_template.py:237 +msgid "le pliego de condiciones" +msgstr "le pliego de condiciones" + +#: scripts/_lib/map_template.py:236 scripts/_lib/map_template.py:238 +msgid "le caderno de especificações" +msgstr "" + +#: scripts/_lib/map_template.py:239 +msgid "Dénomination géographique complémentaire de" +msgstr "Dénomination géographique complémentaire de" + +#: scripts/_lib/map_template.py:240 +msgid "À propos" +msgstr "À propos" + +#: scripts/_lib/map_template.py:241 +msgid "À propos d'Open Wine Map" +msgstr "À propos d'Open Wine Map" + +#: scripts/_lib/map_template.py:243 +msgid "" +"Carte de référence des appellations viticoles, générée automatiquement à " +"partir des registres publics : le registre de l'Union européenne des AOP " +"et IGP, les régulateurs nationaux, le répertoire fédéral suisse des AOC " +"cantonales et le registre britannique des indications géographiques." +msgstr "" +"Carte de référence des appellations viticoles, générée automatiquement à " +"partir des registres publics : le registre de l'Union européenne des AOP " +"et IGP, les régulateurs nationaux, le répertoire fédéral suisse des AOC " +"cantonales et le registre britannique des indications géographiques." + +#: scripts/_lib/map_template.py:249 +msgid "" +"Une partie du texte est produite par des modèles de langage : les repères" +" de terroir sont dégagés du texte du régulateur et traduits par Claude " +"(Sonnet 4.6) ; les extraits Wikipedia des infobulles de cépages et de " +"styles sont traduits pour l'essentiel par Mistral Small 3.2, exécuté " +"localement, et pour quelques-uns par Claude ; les résumés des cahiers des" +" charges ont été traduits par un traducteur humain, à l'exception d'un " +"petit reliquat traduit automatiquement. Chaque élément porte sa propre " +"ligne d'attribution dans le panneau." +msgstr "" +"Une partie du texte est produite par des modèles de langage : les repères" +" de terroir sont dégagés du texte du régulateur et traduits par Claude " +"(Sonnet 4.6) ; les extraits Wikipedia des infobulles de cépages et de " +"styles sont traduits pour l'essentiel par Mistral Small 3.2, exécuté " +"localement, et pour quelques-uns par Claude ; les résumés des cahiers des" +" charges ont été traduits par un traducteur humain, à l'exception d'un " +"petit reliquat traduit automatiquement. Chaque élément porte sa propre " +"ligne d'attribution dans le panneau." + +#: scripts/_lib/map_template.py:258 +#, python-brace-format +msgid "Réalisé avec ♡ par {devloed}." +msgstr "Réalisé avec ♡ par {devloed}." + +#: scripts/_lib/map_template.py:260 +#, python-brace-format +msgid "" +"Sources : INAO ({inao}) pour les cahiers des charges et les aires " +"parcellaires françaises, IGN ({ign}) pour les contours des communes " +"françaises, le registre des indications géographiques de l'UE, les " +"régulateurs nationaux, Eurostat GISCO et Bétard 2022 pour les autres " +"pays, OpenStreetMap et CARTO pour le fond de carte, Wikipedia " +"({wikipedia}) pour quelques compléments narratifs (CC BY-SA 4.0), et VIVC" +" ({vivc}), le Vitis International Variety Catalogue du Julius Kühn-" +"Institut, pour les noms canoniques et numéros de cépage (citation Röckel " +"et al.). Tout extrait Wikipedia est signalé sur place. Détails et " +"licences dans le {readme}." +msgstr "" +"Sources : INAO ({inao}) pour les cahiers des charges et les aires " +"parcellaires françaises, IGN ({ign}) pour les contours des communes " +"françaises, le registre des indications géographiques de l'UE, les " +"régulateurs nationaux, Eurostat GISCO et Bétard 2022 pour les autres " +"pays, OpenStreetMap et CARTO pour le fond de carte, Wikipedia " +"({wikipedia}) pour quelques compléments narratifs (CC BY-SA 4.0), et VIVC" +" ({vivc}), le Vitis International Variety Catalogue du Julius Kühn-" +"Institut, pour les noms canoniques et numéros de cépage (citation Röckel " +"et al.). Tout extrait Wikipedia est signalé sur place. Détails et " +"licences dans le {readme}." + +#: scripts/_lib/map_template.py:270 +#, python-brace-format +msgid "Suggestions et pull requests bienvenues sur {github}." +msgstr "Suggestions et pull requests bienvenues sur {github}." + +#: scripts/_lib/map_template.py:271 +msgid "ticket GitHub" +msgstr "ticket GitHub" + +#: scripts/_lib/map_template.py:272 +msgid "e-mail" +msgstr "e-mail" + +#: scripts/_lib/map_template.py:273 +msgid "E-mail copié dans le presse-papiers" +msgstr "E-mail copié dans le presse-papiers" + +#: scripts/_lib/map_template.py:275 +#, python-brace-format +msgid "" +"Carte générée automatiquement — des erreurs sont possibles. Signalez-les " +"via {issue} ou {email}." +msgstr "" +"Carte générée automatiquement — des erreurs sont possibles. Signalez-les " +"via {issue} ou {email}." + +#: scripts/_lib/map_template.py:279 +#, python-brace-format +msgid "" +"{c} pays européens cartographiés ({n} appellations : {parents} " +"appellations et {subs} dénominations rattachées). Des itérations " +"supplémentaires affineront la qualité des données et étendront la " +"couverture au-delà de l'UE, de la Suisse et du Royaume-Uni." +msgstr "" +"{c} pays européens cartographiés ({n} appellations : {parents} " +"appellations et {subs} dénominations rattachées). Des itérations " +"supplémentaires affineront la qualité des données et étendront la " +"couverture au-delà de l'UE, de la Suisse et du Royaume-Uni." + +#: scripts/_lib/map_template.py:284 +msgid "Toutes les appellations" +msgstr "Toutes les appellations" + +#: scripts/_lib/map_template.py:285 +msgid "Toutes les appellations viticoles — Open Wine Map" +msgstr "Toutes les appellations viticoles — Open Wine Map" + +#: scripts/_lib/map_template.py:287 +#, python-brace-format +msgid "" +"Liste des {n} appellations viticoles cartographiées sur Open Wine Map, " +"classées par pays : AOP et IGP de l'UE avec leurs termes traditionnels " +"(AOC, DOCG, DOQ…), AOC suisses et IG britanniques." +msgstr "" +"Liste des {n} appellations viticoles cartographiées sur Open Wine Map, " +"classées par pays : AOP et IGP de l'UE avec leurs termes traditionnels " +"(AOC, DOCG, DOQ…), AOC suisses et IG britanniques." + +#: scripts/_lib/map_template.py:292 +#, python-brace-format +msgid "" +"Les {n} appellations ci-dessous sont classées par pays. Retour à la " +"{map_link}." +msgstr "" +"Les {n} appellations ci-dessous sont classées par pays. Retour à la " +"{map_link}." + +#: scripts/_lib/map_template.py:295 +msgid "carte interactive" +msgstr "carte interactive" + +#: scripts/_lib/map_template.py:296 +msgid "Liste des appellations par pays" +msgstr "Liste des appellations par pays" + +#: scripts/_lib/map_template.py:297 +msgid "Appellation parente" +msgstr "Appellation parente" + +#: scripts/_lib/map_template.py:298 +msgid "Dénominations rattachées" +msgstr "Dénominations rattachées" + +#: scripts/_lib/map_template.py:300 +#, python-brace-format +msgid "Parcourir la liste complète des appellations : {browse_link}." +msgstr "Parcourir la liste complète des appellations : {browse_link}." + +#: scripts/_lib/map_template.py:302 +#, python-brace-format +msgid "Données mises à jour le {date}." +msgstr "Données mises à jour le {date}." + +#: scripts/_lib/map_template.py:317 +msgid "BOURGOGNE" +msgstr "Bourgogne" + +#: scripts/_lib/map_template.py:318 +msgid "BEAUJOLAIS" +msgstr "Beaujolais" + +#: scripts/_lib/map_template.py:319 +msgid "JURA" +msgstr "Jura" + +#: scripts/_lib/map_template.py:320 +msgid "SAVOIE" +msgstr "Savoie" + +#: scripts/_lib/map_template.py:321 +msgid "BUGEY" +msgstr "Bugey" + +#: scripts/_lib/map_template.py:322 +msgid "ALSACE ET EST" +msgstr "Alsace et Est" + +#: scripts/_lib/map_template.py:323 +msgid "VAL DE LOIRE" +msgstr "Val de Loire" + +#: scripts/_lib/map_template.py:324 +msgid "SUD-OUEST" +msgstr "Sud-Ouest" + +#: scripts/_lib/map_template.py:325 +msgid "VALLEE DU RHÔNE" +msgstr "Vallée du Rhône" + +#: scripts/_lib/map_template.py:326 +msgid "LANGUEDOC-ROUSSILLON" +msgstr "Languedoc-Roussillon" + +#: scripts/_lib/map_template.py:327 +msgid "TOULOUSE-PYRENEES" +msgstr "Toulouse-Pyrénées" + +#: scripts/_lib/map_template.py:328 +msgid "PROVENCE-CORSE" +msgstr "Provence-Corse" + +#: scripts/_lib/map_template.py:329 +msgid "CHAMPAGNE" +msgstr "Champagne" + +#: scripts/_lib/map_template.py:330 +msgid "EAUX-DE-VIE DE CIDRE" +msgstr "Eaux-de-vie de cidre" + +#: scripts/_lib/map_template.py:331 +msgid "VIN DOUX NATURELS" +msgstr "Vins doux naturels" + +#: scripts/_lib/map_template.py:332 +msgid "COGNAC" +msgstr "Cognac" + +#: scripts/_lib/map_template.py:333 +msgid "ARMAGNAC" +msgstr "Armagnac" + +#: scripts/_lib/map_template.py:334 +msgid "RHUM" +msgstr "Rhum" + +#: scripts/_lib/map_template.py:374 +msgid "France" +msgstr "France" + +#: scripts/_lib/map_template.py:375 +msgid "Espagne" +msgstr "Espagne" + +#: scripts/_lib/map_template.py:376 +msgid "Portugal" +msgstr "Portugal" + +#: scripts/_lib/map_template.py:377 +msgid "Italie" +msgstr "Italie" + +#: scripts/_lib/map_template.py:378 +msgid "Autriche" +msgstr "Autriche" + +#: scripts/_lib/map_template.py:379 +msgid "Slovénie" +msgstr "Slovénie" + +#: scripts/_lib/map_template.py:380 +msgid "Croatie" +msgstr "Croatie" + +#: scripts/_lib/map_template.py:381 +msgid "Hongrie" +msgstr "Hongrie" + +#: scripts/_lib/map_template.py:382 +msgid "Roumanie" +msgstr "Roumanie" + +#: scripts/_lib/map_template.py:383 +msgid "Bulgarie" +msgstr "Bulgarie" + +#: scripts/_lib/map_template.py:384 +msgid "Grèce" +msgstr "Grèce" + +#: scripts/_lib/map_template.py:385 +msgid "Allemagne" +msgstr "Allemagne" + +#: scripts/_lib/map_template.py:386 +msgid "Slovaquie" +msgstr "Slovaquie" + +#: scripts/_lib/map_template.py:387 +msgid "Suisse" +msgstr "Suisse" + +#: scripts/_lib/map_template.py:388 +msgid "Tchéquie" +msgstr "Tchéquie" + +#: scripts/_lib/map_template.py:389 +msgid "Luxembourg" +msgstr "Luxembourg" + +#: scripts/_lib/map_template.py:390 +msgid "Belgique" +msgstr "Belgique" + +#: scripts/_lib/map_template.py:391 +msgid "Pays-Bas" +msgstr "Pays-Bas" + +#: scripts/_lib/map_template.py:392 +msgid "Malte" +msgstr "Malte" + +#: scripts/_lib/map_template.py:393 +msgid "Chypre" +msgstr "Chypre" + +#: scripts/_lib/map_template.py:394 +msgid "Royaume-Uni" +msgstr "Royaume-Uni" + +#: scripts/_lib/style_taxonomy.py:332 scripts/_lib/style_taxonomy.py:365 +msgid "doux" +msgstr "doux" + +#: scripts/_lib/style_taxonomy.py:333 scripts/_lib/style_taxonomy.py:366 +msgid "autres" +msgstr "autres" + +#: scripts/_lib/style_taxonomy.py:334 +msgid "clairet" +msgstr "clairet" + +#: scripts/_lib/style_taxonomy.py:334 +msgid "primeur" +msgstr "primeur" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "mousseux de qualité" +msgstr "mousseux de qualité" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "crémant" +msgstr "crémant" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "méthode ancestrale" +msgstr "méthode ancestrale" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode traditionnelle" +msgstr "méthode traditionnelle" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode Charmat" +msgstr "méthode Charmat" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode dioise" +msgstr "méthode dioise" + +#: scripts/_lib/style_taxonomy.py:337 +msgid "pétillant" +msgstr "pétillant" + +#: scripts/_lib/style_taxonomy.py:338 +msgid "vin muté" +msgstr "vin muté" + +#: scripts/_lib/style_taxonomy.py:338 +msgid "vendanges tardives (catégorie)" +msgstr "vendanges tardives (catégorie)" + +#: scripts/_lib/style_taxonomy.py:339 +msgid "vin de raisins passerillés" +msgstr "vin de raisins passerillés" + +#: scripts/_lib/style_taxonomy.py:340 +msgid "demi-doux" +msgstr "demi-doux" + +#: scripts/_lib/style_taxonomy.py:340 +msgid "vin de glace" +msgstr "vin de glace" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "vin doux naturel" +msgstr "vin doux naturel" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "vin de liqueur" +msgstr "vin de liqueur" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "mistelle" +msgstr "mistelle" + +#: scripts/_lib/style_taxonomy.py:342 +msgid "vendanges tardives" +msgstr "vendanges tardives" + +#: scripts/_lib/style_taxonomy.py:342 +msgid "uvas sobremaduradas" +msgstr "uvas sobremaduradas" + +#: scripts/_lib/style_taxonomy.py:343 +msgid "vin naturellement doux" +msgstr "vin naturellement doux" + +#: scripts/_lib/style_taxonomy.py:343 +msgid "grains nobles" +msgstr "grains nobles" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "vin de paille" +msgstr "vin de paille" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "uvas pasificadas" +msgstr "uvas pasificadas" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "vin santo" +msgstr "vin santo" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "tranquille" +msgstr "tranquille" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "sur lie" +msgstr "sur lie" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "sec" +msgstr "sec" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "demi-sec" +msgstr "demi-sec" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "rancio" +msgstr "rancio" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "vin jaune" +msgstr "vin jaune" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "generoso" +msgstr "generoso" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "fino" +msgstr "fino" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "manzanilla" +msgstr "manzanilla" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "amontillado" +msgstr "amontillado" + +#: scripts/_lib/style_taxonomy.py:348 +msgid "oloroso" +msgstr "oloroso" + +#: scripts/_lib/style_taxonomy.py:348 +msgid "palo cortado" +msgstr "palo cortado" + +#~ msgid "carte des appellations françaises" +#~ msgstr "carte des appellations françaises" + +#~ msgid "" +#~ "Priorité actuelle : affiner la " +#~ "couverture française — précision des " +#~ "aires, qualité des extraits, climats et" +#~ " lieux-dits. L'extension à d'autres " +#~ "pays viticoles viendra ensuite." +#~ msgstr "" +#~ "Priorité actuelle : affiner la " +#~ "couverture française — précision des " +#~ "aires, qualité des extraits, climats et" +#~ " lieux-dits. L'extension à d'autres " +#~ "pays viticoles viendra ensuite." + +#~ msgid "anglais" +#~ msgstr "" + +#~ msgid "français" +#~ msgstr "(français)" + +#~ msgid "espagnol" +#~ msgstr "(español)" + +#~ msgid "néerlandais" +#~ msgstr "" + +#~ msgid "" +#~ "Couverture actuelle : France et une " +#~ "première version de l'Espagne. Prochaines " +#~ "étapes : amélioration continue de la " +#~ "qualité des données, puis ajout du " +#~ "Portugal." +#~ msgstr "" +#~ "Couverture actuelle : France et une " +#~ "première version de l'Espagne. Prochaines " +#~ "étapes : amélioration continue de la " +#~ "qualité des données, puis ajout du " +#~ "Portugal." + +#~ msgid "moelleux" +#~ msgstr "moelleux" + +#~ msgid "" +#~ "Carte interactive des appellations viticoles" +#~ " européennes (AOC, AOP, IGP, DOP) :" +#~ " cépages, styles et terroir, d'après " +#~ "les registres officiels (INAO, EUR-Lex)." +#~ msgstr "" +#~ "Carte interactive des appellations viticoles" +#~ " européennes (AOC, AOP, IGP, DOP) :" +#~ " cépages, styles et terroir, d'après " +#~ "les registres officiels (INAO, EUR-Lex)." + +#~ msgid "AOC / AOP" +#~ msgstr "AOC / AOP" + +#~ msgid "{n} dans IGP masquées — afficher" +#~ msgstr "{n} dans des IGP masquées — afficher" + +#~ msgid "" +#~ "Carte de référence des appellations " +#~ "viticoles (AOC, AOP, IGP, DOP), générée" +#~ " automatiquement à partir des données " +#~ "publiques." +#~ msgstr "" +#~ "Carte de référence des appellations " +#~ "viticoles (AOC, AOP, IGP, DOP), générée" +#~ " automatiquement à partir des données " +#~ "publiques." + +#~ msgid "" +#~ "Sources : INAO ({inao}) pour les " +#~ "cahiers des charges et les aires " +#~ "parcellaires, IGN ({ign}) pour le fond" +#~ " cartographique, Wikipedia ({wikipedia}) pour " +#~ "quelques compléments narratifs (CC BY-SA" +#~ " 4.0), VIVC ({vivc}) — Vitis " +#~ "International Variety Catalogue, Julius " +#~ "Kühn-Institut — pour les noms " +#~ "canoniques et numéros de cépage " +#~ "(citation Röckel et al.). Tout extrait" +#~ " Wikipedia est signalé sur place. " +#~ "Détails et licences dans le {readme}." +#~ msgstr "" +#~ "Sources : INAO ({inao}) pour les " +#~ "cahiers des charges et les aires " +#~ "parcellaires, IGN ({ign}) pour le fond" +#~ " cartographique, Wikipedia ({wikipedia}) pour " +#~ "quelques compléments narratifs (CC BY-SA" +#~ " 4.0), VIVC ({vivc}) — Vitis " +#~ "International Variety Catalogue, Julius " +#~ "Kühn-Institut — pour les noms " +#~ "canoniques et numéros de cépage " +#~ "(citation Röckel et al.). Tout extrait" +#~ " Wikipedia est signalé sur place. " +#~ "Détails et licences dans le {readme}." + +#~ msgid "" +#~ "20 pays européens cartographiés : " +#~ "France, Espagne, Portugal, Italie, Autriche," +#~ " Allemagne, Suisse, Slovénie, Croatie, " +#~ "Hongrie, Roumanie, Bulgarie, Grèce, Slovaquie," +#~ " Tchéquie, Luxembourg, Belgique, Pays-Bas," +#~ " Malte et Chypre. Des itérations " +#~ "supplémentaires viendront affiner la qualité" +#~ " des données. La couverture sera " +#~ "étendue au-delà de l'UE et de " +#~ "la Suisse, ainsi qu'aux classifications " +#~ "hors AOP." +#~ msgstr "" +#~ "Extension de la couverture européenne en" +#~ " cours : France, Espagne, Portugal, " +#~ "Italie, Autriche, Allemagne, Slovénie, " +#~ "Croatie, Hongrie, Roumanie, Bulgarie et " +#~ "Grèce sont déjà cartographiés. Quelques " +#~ "itérations supplémentaires viendront affiner " +#~ "la qualité des données existantes ; " +#~ "l'Europe centrale et orientale est à " +#~ "considérer comme une première version, " +#~ "appelée à être améliorée. À plus " +#~ "long terme, des classifications hors AOP" +#~ " pourront être ajoutées." + +#~ msgid "" +#~ "Liste des {n} appellations viticoles " +#~ "européennes cartographiées sur Open Wine " +#~ "Map, classées par pays — AOC, AOP," +#~ " IGP, DOP." +#~ msgstr "" +#~ "Liste des {n} appellations viticoles " +#~ "européennes cartographiées sur Open Wine " +#~ "Map, classées par pays — AOC, AOP," +#~ " IGP, DOP." + +#~ msgid "" +#~ "Chaque appellation porte deux noms, qui" +#~ " désignent deux choses : le terme " +#~ "traditionnel que le régulateur du pays" +#~ " lui rattache, et le régime sous " +#~ "lequel elle est enregistrée. La carte" +#~ " les affiche sous la forme <em>TERME" +#~ " (RÉGIME)</em>, par exemple « DOCG " +#~ "(AOP) » ou « DOQ (AOP) ». " +#~ "Lorsqu'un pays n'a pas de terme " +#~ "propre, seul le régime apparaît. Les " +#~ "AOC suisses sont hors du régime de" +#~ " l'UE et ne portent aucune parenthèse" +#~ " ; le Royaume-Uni enregistre ses " +#~ "appellations dans son propre régime, qui" +#~ " conserve les mots PDO et PGI ;" +#~ " les eaux-de-vie françaises sont " +#~ "des indications géographiques de boissons " +#~ "spiritueuses, pas des AOP viticoles. " +#~ "Survolez un terme pour en lire la" +#~ " définition et la source." +#~ msgstr "" +#~ "Chaque appellation porte deux noms, qui" +#~ " désignent deux choses : le terme " +#~ "traditionnel que le régulateur du pays" +#~ " lui rattache, et le régime sous " +#~ "lequel elle est enregistrée. La carte" +#~ " les affiche sous la forme <em>TERME" +#~ " (RÉGIME)</em>, par exemple « DOCG " +#~ "(AOP) » ou « DOQ (AOP) ». " +#~ "Lorsqu'un pays n'a pas de terme " +#~ "propre, seul le régime apparaît. Les " +#~ "AOC suisses sont hors du régime de" +#~ " l'UE et ne portent aucune parenthèse" +#~ " ; le Royaume-Uni enregistre ses " +#~ "appellations dans son propre régime, qui" +#~ " conserve les mots PDO et PGI ;" +#~ " les eaux-de-vie françaises sont " +#~ "des indications géographiques de boissons " +#~ "spiritueuses, pas des AOP viticoles. " +#~ "Survolez un terme pour en lire la" +#~ " définition et la source." + +#~ msgid "Open Wine Map — carte des appellations" +#~ msgstr "Open Wine Map — carte des appellations" + +#~ msgid "" +#~ "Carte des AOP et IGP viticoles " +#~ "d'Europe (AOC, DOCG, DOQ, DAC…), AOC " +#~ "suisses et IG britanniques : cépages," +#~ " styles et terroir, d'après les " +#~ "registres officiels." +#~ msgstr "" +#~ "Carte des AOP et IGP viticoles " +#~ "d'Europe (AOC, DOCG, DOQ, DAC…), AOC " +#~ "suisses et IG britanniques : cépages," +#~ " styles et terroir, d'après les " +#~ "registres officiels." + diff --git a/locale/messages.pot b/locale/messages.pot new file mode 100644 index 0000000..81c5db0 --- /dev/null +++ b/locale/messages.pot @@ -0,0 +1,1110 @@ +# Translations template for PROJECT. +# Copyright (C) 2026 ORGANIZATION +# This file is distributed under the same license as the PROJECT project. +# FIRST AUTHOR <EMAIL@ADDRESS>, 2026. +# +#, fuzzy +msgid "" +msgstr "" +"Project-Id-Version: PROJECT VERSION\n" +"Report-Msgid-Bugs-To: EMAIL@ADDRESS\n" +"POT-Creation-Date: 2026-09-08 21:10+0200\n" +"PO-Revision-Date: YEAR-MO-DA HO:MI+ZONE\n" +"Last-Translator: FULL NAME <EMAIL@ADDRESS>\n" +"Language-Team: LANGUAGE <LL@li.org>\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.18.0\n" + +#: scripts/_lib/aging_taxonomy.py:184 +msgid "Vieillissement" +msgstr "" + +#: scripts/_lib/aging_taxonomy.py:185 +msgid "Prädikat" +msgstr "" + +#: scripts/_lib/aging_taxonomy.py:186 +msgid "Sélection" +msgstr "" + +#: scripts/_lib/map_template.py:46 +msgid "Open Wine Map — appellations viticoles d'Europe" +msgstr "" + +#: scripts/_lib/map_template.py:47 +msgid "carte des appellations viticoles" +msgstr "" + +#: scripts/_lib/map_template.py:49 +msgid "" +"Carte interactive des appellations viticoles d'Europe : cépages, styles " +"et terroir, d'après les registres officiels." +msgstr "" + +#: scripts/_lib/map_template.py:52 +msgid "Chargement…" +msgstr "" + +#: scripts/_lib/map_template.py:53 +msgid "Recherche" +msgstr "" + +#: scripts/_lib/map_template.py:54 +msgid "nom d'appellation…" +msgstr "" + +#: scripts/_lib/map_template.py:55 +msgid "Recherche d'appellation…" +msgstr "" + +#: scripts/_lib/map_template.py:56 +msgid "Recherche de cépage…" +msgstr "" + +#: scripts/_lib/map_template.py:58 +msgid "Rechercher une appellation, un cépage, une région…" +msgstr "" + +#: scripts/_lib/map_template.py:59 +msgid "Cépage principal uniquement" +msgstr "" + +#: scripts/_lib/map_template.py:60 +#, python-brace-format +msgid "Aucun résultat pour « {q} »" +msgstr "" + +#: scripts/_lib/map_template.py:61 +msgid "Options" +msgstr "" + +#: scripts/_lib/map_template.py:62 +msgid "Filtres actifs" +msgstr "" + +#: scripts/_lib/map_template.py:63 +msgid "Tout sélectionner" +msgstr "" + +#: scripts/_lib/map_template.py:65 +#, python-brace-format +msgid "{p} appellations · {s} dénominations géographiques complémentaires" +msgstr "" + +#: scripts/_lib/map_template.py:67 +#, python-brace-format +msgid "{p} appellations" +msgstr "" + +#: scripts/_lib/map_template.py:68 +#, python-brace-format +msgid "Ouvrir la fiche de {name}" +msgstr "" + +#: scripts/_lib/map_template.py:69 +msgid "Ouvrir la fiche" +msgstr "" + +#: scripts/_lib/map_template.py:70 +msgid "Inclure les spiritueux" +msgstr "" + +#: scripts/_lib/map_template.py:71 +msgid "Style de vin" +msgstr "" + +#: scripts/_lib/map_template.py:72 +msgid "Classement" +msgstr "" + +#: scripts/_lib/map_template.py:73 +msgid "Cépages principaux" +msgstr "" + +#: scripts/_lib/map_template.py:74 +msgid "Cépages accessoires" +msgstr "" + +#: scripts/_lib/map_template.py:75 scripts/_lib/map_template.py:189 +msgid "Cépages" +msgstr "" + +#: scripts/_lib/map_template.py:76 +msgid "Région" +msgstr "" + +#: scripts/_lib/map_template.py:77 +msgid "Appellation" +msgstr "" + +#: scripts/_lib/map_template.py:78 +msgid "Type" +msgstr "" + +#: scripts/_lib/map_template.py:79 +msgid "Origine protégée (AOP, AOC, DOC, DO…)" +msgstr "" + +#: scripts/_lib/map_template.py:80 +msgid "Indication géographique (IGP, IGT, Landwein…)" +msgstr "" + +#: scripts/_lib/map_template.py:81 +msgid "AOP" +msgstr "" + +#: scripts/_lib/map_template.py:82 +msgid "IGP" +msgstr "" + +#: scripts/_lib/map_template.py:83 +msgid "IG spiritueux" +msgstr "" + +#: scripts/_lib/map_template.py:84 +msgid "Type d'appellation" +msgstr "" + +#: scripts/_lib/map_template.py:85 +msgid "AOP (régime britannique)" +msgstr "" + +#: scripts/_lib/map_template.py:86 +msgid "IGP (régime britannique)" +msgstr "" + +#: scripts/_lib/map_template.py:87 +msgid "AOC (Suisse)" +msgstr "" + +#: scripts/_lib/map_template.py:88 +msgid "Source" +msgstr "" + +#: scripts/_lib/map_template.py:89 +msgid "Vue" +msgstr "" + +#: scripts/_lib/map_template.py:90 +msgid "Simple" +msgstr "" + +#: scripts/_lib/map_template.py:91 +msgid "Avancée" +msgstr "" + +#: scripts/_lib/map_template.py:92 +msgid "Thème" +msgstr "" + +#: scripts/_lib/map_template.py:93 +msgid "Clair" +msgstr "" + +#: scripts/_lib/map_template.py:94 +msgid "Sombre" +msgstr "" + +#: scripts/_lib/map_template.py:95 +msgid "Système" +msgstr "" + +#: scripts/_lib/map_template.py:96 +msgid "Afficher les IGP" +msgstr "" + +#: scripts/_lib/map_template.py:97 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:361 +msgid "rouge" +msgstr "" + +#: scripts/_lib/map_template.py:98 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:362 +msgid "blanc" +msgstr "" + +#: scripts/_lib/map_template.py:99 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:363 +msgid "rosé" +msgstr "" + +#: scripts/_lib/map_template.py:100 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:364 +msgid "mousseux" +msgstr "" + +#: scripts/_lib/map_template.py:101 +msgid "moelleux / liquoreux" +msgstr "" + +#: scripts/_lib/map_template.py:102 scripts/_lib/style_taxonomy.py:345 +msgid "oxydatif" +msgstr "" + +#: scripts/_lib/map_template.py:103 +msgid "autre" +msgstr "" + +#: scripts/_lib/map_template.py:104 +msgid "Réinitialiser" +msgstr "" + +#: scripts/_lib/map_template.py:105 +#, python-brace-format +msgid "{n} appellations" +msgstr "" + +#: scripts/_lib/map_template.py:106 +#, python-brace-format +msgid "{n} / {total} appellations" +msgstr "" + +#: scripts/_lib/map_template.py:107 +#, python-brace-format +msgid "{n} masquées dans les IGP · afficher" +msgstr "" + +#: scripts/_lib/map_template.py:108 +msgid "Fermer" +msgstr "" + +#: scripts/_lib/map_template.py:109 +msgid "Détails de l'appellation" +msgstr "" + +#: scripts/_lib/map_template.py:110 +#, python-brace-format +msgid "Retirer le filtre {label}" +msgstr "" + +#: scripts/_lib/map_template.py:111 +msgid "Filtres et options de la carte" +msgstr "" + +#: scripts/_lib/map_template.py:112 +msgid "Langue" +msgstr "" + +#: scripts/_lib/map_template.py:113 +msgid "Carte des appellations viticoles" +msgstr "" + +#: scripts/_lib/map_template.py:114 +msgid "Aller à la carte" +msgstr "" + +#: scripts/_lib/map_template.py:115 +msgid "Styles" +msgstr "" + +#: scripts/_lib/map_template.py:116 +msgid "Variétés d'intérêt" +msgstr "" + +#: scripts/_lib/map_template.py:117 +msgid "Sources" +msgstr "" + +#: scripts/_lib/map_template.py:118 +msgid "Terroir" +msgstr "" + +#: scripts/_lib/map_template.py:119 +#, python-brace-format +msgid "Dűlők (lieux-dits) : {n}" +msgstr "" + +#: scripts/_lib/map_template.py:120 +#, python-brace-format +msgid "Menzioni geografiche aggiuntive (crus) : {n}" +msgstr "" + +#: scripts/_lib/map_template.py:121 +msgid "Facteurs naturels" +msgstr "" + +#: scripts/_lib/map_template.py:122 +msgid "Facteurs humains" +msgstr "" + +#: scripts/_lib/map_template.py:123 +msgid "Caractéristiques du produit" +msgstr "" + +#: scripts/_lib/map_template.py:124 +msgid "Lien terroir / vin" +msgstr "" + +#: scripts/_lib/map_template.py:126 +#, python-brace-format +msgid "" +"Faits dégagés du Lien au terroir par interprétation automatique — voir la" +" {source}." +msgstr "" + +#: scripts/_lib/map_template.py:128 +msgid "source" +msgstr "" + +#: scripts/_lib/map_template.py:129 +msgid "via Wikipedia · CC BY-SA 4.0" +msgstr "" + +#: scripts/_lib/map_template.py:131 +#, python-brace-format +msgid "Citation textuelle du Lien au terroir — voir la {source}." +msgstr "" + +#: scripts/_lib/map_template.py:133 +msgid "à vérifier — texte source court" +msgstr "" + +#: scripts/_lib/map_template.py:135 +#, python-brace-format +msgid "" +"Ces repères décrivent l'appellation englobante {parent} — pas " +"spécifiquement cette dénomination." +msgstr "" + +#: scripts/_lib/map_template.py:138 +msgid "sans région" +msgstr "" + +#: scripts/_lib/map_template.py:139 +#, python-brace-format +msgid "{n} commune(s) INAO" +msgstr "" + +#: scripts/_lib/map_template.py:140 +#, python-brace-format +msgid "{n} commune(s)" +msgstr "" + +#: scripts/_lib/map_template.py:141 +msgid "aire approchée" +msgstr "" + +#: scripts/_lib/map_template.py:142 +msgid "aire approchée (à l'échelle communale)" +msgstr "" + +#: scripts/_lib/map_template.py:144 +#, python-brace-format +msgid "" +"Aire approchée — pas de données parcellaires précises pour cette " +"dénomination ; polygone hérité de {umbrella}." +msgstr "" + +#: scripts/_lib/map_template.py:148 +msgid "" +"Aire approchée — pas de données parcellaires précises pour cette " +"dénomination ; polygone hérité de l'appellation parente." +msgstr "" + +#: scripts/_lib/map_template.py:152 +msgid "" +"Aire approchée — pas de données parcellaires disponibles ; affichée comme" +" l'emprise de la commune où se situe la dénomination." +msgstr "" + +#: scripts/_lib/map_template.py:156 +#, python-brace-format +msgid "" +"Aire issue du lieu-dit cadastral « {lieu_dit} » (commune de {commune}, " +"{source})." +msgstr "" + +#: scripts/_lib/map_template.py:159 +msgid "cadastre.data.gouv.fr" +msgstr "" + +#: scripts/_lib/map_template.py:161 +msgid "" +"Aire approchée — reconstituée à partir des références parcellaires du " +"plan de l'aire délimitée annexé au cahier des charges ; ce n'est pas une " +"limite officielle." +msgstr "" + +#: scripts/_lib/map_template.py:165 +#, python-brace-format +msgid "{n} appellations à ce point" +msgstr "" + +#: scripts/_lib/map_template.py:166 +msgid "Cliquer à nouveau pour parcourir les autres" +msgstr "" + +#: scripts/_lib/map_template.py:167 +msgid "Cahier des charges (BO Agri, PDF)" +msgstr "" + +#: scripts/_lib/map_template.py:168 +msgid "Cahier des charges (registre GI de l'UE, PDF)" +msgstr "" + +#: scripts/_lib/map_template.py:169 +msgid "homologué" +msgstr "" + +#: scripts/_lib/map_template.py:170 +msgid "JORF" +msgstr "" + +#: scripts/_lib/map_template.py:171 +msgid "Texte officiel INAO (show_texte)" +msgstr "" + +#: scripts/_lib/map_template.py:172 +msgid "Fiche produit INAO" +msgstr "" + +#: scripts/_lib/map_template.py:173 +msgid "Site officiel de l'interprofession" +msgstr "" + +#: scripts/_lib/map_template.py:174 +msgid "Cahier des charges (EUR-Lex, document unique)" +msgstr "" + +#: scripts/_lib/map_template.py:175 +msgid "Pliego de condiciones (national, PDF)" +msgstr "" + +#: scripts/_lib/map_template.py:176 +msgid "variétés ajoutées" +msgstr "" + +#: scripts/_lib/map_template.py:177 +msgid "Cahier des charges national (PDF)" +msgstr "" + +#: scripts/_lib/map_template.py:178 +msgid "Spécification du produit (IGP, PDF)" +msgstr "" + +#: scripts/_lib/map_template.py:179 +msgid "Registre régional des cépages (PDF)" +msgstr "" + +#: scripts/_lib/map_template.py:180 +msgid "Registre eAmbrosia (UE)" +msgstr "" + +#: scripts/_lib/map_template.py:181 +msgid "Numéro de dossier" +msgstr "" + +#: scripts/_lib/map_template.py:182 +msgid "Règlement cantonal sur la vigne et le vin" +msgstr "" + +#: scripts/_lib/map_template.py:183 +msgid "Répertoire suisse des AOC (OFAG/BLW)" +msgstr "" + +#: scripts/_lib/map_template.py:184 +msgid "Cahier des charges (GOV.UK, PDF)" +msgstr "" + +#: scripts/_lib/map_template.py:185 +msgid "Registre des IG du Royaume-Uni" +msgstr "" + +#: scripts/_lib/map_template.py:186 +msgid "Légende couleurs" +msgstr "" + +#: scripts/_lib/map_template.py:187 +msgid "Bassin viticole" +msgstr "" + +#: scripts/_lib/map_template.py:188 +msgid "Plus l'aire est petite, plus la teinte est dense." +msgstr "" + +#: scripts/_lib/map_template.py:190 +msgid "principal — variété de la cuvée" +msgstr "" + +#: scripts/_lib/map_template.py:191 +msgid "accessoire — assemblage limité" +msgstr "" + +#: scripts/_lib/map_template.py:192 +msgid "intérêt — observation/conservation" +msgstr "" + +#: scripts/_lib/map_template.py:194 +msgid "" +"Le régulateur portugais (IVV) n'établit pas de distinction " +"principal/accessoire — toutes les castas autorisées sont listées ensemble" +" dans le caderno de especificações." +msgstr "" + +#: scripts/_lib/map_template.py:198 +msgid "" +"Bianchello (ou Biancame) est traité ici comme un cépage distinct, " +"conformément au disciplinare de la DOP Bianchello del Metauro ; le " +"catalogue VIVC le recense comme synonyme du Trebbiano Toscano." +msgstr "" + +#: scripts/_lib/map_template.py:203 +#, python-brace-format +msgid "Open Wine Map n'a pas encore trouvé de {doc} pour cette appellation." +msgstr "" + +#: scripts/_lib/map_template.py:205 +msgid "aidez-nous à le trouver" +msgstr "" + +#: scripts/_lib/map_template.py:211 +#, python-brace-format +msgid "Délimitée par {regulator} dans son {doc}, qui autorise {grapes}." +msgstr "" + +#: scripts/_lib/map_template.py:213 +#, python-brace-format +msgid "Délimitée par {regulator} dans son {doc}{extra}." +msgstr "" + +#: scripts/_lib/map_template.py:214 +#, python-brace-format +msgid "{names} et {n} autres cépages" +msgstr "" + +#: scripts/_lib/map_template.py:215 +msgid "le registre eAmbrosia de l'UE" +msgstr "" + +#: scripts/_lib/map_template.py:216 +#, python-brace-format +msgid "le canton de {canton}" +msgstr "" + +#: scripts/_lib/map_template.py:217 +msgid "(français)" +msgstr "" + +#: scripts/_lib/map_template.py:218 +msgid "Texte source en français" +msgstr "" + +#: scripts/_lib/map_template.py:219 +msgid "(español)" +msgstr "" + +#: scripts/_lib/map_template.py:220 +msgid "Texte source en espagnol" +msgstr "" + +#: scripts/_lib/map_template.py:221 +msgid "(português)" +msgstr "" + +#: scripts/_lib/map_template.py:222 +msgid "Texte source en portugais" +msgstr "" + +#: scripts/_lib/map_template.py:223 +msgid "Filtres" +msgstr "" + +#: scripts/_lib/map_template.py:224 +#, python-brace-format +msgid "Traduit de {wiki} · CC BY-SA 4.0" +msgstr "" + +#: scripts/_lib/map_template.py:225 +msgid "Wikipédia en anglais" +msgstr "" + +#: scripts/_lib/map_template.py:226 +msgid "Wikipédia en français" +msgstr "" + +#: scripts/_lib/map_template.py:227 +msgid "Wikipédia en espagnol" +msgstr "" + +#: scripts/_lib/map_template.py:228 +msgid "Wikipédia en néerlandais" +msgstr "" + +#: scripts/_lib/map_template.py:229 +msgid "Wikipédia en portugais" +msgstr "" + +#: scripts/_lib/map_template.py:230 +msgid "Wikipédia en croate" +msgstr "" + +#: scripts/_lib/map_template.py:231 +msgid "Vitis International Variety Catalogue (Julius Kühn-Institut)" +msgstr "" + +#: scripts/_lib/map_template.py:232 +#, python-brace-format +msgid "VIVC #{id}" +msgstr "" + +#: scripts/_lib/map_template.py:233 +#, python-brace-format +msgid "Traduction automatique depuis {source}" +msgstr "" + +#: scripts/_lib/map_template.py:234 +msgid "le cahier des charges" +msgstr "" + +#: scripts/_lib/map_template.py:235 scripts/_lib/map_template.py:237 +msgid "le pliego de condiciones" +msgstr "" + +#: scripts/_lib/map_template.py:236 scripts/_lib/map_template.py:238 +msgid "le caderno de especificações" +msgstr "" + +#: scripts/_lib/map_template.py:239 +msgid "Dénomination géographique complémentaire de" +msgstr "" + +#: scripts/_lib/map_template.py:240 +msgid "À propos" +msgstr "" + +#: scripts/_lib/map_template.py:241 +msgid "À propos d'Open Wine Map" +msgstr "" + +#: scripts/_lib/map_template.py:243 +msgid "" +"Carte de référence des appellations viticoles, générée automatiquement à " +"partir des registres publics : le registre de l'Union européenne des AOP " +"et IGP, les régulateurs nationaux, le répertoire fédéral suisse des AOC " +"cantonales et le registre britannique des indications géographiques." +msgstr "" + +#: scripts/_lib/map_template.py:249 +msgid "" +"Une partie du texte est produite par des modèles de langage : les repères" +" de terroir sont dégagés du texte du régulateur et traduits par Claude " +"(Sonnet 4.6) ; les extraits Wikipedia des infobulles de cépages et de " +"styles sont traduits pour l'essentiel par Mistral Small 3.2, exécuté " +"localement, et pour quelques-uns par Claude ; les résumés des cahiers des" +" charges ont été traduits par un traducteur humain, à l'exception d'un " +"petit reliquat traduit automatiquement. Chaque élément porte sa propre " +"ligne d'attribution dans le panneau." +msgstr "" + +#: scripts/_lib/map_template.py:258 +#, python-brace-format +msgid "Réalisé avec ♡ par {devloed}." +msgstr "" + +#: scripts/_lib/map_template.py:260 +#, python-brace-format +msgid "" +"Sources : INAO ({inao}) pour les cahiers des charges et les aires " +"parcellaires françaises, IGN ({ign}) pour les contours des communes " +"françaises, le registre des indications géographiques de l'UE, les " +"régulateurs nationaux, Eurostat GISCO et Bétard 2022 pour les autres " +"pays, OpenStreetMap et CARTO pour le fond de carte, Wikipedia " +"({wikipedia}) pour quelques compléments narratifs (CC BY-SA 4.0), et VIVC" +" ({vivc}), le Vitis International Variety Catalogue du Julius Kühn-" +"Institut, pour les noms canoniques et numéros de cépage (citation Röckel " +"et al.). Tout extrait Wikipedia est signalé sur place. Détails et " +"licences dans le {readme}." +msgstr "" + +#: scripts/_lib/map_template.py:270 +#, python-brace-format +msgid "Suggestions et pull requests bienvenues sur {github}." +msgstr "" + +#: scripts/_lib/map_template.py:271 +msgid "ticket GitHub" +msgstr "" + +#: scripts/_lib/map_template.py:272 +msgid "e-mail" +msgstr "" + +#: scripts/_lib/map_template.py:273 +msgid "E-mail copié dans le presse-papiers" +msgstr "" + +#: scripts/_lib/map_template.py:275 +#, python-brace-format +msgid "" +"Carte générée automatiquement — des erreurs sont possibles. Signalez-les " +"via {issue} ou {email}." +msgstr "" + +#: scripts/_lib/map_template.py:279 +#, python-brace-format +msgid "" +"{c} pays européens cartographiés ({n} appellations : {parents} " +"appellations et {subs} dénominations rattachées). Des itérations " +"supplémentaires affineront la qualité des données et étendront la " +"couverture au-delà de l'UE, de la Suisse et du Royaume-Uni." +msgstr "" + +#: scripts/_lib/map_template.py:284 +msgid "Toutes les appellations" +msgstr "" + +#: scripts/_lib/map_template.py:285 +msgid "Toutes les appellations viticoles — Open Wine Map" +msgstr "" + +#: scripts/_lib/map_template.py:287 +#, python-brace-format +msgid "" +"Liste des {n} appellations viticoles cartographiées sur Open Wine Map, " +"classées par pays : AOP et IGP de l'UE avec leurs termes traditionnels " +"(AOC, DOCG, DOQ…), AOC suisses et IG britanniques." +msgstr "" + +#: scripts/_lib/map_template.py:292 +#, python-brace-format +msgid "" +"Les {n} appellations ci-dessous sont classées par pays. Retour à la " +"{map_link}." +msgstr "" + +#: scripts/_lib/map_template.py:295 +msgid "carte interactive" +msgstr "" + +#: scripts/_lib/map_template.py:296 +msgid "Liste des appellations par pays" +msgstr "" + +#: scripts/_lib/map_template.py:297 +msgid "Appellation parente" +msgstr "" + +#: scripts/_lib/map_template.py:298 +msgid "Dénominations rattachées" +msgstr "" + +#: scripts/_lib/map_template.py:300 +#, python-brace-format +msgid "Parcourir la liste complète des appellations : {browse_link}." +msgstr "" + +#: scripts/_lib/map_template.py:302 +#, python-brace-format +msgid "Données mises à jour le {date}." +msgstr "" + +#: scripts/_lib/map_template.py:317 +msgid "BOURGOGNE" +msgstr "" + +#: scripts/_lib/map_template.py:318 +msgid "BEAUJOLAIS" +msgstr "" + +#: scripts/_lib/map_template.py:319 +msgid "JURA" +msgstr "" + +#: scripts/_lib/map_template.py:320 +msgid "SAVOIE" +msgstr "" + +#: scripts/_lib/map_template.py:321 +msgid "BUGEY" +msgstr "" + +#: scripts/_lib/map_template.py:322 +msgid "ALSACE ET EST" +msgstr "" + +#: scripts/_lib/map_template.py:323 +msgid "VAL DE LOIRE" +msgstr "" + +#: scripts/_lib/map_template.py:324 +msgid "SUD-OUEST" +msgstr "" + +#: scripts/_lib/map_template.py:325 +msgid "VALLEE DU RHÔNE" +msgstr "" + +#: scripts/_lib/map_template.py:326 +msgid "LANGUEDOC-ROUSSILLON" +msgstr "" + +#: scripts/_lib/map_template.py:327 +msgid "TOULOUSE-PYRENEES" +msgstr "" + +#: scripts/_lib/map_template.py:328 +msgid "PROVENCE-CORSE" +msgstr "" + +#: scripts/_lib/map_template.py:329 +msgid "CHAMPAGNE" +msgstr "" + +#: scripts/_lib/map_template.py:330 +msgid "EAUX-DE-VIE DE CIDRE" +msgstr "" + +#: scripts/_lib/map_template.py:331 +msgid "VIN DOUX NATURELS" +msgstr "" + +#: scripts/_lib/map_template.py:332 +msgid "COGNAC" +msgstr "" + +#: scripts/_lib/map_template.py:333 +msgid "ARMAGNAC" +msgstr "" + +#: scripts/_lib/map_template.py:334 +msgid "RHUM" +msgstr "" + +#: scripts/_lib/map_template.py:374 +msgid "France" +msgstr "" + +#: scripts/_lib/map_template.py:375 +msgid "Espagne" +msgstr "" + +#: scripts/_lib/map_template.py:376 +msgid "Portugal" +msgstr "" + +#: scripts/_lib/map_template.py:377 +msgid "Italie" +msgstr "" + +#: scripts/_lib/map_template.py:378 +msgid "Autriche" +msgstr "" + +#: scripts/_lib/map_template.py:379 +msgid "Slovénie" +msgstr "" + +#: scripts/_lib/map_template.py:380 +msgid "Croatie" +msgstr "" + +#: scripts/_lib/map_template.py:381 +msgid "Hongrie" +msgstr "" + +#: scripts/_lib/map_template.py:382 +msgid "Roumanie" +msgstr "" + +#: scripts/_lib/map_template.py:383 +msgid "Bulgarie" +msgstr "" + +#: scripts/_lib/map_template.py:384 +msgid "Grèce" +msgstr "" + +#: scripts/_lib/map_template.py:385 +msgid "Allemagne" +msgstr "" + +#: scripts/_lib/map_template.py:386 +msgid "Slovaquie" +msgstr "" + +#: scripts/_lib/map_template.py:387 +msgid "Suisse" +msgstr "" + +#: scripts/_lib/map_template.py:388 +msgid "Tchéquie" +msgstr "" + +#: scripts/_lib/map_template.py:389 +msgid "Luxembourg" +msgstr "" + +#: scripts/_lib/map_template.py:390 +msgid "Belgique" +msgstr "" + +#: scripts/_lib/map_template.py:391 +msgid "Pays-Bas" +msgstr "" + +#: scripts/_lib/map_template.py:392 +msgid "Malte" +msgstr "" + +#: scripts/_lib/map_template.py:393 +msgid "Chypre" +msgstr "" + +#: scripts/_lib/map_template.py:394 +msgid "Royaume-Uni" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:332 scripts/_lib/style_taxonomy.py:365 +msgid "doux" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:333 scripts/_lib/style_taxonomy.py:366 +msgid "autres" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:334 +msgid "clairet" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:334 +msgid "primeur" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "mousseux de qualité" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "crémant" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "méthode ancestrale" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode traditionnelle" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode Charmat" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode dioise" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:337 +msgid "pétillant" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:338 +msgid "vin muté" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:338 +msgid "vendanges tardives (catégorie)" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:339 +msgid "vin de raisins passerillés" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:340 +msgid "demi-doux" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:340 +msgid "vin de glace" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "vin doux naturel" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "vin de liqueur" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "mistelle" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:342 +msgid "vendanges tardives" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:342 +msgid "uvas sobremaduradas" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:343 +msgid "vin naturellement doux" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:343 +msgid "grains nobles" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "vin de paille" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "uvas pasificadas" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "vin santo" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "tranquille" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "sur lie" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "sec" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "demi-sec" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "rancio" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "vin jaune" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "generoso" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "fino" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "manzanilla" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "amontillado" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:348 +msgid "oloroso" +msgstr "" + +#: scripts/_lib/style_taxonomy.py:348 +msgid "palo cortado" +msgstr "" + diff --git a/locale/nl/LC_MESSAGES/messages.po b/locale/nl/LC_MESSAGES/messages.po new file mode 100644 index 0000000..ad9e316 --- /dev/null +++ b/locale/nl/LC_MESSAGES/messages.po @@ -0,0 +1,1353 @@ +# Dutch translations for Open Wine Map. Hand-seeded; review and amend in +# place. +msgid "" +msgstr "" +"Project-Id-Version: open-wine-map\n" +"Report-Msgid-Bugs-To: EMAIL@ADDRESS\n" +"POT-Creation-Date: 2026-09-08 21:10+0200\n" +"PO-Revision-Date: YEAR-MO-DA HO:MI+ZONE\n" +"Last-Translator: FULL NAME <EMAIL@ADDRESS>\n" +"Language: nl\n" +"Language-Team: nl <LL@li.org>\n" +"Plural-Forms: nplurals=2; plural=(n != 1);\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.18.0\n" + +#: scripts/_lib/aging_taxonomy.py:184 +msgid "Vieillissement" +msgstr "Rijping" + +#: scripts/_lib/aging_taxonomy.py:185 +msgid "Prädikat" +msgstr "Prädikat" + +#: scripts/_lib/aging_taxonomy.py:186 +msgid "Sélection" +msgstr "Selectie" + +#: scripts/_lib/map_template.py:46 +msgid "Open Wine Map — appellations viticoles d'Europe" +msgstr "Open Wine Map — Europese wijnappellaties" +#: scripts/_lib/map_template.py:47 +msgid "carte des appellations viticoles" +msgstr "Kaart van wijnappellaties" + +#: scripts/_lib/map_template.py:49 +msgid "" +"Carte interactive des appellations viticoles d'Europe : cépages, styles " +"et terroir, d'après les registres officiels." +msgstr "Interactieve kaart van de wijnappellaties van Europa: druivenrassen, stijlen en terroir, uit de officiële registers." +#: scripts/_lib/map_template.py:52 +msgid "Chargement…" +msgstr "Laden…" + +#: scripts/_lib/map_template.py:53 +msgid "Recherche" +msgstr "Zoeken" + +#: scripts/_lib/map_template.py:54 +msgid "nom d'appellation…" +msgstr "naam van appellatie…" + +#: scripts/_lib/map_template.py:55 +msgid "Recherche d'appellation…" +msgstr "Appellatie zoeken…" + +#: scripts/_lib/map_template.py:56 +msgid "Recherche de cépage…" +msgstr "Druif zoeken…" + +#: scripts/_lib/map_template.py:58 +msgid "Rechercher une appellation, un cépage, une région…" +msgstr "Zoek een appellatie, druif of streek…" + +#: scripts/_lib/map_template.py:59 +msgid "Cépage principal uniquement" +msgstr "Alleen hoofddruif" + +#: scripts/_lib/map_template.py:60 +#, python-brace-format +msgid "Aucun résultat pour « {q} »" +msgstr "Geen resultaten voor ‘{q}’" + +#: scripts/_lib/map_template.py:61 +msgid "Options" +msgstr "Opties" + +#: scripts/_lib/map_template.py:62 +msgid "Filtres actifs" +msgstr "Actieve filters" + +#: scripts/_lib/map_template.py:63 +msgid "Tout sélectionner" +msgstr "Alles selecteren" + +#: scripts/_lib/map_template.py:65 +#, python-brace-format +msgid "{p} appellations · {s} dénominations géographiques complémentaires" +msgstr "{p} appellaties · {s} aanvullende geografische benamingen" + +#: scripts/_lib/map_template.py:67 +#, python-brace-format +msgid "{p} appellations" +msgstr "{p} appellaties" + +#: scripts/_lib/map_template.py:68 +#, python-brace-format +msgid "Ouvrir la fiche de {name}" +msgstr "Open de fiche van {name}" + +#: scripts/_lib/map_template.py:69 +msgid "Ouvrir la fiche" +msgstr "Open fiche" + +#: scripts/_lib/map_template.py:70 +msgid "Inclure les spiritueux" +msgstr "Gedistilleerd tonen" + +#: scripts/_lib/map_template.py:71 +msgid "Style de vin" +msgstr "Wijnstijl" + +#: scripts/_lib/map_template.py:72 +msgid "Classement" +msgstr "Classificatie" + +#: scripts/_lib/map_template.py:73 +msgid "Cépages principaux" +msgstr "Hoofddruivenrassen" + +#: scripts/_lib/map_template.py:74 +msgid "Cépages accessoires" +msgstr "Aanvullende druivenrassen" + +#: scripts/_lib/map_template.py:75 scripts/_lib/map_template.py:189 +msgid "Cépages" +msgstr "Druivenrassen" + +#: scripts/_lib/map_template.py:76 +msgid "Région" +msgstr "Regio" + +#: scripts/_lib/map_template.py:77 +msgid "Appellation" +msgstr "Appellatie" + +#: scripts/_lib/map_template.py:78 +msgid "Type" +msgstr "Type" + +#: scripts/_lib/map_template.py:79 +msgid "Origine protégée (AOP, AOC, DOC, DO…)" +msgstr "Beschermde oorsprong (BOB, AOC, DOC, DO…)" + +#: scripts/_lib/map_template.py:80 +msgid "Indication géographique (IGP, IGT, Landwein…)" +msgstr "Geografische aanduiding (BGA, IGT, Landwein…)" + +#: scripts/_lib/map_template.py:81 +msgid "AOP" +msgstr "BOB" + +#: scripts/_lib/map_template.py:82 +msgid "IGP" +msgstr "BGA" + +#: scripts/_lib/map_template.py:83 +msgid "IG spiritueux" +msgstr "GA gedistilleerde drank" + +#: scripts/_lib/map_template.py:84 +msgid "Type d'appellation" +msgstr "Type appellatie" + +#: scripts/_lib/map_template.py:85 +msgid "AOP (régime britannique)" +msgstr "BOB (Britse regeling)" + +#: scripts/_lib/map_template.py:86 +msgid "IGP (régime britannique)" +msgstr "BGA (Britse regeling)" + +#: scripts/_lib/map_template.py:87 +msgid "AOC (Suisse)" +msgstr "AOC (Zwitserland)" + +#: scripts/_lib/map_template.py:88 +msgid "Source" +msgstr "Bron" + +#: scripts/_lib/map_template.py:89 +msgid "Vue" +msgstr "Weergave" + +#: scripts/_lib/map_template.py:90 +msgid "Simple" +msgstr "Eenvoudig" + +#: scripts/_lib/map_template.py:91 +msgid "Avancée" +msgstr "Geavanceerd" + +#: scripts/_lib/map_template.py:92 +msgid "Thème" +msgstr "Thema" + +#: scripts/_lib/map_template.py:93 +msgid "Clair" +msgstr "Licht" + +#: scripts/_lib/map_template.py:94 +msgid "Sombre" +msgstr "Donker" + +#: scripts/_lib/map_template.py:95 +msgid "Système" +msgstr "Systeem" + +#: scripts/_lib/map_template.py:96 +msgid "Afficher les IGP" +msgstr "BGA's tonen" + +#: scripts/_lib/map_template.py:97 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:361 +msgid "rouge" +msgstr "rood" + +#: scripts/_lib/map_template.py:98 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:362 +msgid "blanc" +msgstr "wit" + +#: scripts/_lib/map_template.py:99 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:363 +msgid "rosé" +msgstr "rosé" + +#: scripts/_lib/map_template.py:100 scripts/_lib/style_taxonomy.py:332 +#: scripts/_lib/style_taxonomy.py:364 +msgid "mousseux" +msgstr "mousserend" + +#: scripts/_lib/map_template.py:101 +msgid "moelleux / liquoreux" +msgstr "zoet / dessertwijn" + +#: scripts/_lib/map_template.py:102 scripts/_lib/style_taxonomy.py:345 +msgid "oxydatif" +msgstr "oxidatief" + +#: scripts/_lib/map_template.py:103 +msgid "autre" +msgstr "andere" + +#: scripts/_lib/map_template.py:104 +msgid "Réinitialiser" +msgstr "Resetten" + +#: scripts/_lib/map_template.py:105 +#, python-brace-format +msgid "{n} appellations" +msgstr "{n} appellaties" + +#: scripts/_lib/map_template.py:106 +#, python-brace-format +msgid "{n} / {total} appellations" +msgstr "{n} / {total} appellaties" + +#: scripts/_lib/map_template.py:107 +#, python-brace-format +msgid "{n} masquées dans les IGP · afficher" +msgstr "{n} verborgen in BGA's · tonen" + +#: scripts/_lib/map_template.py:108 +msgid "Fermer" +msgstr "Sluiten" + +#: scripts/_lib/map_template.py:109 +msgid "Détails de l'appellation" +msgstr "Details van de appellatie" + +#: scripts/_lib/map_template.py:110 +#, python-brace-format +msgid "Retirer le filtre {label}" +msgstr "Filter {label} verwijderen" + +#: scripts/_lib/map_template.py:111 +msgid "Filtres et options de la carte" +msgstr "Filters en kaartopties" + +#: scripts/_lib/map_template.py:112 +msgid "Langue" +msgstr "Taal" + +#: scripts/_lib/map_template.py:113 +msgid "Carte des appellations viticoles" +msgstr "Kaart van wijnappellaties" + +#: scripts/_lib/map_template.py:114 +msgid "Aller à la carte" +msgstr "Naar de kaart" + +#: scripts/_lib/map_template.py:115 +msgid "Styles" +msgstr "Stijlen" + +#: scripts/_lib/map_template.py:116 +msgid "Variétés d'intérêt" +msgstr "Variëteiten van belang" + +#: scripts/_lib/map_template.py:117 +msgid "Sources" +msgstr "Bronnen" + +#: scripts/_lib/map_template.py:118 +msgid "Terroir" +msgstr "Terroir" + +#: scripts/_lib/map_template.py:119 +#, python-brace-format +msgid "Dűlők (lieux-dits) : {n}" +msgstr "Dűlők (wijngaarden): {n}" + +#: scripts/_lib/map_template.py:120 +#, python-brace-format +msgid "Menzioni geografiche aggiuntive (crus) : {n}" +msgstr "Aanvullende geografische aanduidingen (crus): {n}" + +#: scripts/_lib/map_template.py:121 +msgid "Facteurs naturels" +msgstr "Natuurlijke factoren" + +#: scripts/_lib/map_template.py:122 +msgid "Facteurs humains" +msgstr "Menselijke factoren" + +#: scripts/_lib/map_template.py:123 +msgid "Caractéristiques du produit" +msgstr "Producteigenschappen" + +#: scripts/_lib/map_template.py:124 +msgid "Lien terroir / vin" +msgstr "Verband terroir / wijn" + +#: scripts/_lib/map_template.py:126 +#, python-brace-format +msgid "" +"Faits dégagés du Lien au terroir par interprétation automatique — voir la" +" {source}." +msgstr "" +"Feiten afgeleid uit de terroir-bandsectie (Lien au terroir) door " +"automatische interpretatie — zie de {source}." + +#: scripts/_lib/map_template.py:128 +msgid "source" +msgstr "bron" + +#: scripts/_lib/map_template.py:129 +msgid "via Wikipedia · CC BY-SA 4.0" +msgstr "via Wikipedia · CC BY-SA 4.0" + +#: scripts/_lib/map_template.py:131 +#, python-brace-format +msgid "Citation textuelle du Lien au terroir — voir la {source}." +msgstr "" +"Letterlijk citaat uit de terroir-bandsectie (Lien au terroir) — zie de " +"{source}." + +#: scripts/_lib/map_template.py:133 +msgid "à vérifier — texte source court" +msgstr "te controleren — korte brontekst" + +#: scripts/_lib/map_template.py:135 +#, python-brace-format +msgid "" +"Ces repères décrivent l'appellation englobante {parent} — pas " +"spécifiquement cette dénomination." +msgstr "" +"Deze kenmerken beschrijven de overkoepelende appellatie {parent} — niet " +"specifiek deze denominatie." + +#: scripts/_lib/map_template.py:138 +msgid "sans région" +msgstr "geen regio" + +#: scripts/_lib/map_template.py:139 +#, python-brace-format +msgid "{n} commune(s) INAO" +msgstr "{n} INAO-gemeente(n)" + +#: scripts/_lib/map_template.py:140 +#, python-brace-format +msgid "{n} commune(s)" +msgstr "{n} gemeente(n)" + +#: scripts/_lib/map_template.py:141 +msgid "aire approchée" +msgstr "benaderd gebied" + +#: scripts/_lib/map_template.py:142 +msgid "aire approchée (à l'échelle communale)" +msgstr "benaderd gebied (gemeenteniveau)" + +#: scripts/_lib/map_template.py:144 +#, python-brace-format +msgid "" +"Aire approchée — pas de données parcellaires précises pour cette " +"dénomination ; polygone hérité de {umbrella}." +msgstr "" +"Benaderd gebied — geen perceelnauwkeurige data beschikbaar voor deze " +"benaming; polygoon overgenomen van {umbrella}." + +#: scripts/_lib/map_template.py:148 +msgid "" +"Aire approchée — pas de données parcellaires précises pour cette " +"dénomination ; polygone hérité de l'appellation parente." +msgstr "" +"Benaderd gebied — geen perceelnauwkeurige data beschikbaar voor deze " +"benaming; polygoon overgenomen van de bovenliggende appellatie." + +#: scripts/_lib/map_template.py:152 +msgid "" +"Aire approchée — pas de données parcellaires disponibles ; affichée comme" +" l'emprise de la commune où se situe la dénomination." +msgstr "" +"Benaderd gebied — geen perceelgegevens beschikbaar; weergegeven als de " +"omtrek van de gemeente waarin de benaming ligt." + +#: scripts/_lib/map_template.py:156 +#, python-brace-format +msgid "" +"Aire issue du lieu-dit cadastral « {lieu_dit} » (commune de {commune}, " +"{source})." +msgstr "" +"Gebied afgeleid van het kadastrale gehucht „{lieu_dit}” (gemeente " +"{commune}, {source})." + +#: scripts/_lib/map_template.py:159 +msgid "cadastre.data.gouv.fr" +msgstr "cadastre.data.gouv.fr" + +#: scripts/_lib/map_template.py:161 +msgid "" +"Aire approchée — reconstituée à partir des références parcellaires du " +"plan de l'aire délimitée annexé au cahier des charges ; ce n'est pas une " +"limite officielle." +msgstr "" +"Bij benadering — gereconstrueerd op basis van de perceelreferenties op " +"het plan van het afgebakende gebied bij het productdossier; dit is geen " +"officiële grens." + +#: scripts/_lib/map_template.py:165 +#, python-brace-format +msgid "{n} appellations à ce point" +msgstr "{n} appellaties op dit punt" + +#: scripts/_lib/map_template.py:166 +msgid "Cliquer à nouveau pour parcourir les autres" +msgstr "Klik opnieuw om door de andere te bladeren" + +#: scripts/_lib/map_template.py:167 +msgid "Cahier des charges (BO Agri, PDF)" +msgstr "Productdossier (BO Agri, PDF)" + +#: scripts/_lib/map_template.py:168 +msgid "Cahier des charges (registre GI de l'UE, PDF)" +msgstr "Productdossier (EU GI-register, PDF)" + +#: scripts/_lib/map_template.py:169 +msgid "homologué" +msgstr "goedgekeurd" + +#: scripts/_lib/map_template.py:170 +msgid "JORF" +msgstr "JORF" + +#: scripts/_lib/map_template.py:171 +msgid "Texte officiel INAO (show_texte)" +msgstr "Officiële INAO-tekst (show_texte)" + +#: scripts/_lib/map_template.py:172 +msgid "Fiche produit INAO" +msgstr "INAO-productfiche" + +#: scripts/_lib/map_template.py:173 +msgid "Site officiel de l'interprofession" +msgstr "Officiële site van de wijnorganisatie" + +#: scripts/_lib/map_template.py:174 +msgid "Cahier des charges (EUR-Lex, document unique)" +msgstr "Productspecificatie (EUR-Lex, enig document)" + +#: scripts/_lib/map_template.py:175 +msgid "Pliego de condiciones (national, PDF)" +msgstr "Nationaal pliego de condiciones (PDF)" + +#: scripts/_lib/map_template.py:176 +msgid "variétés ajoutées" +msgstr "rassen toegevoegd" + +#: scripts/_lib/map_template.py:177 +msgid "Cahier des charges national (PDF)" +msgstr "Nationaal productdossier (PDF)" + +#: scripts/_lib/map_template.py:178 +msgid "Spécification du produit (IGP, PDF)" +msgstr "Productdossier (BGA, PDF)" + +#: scripts/_lib/map_template.py:179 +msgid "Registre régional des cépages (PDF)" +msgstr "Regionaal druivenrasregister (PDF)" + +#: scripts/_lib/map_template.py:180 +msgid "Registre eAmbrosia (UE)" +msgstr "eAmbrosia-register (EU)" + +#: scripts/_lib/map_template.py:181 +msgid "Numéro de dossier" +msgstr "Dossiernummer" + +#: scripts/_lib/map_template.py:182 +msgid "Règlement cantonal sur la vigne et le vin" +msgstr "Kantonaal wijnreglement" + +#: scripts/_lib/map_template.py:183 +msgid "Répertoire suisse des AOC (OFAG/BLW)" +msgstr "Zwitsers AOC-register (OFAG/BLW)" + +#: scripts/_lib/map_template.py:184 +msgid "Cahier des charges (GOV.UK, PDF)" +msgstr "Productdossier (GOV.UK, PDF)" + +#: scripts/_lib/map_template.py:185 +msgid "Registre des IG du Royaume-Uni" +msgstr "Register van geografische aanduidingen (VK)" + +#: scripts/_lib/map_template.py:186 +msgid "Légende couleurs" +msgstr "Legende" + +#: scripts/_lib/map_template.py:187 +msgid "Bassin viticole" +msgstr "Wijnstreek" + +#: scripts/_lib/map_template.py:188 +msgid "Plus l'aire est petite, plus la teinte est dense." +msgstr "Hoe kleiner het gebied, hoe dichter de tint." + +#: scripts/_lib/map_template.py:190 +msgid "principal — variété de la cuvée" +msgstr "hoofdras — variëteit van de cuvée" + +#: scripts/_lib/map_template.py:191 +msgid "accessoire — assemblage limité" +msgstr "bijras — beperkte assemblage" + +#: scripts/_lib/map_template.py:192 +msgid "intérêt — observation/conservation" +msgstr "interesse — observatie/behoud" + +#: scripts/_lib/map_template.py:194 +msgid "" +"Le régulateur portugais (IVV) n'établit pas de distinction " +"principal/accessoire — toutes les castas autorisées sont listées ensemble" +" dans le caderno de especificações." +msgstr "" +"De Portugese regelgever (IVV) maakt geen onderscheid tussen hoofdrassen " +"en bijrassen — alle toegelaten druivenrassen worden samen vermeld in de " +"caderno de especificações." + +#: scripts/_lib/map_template.py:198 +msgid "" +"Bianchello (ou Biancame) est traité ici comme un cépage distinct, " +"conformément au disciplinare de la DOP Bianchello del Metauro ; le " +"catalogue VIVC le recense comme synonyme du Trebbiano Toscano." +msgstr "" +"Bianchello (ook Biancame) wordt hier als een afzonderlijk ras behandeld, " +"in overeenstemming met het disciplinare van de DOP Bianchello del " +"Metauro; de VIVC-catalogus vermeldt het als synoniem van Trebbiano " +"Toscano." + +#: scripts/_lib/map_template.py:203 +#, python-brace-format +msgid "Open Wine Map n'a pas encore trouvé de {doc} pour cette appellation." +msgstr "Open Wine Map heeft nog geen {doc} gevonden voor deze benaming." + +#: scripts/_lib/map_template.py:205 +msgid "aidez-nous à le trouver" +msgstr "help ons het te vinden" + +#: scripts/_lib/map_template.py:211 +#, python-brace-format +msgid "Délimitée par {regulator} dans son {doc}, qui autorise {grapes}." +msgstr "Afgebakend door {regulator} in zijn {doc}, dat {grapes} toestaat." + +#: scripts/_lib/map_template.py:213 +#, python-brace-format +msgid "Délimitée par {regulator} dans son {doc}{extra}." +msgstr "Afgebakend door {regulator} in zijn {doc}{extra}." + +#: scripts/_lib/map_template.py:214 +#, python-brace-format +msgid "{names} et {n} autres cépages" +msgstr "{names} en {n} andere druivenrassen" + +#: scripts/_lib/map_template.py:215 +msgid "le registre eAmbrosia de l'UE" +msgstr "het eAmbrosia-register van de EU" + +#: scripts/_lib/map_template.py:216 +#, python-brace-format +msgid "le canton de {canton}" +msgstr "het kanton {canton}" + +#: scripts/_lib/map_template.py:217 +msgid "(français)" +msgstr "(bron: Frans)" + +#: scripts/_lib/map_template.py:218 +msgid "Texte source en français" +msgstr "Brontekst in het Frans" + +#: scripts/_lib/map_template.py:219 +msgid "(español)" +msgstr "(bron: Spaans)" + +#: scripts/_lib/map_template.py:220 +msgid "Texte source en espagnol" +msgstr "Brontekst in het Spaans" + +#: scripts/_lib/map_template.py:221 +msgid "(português)" +msgstr "(bron: Portugees)" + +#: scripts/_lib/map_template.py:222 +msgid "Texte source en portugais" +msgstr "Brontekst in het Portugees" + +#: scripts/_lib/map_template.py:223 +msgid "Filtres" +msgstr "Filters" + +#: scripts/_lib/map_template.py:224 +#, python-brace-format +msgid "Traduit de {wiki} · CC BY-SA 4.0" +msgstr "Vertaald uit {wiki} · CC BY-SA 4.0" + +#: scripts/_lib/map_template.py:225 +msgid "Wikipédia en anglais" +msgstr "Engelstalige Wikipedia" + +#: scripts/_lib/map_template.py:226 +msgid "Wikipédia en français" +msgstr "Franstalige Wikipedia" + +#: scripts/_lib/map_template.py:227 +msgid "Wikipédia en espagnol" +msgstr "Spaanstalige Wikipedia" + +#: scripts/_lib/map_template.py:228 +msgid "Wikipédia en néerlandais" +msgstr "Nederlandstalige Wikipedia" + +#: scripts/_lib/map_template.py:229 +msgid "Wikipédia en portugais" +msgstr "Portugeestalige Wikipedia" + +#: scripts/_lib/map_template.py:230 +msgid "Wikipédia en croate" +msgstr "Kroatische Wikipedia" + +#: scripts/_lib/map_template.py:231 +msgid "Vitis International Variety Catalogue (Julius Kühn-Institut)" +msgstr "Vitis International Variety Catalogue (Julius Kühn-Institut)" + +#: scripts/_lib/map_template.py:232 +#, python-brace-format +msgid "VIVC #{id}" +msgstr "VIVC #{id}" + +#: scripts/_lib/map_template.py:233 +#, python-brace-format +msgid "Traduction automatique depuis {source}" +msgstr "Automatisch vertaald uit {source}" + +#: scripts/_lib/map_template.py:234 +msgid "le cahier des charges" +msgstr "het cahier des charges" + +#: scripts/_lib/map_template.py:235 scripts/_lib/map_template.py:237 +msgid "le pliego de condiciones" +msgstr "de productspecificatie" + +#: scripts/_lib/map_template.py:236 scripts/_lib/map_template.py:238 +msgid "le caderno de especificações" +msgstr "de caderno de especificações" + +#: scripts/_lib/map_template.py:239 +msgid "Dénomination géographique complémentaire de" +msgstr "Deelgebied van" + +#: scripts/_lib/map_template.py:240 +msgid "À propos" +msgstr "Over" + +#: scripts/_lib/map_template.py:241 +msgid "À propos d'Open Wine Map" +msgstr "Over Open Wine Map" + +#: scripts/_lib/map_template.py:243 +msgid "" +"Carte de référence des appellations viticoles, générée automatiquement à " +"partir des registres publics : le registre de l'Union européenne des AOP " +"et IGP, les régulateurs nationaux, le répertoire fédéral suisse des AOC " +"cantonales et le registre britannique des indications géographiques." +msgstr "" +"Referentiekaart van wijnappellaties, automatisch gegenereerd uit openbare" +" registers: het EU-register van BOB's en BGA's, de nationale regulatoren," +" het Zwitserse federale repertorium van kantonnale AOC's en het Britse " +"register van geografische aanduidingen." + +#: scripts/_lib/map_template.py:249 +msgid "" +"Une partie du texte est produite par des modèles de langage : les repères" +" de terroir sont dégagés du texte du régulateur et traduits par Claude " +"(Sonnet 4.6) ; les extraits Wikipedia des infobulles de cépages et de " +"styles sont traduits pour l'essentiel par Mistral Small 3.2, exécuté " +"localement, et pour quelques-uns par Claude ; les résumés des cahiers des" +" charges ont été traduits par un traducteur humain, à l'exception d'un " +"petit reliquat traduit automatiquement. Chaque élément porte sa propre " +"ligne d'attribution dans le panneau." +msgstr "" +"Een deel van de tekst wordt door taalmodellen gemaakt: de terroirnotities" +" worden uit de tekst van de regulator gehaald en door Claude (Sonnet 4.6)" +" vertaald; de Wikipedia-fragmenten in de tooltips voor druivenrassen en " +"stijlen zijn grotendeels vertaald door Mistral Small 3.2, lokaal " +"uitgevoerd, en enkele door Claude; de samenvattingen van de cahiers des " +"charges zijn door een menselijke vertaler vertaald, op een klein " +"machinaal vertaald restant na. Elk onderdeel heeft zijn eigen " +"bronvermelding in het paneel." + +#: scripts/_lib/map_template.py:258 +#, python-brace-format +msgid "Réalisé avec ♡ par {devloed}." +msgstr "Met ♡ gemaakt door {devloed}." + +#: scripts/_lib/map_template.py:260 +#, python-brace-format +msgid "" +"Sources : INAO ({inao}) pour les cahiers des charges et les aires " +"parcellaires françaises, IGN ({ign}) pour les contours des communes " +"françaises, le registre des indications géographiques de l'UE, les " +"régulateurs nationaux, Eurostat GISCO et Bétard 2022 pour les autres " +"pays, OpenStreetMap et CARTO pour le fond de carte, Wikipedia " +"({wikipedia}) pour quelques compléments narratifs (CC BY-SA 4.0), et VIVC" +" ({vivc}), le Vitis International Variety Catalogue du Julius Kühn-" +"Institut, pour les noms canoniques et numéros de cépage (citation Röckel " +"et al.). Tout extrait Wikipedia est signalé sur place. Détails et " +"licences dans le {readme}." +msgstr "" +"Bronnen: INAO ({inao}) voor de Franse cahiers des charges en " +"perceelsgrenzen, IGN ({ign}) voor de Franse gemeentegrenzen, het EU-" +"register van geografische aanduidingen, de nationale regulatoren, " +"Eurostat GISCO en Bétard 2022 voor de andere landen, OpenStreetMap en " +"CARTO voor de basiskaart, Wikipedia ({wikipedia}) voor enkele narratieve " +"aanvullingen (CC BY-SA 4.0), en VIVC ({vivc}), de Vitis International " +"Variety Catalogue van het Julius Kühn-Institut, voor de canonieke " +"druivennamen en rasnummers (citaat: Röckel et al.). Elk Wikipedia-" +"fragment wordt ter plaatse vermeld. Details en licenties in de {readme}." + +#: scripts/_lib/map_template.py:270 +#, python-brace-format +msgid "Suggestions et pull requests bienvenues sur {github}." +msgstr "Issues en pull requests welkom op {github}." + +#: scripts/_lib/map_template.py:271 +msgid "ticket GitHub" +msgstr "GitHub-issue" + +#: scripts/_lib/map_template.py:272 +msgid "e-mail" +msgstr "e-mail" + +#: scripts/_lib/map_template.py:273 +msgid "E-mail copié dans le presse-papiers" +msgstr "E-mail gekopieerd naar klembord" + +#: scripts/_lib/map_template.py:275 +#, python-brace-format +msgid "" +"Carte générée automatiquement — des erreurs sont possibles. Signalez-les " +"via {issue} ou {email}." +msgstr "" +"Automatisch gegenereerde kaart — fouten zijn mogelijk. Meld ze via " +"{issue} of {email}." + +#: scripts/_lib/map_template.py:279 +#, python-brace-format +msgid "" +"{c} pays européens cartographiés ({n} appellations : {parents} " +"appellations et {subs} dénominations rattachées). Des itérations " +"supplémentaires affineront la qualité des données et étendront la " +"couverture au-delà de l'UE, de la Suisse et du Royaume-Uni." +msgstr "" +"{c} Europese landen in kaart gebracht ({n} vermeldingen: {parents} " +"appellaties en {subs} gekoppelde benamingen). Volgende iteraties " +"verfijnen de datakwaliteit en breiden de dekking uit tot buiten de EU, " +"Zwitserland en het Verenigd Koninkrijk." + +#: scripts/_lib/map_template.py:284 +msgid "Toutes les appellations" +msgstr "Alle appellaties" + +#: scripts/_lib/map_template.py:285 +msgid "Toutes les appellations viticoles — Open Wine Map" +msgstr "Alle wijnappellaties — Open Wine Map" + +#: scripts/_lib/map_template.py:287 +#, python-brace-format +msgid "" +"Liste des {n} appellations viticoles cartographiées sur Open Wine Map, " +"classées par pays : AOP et IGP de l'UE avec leurs termes traditionnels " +"(AOC, DOCG, DOQ…), AOC suisses et IG britanniques." +msgstr "" +"Lijst van de {n} wijnappellaties op Open Wine Map, per land: BOB's en " +"BGA's van de EU met hun traditionele aanduidingen (AOC, DOCG, DOQ…), " +"Zwitserse AOC's en Britse GA's." + +#: scripts/_lib/map_template.py:292 +#, python-brace-format +msgid "" +"Les {n} appellations ci-dessous sont classées par pays. Retour à la " +"{map_link}." +msgstr "" +"De {n} onderstaande appellaties zijn gegroepeerd per land. Terug naar de " +"{map_link}." + +#: scripts/_lib/map_template.py:295 +msgid "carte interactive" +msgstr "interactieve kaart" + +#: scripts/_lib/map_template.py:296 +msgid "Liste des appellations par pays" +msgstr "Lijst van appellaties per land" + +#: scripts/_lib/map_template.py:297 +msgid "Appellation parente" +msgstr "Bovenliggende appellatie" + +#: scripts/_lib/map_template.py:298 +msgid "Dénominations rattachées" +msgstr "Gekoppelde benamingen" + +#: scripts/_lib/map_template.py:300 +#, python-brace-format +msgid "Parcourir la liste complète des appellations : {browse_link}." +msgstr "Blader door de volledige lijst van appellaties: {browse_link}." + +#: scripts/_lib/map_template.py:302 +#, python-brace-format +msgid "Données mises à jour le {date}." +msgstr "Gegevens bijgewerkt op {date}." + +#: scripts/_lib/map_template.py:317 +msgid "BOURGOGNE" +msgstr "Bourgogne" + +#: scripts/_lib/map_template.py:318 +msgid "BEAUJOLAIS" +msgstr "Beaujolais" + +#: scripts/_lib/map_template.py:319 +msgid "JURA" +msgstr "Jura" + +#: scripts/_lib/map_template.py:320 +msgid "SAVOIE" +msgstr "Savoie" + +#: scripts/_lib/map_template.py:321 +msgid "BUGEY" +msgstr "Bugey" + +#: scripts/_lib/map_template.py:322 +msgid "ALSACE ET EST" +msgstr "Elzas en Oost" + +#: scripts/_lib/map_template.py:323 +msgid "VAL DE LOIRE" +msgstr "Loirestreek" + +#: scripts/_lib/map_template.py:324 +msgid "SUD-OUEST" +msgstr "Zuidwest-Frankrijk" + +#: scripts/_lib/map_template.py:325 +msgid "VALLEE DU RHÔNE" +msgstr "Rhônestreek" + +#: scripts/_lib/map_template.py:326 +msgid "LANGUEDOC-ROUSSILLON" +msgstr "Languedoc-Roussillon" + +#: scripts/_lib/map_template.py:327 +msgid "TOULOUSE-PYRENEES" +msgstr "Toulouse-Pyreneeën" + +#: scripts/_lib/map_template.py:328 +msgid "PROVENCE-CORSE" +msgstr "Provence-Corsica" + +#: scripts/_lib/map_template.py:329 +msgid "CHAMPAGNE" +msgstr "Champagne" + +#: scripts/_lib/map_template.py:330 +msgid "EAUX-DE-VIE DE CIDRE" +msgstr "Cider-eaux-de-vie" + +#: scripts/_lib/map_template.py:331 +msgid "VIN DOUX NATURELS" +msgstr "Natuurlijk zoete wijnen" + +#: scripts/_lib/map_template.py:332 +msgid "COGNAC" +msgstr "Cognac" + +#: scripts/_lib/map_template.py:333 +msgid "ARMAGNAC" +msgstr "Armagnac" + +#: scripts/_lib/map_template.py:334 +msgid "RHUM" +msgstr "Rum" + +#: scripts/_lib/map_template.py:374 +msgid "France" +msgstr "Frankrijk" + +#: scripts/_lib/map_template.py:375 +msgid "Espagne" +msgstr "Spanje" + +#: scripts/_lib/map_template.py:376 +msgid "Portugal" +msgstr "Portugal" + +#: scripts/_lib/map_template.py:377 +msgid "Italie" +msgstr "Italië" + +#: scripts/_lib/map_template.py:378 +msgid "Autriche" +msgstr "Oostenrijk" + +#: scripts/_lib/map_template.py:379 +msgid "Slovénie" +msgstr "Slovenië" + +#: scripts/_lib/map_template.py:380 +msgid "Croatie" +msgstr "Kroatië" + +#: scripts/_lib/map_template.py:381 +msgid "Hongrie" +msgstr "Hongarije" + +#: scripts/_lib/map_template.py:382 +msgid "Roumanie" +msgstr "Roemenië" + +#: scripts/_lib/map_template.py:383 +msgid "Bulgarie" +msgstr "Bulgarije" + +#: scripts/_lib/map_template.py:384 +msgid "Grèce" +msgstr "Griekenland" + +#: scripts/_lib/map_template.py:385 +msgid "Allemagne" +msgstr "Duitsland" + +#: scripts/_lib/map_template.py:386 +msgid "Slovaquie" +msgstr "Slowakije" + +#: scripts/_lib/map_template.py:387 +msgid "Suisse" +msgstr "Zwitserland" + +#: scripts/_lib/map_template.py:388 +msgid "Tchéquie" +msgstr "Tsjechië" + +#: scripts/_lib/map_template.py:389 +msgid "Luxembourg" +msgstr "Luxemburg" + +#: scripts/_lib/map_template.py:390 +msgid "Belgique" +msgstr "België" + +#: scripts/_lib/map_template.py:391 +msgid "Pays-Bas" +msgstr "Nederland" + +#: scripts/_lib/map_template.py:392 +msgid "Malte" +msgstr "Malta" + +#: scripts/_lib/map_template.py:393 +msgid "Chypre" +msgstr "Cyprus" + +#: scripts/_lib/map_template.py:394 +msgid "Royaume-Uni" +msgstr "Verenigd Koninkrijk" + +#: scripts/_lib/style_taxonomy.py:332 scripts/_lib/style_taxonomy.py:365 +msgid "doux" +msgstr "zoet" + +#: scripts/_lib/style_taxonomy.py:333 scripts/_lib/style_taxonomy.py:366 +msgid "autres" +msgstr "overige" + +#: scripts/_lib/style_taxonomy.py:334 +msgid "clairet" +msgstr "clairet" + +#: scripts/_lib/style_taxonomy.py:334 +msgid "primeur" +msgstr "primeur" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "mousseux de qualité" +msgstr "kwaliteitsmousserend" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "crémant" +msgstr "crémant" + +#: scripts/_lib/style_taxonomy.py:335 +msgid "méthode ancestrale" +msgstr "méthode ancestrale" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode traditionnelle" +msgstr "traditionele methode" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode Charmat" +msgstr "charmatmethode" + +#: scripts/_lib/style_taxonomy.py:336 +msgid "méthode dioise" +msgstr "méthode dioise" + +#: scripts/_lib/style_taxonomy.py:337 +msgid "pétillant" +msgstr "parelend" + +#: scripts/_lib/style_taxonomy.py:338 +msgid "vin muté" +msgstr "versterkte wijn" + +#: scripts/_lib/style_taxonomy.py:338 +msgid "vendanges tardives (catégorie)" +msgstr "late oogst (categorie)" + +#: scripts/_lib/style_taxonomy.py:339 +msgid "vin de raisins passerillés" +msgstr "rozijnenwijn" + +#: scripts/_lib/style_taxonomy.py:340 +msgid "demi-doux" +msgstr "halfzoet" + +#: scripts/_lib/style_taxonomy.py:340 +msgid "vin de glace" +msgstr "ijswijn" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "vin doux naturel" +msgstr "vin doux naturel" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "vin de liqueur" +msgstr "vin de liqueur" + +#: scripts/_lib/style_taxonomy.py:341 +msgid "mistelle" +msgstr "mistelle" + +#: scripts/_lib/style_taxonomy.py:342 +msgid "vendanges tardives" +msgstr "late oogst" + +#: scripts/_lib/style_taxonomy.py:342 +msgid "uvas sobremaduradas" +msgstr "overrijpe druiven" + +#: scripts/_lib/style_taxonomy.py:343 +msgid "vin naturellement doux" +msgstr "natuurzoete wijn" + +#: scripts/_lib/style_taxonomy.py:343 +msgid "grains nobles" +msgstr "grains nobles" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "vin de paille" +msgstr "vin de paille" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "uvas pasificadas" +msgstr "gedroogde druiven" + +#: scripts/_lib/style_taxonomy.py:344 +msgid "vin santo" +msgstr "vin santo" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "tranquille" +msgstr "stil" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "sur lie" +msgstr "sur lie" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "sec" +msgstr "droog" + +#: scripts/_lib/style_taxonomy.py:345 +msgid "demi-sec" +msgstr "halfdroog" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "rancio" +msgstr "rancio" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "vin jaune" +msgstr "vin jaune" + +#: scripts/_lib/style_taxonomy.py:346 +msgid "generoso" +msgstr "generoso" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "fino" +msgstr "fino" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "manzanilla" +msgstr "manzanilla" + +#: scripts/_lib/style_taxonomy.py:347 +msgid "amontillado" +msgstr "amontillado" + +#: scripts/_lib/style_taxonomy.py:348 +msgid "oloroso" +msgstr "oloroso" + +#: scripts/_lib/style_taxonomy.py:348 +msgid "palo cortado" +msgstr "palo cortado" + +#~ msgid "carte des appellations françaises" +#~ msgstr "kaart van Franse appellaties" + +#~ msgid "" +#~ "Priorité actuelle : affiner la " +#~ "couverture française — précision des " +#~ "aires, qualité des extraits, climats et" +#~ " lieux-dits. L'extension à d'autres " +#~ "pays viticoles viendra ensuite." +#~ msgstr "" +#~ "Huidige prioriteit: de Franse dekking " +#~ "aanscherpen — nauwkeurigheid van de " +#~ "gebieden, kwaliteit van de extracten, " +#~ "climats en lieux-dits. Uitbreiding naar" +#~ " andere wijnlanden volgt daarna." + +#~ msgid "anglais" +#~ msgstr "" + +#~ msgid "français" +#~ msgstr "(Frans)" + +#~ msgid "espagnol" +#~ msgstr "(Spaans)" + +#~ msgid "néerlandais" +#~ msgstr "" + +#~ msgid "" +#~ "Couverture actuelle : France et une " +#~ "première version de l'Espagne. Prochaines " +#~ "étapes : amélioration continue de la " +#~ "qualité des données, puis ajout du " +#~ "Portugal." +#~ msgstr "" +#~ "Huidige dekking: Frankrijk en een eerste" +#~ " versie van Spanje. Volgende stappen: " +#~ "doorlopende verbetering van de datakwaliteit" +#~ " en daarna Portugal." + +#~ msgid "moelleux" +#~ msgstr "halfzoet" + +#~ msgid "" +#~ "Carte interactive des appellations viticoles" +#~ " européennes (AOC, AOP, IGP, DOP) :" +#~ " cépages, styles et terroir, d'après " +#~ "les registres officiels (INAO, EUR-Lex)." +#~ msgstr "" +#~ "Interactieve kaart van Europese " +#~ "wijnappellaties (AOC, AOP, IGP, DOP): " +#~ "druivenrassen, stijlen en terroir, op " +#~ "basis van officiële registers (INAO, " +#~ "EUR-Lex)." + +#~ msgid "AOC / AOP" +#~ msgstr "AOC / AOP" + +#~ msgid "{n} dans IGP masquées — afficher" +#~ msgstr "{n} verborgen in IGP's — toon" + +#~ msgid "" +#~ "Carte de référence des appellations " +#~ "viticoles (AOC, AOP, IGP, DOP), générée" +#~ " automatiquement à partir des données " +#~ "publiques." +#~ msgstr "" +#~ "Open referentiekaart van wijnappellaties (AOC," +#~ " AOP, IGP, DOP), automatisch gegenereerd" +#~ " op basis van publieke data." + +#~ msgid "" +#~ "Sources : INAO ({inao}) pour les " +#~ "cahiers des charges et les aires " +#~ "parcellaires, IGN ({ign}) pour le fond" +#~ " cartographique, Wikipedia ({wikipedia}) pour " +#~ "quelques compléments narratifs (CC BY-SA" +#~ " 4.0), VIVC ({vivc}) — Vitis " +#~ "International Variety Catalogue, Julius " +#~ "Kühn-Institut — pour les noms " +#~ "canoniques et numéros de cépage " +#~ "(citation Röckel et al.). Tout extrait" +#~ " Wikipedia est signalé sur place. " +#~ "Détails et licences dans le {readme}." +#~ msgstr "" +#~ "Bronnen: INAO ({inao}) voor de cahiers" +#~ " des charges en de perceelsgrenzen, " +#~ "IGN ({ign}) voor de basiskaart, " +#~ "Wikipedia ({wikipedia}) voor enkele narratieve" +#~ " aanvullingen (CC BY-SA 4.0), VIVC" +#~ " ({vivc}) — Vitis International Variety " +#~ "Catalogue, Julius Kühn-Institut — voor" +#~ " de canonieke namen en druivenrasnummers" +#~ " (citaat Röckel et al.). Elk " +#~ "Wikipedia-fragment wordt ter plaatse " +#~ "vermeld. Details en licenties in de " +#~ "{readme}." + +#~ msgid "" +#~ "20 pays européens cartographiés : " +#~ "France, Espagne, Portugal, Italie, Autriche," +#~ " Allemagne, Suisse, Slovénie, Croatie, " +#~ "Hongrie, Roumanie, Bulgarie, Grèce, Slovaquie," +#~ " Tchéquie, Luxembourg, Belgique, Pays-Bas," +#~ " Malte et Chypre. Des itérations " +#~ "supplémentaires viendront affiner la qualité" +#~ " des données. La couverture sera " +#~ "étendue au-delà de l'UE et de " +#~ "la Suisse, ainsi qu'aux classifications " +#~ "hors AOP." +#~ msgstr "" +#~ "20 Europese landen al in kaart " +#~ "gebracht: Frankrijk, Spanje, Portugal, Italië," +#~ " Oostenrijk, Duitsland, Zwitserland, Slovenië," +#~ " Kroatië, Hongarije, Roemenië, Bulgarije, " +#~ "Griekenland, Slowakije, Tsjechië, Luxemburg, " +#~ "België, Nederland, Malta en Cyprus. " +#~ "Verdere iteraties zullen de datakwaliteit " +#~ "blijven verbeteren. De dekking wordt " +#~ "uitgebreid buiten de EU en Zwitserland," +#~ " evenals naar niet-BOB-classificaties." + +#~ msgid "" +#~ "Liste des {n} appellations viticoles " +#~ "européennes cartographiées sur Open Wine " +#~ "Map, classées par pays — AOC, AOP," +#~ " IGP, DOP." +#~ msgstr "" +#~ "Lijst van {n} Europese wijnappellaties " +#~ "op Open Wine Map, gerangschikt per " +#~ "land — AOC, AOP, IGP, DOP." + +#~ msgid "" +#~ "Chaque appellation porte deux noms, qui" +#~ " désignent deux choses : le terme " +#~ "traditionnel que le régulateur du pays" +#~ " lui rattache, et le régime sous " +#~ "lequel elle est enregistrée. La carte" +#~ " les affiche sous la forme <em>TERME" +#~ " (RÉGIME)</em>, par exemple « DOCG " +#~ "(AOP) » ou « DOQ (AOP) ». " +#~ "Lorsqu'un pays n'a pas de terme " +#~ "propre, seul le régime apparaît. Les " +#~ "AOC suisses sont hors du régime de" +#~ " l'UE et ne portent aucune parenthèse" +#~ " ; le Royaume-Uni enregistre ses " +#~ "appellations dans son propre régime, qui" +#~ " conserve les mots PDO et PGI ;" +#~ " les eaux-de-vie françaises sont " +#~ "des indications géographiques de boissons " +#~ "spiritueuses, pas des AOP viticoles. " +#~ "Survolez un terme pour en lire la" +#~ " définition et la source." +#~ msgstr "" +#~ "Elke appellatie draagt twee namen, en" +#~ " die betekenen twee verschillende dingen:" +#~ " de traditionele aanduiding die de " +#~ "regulator van het land eraan koppelt," +#~ " en de regeling waaronder ze is " +#~ "geregistreerd. De kaart toont ze als " +#~ "<em>AANDUIDING (REGELING)</em>, bijvoorbeeld “DOCG" +#~ " (BOB)” of “DOQ (BOB)”. Heeft een " +#~ "land geen eigen aanduiding, dan staat" +#~ " er alleen de regeling. Zwitserse " +#~ "AOC's vallen buiten de EU-regeling " +#~ "en krijgen geen haakjes; het Verenigd" +#~ " Koninkrijk registreert onder een eigen " +#~ "regeling, die de woorden PDO en " +#~ "PGI behoudt; Franse eaux-de-vie " +#~ "zijn geografische aanduidingen voor " +#~ "gedistilleerde dranken, geen BOB's voor " +#~ "wijn. Beweeg de muis over een " +#~ "aanduiding voor de definitie en de " +#~ "bron." + +#~ msgid "Open Wine Map — carte des appellations" +#~ msgstr "Open Wine Map — appellatiekaart" + +#~ msgid "" +#~ "Carte des AOP et IGP viticoles " +#~ "d'Europe (AOC, DOCG, DOQ, DAC…), AOC " +#~ "suisses et IG britanniques : cépages," +#~ " styles et terroir, d'après les " +#~ "registres officiels." +#~ msgstr "" +#~ "Kaart van Europese wijn-BOB's en " +#~ "-BGA's (AOC, DOCG, DOQ, DAC…), Zwitserse" +#~ " AOC's en Britse GA's: druivenrassen, " +#~ "stijlen en terroir uit de officiële " +#~ "registers." + diff --git a/raw/wikipedia/aoc_overrides.json b/raw/wikipedia/aoc_overrides.json index 4ec1cc8..62a2580 100644 --- a/raw/wikipedia/aoc_overrides.json +++ b/raw/wikipedia/aoc_overrides.json @@ -506,6 +506,10 @@ "missing": true, "note": "researched 2026-05; no dedicated wine article on de.wikipedia.org" }, + "tirol": { + "missing": true, + "note": "review 2026-09-12: the cascade bound de.wikipedia 'Toro (Weinbaugebiet)' — the Spanish DO, not the Austrian Tirol g.U.; no dedicated wine article on de.wikipedia.org" + }, "traisental": { "missing": true, "note": "researched 2026-05; \"Traisen (Fluss)\" / Traisental is the river/valley geographic article" @@ -2287,6 +2291,10 @@ "missing": true, "note": "researched 2026-05; only listed inside Vini a indicazione geografica tipica (Bolzano province IGT entry); no standalone article" }, + "montecastelli": { + "missing": true, + "note": "review 2026-09-12: the cascade bound 'Montecastelli Pisano' — the village article, not the IGT; its one wiki-only bullet described the village hill" + }, "montello-rosso": { "page_url": "https://it.wikipedia.org/wiki/Montello_rosso", "verification_quote": "Il Montello rosso o Montello è una DOCG riservata a un vino la cui produzione è consentita nella provincia di Treviso, in due comprensori collinari limitrofi posti ai piedi delle Dolomiti.", diff --git a/scripts/01_scrape_cahiers.py b/scripts/01_scrape_cahiers.py index 25d3c6d..a942bce 100644 --- a/scripts/01_scrape_cahiers.py +++ b/scripts/01_scrape_cahiers.py @@ -51,6 +51,7 @@ from _lib import eambrosia_register as er # noqa: E402 from _lib.fr import register_cahier as rc # noqa: E402 +from _lib.fr import register_match as rm # noqa: E402 RAW = ROOT / "raw" SIQO_CSV = RAW / "inao" / "siqo-referentiel.csv" @@ -524,15 +525,52 @@ def _register_tier( return "register", 0 +def _prefer_cahier_ids() -> set[str]: + """id_appellations pinned `prefer_cahier: true` in the checked-in + scripts/_lib/fr/register_overrides.json — bind the register's cahier + even though BO Agri serves a PDF, because that PDF is verifiably another + appellation's cahier.""" + try: + data = json.loads(rm.OVERRIDES_PATH.read_text(encoding="utf-8")) + except (OSError, ValueError): + return set() + return {k for k, v in data.items() if isinstance(v, dict) and v.get("prefer_cahier")} + + +def _process_prefer_register( + app: Appellation, manifest: dict, prior: dict, register: RegisterTier | None, +) -> tuple[str, int]: + """Curator pin: bind this appellation to the eAmbrosia register's cahier + ahead of BO Agri. Deliberately bypasses `_has_usable_cahier`: that guard + protects working resolutions from a transient INAO outage, and a pin is + the opposite of transient — the PDF on disk is the wrong cahier.""" + if register is None or not register.enabled: + print(f"[register] {app.name}: prefer_cahier pinned but the register tier is " + f"inert — run scripts/01d_resolve_register.py", file=sys.stderr) + return "missed", 0 + cahier = register.cahier_for(app.id_appellation) + if cahier is None: + print(f"[register] {app.name}: prefer_cahier pinned but no attachment", file=sys.stderr) + return "missed", 0 + meta = _stub_meta(app, prior) + meta["manual_override_note"] = "prefer_cahier pin (scripts/_lib/fr/register_overrides.json)" + manifest[app.id_appellation] = _apply_register( + meta, app, cahier, register.resolved.get(app.id_appellation, {})) + return "register", 0 + + def _process_app( session: requests.Session, app: Appellation, manifest: dict, overrides: dict, delay: float, register: RegisterTier | None = None, + prefer_cahier: set[str] = frozenset(), ) -> tuple[str, int]: """Resolve and download `app`'s cahier(s). Returns (status, alt_count) where status is one of: missed, cached, fetched, legifrance-only, override-only, register. """ prior = manifest.get(app.id_appellation, {}) + if app.id_appellation in prefer_cahier: + return _process_prefer_register(app, manifest, prior, register) override = overrides.get(app.id_appellation) has_override_urls = bool(override and override.get("boagri_urls")) @@ -688,6 +726,11 @@ def main() -> int: f"run scripts/01d_resolve_register.py to enable the register tier", file=sys.stderr) + prefer_cahier = _prefer_cahier_ids() + if prefer_cahier: + print(f"[register] {len(prefer_cahier)} prefer_cahier pin(s) in " + f"{rm.OVERRIDES_PATH.relative_to(ROOT)}", file=sys.stderr) + fetched = cached = missed = extra = 0 counters = { "fetched": 0, "cached": 0, "missed": 0, @@ -697,7 +740,8 @@ def main() -> int: try: for app in tqdm(appellations, desc="cahiers", leave=False): status, alt_count = _process_app( - session, app, manifest, overrides, args.delay, register + session, app, manifest, overrides, args.delay, register, + prefer_cahier=prefer_cahier, ) counters[status] += 1 extra += alt_count diff --git a/scripts/02_extract_cahiers.py b/scripts/02_extract_cahiers.py index 80b413a..c10e1a1 100644 --- a/scripts/02_extract_cahiers.py +++ b/scripts/02_extract_cahiers.py @@ -117,9 +117,35 @@ + r")" + r"\s*(?:\(\d+\))?" + r"(?P<after>(?:[^:\n]*\n?){0,4}?):" - + r"(?P<communes>[^\n]*(?:\n(?!\s*\n|\s*-?\s*(?i:D[ée]partement|Dans\s+(?:le|les)\s+d[ée]partement)|\s*\d°|\s*[IVX]+\s*\.\s*-)[^\n]*)*)" + + r"(?P<communes>[^\n]*(?:\n(?!\s*\n|\s*-?\s*(?i:D[ée]partement|Dans\s+(?:le|les)\s+d[ée]partement)|\s*\d[ \t]*(?:°|[-–][ \t]*[A-Za-zÀ-ÿ])|\s*[IVX]+\s*\.\s*-)[^\n]*)*)" ) DEPT_HEADER_RE = re.compile(_DEPT_HEADER_PATTERN, re.MULTILINE) +# Sub-block headers inside section IV: "1° - Aire géographique", +# "1°- Aire parcellaire délimitée", plus the degree-less "1 - Aire +# géographique" / "1) Aire …" forms the 2024 PNOCDC republications use +# (Pouilly-Loché). The marker (°, dash or paren) is mandatory so a body +# line that merely starts with a digit ("3 communes …") never opens a +# block. Without the split the whole section is scanned and the aire de +# proximité immédiate list is taken as the aire géographique. +# Intra-line gaps are [ \t] only: a page-number line (" 2") followed by a +# form feed and "- Département du Rhône : …" must not glue into one header. +_AIRE_BLOCK_HEADER_PATTERN = ( + r"^[ \t\x0c]*(\d[ \t]*(?:°[ \t]*[-–)]?|[-–]|\))[ \t]*[A-Za-zÀ-ÿ][^\n]*)$" +) +# Sentence-form aire, used when a block carries no "Département de X :" +# list: "… sont assurés sur le territoire de la commune de Mâcon du +# département de Saône-et-Loire" / "des communes de A, B et C du +# département de l'Yonne". One match per (commune list, département). +_AIRE_SENTENCE_PATTERN = ( + r"(?:territoire\s+)?(?i:de\s+la\s+commune|des\s+communes)\s+(?i:de\s+|d['’]\s*)" + r"(?P<communes>[^:;.]+?)\s+" + r"(?i:du|dans\s+le)\s+(?i:d[ée]partement)\s+(?i:" + _ART + r")" + r"(?P<dept>" + r"[A-ZÀÂÄÉÈÊËÎÏÔÖÙÛÜŸ][\wÀ-ÿ'’]*(?:-[\wÀ-ÿ'’]+)*" + r"(?:\s+[A-ZÀÂÄÉÈÊËÎÏÔÖÙÛÜŸ][\wÀ-ÿ'’]*(?:-[\wÀ-ÿ'’]+)*)*" + r")" +) +AIRE_SENTENCE_RE = re.compile(_AIRE_SENTENCE_PATTERN) COG_YEAR_RE = re.compile(r"code officiel g[ée]ographique de l['’]ann[ée]e\s+(\d{4})") @@ -746,16 +772,33 @@ def parse_communes(field: str) -> list[str]: return out +def _clean_commune_tokens(tokens: list[str]) -> list[str]: + """Drop the prose that rides along with a sentence-form commune list. + + A commune name starts with a capital and never contains a sentence + break, but the sentence form leaks asides into the split tokens: + "sur la base du code officiel géographique …" (Barsac), "située" + (Loupiac: "la commune de Loupiac, située dans le département"). Applied + to the sentence path only — the list form ("Département de X : …") + legitimately carries lowercase-initial tokens in IGP layouts + ("l'ensemble des communes …"), so it is left as it was. + """ + out: list[str] = [] + for c in tokens: + if ". " in c: + c = c.split(". ", 1)[0].strip() + if not c or not c[:1].isupper() or "code officiel" in c.lower(): + continue + out.append(c) + return out + + def extract_aire(section_iv: str) -> dict: """Parse section IV. Returns geographique/proximite_immediate commune lists.""" cog_match = COG_YEAR_RE.search(section_iv) cog_year = int(cog_match.group(1)) if cog_match else None - blocks = re.split( - r"^\s*(\d°\s*[-–]?\s*[A-Za-zÀ-ÿ][^\n]*)$", - section_iv, - flags=re.MULTILINE, - ) + blocks = re.split(_AIRE_BLOCK_HEADER_PATTERN, section_iv, flags=re.MULTILINE) # blocks: [pre, header1, body1, header2, body2, ...] by_block: dict[str, str] = {} for i in range(1, len(blocks) - 1, 2): @@ -773,6 +816,12 @@ def by_dept(text: str) -> dict[str, list[str]]: communes = parse_communes(communes_raw) if communes: result[dept].extend(communes) + if not result: + for m in AIRE_SENTENCE_RE.finditer(text): + dept = re.sub(r"\s+", " ", m.group("dept")).strip(" '’.,:") + communes = _clean_commune_tokens(parse_communes(m.group("communes"))) + if communes: + result[dept].extend(communes) return dict(result) aire_geo_text = next( @@ -1581,9 +1630,22 @@ def main() -> int: dgc_emitted += 1 set_pliego_context(None) - stubs = emit_stub_records( - siqo_denoms, siqo_categories, manifest, index, slug_map, OUT_DIR - ) + if args.only or args.limit: + # A partial run must leave every unselected record untouched. The + # stub pass walks the whole SIQO referentiel and would rewrite + # every record outside the selection as `no-extract` (it once + # stubbed 1,131 records on a 61-name `--only` run), and the index + # would shrink to the selection — so skip the stub pass and merge + # the fresh entries into the index on disk instead. + stubs = 0 + if INDEX_PATH.exists(): + merged = json.loads(INDEX_PATH.read_text(encoding="utf-8")) + merged.update(index) + index = merged + else: + stubs = emit_stub_records( + siqo_denoms, siqo_categories, manifest, index, slug_map, OUT_DIR + ) INDEX_PATH.write_text(json.dumps(index, ensure_ascii=False, indent=2, sort_keys=True), encoding="utf-8") unknowns_path = ROOT / "raw" / "inao" / "extraction-unknowns.json" diff --git a/scripts/02b_fetch_aoc_lexicon.py b/scripts/02b_fetch_aoc_lexicon.py index a23711f..d7de0db 100644 --- a/scripts/02b_fetch_aoc_lexicon.py +++ b/scripts/02b_fetch_aoc_lexicon.py @@ -684,6 +684,11 @@ def main() -> int: ap.add_argument("--refresh", action="store_true", help="re-fetch even if cached") ap.add_argument("--throttle", type=float, default=0.2, help="seconds between API calls") ap.add_argument("--limit", type=int, default=0, help="cap on entries to process (0 = all)") + ap.add_argument( + "--only", action="append", default=[], + help="restrict to these slugs (repeatable) — with --refresh, re-resolves just them " + "(e.g. after pinning a slug in aoc_overrides.json)", + ) args = ap.parse_args() cfg = LANG_CONFIG[args.lang] @@ -699,6 +704,11 @@ def main() -> int: return 1 targets = collect_targets(source_dir) + if args.only: + wanted = set(args.only) + targets = [t for t in targets if t[0] in wanted] + for missing_slug in sorted(wanted - {t[0] for t in targets}): + print(f" --only {missing_slug}: no such non-DGC entry in {source_dir}", file=sys.stderr) if args.limit: targets = targets[: args.limit] print( diff --git a/scripts/02d_extract_terroir_facts.py b/scripts/02d_extract_terroir_facts.py index 67a72e1..b71e550 100644 --- a/scripts/02d_extract_terroir_facts.py +++ b/scripts/02d_extract_terroir_facts.py @@ -52,7 +52,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -61,6 +60,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_chapters import is_shared, own_chapter # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "inao" / "cahier-extracted" WIKI_AOCS = ROOT / "raw" / "wikipedia" / "aocs" / "fr" @@ -73,8 +80,12 @@ # Cahier section X anchors. Validated 100% against the 6-AOC eval sample; # fall back to a flat slice when the slicer returns < 2 sub-sections. -TOP_RE = re.compile(r"\b([1-9])°\s*[-–]\s*([A-ZÀ-Ý][^\n]{5,80})") +# "l°" / "I°" are pdftotext's OCR of "1°" (Pouilly-Vinzelles); the top-level +# key is normalised in _spans_by_top. +TOP_RE = re.compile(r"\b([1-9lI])°\s*[-–]\s*([A-ZÀ-Ý][^\n]{5,80})") SUB_RE = re.compile(r"\b([a-c])\)\s*[-–]?\s*([A-ZÀ-Ý][^\n]{5,80})") +_TOP_OCR = {"l": "1", "I": "1"} +MIN_SLICE_CHARS = 200 SUBSECTIONS = [ { @@ -149,43 +160,28 @@ - Les citations sont VERBATIM (copiées-collées) de leur source respective. NE JAMAIS attribuer à une source un texte qui n'y figure pas. - Aucun jugement de valeur (« exceptionnel », « remarquable », « prestigieux »...). - Aucune inférence externe. Aucun chiffre absent des deux sources. -- Maximum {max_bullets} puces, ≤ 140 caractères chacune. +- Maximum {max_bullets} puces ; chaque puce est une phrase complète d'environ 120 à 220 caractères — jamais un fragment télégraphique. - Si ni le cahier ni Wikipedia ne contiennent de fait notable concret pour cette sous-section, retourne une liste vide. Réponds UNIQUEMENT en JSON, sans texte avant ou après : {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Utilise une chaîne vide "" pour la citation absente.""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - """Longest-contiguous-match coverage of `quote` in `source` (0.0–1.0).""" - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _spans_by_top(lien: str, tops: list[re.Match]) -> dict[str, tuple[int, int]]: """Map top-level numeral ('1'/'2'/'3') → (start, end) span in lien.""" out: dict[str, tuple[int, int]] = {} for i, m in enumerate(tops): end = tops[i + 1].start() if i + 1 < len(tops) else len(lien) - out[m.group(1)] = (m.start(), end) + out.setdefault(_TOP_OCR.get(m.group(1), m.group(1)), (m.start(), end)) return out @@ -199,6 +195,11 @@ def _split_zone_geographique( if not sub_in_1: return {"facteurs_naturels": (s1, e1)} out: dict[str, tuple[int, int]] = {} + first = sub_in_1[0] + if first.group(1) != "a" and first.start() - s1 >= MIN_SLICE_CHARS: + # The a) heading lost its letter ("- Description des facteurs naturels", + # Floc de Gascogne): the text before b) is the natural factors. + out["facteurs_naturels"] = (s1, first.start()) for i, m in enumerate(sub_in_1): end = sub_in_1[i + 1].start() if i + 1 < len(sub_in_1) else e1 if m.group(1) == "a": @@ -222,6 +223,11 @@ def slice_section_x(lien: str) -> dict[str, str]: if "1" in by_top: s1, e1 = by_top["1"] spans.update(_split_zone_geographique(s1, e1, subs)) + elif tops[0].start() >= MIN_SLICE_CHARS: + # No "1°" heading at all — the lien opens straight at "a) - Description + # des facteurs naturels" (Menetou-Salon): everything before the first + # numbered heading is section 1. + spans.update(_split_zone_geographique(0, tops[0].start(), subs)) if "2" in by_top: spans["produit"] = by_top["2"] if "3" in by_top: @@ -355,17 +361,19 @@ def write_cache( wiki_meta: dict, translator_id: str, translator_kind: str, + counts: dict | None = None, ) -> None: payload = { "slug": slug, "facts": facts, **cahier_meta, **wiki_meta, + **(counts or {}), "translator": translator_id, "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(slug), payload) + write_source_cache(cache_path(slug), payload) def _job_from_record(rec: dict) -> dict | None: @@ -379,6 +387,18 @@ def _job_from_record(rec: dict) -> dict | None: lien = (rec.get("lien_au_terroir") or "").strip() if len(lien) < MIN_CAHIER_CHARS: return None + if is_shared(lien): + # One cahier for the 51 Alsace grands crus: the lien repeats a + # chapter per cru, and slicing by section number alone grades every + # cru against the last chapter (Zotzenberg). Restrict to the record's + # own chapter; a cru without one is skipped, never grounded on + # another cru's text. + window = own_chapter(lien, rec.get("name") or "") + if window is None: + print(f"[02d] {slug}: shared cahier but no own chapter for " + f"{rec.get('name')!r} — skipped", file=sys.stderr) + return None + lien = lien[window[0]:window[1]].strip() slices = slice_section_x(lien) if len(slices) < 2: slices = {"facteurs_naturels": lien} @@ -432,10 +452,12 @@ def build_prompt(spec: dict, wiki_hint: str) -> str: ) -def extract_one_aoc(provider, job: dict) -> tuple[list[dict], list[str]]: - """Run all four sub-section calls for one AOC, return (kept_facts, errors).""" +def extract_one_aoc(provider, job: dict) -> tuple[list[dict], list[str], dict]: + """Run all four sub-section calls for one AOC, return (kept_facts, errors, + counts) — counts = {n_dropped, n_deduped, n_unearned_interactions}.""" facts: list[dict] = [] errors: list[str] = [] + n_dropped = 0 for spec in SUBSECTIONS: sub_key = spec["key"] cahier_text = job["slices"].get(sub_key, "") @@ -443,6 +465,7 @@ def extract_one_aoc(provider, job: dict) -> tuple[list[dict], list[str]]: continue wiki_hint = job["wiki_hints"].get(sub_key, "") system = build_prompt(spec, wiki_hint) + system = with_feedback(system, job["slug"]) try: raw = provider.chat(system=system, user=cahier_text, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 @@ -457,7 +480,15 @@ def extract_one_aoc(provider, job: dict) -> tuple[list[dict], list[str]]: if classified is not None: classified["subsection"] = sub_key facts.append(classified) - return facts, errors + else: + n_dropped += 1 + deduped = dedupe_facts(facts) + earned = earn_interactions(deduped.kept, "fr") + kept = earned.kept + normalize_facts(kept) + counts = {"n_dropped": n_dropped, "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped)} + return kept, errors, counts # ─────────────────────────────────────────────────── round-trip (manual) ── @@ -486,7 +517,7 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": spec["label"], "topics": spec["topics"], "max_bullets": spec["max_bullets"], - "system_prompt": build_prompt(spec, wiki_hint), + "system_prompt": with_feedback(build_prompt(spec, wiki_hint), job["slug"]), "cahier_text": cahier_text, "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -665,7 +696,7 @@ def _select_jobs(refresh: bool, limit: int, slugs: list[str] | None) -> list[dic def _make_provider(args) -> tuple[object | None, str]: """Returns (provider, translator_id). provider is None for manual mode.""" return providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) @@ -685,7 +716,7 @@ def _print_manual_listing(jobs: list[dict]) -> int: def _process_one_job(provider, translator_id: str, job: dict) -> tuple[int, int]: """Run one AOC extraction + cache write. Returns (ok, err) where each is 0 or 1. Errors are printed to stderr; exceptions surface to the caller.""" - facts, errors = extract_one_aoc(provider, job) + facts, errors, counts = extract_one_aoc(provider, job) if errors and not facts: for e in errors[:4]: print(f" err {job['slug']}: {e[:160]}", file=sys.stderr) @@ -697,6 +728,7 @@ def _process_one_job(provider, translator_id: str, job: dict) -> tuple[int, int] wiki_meta=job["wiki_meta"], translator_id=translator_id, translator_kind=provider.kind, + counts=counts, ) return 1, 0 @@ -769,7 +801,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") jobs = _select_jobs(refresh=args.refresh, limit=args.limit, slugs=args.slug) if not jobs: print("[02d] batch: nothing to do.", file=sys.stderr) @@ -785,7 +817,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-fr.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) write_manifest( n_jobs=len(jobs), ok=stats.get("ok", 0), err=stats.get("err", 0), cached=0, diff --git a/scripts/02d_verify_terroir_facts.py b/scripts/02d_verify_terroir_facts.py new file mode 100644 index 0000000..9e5b384 --- /dev/null +++ b/scripts/02d_verify_terroir_facts.py @@ -0,0 +1,355 @@ +"""Claim-support gate over the stage-02d terroir-fact caches (all +countries) — review 2026-09-12 recommendations R2 (claim-support gate), +R3 (earned `interactions`), R7 (sub-section per fact), R8 (semantic +dedupe). Runs between 02d and 02e: + + 02d --refresh → 02d_verify → 02e → 02e_verify → 04 + +Per record, ONE model request grades every bullet against the exact +source text stage 02d graded against (`_lib/terroir_sources`, through the +country's own 02d module), the per-sub-section Wikipedia hints and the +record's review feedback (`raw/terroir-facts-feedback/<slug>.json`: +verified-misleading claims become explicit checks, record cautions +disqualify pasted or mis-bound text). Verdicts are applied by +`_lib/terroir_gate.apply_verdicts`: supported bullets stay, over-claiming +bullets are replaced by the model's narrower rewrite (guarded — no new +numbers, no arrows), unsupported / foreign / tautological / restated +bullets are dropped, clearly misfiled bullets move sub-section. + +Writes, per gated record (through the per-run backup, so +`scripts/rollback_terroir_facts.py --run <id>` undoes it): + raw/terroir-facts/<slug>.json facts (with `support` per fact) + + a `gate` block (counts, dropped + bullets, shas the gate keyed on) + raw/translations/terroir-facts/… index-aligned prune for pure drops; + re-keyed `pending:` when a bullet + was rewritten so 02e re-translates + raw/terroir-facts-feedback/<slug> a `history` entry (kind "gate") + raw/terroir-facts/manifest-gate.json + tmp/terroir-facts-review/gate-<run>.json + +Incremental: a record is gated when it has no `gate` block or its facts +or source sha changed since (a 02d re-run, a post-pass). `--refresh` +re-gates everything selected; `--dry-run` grades and writes the report +only (with `--sample N` this is the cheap LLM audit of R9). + +Providers: anthropic — default `claude-opus-5` with adaptive thinking +(`providers.STAGE_DEFAULTS["gate"]`; `--model` / `--thinking` override) / +mistral / ollama; `--batch` submits every selected record as one +Batch-API job (resumable via raw/.batch/02d-verify.json). + +Usage: + .venv/bin/python scripts/02d_verify_terroir_facts.py --batch --provider anthropic + .venv/bin/python scripts/02d_verify_terroir_facts.py --country it --only barolo --provider anthropic + .venv/bin/python scripts/02d_verify_terroir_facts.py --sample 50 --dry-run --provider anthropic +""" + +from __future__ import annotations + +import argparse +import json +import random +import sys +import time +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +from tqdm import tqdm + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib import batch, cache, providers, terroir_backup # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import ( # noqa: E402 + TERROIR, + prune_translations, + sync_translation_meta, + write_source_cache, +) +from _lib.terroir_dedupe import facts_sha # noqa: E402 +from _lib.terroir_feedback import append_history, load_feedback # noqa: E402 +from _lib.terroir_gate import ( # noqa: E402 + GATE_VERSION, + SYSTEM, + apply_verdicts, + build_user_message, + needs_gate, + parse_verdicts, +) +from _lib.terroir_sources import COUNTRIES, Sources, resolve_sources # noqa: E402 + +MANIFEST = TERROIR / "manifest-gate.json" +REPORT_DIR = ROOT / "tmp" / "terroir-facts-review" +BATCH_SIDECAR = ROOT / "raw" / ".batch" / "02d-verify.json" +MAX_TOKENS = 8000 # Opus 5 with adaptive thinking: thinking + the JSON reply + + +def log(msg: str) -> None: + print(f"[02d-verify] {msg}", file=sys.stderr) + + +# ────────────────────────────────────────────────────────── selection ── + + +def select_records( + *, countries: set[str] | None, only: set[str] | None, sample: int, limit: int, refresh: bool, +) -> list[tuple[Path, dict]]: + out: list[tuple[Path, dict]] = [] + for p in sorted(TERROIR.glob("*.json")): + if p.name.startswith("manifest"): + continue + if only and p.stem not in only: + continue + d = cache.read_json_or_none(p) + if not d or d.get("mode") == "verbatim": + continue + if countries and (d.get("country") or "fr") not in countries: + continue + if not needs_gate(d, refresh=refresh): + continue + d["slug"] = d.get("slug") or p.stem + out.append((p, d)) + if sample and len(out) > sample: + random.seed(0) + out = sorted(random.sample(out, sample), key=lambda t: t[0]) + if limit: + out = out[:limit] + return out + + +class SourceResolver: + def __init__(self) -> None: + self._by_country: dict[str, dict[str, Sources] | None] = {} + self.failed: dict[str, str] = {} + + def get(self, country: str) -> dict[str, Sources] | None: + if country not in self._by_country: + if country not in COUNTRIES: + self.failed[country] = "no stage-02d module" + self._by_country[country] = None + else: + log(f"{country}: resolving stage-02d sources …") + try: + self._by_country[country] = resolve_sources(country) + except Exception as e: # noqa: BLE001 + self.failed[country] = repr(e) + log(f"{country}: FAILED to resolve sources: {e!r}") + self._by_country[country] = None + return self._by_country[country] + + +# ─────────────────────────────────────────────────────────── one record ── + + +def gate_record( + provider, model_id: str, path: Path, d: dict, src: Sources, *, run: str, dry_run: bool, +) -> dict: + """Grade + apply for one record. Returns the report row; writes the + cache / translations / feedback history unless `dry_run`.""" + slug = d["slug"] + country = d.get("country") or "fr" + source_lang = d.get("source_lang") or ("fr" if country == "fr" else country) + facts = d.get("facts") or [] + fb = load_feedback(slug) + user = build_user_message( + name=d.get("name") or slug, country=country, source_lang=source_lang, + cahier=src.cahier, hints=src.hints, facts=facts, feedback=fb, + ) + try: + raw = provider.chat(system=mark_cached(SYSTEM), user=user, max_tokens=MAX_TOKENS, num_ctx=32768) + except Exception as e: # noqa: BLE001 + return {"slug": slug, "country": country, "status": "error", "error": str(e)[:200]} + verdicts, err = parse_verdicts(raw, len(facts)) + if verdicts is None: + return {"slug": slug, "country": country, "status": "parse_error", "error": err} + + res = apply_verdicts(facts, verdicts, source=src.cahier, source_lang=source_lang, run=run, model=model_id) + old_sha = facts_sha(facts) + new_sha = facts_sha(res["facts"]) + row = { + "slug": slug, "country": country, "status": "ok", + "n_before": len(facts), "n_after": len(res["facts"]), + "n_supported": sum(1 for f in res["facts"] if f["support"]["verdict"] == "supported"), + "n_rewritten": len(res["rewritten"]), "n_dropped": len(res["dropped"]), + "n_moved": len(res["moved"]), "n_rejected_rewrites": len(res["rejected_rewrites"]), + "n_cosmetic_rewrites": len(res["cosmetic_rewrites"]), + "n_missing_rewrites": len(res["missing_rewrites"]), + "dropped": res["dropped"], "rewritten": res["rewritten"], "moved": res["moved"], + "rejected_rewrites": res["rejected_rewrites"], + "cosmetic_rewrites": res["cosmetic_rewrites"], "missing_rewrites": res["missing_rewrites"], + "verdicts": [{"i": i, **v} for i, v in enumerate(verdicts)], + } + if dry_run: + return row + + d["facts"] = res["facts"] + d["gate"] = { + "version": GATE_VERSION, "run": run, "model": model_id, + "at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "facts_sha_before": old_sha, "facts_sha_after": new_sha, + "cahier_source_sha": d.get("cahier_source_sha"), + "wiki_source_revision": d.get("wiki_source_revision"), + "n_supported": row["n_supported"], "n_rewritten": row["n_rewritten"], + "n_dropped": row["n_dropped"], "n_moved": row["n_moved"], + "n_rejected_rewrites": row["n_rejected_rewrites"], + "n_cosmetic_rewrites": row["n_cosmetic_rewrites"], + "n_missing_rewrites": row["n_missing_rewrites"], + "dropped": res["dropped"], + } + write_source_cache(path, d) + # Translations: index-aligned prune for drops; a rewrite changes the text, + # so the caches are re-keyed `pending:` — still length-aligned for the + # interim render, but stale for 02e, which re-translates the record. + if res["dropped"] or res["text_changed"]: + key = new_sha if not res["text_changed"] else f"pending:{new_sha}" + pruned, stale = prune_translations(slug, old_sha, len(facts), res["kept_indices"], key, dry_run=False) + row["translations_pruned"] = pruned + row["translations_stale"] = [s["lang"] for s in stale] + if res["moved"] and not res["text_changed"]: + sync_translation_meta(slug, res["facts"], dry_run=False) + if res["dropped"] or res["rewritten"] or res["moved"]: + append_history(slug, { + "run": run, "kind": "gate", "model": model_id, + "dropped": [{"bullet": x["bullet"], "note": x["note"]} for x in res["dropped"]], + "rewritten": [{"from": x["from"], "to": x["to"]} for x in res["rewritten"]], + "moved": [{"from": x["from"], "to": x["to"], "index": x["index"]} for x in res["moved"]], + }) + return row + + +# ────────────────────────────────────────────────────────────── main ── + + +def _summary(rows: list[dict]) -> dict: + ok = [r for r in rows if r.get("status") == "ok"] + by_country: dict[str, Counter] = {} + for r in ok: + c = by_country.setdefault(r["country"], Counter()) + c["records"] += 1 + for k in ("n_before", "n_after", "n_supported", "n_rewritten", "n_dropped", "n_moved", + "n_rejected_rewrites", "n_cosmetic_rewrites", "n_missing_rewrites"): + c[k] += r[k] + tot = Counter() + for c in by_country.values(): + tot.update(c) + return { + "records_selected": len(rows), "records_ok": len(ok), + "records_error": sum(1 for r in rows if r.get("status") == "error"), + "records_parse_error": sum(1 for r in rows if r.get("status") == "parse_error"), + "records_no_source": sum(1 for r in rows if r.get("status") == "no_source"), + "totals": dict(tot), + "bullets_dropped_share": round(tot["n_dropped"] / tot["n_before"], 4) if tot["n_before"] else 0, + "bullets_rewritten_share": round(tot["n_rewritten"] / tot["n_before"], 4) if tot["n_before"] else 0, + "by_country": {k: dict(v) for k, v in sorted(by_country.items())}, + } + + +def run_gate(provider, model_id: str, selected: list[tuple[Path, dict]], resolver: SourceResolver, + *, run: str, dry_run: bool, quiet: bool) -> list[dict]: + rows: list[dict] = [] + for path, d in tqdm(selected, desc="02d-verify", leave=False, disable=quiet): + country = d.get("country") or "fr" + sources = resolver.get(country) + src = sources.get(d["slug"]) if sources else None + if src is None: + rows.append({"slug": d["slug"], "country": country, "status": "no_source"}) + continue + row = gate_record(provider, model_id, path, d, src, run=run, dry_run=dry_run) + rows.append(row) + if not quiet and row.get("status") == "ok": + log(f"{d['slug']:40} {country:2} {row['n_before']:2}→{row['n_after']:2} " + f"drop={row['n_dropped']} rewrite={row['n_rewritten']} move={row['n_moved']}" + + (f" rejected={row['n_rejected_rewrites']}" if row['n_rejected_rewrites'] else "")) + elif not quiet: + log(f"{d['slug']:40} {country:2} {row.get('status')}: {row.get('error', '')[:120]}") + return rows + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--provider", default="anthropic", choices=("anthropic", "mistral", "ollama")) + ap.add_argument("--model", default=None, help="default: providers.STAGE_DEFAULTS['gate']") + ap.add_argument("--thinking", default=None, choices=("disabled", "adaptive"), + help="anthropic thinking mode (default: the stage default, adaptive on Opus 5)") + ap.add_argument("--ollama-url", default=providers.DEFAULT_OLLAMA_URL) + ap.add_argument("--mistral-url", default=providers.DEFAULT_MISTRAL_URL) + ap.add_argument("--country", action="append", default=None, help="restrict to a country code (repeatable)") + ap.add_argument("--only", action="append", default=None, help="restrict to these slugs (repeatable)") + ap.add_argument("--only-file", default=None, help="JSON list (or {\"slugs\": [...]}) of slugs to restrict to") + ap.add_argument("--sample", type=int, default=0, help="random sample of N records (seed 0)") + ap.add_argument("--limit", type=int, default=0) + ap.add_argument("--refresh", action="store_true", help="re-gate records that already carry a gate block") + ap.add_argument("--dry-run", action="store_true", help="grade and report only; write no cache") + ap.add_argument("--batch", action="store_true", help="submit as one provider Batch-API job (resumable)") + ap.add_argument("--report", default=None, help="report path (default tmp/terroir-facts-review/gate-<run>.json)") + ap.add_argument("--quiet", action="store_true") + args = ap.parse_args() + if args.only_file: + data = json.loads(Path(args.only_file).read_text(encoding="utf-8")) + args.only = list(args.only or []) + list(data.get("slugs") if isinstance(data, dict) else data) + + run = terroir_backup.run_id() + selected = select_records( + countries=set(args.country) if args.country else None, + only=set(args.only) if args.only else None, + sample=args.sample, limit=args.limit, refresh=args.refresh, + ) + if not selected: + log("nothing to do — every selected record is gated against its current facts.") + return 0 + resolver = SourceResolver() + t0 = time.monotonic() + rows: list[dict] = [] + + batch_stats: dict | None = None + if args.batch: + if not batch.supports(args.provider): + log("--batch requires --provider anthropic|mistral") + return 1 + model_id = args.model or batch.default_model(args.provider, stage="gate") + thinking = args.thinking or batch.default_thinking(args.provider, stage="gate") + log(f"batch: {len(selected)} records (provider={args.provider}, model={model_id}, " + f"thinking={thinking}, dry_run={args.dry_run}, run={run})") + + def run_loop(prov): + nonlocal rows + collecting = getattr(prov, "kind", "") == "collecting" + rows = run_gate(prov, model_id, selected, resolver, run=run, + dry_run=args.dry_run or collecting, quiet=collecting or args.quiet) + + batch_stats = batch.run_two_pass(provider=args.provider, model=model_id, sidecar=BATCH_SIDECAR, + run_loop=run_loop, thinking=thinking) + kind = f"{args.provider}-api" + else: + provider, model_id = providers.make_provider( + args.provider, model=args.model, ollama_url=args.ollama_url, mistral_url=args.mistral_url, + stage="gate", thinking=args.thinking, + ) + log(f"{len(selected)} records (provider={args.provider}, model={model_id}, dry_run={args.dry_run}, run={run})") + rows = run_gate(provider, model_id, selected, resolver, run=run, dry_run=args.dry_run, quiet=args.quiet) + kind = provider.kind + + summary = _summary(rows) + if batch_stats: + summary["batch"] = batch_stats + summary.update({ + "run": run, "model": model_id, "provider_kind": kind, "dry_run": args.dry_run, + "gate_version": GATE_VERSION, "elapsed_seconds": round(time.monotonic() - t0, 1), + "source_unresolved_countries": resolver.failed, + }) + report = Path(args.report) if args.report else REPORT_DIR / f"gate-{run}{'-dryrun' if args.dry_run else ''}.json" + report.parent.mkdir(parents=True, exist_ok=True) + report.write_text(json.dumps({"summary": summary, "records": rows}, ensure_ascii=False, indent=1, default=str) + "\n", + encoding="utf-8") + if not args.dry_run: + cache.write_json(MANIFEST, { + "generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), **summary, + }, sort_keys=True) + print(json.dumps(summary, ensure_ascii=False, indent=2), file=sys.stderr) + log(f"report → {report.relative_to(ROOT) if report.is_relative_to(ROOT) else report}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/02e_translate_terroir_facts.py b/scripts/02e_translate_terroir_facts.py index 14d4640..4aa842e 100644 --- a/scripts/02e_translate_terroir_facts.py +++ b/scripts/02e_translate_terroir_facts.py @@ -50,7 +50,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 -from _lib.translation_glossary import glossary_for # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -64,6 +66,8 @@ "pt": "Portuguese", } +PROPER_NOUNS = """named geological formations (Marnes à exogyra virgula, Calcaire du Barrois, Poudingue de Jurançon, tuffeau, llicorella, albariza) and named local soils (caillottes, chailloux); named winds (Mistral, Bise, Tramontane, foehn, cierzo, levante)""" + def build_system_prompt(*, source_lang: str, target_lang: str) -> str: source_name = LOCALE_NAME.get(source_lang, "French") @@ -72,13 +76,13 @@ def build_system_prompt(*, source_lang: str, target_lang: str) -> str: Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve {source_name} proper nouns verbatim: appellation names, region names, commune names, grape variety names, named geological formations (e.g. "Marnes à exogyra virgula", "Calcaire du Barrois", "Poudingue de Jurançon", "tuffeau", "llicorella", "albariza"), named winds (e.g. "Mistral", "Bise", "Tramontane", "foehn", "cierzo", "levante"), local soil/landscape names ("caillottes", "chailloux", "restanques", "chaillées"). - Geological era labels: translate to the standard {target_name} form if it exists (e.g. Kimméridgien → Kimmeridgian in EN, Kimmeridgiense in ES). When unsure, keep the source-language form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" - glossary = glossary_for(target_lang) - return base + "\n\n" + glossary if glossary else base + return translation_system_prompt( + base, source_lang=source_lang, target_lang=target_lang, proper_nouns=PROPER_NOUNS, + ) # ─────────────────────────────────────────────────────────────── helpers ── @@ -137,7 +141,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -202,8 +206,10 @@ def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: source_lang = (job.get("fr_data") or {}).get("source_lang") or "fr" system = build_system_prompt(source_lang=source_lang, target_lang=job["lang"]) user = build_user_prompt(job["fr_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["fr_facts"])) @@ -342,6 +348,7 @@ def _build_argparser() -> argparse.ArgumentParser: "For Anthropic, respect your account's RPM/concurrency limits." ), ) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true", help="re-translate even if cached") ap.add_argument( "--batch", action="store_true", @@ -381,7 +388,7 @@ def _dispatch_emit_or_import(args, languages: tuple[str, ...]) -> int | None: def _make_provider(args) -> tuple[object | None, str]: return providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) @@ -450,8 +457,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -483,6 +492,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] diff --git a/scripts/02e_verify_terroir_facts.py b/scripts/02e_verify_terroir_facts.py new file mode 100644 index 0000000..b5b7a24 --- /dev/null +++ b/scripts/02e_verify_terroir_facts.py @@ -0,0 +1,311 @@ +"""Translation back-check over the stage-02e terroir-fact caches (every +country, every target locale) — review 2026-09-12 recommendation R6. +Runs after 02e: + + 02d --refresh → 02d_verify → 02e → 02e_verify → 04 + +Per (record, locale) ONE model request compares every translated bullet +with its source-language bullet (`_lib/terroir_backcheck`): changed +numbers, dropped or upgraded hedges, wrong entities and back-formed +names, watch-list false friends (generoso → generous, tirage → +disgorgement, climat → climate, Lehm → clay), untranslated common nouns +and missing exonyms (the deterministic `exonyms.exonym_hits` detector +feeds the model its hits). Fixes are applied under the gate's guards (no +new numbers, no arrows, sane length); every checked bullet carries +`check` ({verdict, issue[, original]}) and the cache a `backcheck` block +keyed on the source and translated shas, so the pass is incremental — +a re-translated record is re-checked, an unchanged one is not. + +Only caches aligned with their source (same `source_facts_sha`, same +length) are checked; a stale one is left for 02e. Writes go through the +per-run backup (`scripts/rollback_terroir_facts.py --run <id>`). + +Usage: + .venv/bin/python scripts/02e_verify_terroir_facts.py --batch --provider anthropic + .venv/bin/python scripts/02e_verify_terroir_facts.py --lang en --only rueda --provider anthropic + .venv/bin/python scripts/02e_verify_terroir_facts.py --sample 40 --dry-run --provider anthropic +""" + +from __future__ import annotations + +import argparse +import json +import random +import sys +import time +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +from tqdm import tqdm + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib import batch, cache, providers, terroir_backup # noqa: E402 +from _lib.exonyms import gi_forms_from_names # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_backcheck import ( # noqa: E402 + BACKCHECK_VERSION, + apply_fixes, + build_user_message, + parse_checks, + system_prompt, +) +from _lib.terroir_cache import LANGS, TERROIR, TRANSLATIONS, write_translation_cache # noqa: E402 +from _lib.terroir_dedupe import facts_sha # noqa: E402 +from _lib.terroir_feedback import append_history, load_feedback # noqa: E402 + +MANIFEST = TERROIR / "manifest-backcheck.json" +REPORT_DIR = ROOT / "tmp" / "terroir-facts-review" +BATCH_SIDECAR = ROOT / "raw" / ".batch" / "02e-verify.json" +MAX_TOKENS = 4000 + + +def log(msg: str) -> None: + print(f"[02e-verify] {msg}", file=sys.stderr) + + +def _translated_sha(facts: list[dict]) -> str: + return facts_sha(facts) + + +def needs_check(t: dict, src: dict, *, refresh: bool) -> str | None: + """None when the cache is due; else why it is skipped.""" + if t.get("mode") == "verbatim" or not t.get("facts"): + return "verbatim-or-empty" + sfacts = src.get("facts") or [] + if not sfacts or t.get("source_facts_sha") != facts_sha(sfacts) or len(t["facts"]) != len(sfacts): + return "stale" + if refresh: + return None + b = t.get("backcheck") or {} + if not b: + return None + if ( + b.get("version") == BACKCHECK_VERSION + and b.get("source_facts_sha") == t.get("source_facts_sha") + and b.get("translated_sha") == _translated_sha(t["facts"]) + ): + return "checked" + return None + + +def select( + *, langs: list[str], countries: set[str] | None, only: set[str] | None, sample: int, limit: int, refresh: bool, +) -> tuple[list[tuple[str, str, Path, dict, dict]], Counter]: + """[(slug, lang, path, translation, source)] due for a check.""" + skipped: Counter = Counter() + out: list[tuple[str, str, Path, dict, dict]] = [] + sources: dict[str, dict | None] = {} + for lang in langs: + d = TRANSLATIONS / lang + if not d.exists(): + continue + for p in sorted(d.glob("*.json")): + slug = p.stem + if only and slug not in only: + continue + if slug not in sources: + sources[slug] = cache.read_json_or_none(TERROIR / f"{slug}.json") + src = sources[slug] + if not src or src.get("mode") == "verbatim": + skipped["no-source"] += 1 + continue + if countries and (src.get("country") or "fr") not in countries: + continue + t = cache.read_json_or_none(p) + if not t: + skipped["unreadable"] += 1 + continue + why = needs_check(t, src, refresh=refresh) + if why: + skipped[why] += 1 + continue + out.append((slug, lang, p, t, src)) + if sample and len(out) > sample: + random.seed(0) + out = sorted(random.sample(out, sample), key=lambda x: (x[0], x[1])) + if limit: + out = out[:limit] + return out, skipped + + +def check_one( + provider, model_id: str, item: tuple[str, str, Path, dict, dict], *, run: str, dry_run: bool, + gi_forms: frozenset[str], +) -> dict: + slug, lang, path, t, src = item + country = src.get("country") or "fr" + source_lang = src.get("source_lang") or ("fr" if country == "fr" else country) + sfacts = src.get("facts") or [] + user = build_user_message( + name=src.get("name") or slug, source_lang=source_lang, target_lang=lang, + source_facts=sfacts, translated=t["facts"], feedback=load_feedback(slug), gi_forms=gi_forms, + slug=slug, + ) + try: + raw = provider.chat(system=mark_cached(system_prompt()), user=user, max_tokens=MAX_TOKENS, num_ctx=16384) + except Exception as e: # noqa: BLE001 + return {"slug": slug, "lang": lang, "country": country, "status": "error", "error": str(e)[:200]} + checks, err = parse_checks(raw, len(t["facts"])) + if checks is None: + return {"slug": slug, "lang": lang, "country": country, "status": "parse_error", "error": err} + res = apply_fixes(t["facts"], sfacts, checks, lang=lang, run=run, model=model_id) + row = { + "slug": slug, "lang": lang, "country": country, "status": "ok", "n": len(t["facts"]), + "n_fixed": len(res["fixed"]), "n_rejected": len(res["rejected"]), + "n_missing_fixes": len(res["missing_fixes"]), + "n_flagged": sum(1 for c in checks if c["verdict"] == "fix"), + "fixed": res["fixed"], "rejected": res["rejected"], + "issues": [{"i": i, **c} for i, c in enumerate(checks) if c["verdict"] == "fix"], + } + if dry_run: + return row + t["facts"] = res["facts"] + t["backcheck"] = { + "version": BACKCHECK_VERSION, "run": run, "model": model_id, + "at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "source_facts_sha": t.get("source_facts_sha"), "translated_sha": _translated_sha(res["facts"]), + "n_fixed": len(res["fixed"]), "n_rejected": len(res["rejected"]), + "n_missing_fixes": len(res["missing_fixes"]), + } + write_translation_cache(path, t) + if res["fixed"]: + append_history(slug, { + "run": run, "kind": "backcheck", "lang": lang, "model": model_id, + "fixed": [{"from": f["from"], "to": f["to"], "issue": f["issue"]} for f in res["fixed"]], + }) + return row + + +def _summary(rows: list[dict]) -> dict: + ok = [r for r in rows if r.get("status") == "ok"] + by_lang: dict[str, Counter] = {} + for r in ok: + c = by_lang.setdefault(r["lang"], Counter()) + c["caches"] += 1 + for k in ("n", "n_fixed", "n_rejected", "n_flagged", "n_missing_fixes"): + c[k] += r[k] + tot = Counter() + for c in by_lang.values(): + tot.update(c) + return { + "caches_selected": len(rows), "caches_ok": len(ok), + "caches_error": sum(1 for r in rows if r.get("status") == "error"), + "caches_parse_error": sum(1 for r in rows if r.get("status") == "parse_error"), + "totals": dict(tot), + "bullets_fixed_share": round(tot["n_fixed"] / tot["n"], 4) if tot["n"] else 0, + "by_lang": {k: dict(v) for k, v in sorted(by_lang.items())}, + } + + +def run_checks(provider, model_id, items, *, run, dry_run, quiet, gi_forms) -> list[dict]: + rows: list[dict] = [] + for item in tqdm(items, desc="02e-verify", leave=False, disable=quiet): + row = check_one(provider, model_id, item, run=run, dry_run=dry_run, gi_forms=gi_forms) + rows.append(row) + if not quiet: + if row.get("status") == "ok": + log(f"{row['slug']:40} {row['lang']} n={row['n']:2} fixed={row['n_fixed']}" + + (f" rejected={row['n_rejected']}" if row["n_rejected"] else "")) + else: + log(f"{row['slug']:40} {row['lang']} {row.get('status')}: {row.get('error', '')[:120]}") + return rows + + +def _gi_forms() -> frozenset[str]: + names = [] + for p in TERROIR.glob("*.json"): + if p.name.startswith("manifest"): + continue + d = cache.read_json_or_none(p) + if d and d.get("name"): + names.append(d["name"]) + return gi_forms_from_names(names) + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--provider", default="anthropic", choices=("anthropic", "mistral", "ollama")) + ap.add_argument("--model", default=None, help="default: providers.STAGE_DEFAULTS['backcheck']") + ap.add_argument("--thinking", default=None, choices=("disabled", "adaptive")) + ap.add_argument("--ollama-url", default=providers.DEFAULT_OLLAMA_URL) + ap.add_argument("--mistral-url", default=providers.DEFAULT_MISTRAL_URL) + ap.add_argument("--lang", action="append", default=None, help="target locale(s); default all four") + ap.add_argument("--country", action="append", default=None) + ap.add_argument("--only", action="append", default=None) + ap.add_argument("--only-file", default=None, help="JSON list (or {\"slugs\": [...]}) of slugs to restrict to") + ap.add_argument("--sample", type=int, default=0) + ap.add_argument("--limit", type=int, default=0) + ap.add_argument("--refresh", action="store_true") + ap.add_argument("--dry-run", action="store_true") + ap.add_argument("--batch", action="store_true") + ap.add_argument("--report", default=None) + ap.add_argument("--quiet", action="store_true") + args = ap.parse_args() + if args.only_file: + data = json.loads(Path(args.only_file).read_text(encoding="utf-8")) + args.only = list(args.only or []) + list(data.get("slugs") if isinstance(data, dict) else data) + + run = terroir_backup.run_id() + langs = args.lang or list(LANGS) + items, skipped = select( + langs=langs, countries=set(args.country) if args.country else None, + only=set(args.only) if args.only else None, sample=args.sample, limit=args.limit, refresh=args.refresh, + ) + log(f"{len(items)} translation caches due (skipped: {dict(skipped)})") + if not items: + return 0 + gi_forms = _gi_forms() + t0 = time.monotonic() + rows: list[dict] = [] + batch_stats: dict | None = None + if args.batch: + if not batch.supports(args.provider): + log("--batch requires --provider anthropic|mistral") + return 1 + model_id = args.model or batch.default_model(args.provider, stage="backcheck") + thinking = args.thinking or batch.default_thinking(args.provider, stage="backcheck") + log(f"batch: {len(items)} caches (provider={args.provider}, model={model_id}, thinking={thinking}, " + f"dry_run={args.dry_run}, run={run})") + + def run_loop(prov): + nonlocal rows + collecting = getattr(prov, "kind", "") == "collecting" + rows = run_checks(prov, model_id, items, run=run, dry_run=args.dry_run or collecting, + quiet=collecting or args.quiet, gi_forms=gi_forms) + + batch_stats = batch.run_two_pass(provider=args.provider, model=model_id, sidecar=BATCH_SIDECAR, + run_loop=run_loop, thinking=thinking) + kind = f"{args.provider}-api" + else: + provider, model_id = providers.make_provider( + args.provider, model=args.model, ollama_url=args.ollama_url, mistral_url=args.mistral_url, + stage="backcheck", thinking=args.thinking, + ) + rows = run_checks(provider, model_id, items, run=run, dry_run=args.dry_run, quiet=args.quiet, gi_forms=gi_forms) + kind = provider.kind + + summary = _summary(rows) + if batch_stats: + summary["batch"] = batch_stats + summary.update({ + "run": run, "model": model_id, "provider_kind": kind, "dry_run": args.dry_run, + "version": BACKCHECK_VERSION, "elapsed_seconds": round(time.monotonic() - t0, 1), + "skipped": dict(skipped), + }) + report = Path(args.report) if args.report else REPORT_DIR / f"backcheck-{run}{'-dryrun' if args.dry_run else ''}.json" + report.parent.mkdir(parents=True, exist_ok=True) + report.write_text(json.dumps({"summary": summary, "caches": rows}, ensure_ascii=False, indent=1, default=str) + "\n", + encoding="utf-8") + if not args.dry_run: + cache.write_json(MANIFEST, {"generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), **summary}, + sort_keys=True) + print(json.dumps(summary, ensure_ascii=False, indent=2), file=sys.stderr) + log(f"report → {report.relative_to(ROOT) if report.is_relative_to(ROOT) else report}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/04_build_maps.py b/scripts/04_build_maps.py index f711ab7..892cc12 100644 --- a/scripts/04_build_maps.py +++ b/scripts/04_build_maps.py @@ -92,6 +92,7 @@ from _lib.de.region import derive_region as derive_de_region from _lib.env import load_dotenv from _lib.es.geometry import ESPolygonIndex +from _lib.es.national_term import es_term_for from _lib.es.region import ( derive_ccaa as derive_es_ccaa, ) @@ -110,6 +111,14 @@ union_from_insee, ) from _lib.geometry_overrides import ClipResult, GeometryOverrides +from _lib.gi_terms import ( + build_term_tree, + build_terms_info, + class_key, + classification_label, + derive_eu_scheme, + resolve_national_term, +) from _lib.gr.geometry import GRPolygonIndex from _lib.gr.region import derive_region as derive_gr_region from _lib.hr.geometry import HRPolygonIndex @@ -119,6 +128,7 @@ from _lib.i18n import LOCALES, compile_catalogs, load_translations from _lib.it.comune import ITCommuneIndex from _lib.it.geometry import ITPolygonIndex +from _lib.it.national_term import it_term_for from _lib.it.region import derive_regione as derive_it_regione from _lib.it.zones import ITZoneIndex from _lib.lexicon_loading import ( @@ -166,6 +176,7 @@ taxonomy_dfs_order as _taxonomy_dfs_order, ) from _lib.summaries import derive_summary +from _lib.terroir_normalize import normalize_aocs from shapely.geometry import mapping, shape from tqdm import tqdm from unidecode import unidecode as _unidecode @@ -587,6 +598,19 @@ def walk(o): return minx, miny, maxx, maxy +# Villages (simple-mode) polygon vs parcellaire bbox ratio above which a +# cahier-text commune union is treated as an extraction artefact and +# narrowed to parcel-bearing communes. Corpus median is ~2x; legitimate +# single-commune aires around tiny crus reach ~1000x but never pass the +# commune-count gate. Pouilly-Loché's bad aire was ~44,000x. +VILLAGES_BBOX_GUARD_RATIO = 20.0 + + +def _bbox_area(g) -> float: + minx, miny, maxx, maxy = g.bounds + return float((maxx - minx) * (maxy - miny)) + + def communes_containing(needle, insee_idx: dict[str, dict]) -> set[str]: """Return the INSEE codes of every IGN commune intersecting `needle`. @@ -1358,6 +1382,9 @@ def main() -> int: file=sys.stderr, ) + # (eu_scheme, national_term) per slug — parents run first, so a + # sub-denomination without a term of its own can fall back on its parent. + term_by_slug: dict[str, tuple[str, str]] = {} for record in tqdm(extracted_records, desc="union", leave=False): is_sub_denomination = bool(record.get("is_sub_denomination")) country = record.get("country") or "fr" @@ -2200,6 +2227,42 @@ def main() -> int: else: v_geom, v_stats = union_for_appellation(record, commune_idx) v_source = "communes" + # Guard: a cahier-text aire that dwarfs the parcellaire polygon is + # an extraction artefact (an aire de proximité list read as the + # aire — Pouilly-Loché's 2024 cahier drew the AOC across all of + # Burgundy in simple mode), not a production area. Narrow the + # text-derived communes to those holding parcels, exactly as the + # aires-CSV path does, and say so loudly. Dormant for correct + # records: a single-commune aire around a tiny cru never trips + # the count gate, and a multi-commune aire whose communes all + # hold parcels keeps every commune. + if ( + v_source == "communes" + and geom_source == "parcellaire" + and v_geom is not None + and not v_geom.is_empty + and geom is not None + and not geom.is_empty + and v_stats.get("matched", 0) > 3 + and _bbox_area(v_geom) > VILLAGES_BBOX_GUARD_RATIO * _bbox_area(geom) + ): + text_codes = cahier_insee(record, commune_idx) + narrowed = { + c for c in text_codes + if c in insee_idx and shape(insee_idx[c]).intersects(geom) + } + if narrowed and len(narrowed) < len(text_codes): + n_geom, n_stats = union_from_insee(narrowed, insee_idx) + if n_geom is not None and not n_geom.is_empty: + ratio = _bbox_area(v_geom) / max(_bbox_area(geom), 1e-12) + print( + f"[villages-guard] {record['slug']}: cahier-text aire " + f"({len(text_codes)} communes) spans {ratio:.0f}x the " + f"parcellaire bbox — narrowed to {len(narrowed)} " + f"parcel-bearing commune(s); check the stage-02 aire", + file=sys.stderr, + ) + v_geom, v_stats = n_geom, n_stats # Drop curator-reviewed spurious parts (geometry-outlier overrides). # Applied to both the detail and village geometries; for ES/PT/IT/AT @@ -2328,6 +2391,16 @@ def main() -> int: mvt_kind = "IGP" else: mvt_kind = raw_kind if raw_kind != "STUB" else "AOC" + eu_scheme = derive_eu_scheme(record, mvt_kind) + national_term = resolve_national_term( + record, mvt_kind, it_term_for=it_term_for, es_term_for=es_term_for + ) + if not national_term and is_sub_denomination: + parent_axes = term_by_slug.get(record.get("parent_slug") or "") + if parent_axes: + national_term = parent_axes[1] + term_by_slug[record["slug"]] = (eu_scheme, national_term) + gi_class_key = class_key(eu_scheme, record.get("country") or "fr", national_term) # ES records have `file_number` (e.g. PDO-ES-A0117) instead of FR's # numeric id_appellation; we coalesce so the MVT property carries # *some* stable identifier regardless of country. @@ -2546,6 +2619,9 @@ def main() -> int: "name": record["name"], "name_latin": record.get("name_latin") or "", "kind": mvt_kind, + "eu_scheme": eu_scheme, + "national_term": national_term, + "class_key": gi_class_key, "region": region_value, "categorie": categorie, "is_wine": is_wine, @@ -3501,6 +3577,11 @@ def emit_html( principal_counts: dict[str, int] = {} accessory_counts: dict[str, int] = {} region_counts: dict[str, int] = {} + # Appellation-type facet counts (scheme rows + "<cc>:<term>" rows), parents + # only and wines only, so a DOCG count is the roster count, not roster + + # sottozone, and a DOCa count is Rioja, not Rioja + its subzonas. + gi_class_counts: dict[str, int] = {} + gi_term_display: dict[str, tuple[str, str]] = {} grapes_all_counts: dict[str, int] = {} simple_style_counts: dict[str, int] = {} village_bbox_by_slug: dict[str, list[float]] = {} @@ -3671,6 +3752,9 @@ def emit_html( "name": p["name"], "name_latin": p.get("name_latin") or "", "kind": p["kind"], + "eu_scheme": p.get("eu_scheme") or "", + "national_term": p.get("national_term") or "", + "class_key": p.get("class_key") or "", "region": p["region"], "is_wine": p.get("is_wine", "1") == "1", "is_sub_denomination": p.get("is_sub_denomination", "0") == "1", @@ -3716,6 +3800,13 @@ def emit_html( grapes_all_counts[s] = grapes_all_counts.get(s, 0) + 1 if p["region"]: region_counts[p["region"]] = region_counts.get(p["region"], 0) + 1 + if p.get("is_sub_denomination", "0") != "1" and p.get("is_wine", "1") == "1": + ck = p.get("class_key") or "" + if ck: + gi_class_counts[ck] = gi_class_counts.get(ck, 0) + 1 + toks = ck.strip(";").split(";") + if len(toks) > 1: + gi_term_display[toks[1]] = (p.get("country") or "", p.get("national_term") or "") def sort_facet(d: dict[str, int]) -> list[tuple[str, int]]: return sorted(d.items(), key=lambda kv: (-kv[1], kv[0])) @@ -3794,6 +3885,8 @@ def sort_facet(d: dict[str, int]) -> list[tuple[str, int]]: ] class_descendants = _aging_tax.descendants_map() + facet_term_tree, term_descendants = build_term_tree(gi_class_counts) + facets = dict( layer_url=layer_url, villages_layer_url=villages_layer_url, @@ -3805,6 +3898,15 @@ def sort_facet(d: dict[str, int]) -> list[tuple[str, int]]: class_descendants=class_descendants, facet_styles_simple=facet_styles_simple, facet_regions=sort_facet(region_counts), + facet_term_tree=facet_term_tree, + term_descendants=term_descendants, + term_display=gi_term_display, + corpus_counts={ + "n": len(aocs), + "c": len({r.get("country") for r in aocs.values()}), + "parents": sum(1 for r in aocs.values() if not r.get("is_sub_denomination")), + "subs": sum(1 for r in aocs.values() if r.get("is_sub_denomination")), + }, area_quartiles=(area_q1, area_q3), vivc_by_slug=_load_vivc_by_slug(), ) @@ -3889,6 +3991,7 @@ def gate_classify(rec: dict, has_children: bool = False) -> tuple[str, str | Non # summaries that fix extraction quirks — same provenance as the # cahier, so no marker. aocs_for_lang = {} + lang_labels = build_labels(load_translations(lang).gettext) for slug, rec in aocs.items(): rec_country = rec.get("country") if rec_country in ("ch", "be"): @@ -3930,6 +4033,14 @@ def gate_classify(rec: dict, has_children: bool = False) -> tuple[str, str | Non "note": {"text": note_text, "sources": note_obj.get("sources") or []}, } + # One composer, run once per locale: the JS app and the SSR card + # both read this string rather than assembling term + scheme. + new_rec = { + **new_rec, + "class_label": classification_label( + rec.get("national_term") or "", rec.get("eu_scheme") or "", lang_labels + ), + } aocs_for_lang[slug] = new_rec # Overlay translated terroir-fact bullets only for records whose # source language differs from the current locale (canonical bullets @@ -3958,6 +4069,10 @@ def _src_lang_for(slug: str) -> str: aocs_for_lang = overlay_translated_facts(aocs_for_lang, facts_translations) if facts_drop_idx: aocs_for_lang = apply_inherited_facts_filter(aocs_for_lang, facts_drop_idx) + # Deterministic bullet clean-up (colour codes, VT/SGN, terminal period) + # after the overlay so every locale's text is normalised, and after the + # sibling filter so the index-based drops stay aligned. + aocs_for_lang = normalize_aocs(aocs_for_lang, lang) out = (WIKI / "index.html") if lang == "en" else (WIKI / lang / "index.html") out.parent.mkdir(parents=True, exist_ok=True) # Pass a swapped facets dict so the per-locale `aocs` is the data bundle. @@ -3966,6 +4081,7 @@ def _src_lang_for(slug: str) -> str: entity_out_dir = WIKI / ("en" if lang == "en" else lang) html_out, assets, n_index, n_fold = render_map_html( **per_locale_facets, locale=lang, grapes_info=lex, styles_info=styles_lex, + terms_info=build_terms_info(lang), index_slugs=index_slugs, fold_slugs=fold_slugs, entity_out_dir=entity_out_dir, children_map=children_by_parent, build_date=build_date, ) diff --git a/scripts/_lib/assets/app.js b/scripts/_lib/assets/app.js index 8c121f2..a137344 100644 --- a/scripts/_lib/assets/app.js +++ b/scripts/_lib/assets/app.js @@ -30,6 +30,14 @@ const CLASS_DESCENDANTS = __OWM_class_descendants_json__; const CLASS_LABELS = __OWM_class_labels_json__; const CLASS_SEARCH_TERMS = __OWM_class_search_terms_json__; + // Appellation-type facet: legal scheme rows (pdo / pgi / spirit-gi / uk-* / + // none) over the regulator's traditional-term rows ('<cc>:<term>'), matched + // against the ';'-padded `class_key` MVT property. TERMS_INFO feeds the + // tooltip on the two meta-line tokens (see _lib/gi_terms.py). + const FACET_TERM_TREE = __OWM_term_tree_json__; + const TERM_DESCENDANTS = __OWM_term_descendants_json__; + const TERM_LABELS = __OWM_term_labels_json__; + const TERMS_INFO = __OWM_terms_info_json__; const LABELS = __OWM_labels_json__; // Default document title (the locale homepage title). The tab title tracks // the open appellation so it doesn't get stuck on whichever entity page the @@ -570,6 +578,12 @@ // clipboard on click so the link still works for users without a // configured mailto handler (Firefox silently drops navigation in // that case). + // Which feedback channel gets used. The GitHub link is an outbound click + // Plausible already counts, but clicks there produced no issues, so the + // channel split (github vs e-mail) is the number that matters. + document.querySelectorAll('a[data-feedback]').forEach(a => { + a.addEventListener('click', () => track('Feedback Clicked', { channel: a.dataset.feedback, locale: LANG })); + }); document.querySelectorAll('a.feedback-mail').forEach(a => { const address = () => a.dataset.u + '@' + a.dataset.d; const arm = () => { @@ -616,6 +630,7 @@ styles: new Set(), stylesSimple: new Set(), classifications: new Set(), + appellationType: new Set(), principal: new Set(), accessory: new Set(), grapesAll: new Set(), @@ -662,6 +677,7 @@ if (kind === 'styleSimple') filters.stylesSimple.delete(key); else if (kind === 'style') filters.styles.delete(key); else if (kind === 'classification') filters.classifications.delete(key); + else if (kind === 'appellationType') filters.appellationType.delete(key); else if (kind === 'grapeAll') filters.grapesAll.delete(key); else if (kind === 'principal') filters.principal.delete(key); else if (kind === 'accessory') filters.accessory.delete(key); @@ -695,6 +711,7 @@ 'facet-styles': filters.styles, 'facet-styles-simple': filters.stylesSimple, 'facet-classification': filters.classifications, + 'facet-appellation-type': filters.appellationType, }; for (const [id, set] of Object.entries(sets)) { const el = document.getElementById(id); @@ -832,9 +849,19 @@ } const expandStyles = set => expandTree(set, STYLE_DESCENDANTS); const expandClass = set => expandTree(set, CLASS_DESCENDANTS); + const expandTerm = set => expandTree(set, TERM_DESCENDANTS); + // A record carries one ';'-padded class_key (';pdo;it:docg;'); a pick matches + // when any expanded token is one of its segments. + function matchesTermSet(rec, set) { + if (!set || !set.size) return true; + const ck = rec.class_key || ''; + for (const t of set) if (ck.includes(';' + t + ';')) return true; + return false; + } buildTreeFacet('facet-styles', FACET_STYLES_TREE, filters.styles, STYLE_LABELS, STYLE_DESCENDANTS, 'styles'); buildTreeFacet('facet-classification', FACET_CLASS_TREE, filters.classifications, CLASS_LABELS, CLASS_DESCENDANTS, 'classification'); + buildTreeFacet('facet-appellation-type', FACET_TERM_TREE, filters.appellationType, TERM_LABELS, TERM_DESCENDANTS, 'appellation-type'); buildFacet('facet-styles-simple', FACET_STYLES_SIMPLE, filters.stylesSimple, k => SIMPLE_STYLE_LABELS[k] || k); document.querySelectorAll('.grape-chip-filter').forEach(container => { const role = container.dataset.role || 'all'; @@ -1075,7 +1102,7 @@ lastPanelTrigger = btn || (label && label.querySelector('.open-aoc')) || null; lastStackKey = slug; stackFocusIndex = 0; - renderPanelStack([slug], 0); + renderPanelStack([slug], 0, undefined, 'facet'); track('Appellation Opened', { slug: slug, via: 'facet', locale: LANG }); const b = (viewMode === 'simple' && AOCS[slug].bbox_villages) ? AOCS[slug].bbox_villages : AOCS[slug].bbox; if (b && typeof map.fitBounds === 'function') { @@ -1273,6 +1300,7 @@ filters.q = ''; filters.styles.clear(); filters.stylesSimple.clear(); filters.classifications.clear(); + filters.appellationType.clear(); filters.principal.clear(); filters.accessory.clear(); filters.grapesAll.clear(); filters.appellations.clear(); filters.mainGrapeOnly = false; @@ -1332,6 +1360,8 @@ if (sExpr) parts.push(sExpr); const cExpr = inField('classifications', expandClass(filters.classifications)); if (cExpr) parts.push(cExpr); + const tExpr = inField('class_key', expandTerm(filters.appellationType)); + if (tExpr) parts.push(tExpr); const gExpr = inField(activeGrapeField(), expandGrapeSet(filters.grapesAll)); if (gExpr) parts.push(gExpr); if (filters.appellations.size) { @@ -1349,6 +1379,7 @@ if (styleSet.size && !setIntersects(styleSet, rec.styles || [])) return false; const classSet = expandClass(filters.classifications); if (classSet && classSet.size && !setIntersects(classSet, rec.classifications || [])) return false; + if (!matchesTermSet(rec, expandTerm(filters.appellationType))) return false; if (filters.grapesAll.size && !setIntersects(expandGrapeSet(filters.grapesAll), rec[activeGrapeField()] || [])) return false; if (filters.appellations.size && !filters.appellations.has(slug)) return false; return true; @@ -1365,6 +1396,7 @@ const classSet = expandClass(filters.classifications); if (classSet && classSet.size && !setIntersects(classSet, rec.classifications || [])) return false; } + if (!except.has('appellationType') && !matchesTermSet(rec, expandTerm(filters.appellationType))) return false; if (!except.has('grapesAll') && filters.grapesAll.size && !setIntersects(expandGrapeSet(filters.grapesAll), rec[activeGrapeField()] || [])) return false; if (!except.has('appellations') && filters.appellations.size && !filters.appellations.has(slug)) return false; return true; @@ -1377,6 +1409,7 @@ for (const k of filters.stylesSimple) chips.push({ kind: 'styleSimple', key: k, label: SIMPLE_STYLE_LABELS[k] || k }); for (const k of filters.styles) chips.push({ kind: 'style', key: k, label: STYLE_LABELS[k] || k }); for (const k of filters.classifications) chips.push({ kind: 'classification', key: k, label: CLASS_LABELS[k] || k }); + for (const k of filters.appellationType) chips.push({ kind: 'appellationType', key: k, label: TERM_LABELS[k] || k }); for (const k of filters.grapesAll) chips.push({ kind: 'grapeAll', key: k, label: grapeName(k) }); if (filters.mainGrapeOnly) chips.push({ kind: 'mainGrapeOnly', key: '1', label: LABELS.main_grape_only_label }); // Collapse a fully-selected subtree into one chip — a whole country first, @@ -1434,6 +1467,7 @@ const map_ = { styles: filters.stylesSimple.size + filters.styles.size, classification: filters.classifications.size, + 'appellation-type': filters.appellationType.size, grapes: filters.grapesAll.size, appellations: filters.appellations.size, regions: regionsSelectedCount(), @@ -1505,6 +1539,25 @@ lbl.classList.toggle('facet-unavailable', n === 0 && !inp.checked); }); } + const termEl = document.getElementById('facet-appellation-type'); + if (termEl) { + const except = new Set(['appellationType']); + const termCounts = new Map(); + for (const slug in AOCS) { + const rec = AOCS[slug]; + if (rec.is_sub_denomination || !rec.class_key) continue; + if (!matchesExceptFacets(rec, slug, except)) continue; + for (const node in TERM_DESCENDANTS) { + if (matchesTermSet(rec, new Set(TERM_DESCENDANTS[node]))) termCounts.set(node, (termCounts.get(node) || 0) + 1); + } + } + termEl.querySelectorAll('label').forEach(lbl => { + const inp = lbl.querySelector('input[type=checkbox]'); if (!inp) return; + const n = termCounts.get(inp.dataset.key) || 0; + const c = lbl.querySelector('.count'); if (c) c.textContent = String(n); + lbl.classList.toggle('facet-unavailable', n === 0 && !inp.checked); + }); + } const appEl = document.getElementById('facet-appellations'); if (appEl) { const except = new Set(['appellations']); @@ -1672,7 +1725,7 @@ lastPanelTrigger = document.getElementById('omni'); lastStackKey = key; stackFocusIndex = 0; - renderPanelStack([key], 0); + renderPanelStack([key], 0, undefined, 'omnisearch'); track('Appellation Opened', { slug: key, via: 'omnisearch', locale: LANG }); const b = (viewMode === 'simple' && AOCS[key].bbox_villages) ? AOCS[key].bbox_villages : AOCS[key].bbox; if (b && typeof map.fitBounds === 'function') map.fitBounds([[b[0], b[1]], [b[2], b[3]]], { padding: 40, maxZoom: 11, duration: 500 }); @@ -2325,7 +2378,7 @@ return ` <div class="${klass}"> <h1>${nameWithLatin(r)}</h1> - <div class="meta">${countrySeg}${r.kind}${regionSeg}${metaTail}</div> + <div class="meta">${countrySeg}${renderClassification(r)}${regionSeg}${metaTail}</div> ${dgcLine} ${approxLine} ${stubLine} @@ -2367,13 +2420,36 @@ // Tab title for an open appellation — mirrors the server-rendered entity // <title> (see _build_entity_meta) so navigating within the SPA and landing // on a pre-rendered /<locale>/<slug> page show the same title. + // The two naming tokens of the meta line — traditional term first, legal + // scheme in brackets — from the precomputed per-locale `class_label` (one + // composer: _lib/gi_terms.classification_label). Each token is a span the + // pill tooltip targets when TERMS_INFO has a definition for it. Mirrors + // classification_html in content_block.py. + function renderClassification(r) { + const label = r.class_label || ''; + if (!label) return escapeHtml(r.kind || ''); + const term = r.national_term || ''; + const scheme = r.eu_scheme || ''; + const span = (text, key, cls) => { + const has = TERMS_INFO[key] && (TERMS_INFO[key].note || TERMS_INFO[key].full); + const extra = has ? ' has-info" tabindex="0' : ''; + return `<span class="${cls}${extra}" data-key="${escapeAttr(key)}">${escapeHtml(text)}</span>`; + }; + const termKey = r.class_key ? (r.class_key.split(';').filter(Boolean)[1] || '') : ''; + if (term && label.startsWith(term) && label !== term) { + return span(term, termKey, 'gi-term') + ' ' + span(label.slice(term.length).trim(), scheme, 'gi-scheme'); + } + if (term && label === term) return span(term, termKey, 'gi-term'); + return span(label, scheme, 'gi-scheme'); + } + function docTitleFor(slug) { const r = AOCS[slug]; if (!r) return DEFAULT_TITLE; const region = r.region ? regionLabel(r.region) : ''; const country = COUNTRY_LABELS[r.country] || ''; const geo = [region, country].filter(Boolean).join(', '); - const head = [r.kind, geo].filter(Boolean).join(' · '); + const head = [r.class_label || r.kind, geo].filter(Boolean).join(' · '); return r.name + (head ? ' — ' + head : '') + ' · Open Wine Map'; } @@ -2416,7 +2492,7 @@ + `</div>`; } - function renderPanelStack(slugs, focusIndex, doTrack) { + function renderPanelStack(slugs, focusIndex, doTrack, via) { if (!slugs.length) return; const sorted = slugs .filter(s => AOCS[s]) @@ -2450,8 +2526,15 @@ } // Popularity signal: the appellation brought to the front of the stack. // doTrack is suppressed for the localStorage restore (fires on every - // reload / language switch — not a fresh view). - if (doTrack !== false) { + // reload / language switch — not a fresh view). The page-load open of a + // /<lang>/<slug> landing is not tracked either: the pageview already + // records that slug, and a custom event fired on load made every entity + // landing a non-bounce by construction and let a session start with a + // custom event (empty entry page) whenever the pageview was deferred. + // `via` separates a map click from stack cycling and from the explicit + // facet / omnisearch / in-panel opens, so the slug breakdown reads per + // intent instead of as one popularity list. + if (doTrack !== false && via !== 'landing') { const focusSlug = ordered[0]; const fr = AOCS[focusSlug]; if (fr) { @@ -2461,6 +2544,8 @@ kind: fr.kind || '(none)', region: fr.region || '(none)', stacked: sorted.length > 1 ? 'true' : 'false', + stack_size: String(sorted.length), + via: via || 'map', locale: LANG, }); } @@ -2578,7 +2663,7 @@ if (urlSlug && AOCS[urlSlug]) { lastStackKey = urlSlug; stackFocusIndex = 0; - renderPanelStack([urlSlug], 0); + renderPanelStack([urlSlug], 0, undefined, 'landing'); // Frame the shared appellation, but only when the link carries no // explicit camera hash (respect a co-shared #zoom/lat/lon). Use the // page-entry snapshot, not the live hash — maplibre has already written @@ -2655,11 +2740,17 @@ if (!info || !info.extract) return null; return { info, url: info.page_url || '' }; } + if (el.matches('.gi-term.has-info, .gi-scheme.has-info')) { + const t = TERMS_INFO[el.dataset.key]; + if (!t || !(t.note || t.full)) return null; + const head = [t.full, t.castilian_form ? `(${t.castilian_form})` : ''].filter(Boolean).join(' '); + return { info: { note: [head, t.note].filter(Boolean).join(' — '), sources: t.sources || [] }, url: '' }; + } return null; } const showPillTip = (e) => { - const el = e.target.closest('a.pill.grape.has-info, .pill.style.has-info'); + const el = e.target.closest('a.pill.grape.has-info, .pill.style.has-info, .gi-term.has-info, .gi-scheme.has-info'); if (!el) return; const resolved = resolvePillInfo(el); if (!resolved) return; @@ -2697,6 +2788,10 @@ const vivcLink = `<a href="${escapeAttr(info.vivc_url)}" target="_blank" rel="noopener" title="${escapeAttr(LABELS.vivc_link_title)}">${escapeHtml(vivcLabel)}</a>`; srcBlock += srcBlock ? ` · ${vivcLink}` : vivcLink; } + if (info.sources && info.sources.length) { + const links = info.sources.map(s => `<a href="${escapeAttr(s.url)}" target="_blank" rel="noopener">${escapeHtml(s.label)}</a>`).join(' · '); + srcBlock += (srcBlock ? ' · ' : '') + escapeHtml(LABELS.gi_term_source_label) + ' : ' + links; + } const extPara = hasExtract ? `<p class="ext">${escapeHtml(info.extract)}</p>` : ''; const notePara = info.note ? `<p class="note">${escapeHtml(info.note)}</p>` : ''; const srcDiv = srcBlock ? `<div class="src">${srcBlock}</div>` : ''; @@ -2710,7 +2805,7 @@ }; const hidePillTip = (e) => { - if (e.target.closest('a.pill.grape.has-info, .pill.style.has-info')) scheduleGrapeTipClose(); + if (e.target.closest('a.pill.grape.has-info, .pill.style.has-info, .gi-term.has-info, .gi-scheme.has-info')) scheduleGrapeTipClose(); }; // Show on hover (mouseover) and on keyboard focus (focusin); hide on the // matching mouseout/focusout; Escape dismisses immediately. @@ -2736,7 +2831,7 @@ lastPanelTrigger = null; lastStackKey = ''; stackFocusIndex = 0; - renderPanelStack([slug]); + renderPanelStack([slug], 0, undefined, 'panel-link'); } }); @@ -2793,14 +2888,16 @@ __OWM_source_block__ } if (!slugs.length) { closePanel(false); return; } const key = slugs.slice().sort().join('|'); + let via = 'map'; if (key === lastStackKey && slugs.length > 1) { stackFocusIndex = (stackFocusIndex + 1) % slugs.length; + via = 'cycle'; } else { lastStackKey = key; stackFocusIndex = 0; } lastPanelTrigger = null; - renderPanelStack(slugs, stackFocusIndex); + renderPanelStack(slugs, stackFocusIndex, undefined, via); }); // Re-apply feature-state for any selection restored from localStorage diff --git a/scripts/_lib/batch.py b/scripts/_lib/batch.py index 55bf773..ef49766 100644 --- a/scripts/_lib/batch.py +++ b/scripts/_lib/batch.py @@ -21,6 +21,14 @@ resubmitting an already-processed one — and an interrupted run resumes the in-flight batch (via the sidecar) even after a partial pass 2. Runs are single-threaded. + +Every fetched result carries the provider's `usage` (input / output / +cache tokens); `run_batch` sums them, prices them at the Batch-API rate +(`BATCH_PRICES_USD_PER_M`, 50 % of the list price), prints the line and +appends it to `raw/.batch/costs.jsonl` — one row per batch, keyed by the +stage sidecar — so a run reports its own cost instead of the numbers +being re-read from the API afterwards. `run_two_pass` returns the same +summary in its stats dict for the stage's report. """ from __future__ import annotations @@ -38,6 +46,8 @@ import requests from _lib.env import load_dotenv +from _lib.prompt_cache import system_text +from _lib.providers import effective_thinking, stage_default def _load_dotenv() -> None: @@ -53,11 +63,19 @@ def _load_dotenv() -> None: _DEFAULT_MODEL = {"anthropic": "claude-sonnet-4-6", "mistral": "mistral-medium-latest"} -def default_model(provider: str) -> str: - """The batch-default model id for a provider (overridable with --model).""" +def default_model(provider: str, stage: str | None = None) -> str: + """The batch-default model id for a provider (overridable with --model); + for anthropic the per-stage default from `providers.STAGE_DEFAULTS`.""" + if provider == "anthropic": + return stage_default(stage)[0] return _DEFAULT_MODEL.get(provider, "") +def default_thinking(provider: str, stage: str | None = None) -> str | None: + """The stage's thinking mode for anthropic (None / "disabled" / "adaptive").""" + return stage_default(stage)[1] if provider == "anthropic" else None + + def supports(provider: str) -> bool: return provider in _KIND @@ -91,12 +109,13 @@ def _retry(fn, *, what: str, attempts: int = 5): # ───────────────────────────────────────────── collecting / replay providers ── -def _request_id(system: str, user: str) -> str: +def _request_id(system, user: str) -> str: """Stable content hash of a prompt — the batch `custom_id`. Replay keys on this, so it is order-independent: an interrupted run resumes correctly even when entries were cached (and thus dropped from the job list) in between.""" - return hashlib.sha256(f"{system}\x00{user}".encode()).hexdigest()[:32] + sys_key = system if isinstance(system, str) else json.dumps(system, sort_keys=True, ensure_ascii=False) + return hashlib.sha256(f"{sys_key}\x00{user}".encode()).hexdigest()[:32] class CollectingProvider: @@ -111,7 +130,7 @@ def __init__(self) -> None: self.requests: list[dict] = [] self._seen: set[str] = set() - def chat(self, *, system: str, user: str, max_tokens: int = 1024, **_: object) -> str: + def chat(self, *, system, user: str, max_tokens: int = 1024, cache_phase=None, **_: object) -> str: cid = _request_id(system, user) if cid not in self._seen: self._seen.add(cid) @@ -120,6 +139,7 @@ def chat(self, *, system: str, user: str, max_tokens: int = 1024, **_: object) - "system": system, "user": user, "max_tokens": max_tokens, + "phase": cache_phase, }) return "" @@ -134,7 +154,7 @@ def __init__(self, results: dict, kind: str) -> None: self.results = results self.kind = kind - def chat(self, *, system: str, user: str, **_: object) -> str: + def chat(self, *, system, user: str, **_: object) -> str: cid = _request_id(system, user) r = self.results.get(cid) if r is None: @@ -147,6 +167,88 @@ def chat(self, *, system: str, user: str, **_: object) -> str: return r["text"] +# ─────────────────────────────────────────────────────────────────── usage ── + +# List price, USD per million tokens (input, output); the Batch API bills +# BATCH_DISCOUNT of it. Cache reads bill at 10 % of input, cache writes at +# 125 %. Unknown model → tokens are still summed, cost is None. +BATCH_PRICES_USD_PER_M: dict[str, tuple[float, float]] = { + "claude-opus-5": (5.0, 25.0), + "claude-opus-4-8": (5.0, 25.0), + "claude-opus-4-7": (5.0, 25.0), + "claude-opus-4-6": (5.0, 25.0), + "claude-sonnet-5": (2.0, 10.0), + "claude-sonnet-4-6": (3.0, 15.0), + "claude-haiku-4-5": (1.0, 5.0), +} +BATCH_DISCOUNT = 0.5 +COSTS_LEDGER = Path(__file__).resolve().parents[2] / "raw" / ".batch" / "costs.jsonl" +_USAGE_KEYS = ("input_tokens", "output_tokens", "cache_creation_input_tokens", "cache_read_input_tokens") + + +def _usage_from_anthropic(usage) -> dict: + return {k: int(getattr(usage, k, 0) or 0) for k in _USAGE_KEYS} + + +def _usage_from_mistral(usage: dict | None) -> dict: + u = usage or {} + return {"input_tokens": int(u.get("prompt_tokens") or 0), "output_tokens": int(u.get("completion_tokens") or 0), + "cache_creation_input_tokens": 0, "cache_read_input_tokens": 0} + + +def batch_cost_usd(usage: dict, model: str) -> float | None: + """The Batch-API price of `usage` for `model`, or None for an unpriced model.""" + prices = BATCH_PRICES_USD_PER_M.get(model) + if not prices: + return None + inp, out = prices + per_m = ( + usage.get("input_tokens", 0) * inp + + usage.get("output_tokens", 0) * out + + usage.get("cache_read_input_tokens", 0) * inp * 0.1 + + usage.get("cache_creation_input_tokens", 0) * inp * 1.25 + ) + return round(per_m / 1_000_000 * BATCH_DISCOUNT, 4) + + +def usage_summary(results: dict, model: str) -> dict: + """Sum the per-result `usage` dicts and price them: {n_results, + n_with_usage, input_tokens, output_tokens, cache_creation_input_tokens, + cache_read_input_tokens, cost_usd}.""" + tot = {k: 0 for k in _USAGE_KEYS} + n = 0 + for r in results.values(): + u = r.get("usage") if isinstance(r, dict) else None + if not u: + continue + n += 1 + for k in _USAGE_KEYS: + tot[k] += int(u.get(k) or 0) + return {"n_results": len(results), "n_with_usage": n, **tot, "cost_usd": batch_cost_usd(tot, model)} + + +def _record_cost(*, provider: str, model: str, batch_id: str, sidecar: Path, thinking: str | None, + summary: dict) -> None: + row = { + "at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "stage": sidecar.stem, "provider": provider, "model": model, "thinking": thinking, + "batch_id": batch_id, "run": os.environ.get("OWM_TERROIR_RUN") or None, **summary, + } + try: + COSTS_LEDGER.parent.mkdir(parents=True, exist_ok=True) + with COSTS_LEDGER.open("a", encoding="utf-8") as fh: + fh.write(json.dumps(row, ensure_ascii=False) + "\n") + except OSError as e: # the ledger is bookkeeping, never a reason to fail a run + print(f"[batch] could not append to {COSTS_LEDGER.name}: {e}", file=sys.stderr) + + +def _cost_line(summary: dict) -> str: + cost = summary.get("cost_usd") + return (f"in={summary['input_tokens']:,} out={summary['output_tokens']:,}" + + (f" cache_read={summary['cache_read_input_tokens']:,}" if summary.get("cache_read_input_tokens") else "") + + (f" ≈ ${cost:,.2f}" if cost is not None else " (unpriced model)")) + + # ─────────────────────────────────────────────────────────── anthropic batch ── @@ -161,19 +263,28 @@ def _anthropic_client(): return anthropic.Anthropic(api_key=key) -def _submit_anthropic(model: str, reqs: list[dict]) -> str: +def _anthropic_params(model: str, r: dict, thinking: str | None) -> dict: + params = { + "model": model, + "max_tokens": r["max_tokens"], + "system": r["system"], + "messages": [{"role": "user", "content": r["user"]}], + } + # The Claude 5 family runs adaptive thinking when `thinking` is omitted, + # and the extraction stages' 1,500–2,000-token `max_tokens` budgets were + # sized for text only: each stage passes its `STAGE_DEFAULTS` mode + # ("disabled" for 02d, "adaptive" for the gate); `OWM_BATCH_THINKING` + # overrides for an experiment. + mode = effective_thinking(model, os.environ.get("OWM_BATCH_THINKING") or thinking) + if mode: + params["thinking"] = {"type": mode} + return params + + +def _submit_anthropic(model: str, reqs: list[dict], thinking: str | None = None) -> str: client = _anthropic_client() batch = client.messages.batches.create(requests=[ - { - "custom_id": r["custom_id"], - "params": { - "model": model, - "max_tokens": r["max_tokens"], - "system": r["system"], - "messages": [{"role": "user", "content": r["user"]}], - }, - } - for r in reqs + {"custom_id": r["custom_id"], "params": _anthropic_params(model, r, thinking)} for r in reqs ]) return batch.id @@ -198,7 +309,7 @@ def _fetch_anthropic(batch_id: str, poll_interval: int) -> dict: if res.type == "succeeded": text = "".join(blk.text for blk in res.message.content if getattr(blk, "type", "") == "text").strip() - out[entry.custom_id] = {"text": text} + out[entry.custom_id] = {"text": text, "usage": _usage_from_anthropic(res.message.usage)} else: out[entry.custom_id] = {"error": res.type} return out @@ -226,7 +337,7 @@ def _submit_mistral(model: str, reqs: list[dict]) -> str: "max_tokens": r["max_tokens"], "temperature": 0.2, "messages": [ - {"role": "system", "content": r["system"]}, + {"role": "system", "content": system_text(r["system"])}, {"role": "user", "content": r["user"]}, ], }, @@ -305,16 +416,21 @@ def _fetch_mistral(job_id: str, poll_interval: int) -> dict: continue d = json.loads(line) text, err = _mistral_line_text(d) - out[d.get("custom_id")] = {"text": text} if err is None else {"error": err} + if err is None: + resp = d.get("response") + body = (resp.get("body") or {}) if isinstance(resp, dict) else {} + out[d.get("custom_id")] = {"text": text, "usage": _usage_from_mistral(body.get("usage"))} + else: + out[d.get("custom_id")] = {"error": err} return out # ───────────────────────────────────────────────────────────── orchestration ── -def _submit(provider: str, model: str, reqs: list[dict]) -> str: +def _submit(provider: str, model: str, reqs: list[dict], thinking: str | None = None) -> str: if provider == "anthropic": - return _submit_anthropic(model, reqs) + return _submit_anthropic(model, reqs, thinking) if provider == "mistral": return _submit_mistral(model, reqs) raise ValueError(f"batch unsupported for provider {provider!r}") @@ -327,7 +443,7 @@ def _fetch(provider: str, batch_id: str, poll_interval: int) -> dict: def run_batch(provider: str, model: str, reqs: list[dict], *, sidecar: Path, - poll_interval: int = POLL_INTERVAL_S) -> dict: + poll_interval: int = POLL_INTERVAL_S, thinking: str | None = None) -> dict: """Submit `reqs` to `provider`'s Batch API, poll to completion, return {custom_id: {"text": ...} | {"error": ...}}. If `sidecar` already holds an in-flight batch id for this provider, resume that batch (no resubmit, @@ -350,25 +466,75 @@ def run_batch(provider: str, model: str, reqs: list[dict], *, sidecar: Path, return {} print(f"[batch] submitting {len(reqs)} requests to the {provider} batch " f"API (model={model}, ~50% cheaper than synchronous)", file=sys.stderr) - batch_id = _retry(lambda: _submit(provider, model, reqs), + batch_id = _retry(lambda: _submit(provider, model, reqs, thinking), what=f"{provider} batch submit") sidecar.parent.mkdir(parents=True, exist_ok=True) sidecar.write_text(json.dumps({ "provider": provider, "model": model, "batch_id": batch_id, - "n_requests": len(reqs), + "thinking": thinking, "n_requests": len(reqs), "submitted_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), }, indent=2), encoding="utf-8") print(f"[batch] {provider} batch {batch_id} submitted — id saved to " f"{sidecar.name} (re-run this command to resume if interrupted)", file=sys.stderr) results = _fetch(provider, batch_id, poll_interval) - print(f"[batch] {provider} batch {batch_id} complete — {len(results)} results", - file=sys.stderr) + summary = usage_summary(results, model) + print(f"[batch] {provider} batch {batch_id} complete — {len(results)} results; " + f"{_cost_line(summary)}", file=sys.stderr) + _record_cost(provider=provider, model=model, batch_id=batch_id, sidecar=sidecar, + thinking=thinking, summary=summary) + return results + + +PHASED_ENV = "OWM_BATCH_PHASED" + + +def phased() -> bool: + """Whether requests tagged with a `cache_phase` are submitted as one + batch per phase, in order (default on; OWM_BATCH_PHASED=0 disables).""" + return (os.environ.get(PHASED_ENV) or "1").strip().lower() not in ("0", "off", "false", "no") + + +def _phase_groups(reqs: list[dict]) -> list[tuple[object, list[dict]]]: + """Requests grouped by their `phase`, in order of first appearance; + untagged requests form one group.""" + order: list[object] = [] + groups: dict[object, list[dict]] = {} + for r in reqs: + ph = r.get("phase") + if ph not in groups: + order.append(ph) + groups[ph] = [] + groups[ph].append(r) + return [(ph, groups[ph]) for ph in order] + + +def run_phased(provider: str, model: str, reqs: list[dict], *, sidecar: Path, + poll_interval: int = POLL_INTERVAL_S, thinking: str | None = None) -> dict: + """The requests of a record that share a cached prefix (a 02d record's + four sub-section calls) are processed concurrently inside one batch, + so most of them write the prefix instead of reading it (13–47 % hits + on the cfg-2026-09-14 run). Submitting one batch per `cache_phase`, + each after the previous has ended, makes the first phase write and + the later phases read — the prefix block carries the 1-hour TTL for + this (a read refreshes the timer, so each phase only has to finish + within an hour). Each phase has its own sidecar and resumes on its + own; the merged results feed one replay.""" + groups = _phase_groups(reqs) + if len(groups) < 2: + return run_batch(provider, model, reqs, sidecar=sidecar, poll_interval=poll_interval, thinking=thinking) + results: dict = {} + for i, (ph, group) in enumerate(groups): + side = sidecar.with_name(f"{sidecar.stem}.p{i}{sidecar.suffix}") + print(f"[batch] phase {i + 1}/{len(groups)} ({ph}): {len(group)} requests", file=sys.stderr) + results.update(run_batch(provider, model, group, sidecar=side, poll_interval=poll_interval, + thinking=thinking)) + side.unlink(missing_ok=True) return results def run_two_pass(*, provider: str, model: str, sidecar: Path, run_loop, - poll_interval: int = POLL_INTERVAL_S) -> dict: + poll_interval: int = POLL_INTERVAL_S, thinking: str | None = None) -> dict: """Run a stage's processing loop as a batch. `run_loop(provider)` runs the stage loop once, single-threaded; it is called twice (collect, replay). The stage should enumerate only stale / missing entries — replay matches @@ -380,16 +546,21 @@ def run_two_pass(*, provider: str, model: str, sidecar: Path, run_loop, with contextlib.redirect_stderr(io.StringIO()): run_loop(collector) # pass 1 — collect prompts (stderr muted: "" noise) reqs = collector.requests - if not reqs and not sidecar.exists(): + in_flight = sidecar.exists() or any(sidecar.parent.glob(f"{sidecar.stem}.p*{sidecar.suffix}")) + if not reqs and not in_flight: print("[batch] nothing to do — all entries already processed.", file=sys.stderr) return {"n_requests": 0, "n_results": 0, "n_errored": 0} print(f"[batch] collected {len(reqs)} distinct model requests", file=sys.stderr) - results = run_batch(provider, model, reqs, sidecar=sidecar, - poll_interval=poll_interval) + use_phases = phased() and any(r.get("phase") is not None for r in reqs) + runner = run_phased if use_phases else run_batch + results = runner(provider, model, reqs, sidecar=sidecar, + poll_interval=poll_interval, thinking=thinking) + usage = usage_summary(results, model) run_loop(ReplayProvider(results, kind=_KIND[provider])) # pass 2 — write caches sidecar.unlink(missing_ok=True) # batch fully consumed — clear resume state n_err = sum(1 for r in results.values() if r.get("error")) if n_err: print(f"[batch] {n_err} of {len(results)} requests errored — re-run to " "retry just those (already-done entries are skipped)", file=sys.stderr) - return {"n_requests": len(reqs), "n_results": len(results), "n_errored": n_err} + return {"n_requests": len(reqs), "n_results": len(results), "n_errored": n_err, + "model": model, "usage": usage} diff --git a/scripts/_lib/content_block.py b/scripts/_lib/content_block.py index 27e2220..372bbc1 100644 --- a/scripts/_lib/content_block.py +++ b/scripts/_lib/content_block.py @@ -31,7 +31,7 @@ import hashlib import re import unicodedata -from dataclasses import dataclass +from dataclasses import dataclass, field # Per-jurisdiction regulator-published specification document name, in the # regulator's own language — mirrors STUB_DOC_NAMES in map_template.py's JS. @@ -94,6 +94,9 @@ class RenderCtx: styles_info: dict style_labels: dict github_new_issue_url: str + # Tooltip payload for the two naming tokens (scheme id / '<cc>:<term>'), + # see _lib/gi_terms.build_terms_info. Empty = plain text, no <abbr>. + terms_info: dict = field(default_factory=dict) # ---------------------------------------------------------------- primitives @@ -705,6 +708,43 @@ def _meta_tail(rec: dict, ctx: RenderCtx) -> str: return "" +def classification_html(rec: dict, ctx: RenderCtx) -> str: + """The two naming tokens of the meta line — traditional term first, legal + scheme in brackets — as spans the client tooltip can target, each carrying + its definition as an <abbr title> for no-JS readers. Mirrors + renderClassification in app.js; the label itself is the precomputed + `class_label` (one composer, _lib/gi_terms.classification_label).""" + label = rec.get("class_label") or "" + if not label: + return "" + term = rec.get("national_term") or "" + scheme = rec.get("eu_scheme") or "" + country = rec.get("country") or "" + info = ctx.terms_info or {} + + def _span(text: str, key: str, cls: str) -> str: + entry = info.get(key) or {} + tip = " — ".join(x for x in (entry.get("full"), entry.get("note")) if x) + has = ' has-info" tabindex="0' if tip else "" + inner = f'<abbr title="{esc(tip)}">{esc(text)}</abbr>' if tip else esc(text) + return f'<span class="{cls}{has}" data-key="{esc(key)}">{inner}</span>' + + if term and label.startswith(term) and label != term: + from _lib.gi_terms import term_key + + bracket = label[len(term):].strip() + return ( + _span(term, term_key(country, term), "gi-term") + + " " + + _span(bracket, scheme, "gi-scheme") + ) + if term and label == term: + from _lib.gi_terms import term_key + + return _span(term, term_key(country, term), "gi-term") + return _span(label, scheme, "gi-scheme") + + def _approx_line(rec: dict, ctx: RenderCtx) -> str: lab = ctx.labels gs = rec.get("geom_source") @@ -744,7 +784,7 @@ def render_subappellations(children, ctx: RenderCtx, country: str | None = None) their names enter the indexable surface — each child rendered as real text + a link on the *parent's* own page (which earns ranking for the child name), paired with the parent's JSON-LD ``containsPlace``. ``children`` items are - ``{name, path, kind}`` dicts, pre-resolved by the caller (URL logic lives in + ``{name, path, classification}`` dicts, pre-resolved by the caller (URL logic lives in ``map_template``, not here). The heading is the regulator's own term for the parent's ``country`` (:data:`SUBDENOM_HEADINGS`), falling back to the generic translated ``entity_nav_children`` label.""" @@ -757,7 +797,7 @@ def render_subappellations(children, ctx: RenderCtx, country: str | None = None) name = esc(c.get("name") or "") path = c.get("path") or "" link = f'<a href="{esc(path)}">{name}</a>' if path else name - kind = c.get("kind") or "" + kind = c.get("classification") or "" kind_html = f' <span class="sub-kind">{esc(kind)}</span>' if kind else "" items.append(f"<li>{link}{kind_html}</li>") return ( @@ -855,7 +895,7 @@ def render_content_block(rec: dict, slug: str, ctx: RenderCtx, children=None) -> inner = ( f"<h1>{name_with_latin(rec)}</h1>" - f'<div class="meta">{country_seg}{esc(rec.get("kind") or "")}{region_seg}{meta_tail}</div>' + f'<div class="meta">{country_seg}{classification_html(rec, ctx) or esc(rec.get("kind") or "")}{region_seg}{meta_tail}</div>' f"{dgc_line}{approx_line}{stub_line}" f"{_section(lab['panel_styles_h'], style_chips)}" f"{_section(lab['facet_principal_h'], principal)}" diff --git a/scripts/_lib/es/national_term.py b/scripts/_lib/es/national_term.py new file mode 100644 index 0000000..0ca2dd6 --- /dev/null +++ b/scripts/_lib/es/national_term.py @@ -0,0 +1,166 @@ +"""Spanish national traditional terms (Reg. (EU) 1308/2013 Art. 112(a)) per +EU-registered wine GI, from MAPA's "Listado de denominaciones de origen +protegidas e indicaciones geográficas protegidas de vinos registradas en la +Unión Europea" (column "Término tradicional": DO / DOCa / VP / VC / VT, +keyed by "Nº expediente UE"). Fetched by scripts/es/00_fetch_data.py. + +`pdftotext -layout` row shape (one GI per line, page furniture between): + + DOP Priorat / Priorato DOCa PDO-ES-A1560 + IGP Ribera del Queiles VT PGI-ES-A0083 + +MAPA's column shorthand is expanded to the bottle wording (TERM_DISPLAY). +Curator pins live in national_term_overrides.json (regional-language legal +forms such as DOQ, file-number drift between the listado and eAmbrosia, +listado lag behind a recognition the regulator itself announced). +""" +from __future__ import annotations + +import json +import re +import subprocess +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[3] +LISTADO_URL = ( + "https://www.mapa.gob.es/es/dam/jcr:f9643333-ef75-4a2f-8864-afd1ade63fd1/02_vinos.pdf" +) +LISTADO_DIR = ROOT / "raw" / "es" / "mapa" +LISTADO_FILE = "listado-dop-igp-vinos.pdf" +OVERRIDES_PATH = Path(__file__).with_name("national_term_overrides.json") + +TERM_DISPLAY = { + "DO": "DO", + "DOCa": "DOCa", + "VP": "Vino de Pago", + "VC": "Vino de Calidad", + "VT": "Vino de la Tierra", +} +PDO_TERMS = frozenset({"DO", "DOCa", "DOQ", "Vino de Pago", "Vino de Calidad"}) +PGI_TERMS = frozenset({"Vino de la Tierra"}) + +_ROW_RE = re.compile( + r"^\s*(?P<scheme>DOP|IGP)\s+(?P<name>\S.*?)\s{2,}" + r"(?P<term>DOCa|DO|VP|VC|VT)\s+(?P<file_number>P(?:DO|GI)-ES-[A-Z0-9]+)\s*$" +) +_FILE_NUMBER_RE = re.compile(r"^(?P<scheme>P(?:DO|GI))-ES-[A-Z]*(?P<tail>\d+)$") + +_TERMS_CACHE: dict[str, str] | None = None +_OVERRIDES_CACHE: dict[str, dict] | None = None + + +def parse_listado_text(text: str) -> dict[str, str]: + """{file_number: MAPA shorthand} from `pdftotext -layout` output.""" + rows: dict[str, str] = {} + for line in text.splitlines(): + m = _ROW_RE.match(line) + if not m: + continue + fn, term = m.group("file_number"), m.group("term") + if fn in rows and rows[fn] != term: + raise ValueError(f"listado: {fn} carries both {rows[fn]} and {term}") + rows[fn] = term + return rows + + +def load_es_terms() -> dict[str, str]: + """{file_number: display term}. Empty (with a stderr warning) when the + listado PDF is absent; cached per process.""" + global _TERMS_CACHE + if _TERMS_CACHE is not None: + return _TERMS_CACHE + pdf = LISTADO_DIR / LISTADO_FILE + shown = pdf.relative_to(ROOT) if pdf.is_relative_to(ROOT) else pdf + if not pdf.exists(): + print( + f"[es-national-term] {shown} missing — run " + "scripts/es/00_fetch_data.py; no ES traditional terms", + file=sys.stderr, + ) + _TERMS_CACHE = {} + return _TERMS_CACHE + proc = subprocess.run( + ["pdftotext", "-layout", str(pdf), "-"], + capture_output=True, text=True, check=True, + ) + shorthand = parse_listado_text(proc.stdout) + _TERMS_CACHE = {fn: TERM_DISPLAY[t] for fn, t in shorthand.items()} + print( + f"[es-national-term] {len(_TERMS_CACHE)} GIs from {shown}", + file=sys.stderr, + ) + return _TERMS_CACHE + + +def load_es_term_overrides() -> dict[str, dict]: + global _OVERRIDES_CACHE + if _OVERRIDES_CACHE is None: + data = json.loads(OVERRIDES_PATH.read_text(encoding="utf-8")) + _OVERRIDES_CACHE = {k: v for k, v in data.items() if not k.startswith("_")} + return _OVERRIDES_CACHE + + +def _bridge_file_number(file_number: str, roster: dict[str, str]) -> str: + """Bridge listado ↔ eAmbrosia file-number drift by scheme + numeric tail + (`PDO-ES-A0117` ↔ `PDO-ES-0117`); only an unambiguous hit resolves.""" + m = _FILE_NUMBER_RE.match(file_number) + if not m: + return "" + hits = set() + for fn, term in roster.items(): + r = _FILE_NUMBER_RE.match(fn) + if r and r.group("scheme") == m.group("scheme") and r.group("tail") == m.group("tail"): + hits.add(term) + return next(iter(hits)) if len(hits) == 1 else "" + + +def _check_scheme(term: str, kind: str, label: str) -> None: + if kind == "IGP": + assert term not in PDO_TERMS, f"{label}: PGI carries PDO-only term {term!r}" + elif kind == "DOP": + assert term not in PGI_TERMS, f"{label}: PDO carries PGI-only term {term!r}" + + +def es_term_for(record: dict, kind: str) -> str: + """Display traditional term for one ES record: slug override first, then + the listado roster by exact file_number, then the numeric-tail bridge, + then "". Never a kind-based default — a PGI must not be stamped DO.""" + slug = record.get("slug") or "" + override = load_es_term_overrides().get(slug) + if override and override.get("term"): + term = override["term"] + _check_scheme(term, kind, slug) + return term + file_number = ( + (override or {}).get("file_number") or record.get("file_number") or "" + ) + if not file_number: + return "" + roster = load_es_terms() + term = roster.get(file_number) or _bridge_file_number(file_number, roster) + if term: + _check_scheme(term, kind, slug or file_number) + return term + + +def check_sub_denomination_terms(records: list[dict]) -> list[str]: + """Sub-denominations share the parent's file_number, so they must resolve + to the parent's term (Rioja subzonas → DOCa). Returns one message per + mismatch; empty means consistent.""" + by_slug = {r.get("slug"): r for r in records} + problems = [] + for rec in records: + if not rec.get("is_sub_denomination"): + continue + parent = by_slug.get(rec.get("parent_slug")) + if parent is None: + problems.append(f"{rec.get('slug')}: parent {rec.get('parent_slug')!r} not in corpus") + continue + own = es_term_for(rec, rec.get("kind", "")) + parents_term = es_term_for(parent, parent.get("kind", "")) + if own != parents_term: + problems.append( + f"{rec.get('slug')}: {own!r} != parent {parent.get('slug')} {parents_term!r}" + ) + return problems diff --git a/scripts/_lib/es/national_term_overrides.json b/scripts/_lib/es/national_term_overrides.json new file mode 100644 index 0000000..f000a0a --- /dev/null +++ b/scripts/_lib/es/national_term_overrides.json @@ -0,0 +1,59 @@ +{ + "_doc": "Curator pins for scripts/_lib/es/national_term.py, keyed by record slug. `term` wins outright (display form); `castilian_form` records the Castilian equivalent when the pinned term is a regional-language legal form; `file_number` re-keys the listado lookup when eAmbrosia and the MAPA listado carry different expedientes for the same GI. Every entry cites a public source.", + "priorat": { + "term": "DOQ", + "castilian_form": "DOCa", + "rule": "regional-language legal form wins when the autonomous community's wine law defines it", + "sources": [ + { + "label": "Llei 2/2020, de 5 de març, de la vitivinicultura (Catalunya), art. 4(e) «denominació d'origen qualificada» — in force since 2020-05-09; BOE núm. 71, 17.3.2020", + "url": "https://www.boe.es/buscar/act.php?id=BOE-A-2020-3782" + }, + { + "label": "Llei 15/2002, de 27 de juny, d'ordenació vitivinícola (Catalunya), art. 3.3 — DOGC núm. 3673; repealed by Llei 2/2020", + "url": "https://www.boe.es/buscar/act.php?id=BOE-A-2002-14986" + }, + { + "label": "Ordre ARP/188/2006 — Reglament de la Denominació d'Origen Qualificada Priorat (BOE-A-2009-11168)", + "url": "https://www.boe.es/diario_boe/txt.php?id=BOE-A-2009-11168" + }, + { + "label": "Consell Regulador de la DOQ Priorat", + "url": "https://www.doqpriorat.org/" + }, + { + "label": "MAPA listado (Término tradicional: DOCa, PDO-ES-A1560)", + "url": "https://www.mapa.gob.es/es/dam/jcr:f9643333-ef75-4a2f-8864-afd1ade63fd1/02_vinos.pdf" + } + ] + }, + "tharsys": { + "term": "Vino de Pago", + "file_number": "PDO-ES-02086", + "note": "eAmbrosia registers Tharsys as PDO-ES-02980; the MAPA listado keys the same GI as PDO-ES-02086 (VP). The pliego de condiciones section 8 states the Art. 112(a) traditional term is \"vino de pago\".", + "sources": [ + { + "label": "Pliego de condiciones DOP Tharsys (MAPA, 2025-10-02), section 8 b) Etiquetado", + "url": "https://www.mapa.gob.es/dam/mapa/contenido/alimentacion/temas/calidad-agroalimentaria/2017-calidad-diferenciada/nuevo_denominaciones/pliegos-de-condiciones/pliego-condiciones-vinos/dops/tharsis_2025_10_02.pdf" + }, + { + "label": "MAPA listado (Término tradicional: VP, PDO-ES-02086)", + "url": "https://www.mapa.gob.es/es/dam/jcr:f9643333-ef75-4a2f-8864-afd1ade63fd1/02_vinos.pdf" + } + ] + }, + "urbezo": { + "term": "Vino de Pago", + "note": "The MAPA listado (2026-07-02) still prints DO for PDO-ES-02585; MAPA's own press release of 2024-10-25 announces the EU registration of the DOP \"de vino de pago 'Urbezo'\" — the 26th vino de pago recognised by the EU.", + "sources": [ + { + "label": "MAPA press release, 2024-10-25: La Unión Europea registra la nueva Denominación de Origen Protegida de vino de pago “Urbezo”", + "url": "https://www.mapa.gob.es/es/prensa/ultimas-noticias/detalle_noticias/la-union-europea-registra-la-nueva-denominacion-de-origen-protegida-de-vino-de-pago--urbezo--/fea9b19d-2912-44ad-800a-532512ce6e33" + }, + { + "label": "Pliego de condiciones DOP Urbezo (MAPA, 2024-10-25)", + "url": "https://www.mapa.gob.es/dam/mapa/contenido/alimentacion/temas/calidad-agroalimentaria/2017-calidad-diferenciada/nuevo_denominaciones/pliegos-de-condiciones/pliego-condiciones-vinos/dops/urbezo_2024_10_25_pc.pdf" + } + ] + } +} diff --git a/scripts/_lib/exonyms.py b/scripts/_lib/exonyms.py new file mode 100644 index 0000000..bed775d --- /dev/null +++ b/scripts/_lib/exonyms.py @@ -0,0 +1,139 @@ +"""Geographic exonyms the translation layer must use. + +The 02e rule keeps names verbatim, but a mountain range, river, sea or +region *used as a place* takes the target language's established name +(Vosges → Vogezen, Rhin → Rijn, Appennino → Apennines; English wine +writing keeps Mosel and Tejo for the regions, so those carry no English +form) — found by Boris +on the Dutch Alsace pages, 2026-09-11. `EXONYMS` maps each source form +to its target form per locale; `exonym_hits` flags a translated bullet +that still carries the source form. + +Many region names double as appellation names (Toscana IGT, Bourgogne, +Wien DAC, Steiermark); those are correct verbatim as labels and wrong as +places, and only the model can tell. The detector therefore treats the +two classes differently: a form that never appears in a GI name is +flagged on any whole-word occurrence; a form that does is flagged only +when no appellation marker sits next to it and it is not part of a grape +or institution name (Melon de Bourgogne, Junta de Andalucía). The +detector scopes a re-translation; it never rewrites anything. +""" + +from __future__ import annotations + +import re + +EXONYMS: dict[str, dict[str, str]] = { + "Vosges": {"nl": "Vogezen", "es": "Vosgos"}, + "Rhin": {"nl": "Rijn", "en": "Rhine", "es": "Rin"}, + "Rhein": {"nl": "Rijn", "en": "Rhine", "es": "Rin", "fr": "Rhin"}, + "Pyrénées": {"nl": "Pyreneeën", "en": "Pyrenees", "es": "Pirineos"}, + "Pirineos": {"nl": "Pyreneeën", "en": "Pyrenees", "fr": "Pyrénées"}, + "Méditerranée": {"nl": "Middellandse Zee", "en": "Mediterranean", "es": "Mediterráneo"}, + "Mediterraneo": {"nl": "Middellandse Zee", "en": "Mediterranean", "fr": "Méditerranée", "es": "Mediterráneo"}, + "Atlantique": {"nl": "Atlantische Oceaan", "en": "Atlantic", "es": "Atlántico"}, + "Atlantico": {"nl": "Atlantische Oceaan", "en": "Atlantic", "fr": "Atlantique"}, + "Adriatico": {"nl": "Adriatische Zee", "en": "Adriatic", "fr": "Adriatique", "es": "Adriático"}, + "Tirreno": {"nl": "Tyrreense Zee", "en": "Tyrrhenian", "fr": "Tyrrhénienne", "es": "Tirreno"}, + "Appennino": {"nl": "Apennijnen", "en": "Apennines", "fr": "Apennins", "es": "Apeninos"}, + "Appennini": {"nl": "Apennijnen", "en": "Apennines", "fr": "Apennins", "es": "Apeninos"}, + "Massif Central": {"nl": "Centraal Massief", "es": "Macizo Central"}, + "Alpes": {"nl": "Alpen", "en": "Alps"}, + "Alpi": {"nl": "Alpen", "en": "Alps", "fr": "Alpes", "es": "Alpes"}, + "Alpen": {"en": "Alps", "fr": "Alpes", "es": "Alpes"}, + "Dolomiti": {"nl": "Dolomieten", "en": "Dolomites", "fr": "Dolomites", "es": "Dolomitas"}, + "Donau": {"en": "Danube", "es": "Danubio", "fr": "Danube"}, + "Danube": {"nl": "Donau", "es": "Danubio"}, + "Danube Plain": {"nl": "Donauvlakte", "en": "Danubian Plain", "es": "Llanura del Danubio", "fr": "plaine du Danube"}, + "Garonne": {"es": "Garona"}, + "Loire": {"es": "Loira"}, + "Rhône": {"es": "Ródano"}, + "Tejo": {"fr": "Tage", "es": "Tajo"}, + "Lisboa": {"en": "Lisbon", "fr": "Lisbonne", "nl": "Lissabon"}, + "Bayern": {"en": "Bavaria", "es": "Baviera", "fr": "Bavière", "nl": "Beieren"}, + "Toscana": {"nl": "Toscane", "en": "Tuscany", "fr": "Toscane"}, + "Piemonte": {"nl": "Piëmont", "en": "Piedmont", "fr": "Piémont", "es": "Piamonte"}, + "Sicilia": {"nl": "Sicilië", "en": "Sicily", "fr": "Sicile"}, + "Sardegna": {"nl": "Sardinië", "en": "Sardinia", "fr": "Sardaigne", "es": "Cerdeña"}, + "Lombardia": {"nl": "Lombardije", "en": "Lombardy", "fr": "Lombardie", "es": "Lombardía"}, + "Steiermark": {"nl": "Stiermarken", "en": "Styria", "fr": "Styrie", "es": "Estiria"}, + "Wien": {"nl": "Wenen", "en": "Vienna", "fr": "Vienne", "es": "Viena"}, + "Sachsen": {"nl": "Saksen", "en": "Saxony", "fr": "Saxe", "es": "Sajonia"}, + "Kärnten": {"nl": "Karinthië", "en": "Carinthia", "fr": "Carinthie", "es": "Carintia"}, + "Andalucía": {"nl": "Andalusië", "en": "Andalusia", "fr": "Andalousie"}, + "Cataluña": {"nl": "Catalonië", "en": "Catalonia", "fr": "Catalogne"}, + "Catalunya": {"nl": "Catalonië", "en": "Catalonia", "fr": "Catalogne", "es": "Cataluña"}, + "Castilla": {"nl": "Castilië", "en": "Castile", "fr": "Castille"}, + "Bourgogne": {"nl": "Bourgondië", "en": "Burgundy", "es": "Borgoña"}, + "Alsace": {"nl": "Elzas", "es": "Alsacia"}, + "Moselle": {"nl": "Moezel", "es": "Mosela"}, + "Mosel": {"nl": "Moezel", "fr": "Moselle", "es": "Mosela"}, + "Rodopi": {"en": "Rhodopes", "fr": "Rhodopes", "nl": "Rhodopen", "es": "Ródope"}, + "Rodopite": {"en": "Rhodopes", "fr": "Rhodopes", "nl": "Rhodopen", "es": "Ródope"}, + "Родопи": {"en": "Rhodopes", "fr": "Rhodopes", "nl": "Rhodopen", "es": "Ródope"}, + "Родопите": {"en": "Rhodopes", "fr": "Rhodopes", "nl": "Rhodopen", "es": "Ródope"}, + "Balkan Mountains": {"nl": "Balkangebergte", "fr": "Grand Balkan", "es": "Gran Balcán"}, + "Nördliche Kalkalpen": {"en": "Northern Limestone Alps", "nl": "Noordelijke Kalkalpen", "fr": "Alpes calcaires septentrionales", "es": "Alpes Calcáreos del Norte"}, + "Schwarzwald": {"en": "Black Forest", "nl": "Zwarte Woud", "fr": "Forêt-Noire", "es": "Selva Negra"}, + "Bodensee": {"en": "Lake Constance", "nl": "Bodenmeer", "fr": "lac de Constance", "es": "lago de Constanza"}, + "Neusiedler See": {"en": "Lake Neusiedl", "nl": "Neusiedler See", "fr": "lac de Neusiedl", "es": "lago Neusiedl"}, + "Karpaty": {"en": "Carpathians", "nl": "Karpaten", "fr": "Carpates", "es": "Cárpatos"}, + "Karpaten": {"en": "Carpathians", "nl": "Karpaten", "fr": "Carpates", "es": "Cárpatos"}, + "Carpați": {"en": "Carpathians", "nl": "Karpaten", "fr": "Carpates", "es": "Cárpatos"}, + "Kárpátok": {"en": "Carpathians", "nl": "Karpaten", "fr": "Carpates", "es": "Cárpatos"}, + "Dunărea": {"en": "Danube", "nl": "Donau", "fr": "Danube", "es": "Danubio"}, + "Дунав": {"en": "Danube", "nl": "Donau", "fr": "Danube", "es": "Danubio"}, + "Duna": {"en": "Danube", "nl": "Donau", "fr": "Danube", "es": "Danubio"}, + "Peloponnisos": {"en": "Peloponnese", "nl": "Peloponnesos", "fr": "Péloponnèse", "es": "Peloponeso"}, + "Πελοπόννησος": {"en": "Peloponnese", "nl": "Peloponnesos", "fr": "Péloponnèse", "es": "Peloponeso"}, + "Kriti": {"en": "Crete", "nl": "Kreta", "fr": "Crète", "es": "Creta"}, + "Κρήτη": {"en": "Crete", "nl": "Kreta", "fr": "Crète", "es": "Creta"}, + "Makedonia": {"en": "Macedonia", "nl": "Macedonië", "fr": "Macédoine", "es": "Macedonia"}, + "Thessalia": {"en": "Thessaly", "nl": "Thessalië", "fr": "Thessalie", "es": "Tesalia"}, + "Ipiros": {"en": "Epirus", "nl": "Epirus", "fr": "Épire", "es": "Epiro"}, + "Egeo": {"en": "Aegean", "nl": "Egeïsche Zee", "fr": "mer Égée", "es": "Egeo"}, + "Aigaio": {"en": "Aegean", "nl": "Egeïsche Zee", "fr": "mer Égée", "es": "Egeo"}, + "Jadran": {"en": "Adriatic", "nl": "Adriatische Zee", "fr": "Adriatique", "es": "Adriático"}, + "Sredozemlje": {"en": "Mediterranean", "nl": "Middellandse Zee", "fr": "Méditerranée", "es": "Mediterráneo"}, + "Mediterrâneo": {"en": "Mediterranean", "nl": "Middellandse Zee", "fr": "Méditerranée", "es": "Mediterráneo"}, + "Atlântico": {"en": "Atlantic", "nl": "Atlantische Oceaan", "fr": "Atlantique", "es": "Atlántico"}, +} + +# An appellation label next to the name: the form is the registered name, not a place. +_GI_MARKER = re.compile( + r"\b(DOC|DOCG|IGT|IGP|DOP|AOC|AOP|PDO|PGI|DAC|g\.U\.|g\.g\.A\.|Landwein|grand cru|premier cru|" + r"appellation|denominazione|denominación|Vino de la Tierra|Anbaugebiet)\b", re.I) +# A grape or institution built on the place name. +_NAME_CONTEXT = re.compile(r"\b(Melon|Junta|Generalitat|Lezíria|Ribatejo|Consorzio|Consejo)\b", re.I) + + +def _pattern(form: str) -> re.Pattern: + return re.compile(r"(?<![\w-])" + re.escape(form) + r"(?![\w-])") + + +_PATTERNS = {form: _pattern(form) for form in EXONYMS} + + +def exonym_hits(bullet: str, lang: str, *, gi_forms: frozenset[str] = frozenset()) -> list[str]: + """Source forms in `bullet` that should read as their `lang` exonym. + `gi_forms`: the forms that also occur in an appellation name of the + corpus — flagged only in a place-like context.""" + out: list[str] = [] + for form, targets in EXONYMS.items(): + if lang not in targets: + continue + for m in _PATTERNS[form].finditer(bullet or ""): + ctx = bullet[max(0, m.start() - 45): m.end() + 45] + if form in gi_forms and (_GI_MARKER.search(ctx) or _NAME_CONTEXT.search(ctx)): + continue + if form not in gi_forms and _NAME_CONTEXT.search(ctx): + continue + out.append(form) + break + return out + + +def gi_forms_from_names(names) -> frozenset[str]: + """The exonym source forms that occur as a whole word in any GI name.""" + blob = " ".join(names) + return frozenset(form for form in EXONYMS if _PATTERNS[form].search(blob)) diff --git a/scripts/_lib/fr/register_overrides.json b/scripts/_lib/fr/register_overrides.json index 8bfa319..efd6122 100644 --- a/scripts/_lib/fr/register_overrides.json +++ b/scripts/_lib/fr/register_overrides.json @@ -1,5 +1,5 @@ { - "_README": "Curator pins for the FR SIQO appellation -> eAmbrosia register file-number resolver (scripts/01d_resolve_register.py). Keyed by INAO id_appellation. `file_number` is the register `fileName` (PDO-FR-…/PGI-FR-…); set it to \"\" to record a verified absence, which drops the appellation from the unresolved queue instead of leaving it as noise. Look a name up at https://ec.europa.eu/geographical-indications-register/ — the resolver deliberately refuses to fuzzy-match, so every residue lands here.", + "_README": "Curator pins for the FR SIQO appellation -> eAmbrosia register file-number resolver (scripts/01d_resolve_register.py). Keyed by INAO id_appellation. `file_number` is the register `fileName` (PDO-FR-…/PGI-FR-…); set it to \"\" to record a verified absence, which drops the appellation from the unresolved queue instead of leaving it as noise. Look a name up at https://ec.europa.eu/geographical-indications-register/ — the resolver deliberately refuses to fuzzy-match, so every residue lands here. An entry may also carry `prefer_cahier: true`: stage 01 then binds the appellation to the register's cahier attachment INSTEAD of the BO Agri PDF, bypassing its has-usable-cahier guard — for the case where the BO Agri PDF downloads fine but is verifiably another appellation's cahier (2026-09-11 review: Pierrevert carried Saint-Pourçain's lien, L'Etoile and Grands-Echezeaux carried Bourgogne Passe-tout-grains'). The register attachment name must name the appellation (see the note on each pin).", "335": { "name": "Calvados Domfontais", "file_number": "PGI-FR-01837", @@ -11,5 +11,33 @@ "file_number": "PGI-FR-01836", "register_name": "Marc d'Alsace Gewurztraminer", "note": "Resolves on the all-partition fallback too (SIQO carries no categorie on the manifest row); pinned so the binding is explicit rather than inferred." + }, + "184": { + "name": "Grands-Echezeaux", + "file_number": "PDO-FR-A0583", + "register_name": "Grands-Echezeaux", + "prefer_cahier": true, + "note": "BO Agri serves another appellation's cahier for this AOC (terroir-facts review 2026-09-11); the register attachment 'AGRT1122691D - Grands-Echezeaux CDC publication BO.pdf' is the AOC's own cahier — bind it instead." + }, + "187": { + "name": "L'Etoile", + "file_number": "PDO-FR-A0598", + "register_name": "L'Etoile", + "prefer_cahier": true, + "note": "BO Agri serves another appellation's cahier for this AOC (terroir-facts review 2026-09-11); the register attachment 'AGRT1108018D - l-Etoile CDC homologue.pdf' is the AOC's own cahier — bind it instead." + }, + "290": { + "name": "Pierrevert", + "file_number": "PDO-FR-A0918", + "register_name": "Pierrevert", + "prefer_cahier": true, + "note": "BO Agri serves another appellation's cahier for this AOC (terroir-facts review 2026-09-11); the register attachment 'AGRT1107885D cdc Pierrevert BO.pdf' is the AOC's own cahier — bind it instead." + }, + "144": { + "name": "Bourgogne Passe-tout-grains", + "file_number": "PDO-FR-A0738", + "register_name": "Bourgogne Passe-tout-grains", + "prefer_cahier": true, + "note": "BO Agri serves the AOC Beaujolais cahier for this AOC (terroir-facts review 2026-09-12: 23 mentions of Beaujolais in the lien, 0 of Passe-tout-grains); the register attachment 'CDC_Bourgogne_Passe-tout-grains.pdf' is the AOC's own cahier — bind it instead." } } diff --git a/scripts/_lib/gi_terms.py b/scripts/_lib/gi_terms.py new file mode 100644 index 0000000..f999936 --- /dev/null +++ b/scripts/_lib/gi_terms.py @@ -0,0 +1,231 @@ +"""Two naming axes for an appellation: the legal scheme and the traditional term. + +`eu_scheme` is derived from fields already on the record — the SIQO `signe_ue` +for France, the eAmbrosia kind elsewhere, the GOV.UK register for the UK; Swiss +records sit outside the EU scheme. `national_term` is the EU-registered +traditional term (Reg. (EU) 1308/2013 Art. 112(a); Reg. (EC) 607/2009 Annex +XII) a member state attaches to the GI as a whole — DOCG / DOC / IGT, +DOCa / DOQ / DO / Vino de Pago, AOC, DAC, Landwein … — read from a public +roster (MASAF elenco, MAPA listado, SIQO `signe_fr`) or from the checked-in +ruling table `traditional_terms.json`. Lot-level quality grades +(Qualitätswein, Prädikatswein, kakovostno …) and mere translations of +PDO / PGI (OEM, ZOP, ΠΟΠ, BOB …) are never terms. The stored `kind` token is +untouched: it stays the map's paint / filter key. + +Rendering is `TERM (SCHEME)` with the scheme word in the UI locale — "DOQ +(PDO)", "DOCG (AOP)" — term-only when there is no scheme (Swiss AOC) and +scheme-only when the country has no term. `classification_label` is the +single composer; stage 04 runs it once per locale so the JS app and the SSR +card read the same precomputed string. +""" +from __future__ import annotations + +import json +import sys +from functools import lru_cache +from pathlib import Path +from typing import Callable + +from _lib.grape_lexicon import slugify + +TERMS_PATH = Path(__file__).with_name("traditional_terms.json") + +SCHEMES: tuple[str, ...] = ("pdo", "pgi", "spirit-gi", "uk-pdo", "uk-pgi", "none") + +_SCHEME_LABEL_KEY = { + "pdo": "scheme_pdo", + "uk-pdo": "scheme_pdo", + "pgi": "scheme_pgi", + "uk-pgi": "scheme_pgi", + "spirit-gi": "scheme_spirit_gi", + "none": "", +} + +_KIND_SCHEME = {"AOC": "pdo", "AOP": "pdo", "DOP": "pdo", "IGP": "pgi", "EDV": "spirit-gi"} + +TermResolver = Callable[[dict, str], str] + + +@lru_cache(maxsize=1) +def load_terms_table() -> dict: + return load_terms_table_from(TERMS_PATH) + + +def load_terms_table_from(path: Path) -> dict: + if not path.exists(): + print(f"[gi_terms] {path.name} missing — no traditional terms", file=sys.stderr) + return {} + with path.open(encoding="utf-8") as fh: + table = json.load(fh) + for key, entry in (table.get("terms") or {}).items(): + scheme = entry.get("scheme") + if scheme not in SCHEMES: + raise ValueError(f"traditional_terms.json: term {key!r} has unknown scheme {scheme!r}") + _reject_scheme_abbreviations(table) + return table + + +# Local abbreviations of PDO / PGI are the scheme, not a traditional term; one +# landing in a term slot would render "OEM (PDO)" — the very conflation this +# layer exists to end — so the table refuses to load. +_SCHEME_ABBREVIATIONS = frozenset({ + "pdo", "pgi", "aop", "igp", "dop", "bob", "bga", "g.u.", "g.g.a.", "gu", "gga", + "oem", "ofj", "zop", "zgo", "zoi", "zozp", "chop", "chzo", "znp", "zgu", "pop", "pge", + "ποπ", "πγε", "знп", "згу", +}) + + +def _reject_scheme_abbreviations(table: dict) -> None: + def _check(term: str, where: str) -> None: + if term and term.casefold().replace(" ", "") in _SCHEME_ABBREVIATIONS: + raise ValueError(f"traditional_terms.json: {where} uses the scheme abbreviation {term!r}") + for cc, by_kind in (table.get("constants") or {}).items(): + for kind, term in by_kind.items(): + _check(term, f"constants[{cc}][{kind}]") + for cc, pins in (table.get("pins") or {}).items(): + for key, pin in pins.items(): + _check(pin.get("term") or "", f"pins[{cc}][{key}]") + for key in (table.get("terms") or {}): + _check(key.partition(":")[2], f"terms[{key}]") + + +def _pin_for(record: dict, table: dict) -> dict | None: + pins = (table.get("pins") or {}).get(record.get("country") or "", {}) + if not pins: + return None + for key in (record.get("slug") or "", record.get("file_number") or ""): + if key and key in pins: + return pins[key] + return None + + +def derive_eu_scheme(record: dict, mvt_kind: str) -> str: + country = record.get("country") or "fr" + pin = _pin_for(record, load_terms_table()) + if pin and pin.get("scheme"): + return pin["scheme"] + if country == "ch": + return "none" + if country == "fr": + sue = (record.get("signe_ue") or "").strip().upper() + if sue == "AOP": + return "pdo" + if sue == "IGP": + return "pgi" + if sue == "IG": + return "spirit-gi" + return _KIND_SCHEME.get(mvt_kind, "pdo") + base = _KIND_SCHEME.get(mvt_kind, "pdo") + if country == "gb": + return "uk-" + base + return base + + +def resolve_national_term( + record: dict, + mvt_kind: str, + *, + it_term_for: TermResolver | None = None, + es_term_for: TermResolver | None = None, +) -> str: + table = load_terms_table() + pin = _pin_for(record, table) + if pin is not None and "term" in pin: + return pin["term"] or "" + country = record.get("country") or "fr" + if country == "it" and it_term_for is not None: + return it_term_for(record, mvt_kind) or "" + if country == "es" and es_term_for is not None: + return es_term_for(record, mvt_kind) or "" + return (table.get("constants") or {}).get(country, {}).get(mvt_kind, "") or "" + + +def term_key(country: str, term: str) -> str: + return f"{country}:{slugify(term)}" if term else "" + + +def class_key(eu_scheme: str, country: str, term: str) -> str: + parts = [eu_scheme] + tk = term_key(country, term) + if tk: + parts.append(tk) + return ";" + ";".join(parts) + ";" + + +def scheme_label(eu_scheme: str, labels: dict) -> str: + key = _SCHEME_LABEL_KEY.get(eu_scheme, "") + return labels.get(key, "") if key else "" + + +def classification_label(term: str, eu_scheme: str, labels: dict) -> str: + scheme = scheme_label(eu_scheme, labels) + if term and scheme and term.casefold() != scheme.casefold(): + return f"{term} ({scheme})" + return term or scheme + + +def _localised(block: dict | None, locale: str) -> str: + if not block: + return "" + return block.get(locale) or block.get("en") or "" + + +def build_terms_info(locale: str) -> dict[str, dict]: + """Tooltip payload keyed exactly like the tokens in `class_key`: scheme ids + and `<cc>:<term-slug>`. Each entry: label, full, note, sources[, castilian_form].""" + table = load_terms_table() + out: dict[str, dict] = {} + for scheme, entry in (table.get("schemes") or {}).items(): + out[scheme] = { + "full": _localised(entry.get("full"), locale), + "note": _localised(entry.get("note"), locale), + "sources": entry.get("sources") or [], + } + for key, entry in (table.get("terms") or {}).items(): + country, _, term = key.partition(":") + info = { + "term": term, + "full": entry.get("full") or "", + "scheme": entry.get("scheme") or "", + "note": _localised(entry.get("note"), locale), + "sources": entry.get("sources") or [], + } + if entry.get("castilian_form"): + info["castilian_form"] = entry["castilian_form"] + out[term_key(country, term)] = info + return out + + +SCHEME_ORDER: tuple[str, ...] = ("pdo", "pgi", "spirit-gi", "uk-pdo", "uk-pgi", "none") + + +def build_term_tree(counts: dict[str, int]) -> tuple[list[dict], dict[str, list[str]]]: + """Two-level facet tree from `class_key` → count: scheme rows (depth 0), then + their term rows (depth 1) by count. Same row shape as the classification + tree (`{slug, parent, depth, count}`); `descendants` includes self, which is + what `expandTree` in app.js relies on to widen a scheme pick to its terms.""" + scheme_totals: dict[str, int] = {} + by_scheme: dict[str, dict[str, int]] = {} + for ck, n in counts.items(): + toks = [t for t in ck.strip(";").split(";") if t] + if not toks: + continue + scheme = toks[0] + scheme_totals[scheme] = scheme_totals.get(scheme, 0) + n + if len(toks) > 1: + terms = by_scheme.setdefault(scheme, {}) + terms[toks[1]] = terms.get(toks[1], 0) + n + tree: list[dict] = [] + descendants: dict[str, list[str]] = {} + ordered = sorted( + scheme_totals, + key=lambda s: (SCHEME_ORDER.index(s) if s in SCHEME_ORDER else len(SCHEME_ORDER), s), + ) + for scheme in ordered: + tree.append({"slug": scheme, "parent": None, "depth": 0, "count": scheme_totals[scheme]}) + kids = sorted((by_scheme.get(scheme) or {}).items(), key=lambda kv: (-kv[1], kv[0])) + descendants[scheme] = [scheme] + [k for k, _ in kids] + for k, n in kids: + tree.append({"slug": k, "parent": scheme, "depth": 1, "count": n}) + descendants[k] = [k] + return tree, descendants diff --git a/scripts/_lib/it/masaf.py b/scripts/_lib/it/masaf.py index 4ae0f51..73e9159 100644 --- a/scripts/_lib/it/masaf.py +++ b/scripts/_lib/it/masaf.py @@ -224,33 +224,121 @@ def find_article_offsets(text: str) -> list[tuple[int, int, int]]: return out -def extract_articles(text: str) -> dict[int, str]: - """Carves the pdftotext output into a dict {article_num: body}. - Body excludes the header line and stops at the next article's - header (or EOF). Newer disciplinari sometimes have the header - appear twice (TOC + body); the LAST occurrence is kept since - that's where the real body sits.""" +TOC_BODY_MAX_CHARS = 200 +ANNEX_TITLE_LINES = 3 + + +def _runs(text: str, heads: list[tuple[int, int, int]]) -> list[list[tuple[int, int, int, int]]]: + """Split the header sequence into runs at every restart at Art. 1: + a consolidated disciplinare appends one sub-disciplinare per + sottozona ("ALLEGATO 3 — SOTTOZONA «ALTO TIRINO»", "TITOLO II + «TRENTINO SUPERIORE»"), each numbered from Art. 1 again. Each entry + is (num, header_start, header_end, body_end).""" + runs: list[list[tuple[int, int, int, int]]] = [] + cur: list[tuple[int, int, int, int]] = [] + for i, (num, hs, he) in enumerate(heads): + end = heads[i + 1][1] if i + 1 < len(heads) else len(text) + if cur and num == 1: + runs.append(cur) + cur = [] + cur.append((num, hs, he, end)) + if cur: + runs.append(cur) + return runs + + +def _is_toc(run: list[tuple[int, int, int, int]]) -> bool: + """A table of contents: every "body" between consecutive headers is a + title line, never article text. (When every run looks like this the + caller keeps them all — a two-header fixture is not a TOC.)""" + return all(end - he < TOC_BODY_MAX_CHARS for _num, _hs, he, end in run) + + +def _bodies(text: str, run: list[tuple[int, int, int, int]]) -> dict[int, str]: + """{article_num: body} for one run. A number repeated inside a run (a + cross-reference line that happens to start with "Art. 5", an OCR + duplicate) keeps the occurrence with the longest body.""" + bodies: dict[int, str] = {} + for num, _hs, he, end in run: + body = text[he:end].strip() + if len(body) > len(bodies.get(num, "")): + bodies[num] = body + return bodies + + +_ANNEX_HEADING_RE = re.compile(r"(?i)\b(allegato|sottozona|titolo\s+[ivx]+)\b") + + +def _annex_title(text: str, run_start: int) -> str: + """The heading of an annex — "ALLEGATO 3 «MONTEPULCIANO D'ABRUZZO» + SOTTOZONA «ALTO TIRINO»" — read from the non-blank lines just before + its Art. 1 header (they sit at the tail of the previous run's last + article body). Starts at the first line naming an allegato / + sottozona / titolo when one is in view; page-number lines dropped.""" + lines = [ln.strip() for ln in text[max(0, run_start - 600):run_start].splitlines() if ln.strip()] + lines = [ln for ln in lines if not ln.isdigit()] + tail = lines[-10:] + for i, ln in enumerate(tail): + if _ANNEX_HEADING_RE.match(ln): + return " ".join(tail[i:]) + for i, ln in enumerate(tail[-6:]): + if _ANNEX_HEADING_RE.search(ln) and ln[:1].isupper(): + return " ".join(tail[-6:][i:]) + return " ".join(tail[-ANNEX_TITLE_LINES:]) + + +_DISCIPLINARE_ART1_RE = re.compile(r"(?i)\briservat[aoei]\b") + + +def _main_run_index(text: str, runs: list[list[tuple[int, int, int, int]]]) -> int: + """The run holding the disciplinare proper: the first whose Art. 1 + reads like one ("La denominazione … è riservata ai vini …"). A + ministerial decree bound in front of it (Veneto IGT: five short + articles on the 2008/2009 transition, "approvato con decreto …") + restarts the numbering too but never reserves the name. Falls back + to the longest run.""" + for i, run in enumerate(runs): + art1 = next((text[he:end] for num, _hs, he, end in run if num == 1), "") + if _DISCIPLINARE_ART1_RE.search(art1[:400]): + return i + return max(range(len(runs)), key=lambda i: sum(end - he for _n, _hs, he, end in runs[i])) + + +def extract_article_runs(text: str) -> tuple[dict[int, str], list[dict]]: + """(main articles, annexes) of a MASAF disciplinare. + + `main` is `{article_num: body}` for the first substantive run — the + parent denomination's own text. `annexes` lists every later run + (`{"title", "articles"}`), in document order: the per-sottozona + sub-disciplinari whose Art. 1 / Art. 3 / Art. 9 belong to the + sottozona, not to the parent. Before 2026-09-13 `extract_articles` + kept the LAST occurrence of each article number, so a parent with + annexes took its summary, grape roster, area and terroir link from + its last sottozona (Montepulciano d'Abruzzo → San Martino sulla + Marrucina; Trentino → Valle di Cembra).""" heads = find_article_offsets(text) if not heads: - return {} + return {}, [] + all_runs = _runs(text, heads) + runs = [r for r in all_runs if not _is_toc(r)] or all_runs + first = _main_run_index(text, runs) + runs = runs[first:] + main = _bodies(text, runs[0]) + annexes = [ + {"title": _annex_title(text, run[0][1]), "articles": _bodies(text, run)} + for run in runs[1:] + ] + return main, annexes - # When the same article number appears multiple times (TOC + body), - # take the LAST occurrence — TOC lines have no body content. - last_by_num: dict[int, tuple[int, int, int]] = {} - for tup in heads: - last_by_num[tup[0]] = tup - ordered = sorted(last_by_num.values(), key=lambda t: t[1]) - bodies: dict[int, str] = {} - for i, (num, _hstart, hend) in enumerate(ordered): - end = ordered[i + 1][1] if i + 1 < len(ordered) else len(text) - body = text[hend:end] - # Trim a per-article sub-title on the first non-blank line: - # MASAF disciplinari place "Denominazione e vini" / - # "(Base ampelografica)" / etc. immediately after the header. - # Keep it — downstream consumers may want it as a salience hint. - bodies[num] = body.strip() - return bodies +def extract_articles(text: str) -> dict[int, str]: + """Carves the pdftotext output into `{article_num: body}` for the + parent's own disciplinare (the first substantive run — a table of + contents is skipped, sottozona annexes are left to + `extract_article_runs`). Body excludes the header line and stops at + the next article header (or EOF).""" + main, _annexes = extract_article_runs(text) + return main # Grape extraction from Article 2 ("Base ampelografica"). Two formats @@ -587,21 +675,28 @@ def parse_grapes_with(matcher, article2_body: str, wine_name: str = "") -> dict: "details": [], } name_key = _loose_key(wine_name) - seen: set[str] = set() - hits: list[tuple] = [] # (MatchResult, from_wine_name) + hits: list = [] # first MatchResult per slug, in order + from_name: dict[str, bool] = {} # slug → matched ONLY from phrases restating the wine name for phrase in article2_candidate_phrases(article2_body): hit = matcher(phrase) - if hit is None or hit.slug in seen: + if hit is None: continue if hit.method.startswith("fuzzy"): score = int(hit.method.split(":")[1]) if score < 90 or len(re.sub(r"[\W\d_]", "", phrase)) < 7: continue - seen.add(hit.slug) - hits.append((hit, bool(name_key) and _loose_key(phrase) == name_key)) - - real = [h for h, from_name in hits if not from_name] - keep = real if real else [h for h, _ in hits] + restates_name = bool(name_key) and _loose_key(phrase) == name_key + if hit.slug not in from_name: + hits.append(hit) + from_name[hit.slug] = restates_name + elif not restates_name: + # "Trebbiano d'Abruzzo" (the DOC name) resolves to the grape + # trebbiano-abruzzese before the roster's own "Trebbiano + # abruzzese" does — the later, genuine phrase must vouch for it. + from_name[hit.slug] = False + + real = [h for h in hits if not from_name[h.slug]] + keep = real if real else hits for hit in keep: out["principal"].append(hit.slug) out["details"].append({ @@ -690,10 +785,18 @@ def derive_summary(article1_body: str, max_chars: int = 600) -> str: return cut + ("." if not cut.endswith(".") else "") -def derive_geo_area(article3_body: str, max_chars: int = 4000) -> str: +def cap_at_sentence(body: str, max_chars: int | None) -> str: + """Cut `body` to at most `max_chars` at the last sentence boundary + before the cap; None (or 0) leaves it whole.""" + if max_chars and len(body) > max_chars: + return body[:max_chars].rsplit(".", 1)[0] + "." + return body + + +def derive_geo_area(article3_body: str, max_chars: int | None = 4000) -> str: """Article 3 ('Zona di produzione delle uve') body. Returned trimmed of leading sub-title noise and capped at max_chars so the panel - doesn't drown in commune lists.""" + doesn't drown in commune lists (None = uncapped).""" if not article3_body: return "" text = article3_body.strip() @@ -713,14 +816,14 @@ def derive_geo_area(article3_body: str, max_chars: int = 4000) -> str: skip_subtitle = False kept.append(s) body = "\n".join(kept).strip() - if len(body) > max_chars: - body = body[:max_chars].rsplit(".", 1)[0] + "." - return body + return cap_at_sentence(body, max_chars) -def derive_terroir(article9_body: str, max_chars: int = 4000) -> str: +def derive_terroir(article9_body: str, max_chars: int | None = 4000) -> str: """Same shape as `derive_geo_area` but for Article 9 ('Legame con - l'ambiente geografico').""" + l'ambiente geografico'). The 4,000-char default is the panel length; + the terroir-fact extractor reads the uncapped body (469 of 522 + disciplinari carry an Art. 9 longer than the cap).""" return derive_geo_area(article9_body, max_chars=max_chars) @@ -731,10 +834,11 @@ def derive_terroir(article9_body: str, max_chars: int = 4000) -> str: def pick_terroir_article( - articles: dict[int, str], raw_text: str | None = None + articles: dict[int, str], raw_text: str | None = None, max_chars: int | None = 4000 ) -> tuple[int, str]: """Return (article_number, derived_terroir_body) for the 'Legame - con l'ambiente geografico' section. + con l'ambiente geografico' section, the body capped at `max_chars` + (None = whole article). The canonical MASAF template puts it at Article 9; the older Veneto-IGT template (colli-trevigiani, conselvano, marca- @@ -759,7 +863,7 @@ def pick_terroir_article( for n in candidates: body = articles.get(n, "") if body and _LEGAME_TITLE_RE.search(body[:300]): - return n, derive_terroir(body) + return n, derive_terroir(body, max_chars=max_chars) # Step 2: raw-text fallback (handles concatenated disciplinari). if raw_text: @@ -781,7 +885,7 @@ def pick_terroir_article( end = next_m.start() if next_m else len(raw_text) body = raw_text[start:end].strip() if body and n: - return n, derive_terroir(body) + return n, derive_terroir(body, max_chars=max_chars) # Step 3: established canonical fallback. - return 9, derive_terroir(articles.get(9, "")) + return 9, derive_terroir(articles.get(9, ""), max_chars=max_chars) diff --git a/scripts/_lib/it/national_term.py b/scripts/_lib/it/national_term.py new file mode 100644 index 0000000..d378ab6 --- /dev/null +++ b/scripts/_lib/it/national_term.py @@ -0,0 +1,99 @@ +"""Italian national term (DOC / DOCG / IGT) per appellation. + +The EU register carries only the scheme (PDO / PGI). The traditional term +Italy attaches to each GI under Reg. (EU) 1308/2013 Art. 112(a) — DOC or +DOCG for a DOP, IGT for an IGP — is published by MASAF in the "Elenco +alfabetico dei vini DOP italiani" (stage 00 caches it under +raw/it/masaf-elenchi/). Each roster row carries the eAmbrosia file number, +which is the join key: the register writes it as `PDO-IT-A1896` for older +GIs and `PDO-IT-01896` for post-2023 ones, so both sides are reduced to the +bare numeric tail before joining. + +Curator pins live in national_term_overrides.json (slug-keyed, with cited +sources) for GIs the roster does not yet carry — a term registered after +the roster was compiled — and take precedence over the roster. +""" + +from __future__ import annotations + +import json +import re +import shutil +import subprocess +import sys +from functools import lru_cache +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[3] +ELENCHI_DIR = ROOT / "raw" / "it" / "masaf-elenchi" +DOP_ELENCO_PATH = ELENCHI_DIR / "elenco-dop.pdf" +OVERRIDES_PATH = Path(__file__).with_name("national_term_overrides.json") + +_ROW_RE = re.compile(r"\b(DOCG|DOC)\b\s+(PDO-IT-[A-Z]?\d+)") +_TAIL_RE = re.compile(r"([A-Za-z]*)(\d+)\s*$") + + +def file_number_tail(fn: str) -> str: + """`PDO-IT-A1896` and `PDO-IT-01896` → `1896`.""" + m = _TAIL_RE.search((fn or "").rsplit("-", 1)[-1]) + if not m: + return "" + return m.group(2).lstrip("0") or "0" + + +def parse_elenco_text(text: str) -> dict[str, str]: + """`pdftotext -layout` output of the DOP elenco → {file_number_tail: term}. + + A row's name may wrap over several lines, but the term and the file + number always sit on the same line, so the pair is the row anchor.""" + roster: dict[str, str] = {} + for term, fn in _ROW_RE.findall(text): + tail = file_number_tail(fn) + prev = roster.get(tail) + if prev and prev != term: + print(f"[national_term] WARNING {fn}: {prev} vs {term} in elenco", + file=sys.stderr) + roster[tail] = term + return roster + + +@lru_cache(maxsize=1) +def load_it_terms() -> dict[str, str]: + if not DOP_ELENCO_PATH.exists(): + print(f"[national_term] WARNING {DOP_ELENCO_PATH.relative_to(ROOT)} absent — " + "run scripts/it/00_fetch_data.py; IT DOC/DOCG terms unresolved", + file=sys.stderr) + return {} + if not shutil.which("pdftotext"): + print("[national_term] WARNING pdftotext not on PATH; IT DOC/DOCG terms unresolved", + file=sys.stderr) + return {} + out = subprocess.run( + ["pdftotext", "-layout", str(DOP_ELENCO_PATH), "-"], + capture_output=True, text=True, check=True, + ).stdout + roster = parse_elenco_text(out) + n_docg = sum(1 for t in roster.values() if t == "DOCG") + print(f"[national_term] elenco DOP: {len(roster)} rows " + f"({n_docg} DOCG / {len(roster) - n_docg} DOC)", file=sys.stderr) + return roster + + +@lru_cache(maxsize=1) +def load_it_term_overrides() -> dict[str, dict]: + if not OVERRIDES_PATH.exists(): + return {} + return json.loads(OVERRIDES_PATH.read_text(encoding="utf-8")) + + +def it_term_for(record: dict, kind: str) -> str: + kind = (kind or "").upper() + if kind == "IGP": + return "IGT" + if kind != "DOP": + return "" + overrides = load_it_term_overrides() + for slug in (record.get("slug"), record.get("parent_slug")): + if slug and slug in overrides: + return overrides[slug].get("term", "") + return load_it_terms().get(file_number_tail(record.get("file_number") or ""), "") diff --git a/scripts/_lib/it/national_term_overrides.json b/scripts/_lib/it/national_term_overrides.json new file mode 100644 index 0000000..43dec9f --- /dev/null +++ b/scripts/_lib/it/national_term_overrides.json @@ -0,0 +1,51 @@ +{ + "casauria": { + "term": "DOCG", + "file_number": "PDO-IT-02972", + "sources": [ + { + "label": "Commission Implementing Regulation (EU) 2025/2261 of 4 November 2025 on the registration of the geographical indication Casauria (PDO)", + "url": "https://eur-lex.europa.eu/eli/reg_impl/2025/2261/oj" + }, + { + "label": "MASAF — Domande di riconoscimento vini DOP e IGP e modifica disciplinari 2025 (Casauria DOP (DOCG))", + "url": "https://www.masaf.gov.it/flex/cm/pages/ServeBLOB.php/L/IT/IDPagina/22762" + } + ], + "note": "Row 73 of the MASAF elenco agg. 18.03.2026 already says DOCG; pinned so the term survives a roster rotation that predates the 2025 registration." + }, + "ciro-classico": { + "term": "DOCG", + "file_number": "PDO-IT-03209", + "sources": [ + { + "label": "Commission Implementing Regulation (EU) 2025/1518 of 18 July 2025 on the registration of the geographical indication Cirò Classico (PDO)", + "url": "https://eur-lex.europa.eu/eli/reg_impl/2025/1518/oj" + }, + { + "label": "MASAF — Domande di riconoscimento vini DOP e IGP e modifica disciplinari 2025 (Cirò Classico DOP (DOCG))", + "url": "https://www.masaf.gov.it/flex/cm/pages/ServeBLOB.php/L/IT/IDPagina/22762" + } + ], + "note": "Row 97 of the MASAF elenco agg. 18.03.2026 already says DOCG; pinned so the term survives a roster rotation that predates the 2025 registration." + }, + "valtenesi": { + "term": "DOC", + "file_number": "PDO-IT-A1188", + "sources": [ + { + "label": "Commission Implementing Regulation (EU) 2026/572 (OJ L, 2026/572, 18.3.2026) — registration of Valtènesi (PDO); eAmbrosia euProtectionDate 2026-03-18", + "url": "http://data.europa.eu/eli/reg_impl/2026/572/oj" + }, + { + "label": "Single document Valtènesi (PDO-IT-A1188), OJ C/2025/5016, 16.9.2025", + "url": "http://data.europa.eu/eli/C/2025/5016/oj" + }, + { + "label": "MASAF — Elenco alfabetico dei vini DOP italiani (2014 build), row 387: Valtènesi DOP DOC PDO-IT-A1188", + "url": "https://www.masaf.gov.it/flex/files/e/a/c/D.3b34b403c9bb8667ee4b/Elenco_alfabetico_Vini_DOP_italiani.pdf" + } + ], + "note": "Registered 2026-03-18 — the same day the current MASAF elenco (agg. 18.03.2026) was compiled, which omits it. The disciplinare consolidated by DM 30.11.2011 (MASAF bundle) is the DOC 'Valtènesi'." + } +} diff --git a/scripts/_lib/map_template.py b/scripts/_lib/map_template.py index 9489a70..13e3af5 100644 --- a/scripts/_lib/map_template.py +++ b/scripts/_lib/map_template.py @@ -19,6 +19,8 @@ from collections.abc import Callable from pathlib import Path +from babel.numbers import format_decimal + from _lib.content_block import RenderCtx, esc, render_content_block from _lib.env import carto_basemap_key from _lib.i18n import load_translations @@ -36,12 +38,16 @@ def build_style_labels(_: Callable[[str], str]) -> dict[str, str]: def build_labels(_: Callable[[str], str]) -> dict[str, str]: """All translatable UI strings for the map. msgid is the French source.""" return { - "page_title": _("Open Wine Map — carte des appellations"), + # Title and description are the share-card and SERP surface: a reader + # meeting the site for the first time. The scheme/term acronyms belong + # on the browse page, which actually lists them; here they read as a + # spec sheet. "Appellation", not "region" — region is the facet one + # level up (bassin / regione / Bundesland). + "page_title": _("Open Wine Map — appellations viticoles d'Europe"), "subtitle": _("carte des appellations viticoles"), "meta_description": _( - "Carte interactive des appellations viticoles européennes " - "(AOC, AOP, IGP, DOP) : cépages, styles et terroir, d'après " - "les registres officiels (INAO, EUR-Lex)." + "Carte interactive des appellations viticoles d'Europe : cépages, " + "styles et terroir, d'après les registres officiels." ), "loading": _("Chargement…"), "search_h": _("Recherche"), @@ -70,8 +76,16 @@ def build_labels(_: Callable[[str], str]) -> dict[str, str]: "facet_regions_h": _("Région"), "facet_appellations_h": _("Appellation"), "facet_kind_h": _("Type"), - "kind_aoc": _("AOC / AOP"), - "kind_igp": _("IGP"), + "legend_origin": _("Origine protégée (AOP, AOC, DOC, DO…)"), + "legend_gi": _("Indication géographique (IGP, IGT, Landwein…)"), + "scheme_pdo": _("AOP"), + "scheme_pgi": _("IGP"), + "scheme_spirit_gi": _("IG spiritueux"), + "facet_appellation_type_h": _("Type d'appellation"), + "facet_scheme_uk_pdo": _("AOP (régime britannique)"), + "facet_scheme_uk_pgi": _("IGP (régime britannique)"), + "facet_scheme_none": _("AOC (Suisse)"), + "gi_term_source_label": _("Source"), "view_mode_h": _("Vue"), "view_mode_simple": _("Simple"), "view_mode_advanced": _("Avancée"), @@ -90,7 +104,7 @@ def build_labels(_: Callable[[str], str]) -> dict[str, str]: "reset": _("Réinitialiser"), "count_total": _("{n} appellations"), "count_filtered": _("{n} / {total} appellations"), - "count_hidden_igp_hint": _("{n} dans IGP masquées — afficher"), + "count_hidden_igp_hint": _("{n} masquées dans les IGP · afficher"), "close_aria": _("Fermer"), "panel_aria": _("Détails de l'appellation"), "remove_filter_aria": _("Retirer le filtre {label}"), @@ -226,20 +240,32 @@ def build_labels(_: Callable[[str], str]) -> dict[str, str]: "about_link_label": _("À propos"), "about_h": _("À propos d'Open Wine Map"), "about_lead_html": _( - "Carte de référence des appellations viticoles " - "(AOC, AOP, IGP, DOP), générée automatiquement à partir des " - "données publiques." + "Carte de référence des appellations viticoles, générée automatiquement à " + "partir des registres publics : le registre de l'Union européenne des AOP et " + "IGP, les régulateurs nationaux, le répertoire fédéral suisse des AOC " + "cantonales et le registre britannique des indications géographiques." + ), + "about_llm_html": _( + "Une partie du texte est produite par des modèles de langage : les repères de " + "terroir sont dégagés du texte du régulateur et traduits par Claude (Sonnet " + "4.6) ; les extraits Wikipedia des infobulles de cépages et de styles sont " + "traduits pour l'essentiel par Mistral Small 3.2, exécuté localement, et pour " + "quelques-uns par Claude ; les résumés des cahiers des charges ont été " + "traduits par un traducteur humain, à l'exception d'un petit reliquat traduit " + "automatiquement. Chaque élément porte sa propre ligne d'attribution dans le " + "panneau." ), "about_made_by_html": _("Réalisé avec ♡ par {devloed}."), "about_data_html": _( - "Sources : INAO ({inao}) pour les cahiers des charges et les " - "aires parcellaires, IGN ({ign}) pour le fond cartographique, " - "Wikipedia ({wikipedia}) pour quelques compléments narratifs " - "(CC BY-SA 4.0), VIVC ({vivc}) — Vitis International Variety " - "Catalogue, Julius Kühn-Institut — pour les noms canoniques " - "et numéros de cépage (citation Röckel et al.). Tout extrait " - "Wikipedia est signalé sur place. Détails et licences dans le " - "{readme}." + "Sources : INAO ({inao}) pour les cahiers des charges et les aires " + "parcellaires françaises, IGN ({ign}) pour les contours des communes " + "françaises, le registre des indications géographiques de l'UE, les " + "régulateurs nationaux, Eurostat GISCO et Bétard 2022 pour les autres pays, " + "OpenStreetMap et CARTO pour le fond de carte, Wikipedia ({wikipedia}) pour " + "quelques compléments narratifs (CC BY-SA 4.0), et VIVC ({vivc}), le Vitis " + "International Variety Catalogue du Julius Kühn-Institut, pour les noms " + "canoniques et numéros de cépage (citation Röckel et al.). Tout extrait " + "Wikipedia est signalé sur place. Détails et licences dans le {readme}." ), "about_contrib_html": _("Suggestions et pull requests bienvenues sur {github}."), "feedback_issue_label": _("ticket GitHub"), @@ -250,19 +276,17 @@ def build_labels(_: Callable[[str], str]) -> dict[str, str]: "Signalez-les via {issue} ou {email}." ), "about_roadmap_html": _( - "20 pays européens cartographiés : France, Espagne, Portugal, " - "Italie, Autriche, Allemagne, Suisse, Slovénie, Croatie, " - "Hongrie, Roumanie, Bulgarie, Grèce, Slovaquie, Tchéquie, " - "Luxembourg, Belgique, Pays-Bas, Malte et Chypre. Des " - "itérations supplémentaires viendront affiner la qualité des " - "données. La couverture sera étendue au-delà de l'UE et de " - "la Suisse, ainsi qu'aux classifications hors AOP." + "{c} pays européens cartographiés ({n} appellations : {parents} appellations " + "et {subs} dénominations rattachées). Des itérations supplémentaires " + "affineront la qualité des données et étendront la couverture au-delà de " + "l'UE, de la Suisse et du Royaume-Uni." ), "browse_all_label": _("Toutes les appellations"), "browse_title": _("Toutes les appellations viticoles — Open Wine Map"), "browse_meta_description": _( - "Liste des {n} appellations viticoles européennes cartographiées " - "sur Open Wine Map, classées par pays — AOC, AOP, IGP, DOP." + "Liste des {n} appellations viticoles cartographiées sur Open Wine Map, " + "classées par pays : AOP et IGP de l'UE avec leurs termes traditionnels (AOC, " + "DOCG, DOQ…), AOC suisses et IG britanniques." ), "browse_intro_html": _( "Les {n} appellations ci-dessous sont classées par pays. " @@ -460,13 +484,16 @@ def _ext_link(url: str, label: str) -> str: def _feedback_email_anchor(label: str) -> str: return ( - f'<a href="#" class="feedback-mail" ' + f'<a href="#" class="feedback-mail" data-feedback="email" ' f'data-u="{_FEEDBACK_USER}" data-d="{_FEEDBACK_DOMAIN}">{label}</a>' ) def _build_sidebar_disclaimer(labels: dict[str, str]) -> str: - issue = _ext_link(_GITHUB_NEW_ISSUE_URL, labels["feedback_issue_label"]) + issue = ( + f'<a href="{_GITHUB_NEW_ISSUE_URL}" target="_blank" rel="noopener" ' + f'data-feedback="github">{labels["feedback_issue_label"]}</a>' + ) email = _feedback_email_anchor(labels["feedback_email_label"]) return ( f'<div id="sidebar-disclaimer">' @@ -476,7 +503,12 @@ def _build_sidebar_disclaimer(labels: dict[str, str]) -> str: def _build_about_dialog( - labels: dict[str, str], *, browse_path: str = "", data_updated_html: str = "" + labels: dict[str, str], + *, + browse_path: str = "", + data_updated_html: str = "", + corpus_counts: dict | None = None, + locale: str = "en", ) -> str: devloed = _ext_link(_DEVLOED_URL, "devloed.com") github = _ext_link(_GITHUB_URL, "GitHub") @@ -485,18 +517,25 @@ def _build_about_dialog( wikipedia = _ext_link(_WIKIPEDIA_URL, "fr.wikipedia.org") vivc = _ext_link(_VIVC_URL, "vivc.de") readme = _ext_link(_GITHUB_URL + "#public-data-sources", "README") + counts = { + k: format_decimal(v, locale=locale) for k, v in (corpus_counts or {}).items() + } + roadmap = labels["about_roadmap_html"] + if counts: + roadmap = roadmap.format(**counts) paragraphs = [ labels["about_lead_html"], - labels["about_made_by_html"].format(devloed=devloed), labels["about_data_html"].format( inao=inao, ign=ign, wikipedia=wikipedia, vivc=vivc, readme=readme ), - labels["about_roadmap_html"], + labels["about_llm_html"], + roadmap, labels["about_contrib_html"].format(github=github), ] if browse_path: browse_link = f'<a href="{browse_path}">{esc(labels["browse_all_label"])}</a>' paragraphs.append(labels["about_browse_html"].format(browse_link=browse_link)) + paragraphs.append(labels["about_made_by_html"].format(devloed=devloed)) body = "\n ".join(f"<p>{p}</p>" for p in paragraphs) if data_updated_html: body += f'\n <p class="data-updated">{data_updated_html}</p>' @@ -1101,7 +1140,7 @@ def _build_entity_meta( near-duplicate-of-parent body is ever server-exposed, so the page simply drops out of the index cleanly.)""" name = rec.get("name") or slug - kind = rec.get("kind") or "" + kind = rec.get("class_label") or rec.get("kind") or "" region = region_labels.get(rec.get("region") or "", rec.get("region") or "") country = country_labels.get(rec.get("country") or "", "") self_url = f"{_SITE_BASE_URL}{_entity_path(locale, slug)}" @@ -1248,7 +1287,7 @@ def _render_browse_page(*, locale, labels, country_labels, aocs, index_slugs) -> continue cc = rec.get("country") or "" by_country.setdefault(cc, []).append( - (rec.get("name") or slug, slug, rec.get("kind") or "") + (rec.get("name") or slug, slug, rec.get("class_label") or "") ) def _country_name(cc: str) -> str: @@ -1326,6 +1365,9 @@ def _country_name(cc: str) -> str: # maps.py imports this to emit the complement as the per-slug panel JSON. STARTUP_AOCS_FIELDS = frozenset({ "name", "name_latin", "kind", "region", "country", "is_wine", + # Two naming axes (see _lib/gi_terms.py): read by the panel meta line, + # docTitleFor (pre-hydration) and the appellation-type facet. + "eu_scheme", "national_term", "class_key", "class_label", "styles", "styles_simple", "classifications", "grapes_principal", "grapes_accessory", "grapes_all", "bbox", "bbox_villages", "geom_source", @@ -1350,6 +1392,11 @@ def render( facet_class_tree: list[dict], class_descendants: dict[str, list[str]], facet_regions: list[tuple[str, int]], + facet_term_tree: list[dict] | None = None, + term_descendants: dict[str, list[str]] | None = None, + term_display: dict[str, tuple[str, str]] | None = None, + terms_info: dict | None = None, + corpus_counts: dict | None = None, locale: str = "fr", grapes_info: dict | None = None, styles_info: dict | None = None, @@ -1570,6 +1617,16 @@ def _merged_facet(field: str) -> list[tuple[str, int]]: # one shared, content-hashed stylesheet for the whole corpus. So each of the # ~11.6k pages is a small shell referencing three cached bundles # (data + app + style) instead of inlining ~450 KB. + term_labels = { + "pdo": labels["scheme_pdo"], + "pgi": labels["scheme_pgi"], + "spirit-gi": labels["scheme_spirit_gi"], + "uk-pdo": labels["facet_scheme_uk_pdo"], + "uk-pgi": labels["facet_scheme_uk_pgi"], + "none": labels["facet_scheme_none"], + } + for tk, (cc, term) in (term_display or {}).items(): + term_labels[tk] = f"{_COUNTRY_FLAG_EMOJI.get(cc, '')} {term}".strip() script_kwargs = dict( lang_attr=locale, github_new_issue_url=_GITHUB_NEW_ISSUE_URL, @@ -1586,6 +1643,10 @@ def _merged_facet(field: str) -> list[tuple[str, int]]: accessory_json=json.dumps(facet_accessory_merged, ensure_ascii=False), grapes_all_json=json.dumps(facet_grapes_all_merged, ensure_ascii=False), regions_json=json.dumps(facet_regions, ensure_ascii=False), + term_tree_json=json.dumps(facet_term_tree or [], ensure_ascii=False), + term_descendants_json=json.dumps(term_descendants or {}, ensure_ascii=False), + term_labels_json=json.dumps(term_labels, ensure_ascii=False), + terms_info_json=json.dumps(terms_info or {}, ensure_ascii=False), style_labels_json=json.dumps(style_labels, ensure_ascii=False), simple_style_labels_json=json.dumps(simple_style_labels, ensure_ascii=False), simple_style_buckets_json=json.dumps(simple_style_buckets, ensure_ascii=False), @@ -1640,6 +1701,8 @@ def _fill(**per_page) -> str: home_about_html = _build_about_dialog( labels, browse_path=browse_path, + corpus_counts=corpus_counts, + locale=locale, data_updated_html=( labels["about_updated_html"].format(date=esc(build_date)) if build_date else "" ), @@ -1672,8 +1735,11 @@ def _fill(**per_page) -> str: country_labels=country_labels, country_flag_emoji=_COUNTRY_FLAG_EMOJI, grapes_info=grapes_info or {}, styles_info=styles_info or {}, style_labels=style_labels, github_new_issue_url=_GITHUB_NEW_ISSUE_URL, + terms_info=terms_info or {}, + ) + entity_about_html = _build_about_dialog( + labels, browse_path=browse_path, corpus_counts=corpus_counts, locale=locale ) - entity_about_html = _build_about_dialog(labels, browse_path=browse_path) def _emit(slug: str, meta: dict, ssr: str, has_card: bool) -> None: page = _fill( @@ -1704,7 +1770,7 @@ def _emit(slug: str, meta: dict, ssr: str, has_card: bool) -> None: continue kids = [ {"name": aocs[k].get("name") or k, "path": _entity_path(locale, k), - "kind": aocs[k].get("kind") or ""} + "classification": aocs[k].get("class_label") or ""} for k in (children_map or {}).get(slug, []) if k in aocs ] @@ -2035,6 +2101,9 @@ def _emit(slug: str, meta: dict, ssr: str, has_card: bool) -> None: #panel .body h2 {{ font-size:13px; text-transform:uppercase; letter-spacing:0.04em; color:#934050; margin:18px 0 6px }} #panel .body p {{ margin:0 0 8px }} #panel .meta {{ color:#666; font-size:12px; margin-bottom:8px }} + #panel .meta .gi-scheme {{ opacity:.78; white-space:nowrap }} + .gi-term.has-info, .gi-scheme.has-info {{ cursor:help; text-decoration:underline dotted; text-underline-offset:2px }} + .gi-term abbr, .gi-scheme abbr {{ text-decoration:inherit }} #panel .meta .meta-country {{ display:inline-flex; align-items:center; gap:5px; color:#444 }} #panel .meta .country-flag {{ font-size:13px; line-height:1 }} #panel .meta .country-name {{ font-weight:600 }} @@ -2362,8 +2431,8 @@ def _split_template(t: str) -> tuple[str, str]: <summary>{labels[legend_h]}</summary> <div class="legend-body"> <div class="legend-h">{labels[legend_bassin_h]}</div> - <div class="swatch-row"><span class="sw aoc"></span><span>{labels[kind_aoc]}</span></div> - <div class="swatch-row"><span class="sw igp"></span><span>{labels[kind_igp]}</span></div> + <div class="swatch-row"><span class="sw aoc"></span><span>{labels[legend_origin]}</span></div> + <div class="swatch-row"><span class="sw igp"></span><span>{labels[legend_gi]}</span></div> <div class="hint">{labels[legend_area_hint]}</div> <div class="legend-h">{labels[legend_grapes_h]}</div> <div class="swatch-row"><span class="sw principal"></span><span>{labels[legend_principal]}</span></div> @@ -2411,6 +2480,11 @@ def _split_template(t: str) -> tuple[str, str]: <div class="facet" id="facet-classification"></div> </details> + <details data-modes="advanced" data-facet="appellation-type"> + <summary><span class="facet-label">{labels[facet_appellation_type_h]}</span><span class="facet-badge"></span></summary> + <div class="facet" id="facet-appellation-type"></div> + </details> + <details open data-facet="grapes"> <summary><span class="facet-label">{labels[facet_grapes_h]}</span><span class="facet-badge"></span></summary> <div class="omni-subopt grape-subopt"> diff --git a/scripts/_lib/prompt_cache.py b/scripts/_lib/prompt_cache.py new file mode 100644 index 0000000..87e5a69 --- /dev/null +++ b/scripts/_lib/prompt_cache.py @@ -0,0 +1,102 @@ +"""Prompt caching for the LLM stages (Anthropic only; a no-op elsewhere). + +Where the same text is sent more than once, it is placed FIRST in the +system prompt as its own block carrying `cache_control`, so the second +request onwards reads it from the cache at 0.1× the input price instead +of paying for it again: + +- stage 02d (20 non-FR scripts): the four sub-section calls of a record + each resent the whole lien — Barolo's uncapped Art. 9 alone was + 119 K input tokens per record. The lien is now the leading cached + block; the per-sub-section instructions (which vary) follow it and + the user turn carries only the sub-section request. FR is left alone: + it slices section X per sub-section, so its four calls share nothing. +- the gate, the back-check, the LLM audit and the 21 × 02e scripts: the + system prompt is the same for every record of a batch (per locale for + 02e) — one cached block, read by every request after the first. + +A cache prefix must be byte-identical and long enough (Sonnet 5 / 4.6: +1,024 tokens; Opus 5: 512) — a shorter block silently does not cache and +costs nothing extra. The Batch API processes requests concurrently, so +hits are best-effort (Anthropic quotes 30–98 %); the four calls of a +record are submitted adjacently, and the ledger's cache_read / +cache_creation tokens show the achieved rate per batch. The TTL is +`OWM_CACHE_TTL`: "5m" (default — write premium 1.25×, the four-call +pattern breaks even at a 29 % hit rate), "1h" (write 2×, break-even +70 %, for batches whose requests may sit more than 5 minutes apart), or +"off". +""" + +from __future__ import annotations + +import os + +TTL_ENV = "OWM_CACHE_TTL" + + +def ttl() -> str: + return (os.environ.get(TTL_ENV) or "5m").strip().lower() + + +def cache_control() -> dict | None: + """The `cache_control` value for a cached block, or None when caching is off.""" + t = ttl() + if t in ("off", "0", "none", "false"): + return None + if t == "1h": + return {"type": "ephemeral", "ttl": "1h"} + return {"type": "ephemeral"} + + +def lien_cache_control() -> dict | None: + """The `cache_control` for a block shared by a record's phased requests + (a 02d lien): the 1-hour TTL when the batch runner submits one batch + per phase (`batch.phased()` — the later phases read what the first + wrote, a read refreshes the timer), else the default TTL — inside one + concurrent batch the 2× write would be a loss.""" + cc = cache_control() + if cc is None: + return None + if (os.environ.get("OWM_BATCH_PHASED") or "1").strip().lower() not in ("0", "off", "false", "no"): + return {"type": "ephemeral", "ttl": "1h"} + return cc + + +def cached_system(shared: str, rest: str, *, phased: bool = False) -> list[dict] | str: + """System prompt as [shared block (cached), rest]; plain text when + caching is off or the shared part is empty. `phased=True` marks the + block a record shares across phased batch requests (lien_cache_control).""" + cc = lien_cache_control() if phased else cache_control() + shared = (shared or "").strip() + if not shared: + return rest + if cc is None: + return f"{shared}\n\n{rest}" + return [ + {"type": "text", "text": shared, "cache_control": cc}, + {"type": "text", "text": rest}, + ] + + +def mark_cached(system: str) -> list[dict] | str: + """A system prompt shared by every request of a batch, as one cached block.""" + cc = cache_control() + if cc is None or not system: + return system + return [{"type": "text", "text": system, "cache_control": cc}] + + +def system_text(system) -> str: + """The plain text of a system prompt, whatever its shape — for + providers without caching (Mistral, Ollama), the manual round-trip + files and the batch request hash.""" + if isinstance(system, str): + return system + return "\n\n".join(b.get("text", "") for b in system if isinstance(b, dict)) + + +def split_user_lead(template: str, **fields) -> tuple[str, str]: + """A "Sub-section: {label}\\n\\nText of …:\\n\\n{lien}" template → + (the request line, the document block), each formatted.""" + ask, _, doc = template.partition("\n\n") + return ask.format(**fields), doc.format(**fields) diff --git a/scripts/_lib/providers.py b/scripts/_lib/providers.py index f9a8b66..aae1fcc 100644 --- a/scripts/_lib/providers.py +++ b/scripts/_lib/providers.py @@ -24,7 +24,52 @@ import requests +from _lib.prompt_cache import system_text + DEFAULT_ANTHROPIC_MODEL = "claude-sonnet-4-6" + +# Per-stage Anthropic defaults — the configuration decided 2026-09-14 after +# the paired experiments in docs/review-terroir-facts-2026-09-12.md: +# 02d extraction Sonnet 5, thinking off — equal per-bullet reliability to +# Sonnet 4.6, 31 % more grounded facts, a third cheaper per +# token; the stages' max_tokens were sized for text only. +# gate Opus 5 with adaptive thinking — the independent Opus-5 +# grader caught residuals a Sonnet gate had passed. +# llm-audit Opus 5, adaptive (unchanged). +# 02e / back-check stay on Sonnet 4.6 (translation was not re-tested). +# `thinking` is None (send nothing), "disabled" or "adaptive"; the batch +# library and AnthropicProvider both honour it. OWM_BATCH_THINKING overrides. +STAGE_DEFAULTS: dict[str, tuple[str, str | None]] = { + "02d": ("claude-sonnet-5", "disabled"), + "gate": ("claude-opus-5", "adaptive"), + "audit": ("claude-opus-5", "adaptive"), + "02e": ("claude-sonnet-4-6", None), + "backcheck": ("claude-sonnet-4-6", None), + "02c": ("claude-sonnet-4-6", None), +} + + +_CLAUDE5_PREFIXES = ("claude-sonnet-5", "claude-opus-5", "claude-fable-5", "claude-mythos-5") + + +def effective_thinking(model: str, thinking: str | None) -> str | None: + """The thinking mode to send. The Claude 5 family runs adaptive thinking + when the parameter is omitted (the 4.x models do not), and every stage's + max_tokens budget is sized for the JSON reply alone — on a stage that + sets no mode, a Claude 5 model therefore gets "disabled" explicitly + (2026-09-15: Sonnet 5 on 02e spent the 2,000-token budget on thinking + and 64 % of the replies came back truncated and were rejected).""" + if thinking in ("disabled", "adaptive"): + return thinking + if any((model or "").startswith(px) for px in _CLAUDE5_PREFIXES): + return "disabled" + return None + + +def stage_default(stage: str | None) -> tuple[str, str | None]: + """(model id, thinking mode) for an Anthropic stage; the generic default + for an unknown stage.""" + return STAGE_DEFAULTS.get(stage or "", (DEFAULT_ANTHROPIC_MODEL, None)) DEFAULT_MISTRAL_URL = "https://api.mistral.ai/v1/chat/completions" DEFAULT_MISTRAL_MODEL = "mistral-medium-latest" DEFAULT_OLLAMA_URL = "http://localhost:11434/api/chat" @@ -36,7 +81,7 @@ class AnthropicProvider: kind = "anthropic-api" - def __init__(self, model: str): + def __init__(self, model: str, thinking: str | None = None): try: import anthropic # type: ignore except ImportError as e: @@ -49,14 +94,21 @@ def __init__(self, model: str): raise SystemExit("error: ANTHROPIC_API_KEY environment variable is unset.") self.client = anthropic.Anthropic(api_key=api_key) self.model = model - - def chat(self, *, system: str, user: str, max_tokens: int = 1024, **_: object) -> str: - msg = self.client.messages.create( - model=self.model, - max_tokens=max_tokens, - system=system, - messages=[{"role": "user", "content": user}], - ) + self.thinking = os.environ.get("OWM_BATCH_THINKING") or thinking + + def chat(self, *, system, user: str, max_tokens: int = 1024, **_: object) -> str: + # `system` is a string or a list of text blocks (prompt_cache.cached_system / + # mark_cached) — the SDK takes both. + params = { + "model": self.model, + "max_tokens": max_tokens, + "system": system, + "messages": [{"role": "user", "content": user}], + } + mode = effective_thinking(self.model, self.thinking) + if mode: + params["thinking"] = {"type": mode} + msg = self.client.messages.create(**params) return "".join(b.text for b in msg.content if getattr(b, "type", "") == "text").strip() @@ -84,7 +136,7 @@ def chat(self, *, system: str, user: str, max_tokens: int = 1024, **_: object) - "max_tokens": max_tokens, "temperature": 0.2, "messages": [ - {"role": "system", "content": system}, + {"role": "system", "content": system_text(system)}, {"role": "user", "content": user}, ], }, @@ -107,7 +159,7 @@ def chat(self, *, system: str, user: str, num_ctx: int = 4096, **_: object) -> s json={ "model": self.model, "messages": [ - {"role": "system", "content": system}, + {"role": "system", "content": system_text(system)}, {"role": "user", "content": user}, ], "stream": False, @@ -135,12 +187,17 @@ def make_provider( model: str | None, ollama_url: str = DEFAULT_OLLAMA_URL, mistral_url: str = DEFAULT_MISTRAL_URL, + stage: str | None = None, + thinking: str | None = None, ) -> tuple[object | None, str]: """Return (provider, translator_id) from CLI args. provider is None for - manual mode (caller should run the manual-listing path).""" + manual mode (caller should run the manual-listing path). `stage` picks + the Anthropic model + thinking default (`STAGE_DEFAULTS`); an explicit + `model` / `thinking` wins.""" if provider == "anthropic": - model_id = model or DEFAULT_ANTHROPIC_MODEL - return AnthropicProvider(model_id), model_id + default_model, default_thinking = stage_default(stage) + model_id = model or default_model + return AnthropicProvider(model_id, thinking=thinking or default_thinking), model_id if provider == "mistral": model_id = model or DEFAULT_MISTRAL_MODEL return MistralProvider(model_id, url=mistral_url), model_id diff --git a/scripts/_lib/terroir_backcheck.py b/scripts/_lib/terroir_backcheck.py new file mode 100644 index 0000000..341ac9b --- /dev/null +++ b/scripts/_lib/terroir_backcheck.py @@ -0,0 +1,256 @@ +"""The translation back-check over stage-02e caches (review 2026-09-12, +R6): per (record, target locale) one model request compares every +translated bullet with its source-language bullet and returns a +corrected translation where the rendering changed the meaning — a +number or unit slip (250–290 m → 250–490 m), a dropped, added or +upgraded hedge ("weakly" → "moderately"), a name back-formed from an +adjective ("the Caiata area" from *caiatino*), a false friend or calque +from the watch-list (*generoso* → "generous", *tirage* → "disgorgement", +Burgundian *climat* → "climate", *Lehm* → "clay", *Pintes* → "Pinot"), a +common noun left in the source language, a place name that has an +established exonym. Fixes are applied deterministically (`apply_fixes`) +under the same guards as the gate's rewrites — no number that is in +neither the source bullet nor the current translation, no arrow, sane +length — and every checked bullet carries `check` ({verdict, issue[, +original]}). + +Pure functions here; I/O, batch and cache writing live in +`scripts/02e_verify_terroir_facts.py`. +""" + +from __future__ import annotations + +import json +import re + +from _lib import llm_json +from _lib.exonyms import exonym_hits +from _lib.terroir_dedupe import _numbers +from _lib.terroir_feedback import _clip, _safe +from _lib.terroir_normalize import normalize_bullet +from _lib.terroir_prompts import appellation_context + +BACKCHECK_VERSION = "backcheck-v2" +FIX_MAX_CHARS = 360 + +_LANG_NAME = { + "en": "English", "fr": "French", "es": "Spanish", "nl": "Dutch", "de": "German", "it": "Italian", + "pt": "Portuguese", "el": "Greek", "bg": "Bulgarian", "hu": "Hungarian", "cs": "Czech", "sk": "Slovak", + "sl": "Slovenian", "hr": "Croatian", "ro": "Romanian", +} + +WATCH_LIST = ( + "climat (Burgundian named site — keep 'climat', never 'climate')", + "tirage (the bottling for the second fermentation — never 'disgorgement', which is dégorgement)", + "generoso / vino generoso (a fortified wine — never 'generous')", + "Lehm (loam — not clay, which is Ton)", + "Spritzigkeit / spritzig (a light sparkle — not 'spiciness')", + "capa (Spanish: depth of colour — 'capa alta' is deep colour, never 'layer')", + "tipologia (a wine type / style, never 'typology')", + "Urgestein (crystalline basement / primary rock, never 'primeval rock')", + "Pintes (a Hungarian grape variety, never 'Pinot')", + "Немски / Рейнски ризлинг (Riesling) versus Италиански ризлинг (Welschriesling) — never swap them", + "Rodopi / Родопи → Rhodopes (EN, FR) / Rhodopen (NL) / Ródope (ES); Stara Planina stays 'Stara Planina' (a one-time gloss '(Balkan Mountains)' is fine); Bayern → Bavaria / Baviera / Bavière / Beieren", + "Dutch common noun is 'appellatie(s)', not the French 'appellation(s)' (only the registered term 'appellation d'origine contrôlée / protégée' stays French)", + "a river valley in Dutch is 'de X-vallei' or 'het X-dal' (Marnevallei, het Marnedal) — a blend such as 'Marnedallei' or 'Audedallei' is not a word and must be fixed", +) + +SYSTEM = """You are the back-checker for Open Wine Map's machine-translated terroir facts. Each record's bullets were extracted in the source language from a wine regulator's specification and then translated into the target language. You receive both versions side by side; the SOURCE bullet is authoritative. Your job is to find every place where the translation changed the meaning, and to give the corrected translation. + +Flag a bullet ("fix") when the translation: +- changes, drops or adds a number, a unit, a date, a range bound, a direction (north/south, above/below), a comparative; +- drops, adds or upgrades a hedge or a quantifier ("mainly" → nothing; "weakly" → "moderately"; "often" → "always"; "some" → "all"); +- names the wrong entity: a grape, place, formation, wind, institution or person rendered as a different one, a name back-formed from an adjective ("the Caiata area" for caiatino, which means "of Caiazzo"), an appellation name translated instead of kept; +- uses a false friend or calque from this watch-list, or otherwise gives a wine term a wrong sense: {watch_list}; +- leaves a common noun (a generic soil, rock, climate, harvest or wine-law term) untranslated in the source language, or keeps a source-form geographic name that has an established target-language exonym (named formations, named winds, appellation names and grape names stay verbatim — those are correct); +- garbles grammar so the sentence no longer says what the source says. +Do NOT flag: a legitimate simplification; a different but equivalent number format; a synonym in register; a sentence that is merely awkward but faithful. When in doubt, "ok". + +For a "fix", give "fix": the full corrected bullet in the target language, faithful to the SOURCE bullet, one complete sentence ending with a period, without adding content. Keep the source's proper nouns exactly; transliterate Greek and Cyrillic to the EU-official Latin form. A "fix" verdict MUST carry a non-empty "fix" — if you cannot write the corrected sentence, answer "ok" and put your concern in "issue". + +Answer ONLY with JSON, no text before or after: +{"facts": [{"i": 0, "verdict": "ok|fix", "issue": "short reason, or empty", "fix": ""}, ...]} +One object per bullet, in order, with "i" equal to the bullet's index.""" + + +def system_prompt() -> str: + return SYSTEM.replace("{watch_list}", "; ".join(WATCH_LIST)) + + +def _constraints_block(fb: dict | None) -> str: + if not fb: + return "" + lines: list[str] = [] + for c in fb.get("do_not_claim") or []: + if (c.get("stage") or "extraction") not in ("translation", "both"): + continue + claim = _clip(c.get("claim_en") or c.get("claim_src") or "", 220) + why = _clip(c.get("why") or "", 240) + if claim: + lines.append(f"- A previous review verified this rendering as misleading: «{claim}» — {why}") + return ("PRIOR-REVIEW NOTES FOR THIS RECORD\n" + "\n".join(lines) + "\n\n") if lines else "" + + +def _appellation_note(slug: str | None) -> str: + """For a record whose bullets are shown on sub-denomination pages, the + translation was asked to name the appellation where the source says + "the appellation" (terroir_prompts.appellation_context): tell the + checker so it does not revert that as an added entity.""" + if not slug: + return "" + ctx = appellation_context(slug, for_translation=True) + if not ctx: + return "" + return (ctx.replace("CONTEXT:", "CONTEXT FOR THE CHECK:", 1) + + " Naming the appellation where the SOURCE only says \"the appellation / denomination / zone\" " + "is therefore intended — do not flag it as an added or wrong entity.\n\n") + + +def build_user_message( + *, name: str, source_lang: str, target_lang: str, source_facts: list[dict], + translated: list[dict], feedback: dict | None, gi_forms: frozenset[str] = frozenset(), + slug: str | None = None, +) -> str: + src_name = _LANG_NAME.get(source_lang, source_lang) + tgt_name = _LANG_NAME.get(target_lang, target_lang) + lines = [] + for i, (sf, tf) in enumerate(zip(source_facts, translated)): + hits = exonym_hits(tf.get("bullet") or "", target_lang, gi_forms=gi_forms) + flag = f"\nDETECTOR: source-form place name(s) still present: {', '.join(hits)}" if hits else "" + lines.append( + f"#{i} [{sf.get('subsection') or ''}]\n" + f"SOURCE ({src_name}): {sf.get('bullet') or ''}\n" + f"TRANSLATION ({tgt_name}): {tf.get('bullet') or ''}{flag}" + ) + return _safe( + f"RECORD: {name} — {src_name} → {tgt_name}; {len(translated)} bullets.\n\n" + f"{_constraints_block(feedback)}" + f"{_appellation_note(slug)}" + "BULLETS\n" + "\n\n".join(lines) + ) + + +# ─────────────────────────────────────────────────────────── parsing ── + +_ROW_SPLIT_RE = re.compile(r"\}\s*,\s*\{") +_I_RE = re.compile(r'"i"\s*:\s*(\d+)') +_VERDICT_RE = re.compile(r'"verdict"\s*:\s*"([A-Za-z]+)"') +_ISSUE_RE = re.compile(r'"issue"\s*:\s*"(.*?)"\s*,\s*"fix"', re.S) +_FIX_RE = re.compile(r'"fix"\s*:\s*"(.*?)"\s*(?:,\s*"|\}|$)', re.S) + + +def _recover_rows(s: str) -> list[dict]: + start = s.find('"facts"') + body = s[start:] if start >= 0 else s + rows: list[dict] = [] + for chunk in _ROW_SPLIT_RE.split(body): + m_i, m_v = _I_RE.search(chunk), _VERDICT_RE.search(chunk) + if not (m_i and m_v): + continue + m_is, m_f = _ISSUE_RE.search(chunk), _FIX_RE.search(chunk) + rows.append({ + "i": int(m_i.group(1)), "verdict": m_v.group(1), + "issue": m_is.group(1).replace('\\"', '"') if m_is else "", + "fix": m_f.group(1).replace('\\"', '"') if m_f else "", + }) + return rows + + +def parse_checks(raw: str, n: int) -> tuple[list[dict] | None, str | None]: + """A list aligned on bullet index ({verdict, issue, fix}); a bullet the + reply omits is `ok`. (None, error) for a reply with no gradable rows.""" + s = llm_json.strip_fences(raw or "") + rows = None + try: + data = json.loads(s) + rows = data.get("facts") if isinstance(data, dict) else None + except ValueError: + m = re.search(r"\{.*\}", s, re.S) + if not m: + return None, "no JSON object in reply" + try: + data = json.loads(m.group(0)) + rows = data.get("facts") if isinstance(data, dict) else None + except ValueError: + rows = _recover_rows(m.group(0)) or None + if rows is None: + return None, "unparseable JSON and no recoverable rows" + if not isinstance(rows, list): + return None, "no `facts` list" + out = [{"verdict": "ok", "issue": "", "fix": ""} for _ in range(n)] + seen = 0 + for r in rows: + if not isinstance(r, dict): + continue + try: + i = int(r.get("i")) + except (TypeError, ValueError): + continue + if not 0 <= i < n: + continue + verdict = str(r.get("verdict") or "ok").strip().lower() + out[i] = { + "verdict": "fix" if verdict == "fix" else "ok", + "issue": " ".join(str(r.get("issue") or "").split())[:300], + "fix": " ".join(str(r.get("fix") or "").split()), + } + seen += 1 + if seen == 0 and n: + return None, "reply graded none of the bullets" + return out, None + + +# ──────────────────────────────────────────────────────── application ── + + +def fix_ok(source_bullet: str, current: str, fix: str) -> str | None: + fx = (fix or "").strip() + if not fx: + return "empty" + if fx == (current or "").strip(): + return "unchanged" + if "→" in fx: + return "arrow" + if len(fx) > FIX_MAX_CHARS or len(fx) < 15: + return "length" + new_nums = _numbers(fx) - _numbers(source_bullet) - _numbers(current) + if new_nums: + return f"new numbers {sorted(new_nums)}" + return None + + +def apply_fixes( + translated: list[dict], source_facts: list[dict], checks: list[dict], *, lang: str, run: str, model: str, +) -> dict: + """Apply the back-check to a translation cache's facts (in place on + copies). Returns {facts, fixed: [{index, from, to, issue}], rejected: + [{index, reason}], missing_fixes: [{index, issue}]}.""" + out: list[dict] = [] + fixed: list[dict] = [] + rejected: list[dict] = [] + missing: list[dict] = [] + for i, (tf, sf, c) in enumerate(zip(translated, source_facts, checks)): + new = dict(tf) + check = {"verdict": c["verdict"], "issue": c.get("issue") or "", "version": BACKCHECK_VERSION, + "run": run, "model": model} + if c["verdict"] == "fix": + reason = fix_ok(sf.get("bullet") or "", tf.get("bullet") or "", c.get("fix") or "") + if reason == "empty": + # The model flagged a concern it could not phrase (512 did in + # r1): keep the translation, keep the issue, list the case. + check["verdict"] = "ok" + check["fix_missing"] = True + missing.append({"index": i, "issue": check["issue"]}) + elif reason is None: + fx = normalize_bullet(c["fix"], lang) + check["original"] = tf.get("bullet") or "" + new["bullet"] = fx + fixed.append({"index": i, "from": tf.get("bullet") or "", "to": fx, "issue": check["issue"]}) + else: + check["verdict"] = "fix-rejected" + check["rejected_reason"] = reason + check["proposed_fix"] = c.get("fix") or "" + rejected.append({"index": i, "reason": reason}) + new["check"] = check + out.append(new) + return {"facts": out, "fixed": fixed, "rejected": rejected, "missing_fixes": missing} diff --git a/scripts/_lib/terroir_backup.py b/scripts/_lib/terroir_backup.py new file mode 100644 index 0000000..82d5eaf --- /dev/null +++ b/scripts/_lib/terroir_backup.py @@ -0,0 +1,187 @@ +"""Per-run snapshots of the terroir-fact caches, so any 02d / 02e / +post-pass write can be rolled back. + +Every write to `raw/terroir-facts/<slug>.json` or +`raw/translations/terroir-facts/<lang>/<slug>.json` goes through +`terroir_cache.write_source_cache` / `write_translation_cache`, which +call `snapshot_slug(slug)` first. The first time a slug is touched in a +run, its current source cache AND all four translation caches (whatever +exists) are copied to + + raw/terroir-facts-backup/<run>/source/<slug>.json + raw/terroir-facts-backup/<run>/translations/<lang>/<slug>.json + raw/terroir-facts-backup/<run>/entries/<slug>.json (what existed) + raw/terroir-facts-backup/<run>/run.json (started_at, argv) + +Source and translations are snapshotted together because stage 04 +matches them by index and `source_facts_sha`: restoring one without the +other leaves a misaligned pair. The per-slug entry records which files +existed so `scripts/rollback_terroir_facts.py` can also delete files the +run created; one file per slug means several per-country processes can +share a run id without racing on a manifest. + +The run id comes from `OWM_TERROIR_RUN` (the orchestrator sets one id +for 02d → gate → 02e so the whole chain is one rollback unit); a script +run on its own gets a timestamp id for its process, printed on stderr. +Never on the map, never in git (raw/ is ignored). +""" + +from __future__ import annotations + +import json +import os +import shutil +import sys +import threading +from datetime import datetime, timezone +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +TERROIR = ROOT / "raw" / "terroir-facts" +TRANSLATIONS = ROOT / "raw" / "translations" / "terroir-facts" +BACKUP_ROOT = ROOT / "raw" / "terroir-facts-backup" +LANGS = ("en", "fr", "es", "nl") +RUN_ENV = "OWM_TERROIR_RUN" + +_lock = threading.Lock() +_run_id: str | None = None +_done: set[str] = set() + + +def run_id() -> str: + """The active run id: `OWM_TERROIR_RUN`, else a per-process timestamp.""" + global _run_id + if _run_id is None: + _run_id = os.environ.get(RUN_ENV) or datetime.now(timezone.utc).strftime("%Y-%m-%dT%H%M%S") + return _run_id + + +def run_dir(run: str | None = None) -> Path: + return BACKUP_ROOT / (run or run_id()) + + +def _entries_dir(run: str | None = None) -> Path: + return run_dir(run) / "entries" + + +def _run_meta_path(run: str | None = None) -> Path: + return run_dir(run) / "run.json" + + +def load_manifest(run: str | None = None) -> dict: + """{run, started_at, argv, slugs: {slug: entry}} — the per-slug entry + files aggregated (one file per slug, so concurrent per-country + processes sharing a run id never race on a single manifest).""" + run = run or run_id() + meta: dict = {"run": run, "started_at": None} + mp = _run_meta_path(run) + if mp.exists(): + try: + meta.update(json.loads(mp.read_text(encoding="utf-8"))) + except (ValueError, OSError): + pass + slugs: dict[str, dict] = {} + ed = _entries_dir(run) + if ed.exists(): + for ep in sorted(ed.glob("*.json")): + try: + slugs[ep.stem] = json.loads(ep.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + meta["slugs"] = slugs + return meta + + +def _write_entry(slug: str, entry: dict, run: str | None = None) -> None: + ep = _entries_dir(run) / f"{slug}.json" + ep.parent.mkdir(parents=True, exist_ok=True) + ep.write_text(json.dumps(entry, ensure_ascii=False, indent=1, sort_keys=True) + "\n", encoding="utf-8") + + +def _ensure_run_meta(at: str) -> None: + mp = _run_meta_path() + if mp.exists(): + return + mp.parent.mkdir(parents=True, exist_ok=True) + mp.write_text(json.dumps({"run": run_id(), "started_at": at, "argv": sys.argv[:6]}, indent=1) + "\n", + encoding="utf-8") + print(f"[terroir-backup] run {run_id()} → {run_dir().relative_to(ROOT)}", file=sys.stderr) + + +def snapshot_slug(slug: str, *, note: str = "") -> bool: + """Copy the slug's current source + translation caches into the run dir, + once per run. Returns True when a snapshot was taken now.""" + with _lock: + if slug in _done: + return False + if (_entries_dir() / f"{slug}.json").exists(): + _done.add(slug) + return False + rd = run_dir() + entry: dict = { + "at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "source": False, "translations": [], + } + src = TERROIR / f"{slug}.json" + if src.exists(): + dst = rd / "source" / f"{slug}.json" + dst.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(src, dst) + entry["source"] = True + for lang in LANGS: + tp = TRANSLATIONS / lang / f"{slug}.json" + if tp.exists(): + dst = rd / "translations" / lang / f"{slug}.json" + dst.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(tp, dst) + entry["translations"].append(lang) + if note: + entry["note"] = note + _ensure_run_meta(entry["at"]) + _write_entry(slug, entry) + _done.add(slug) + return True + + +def list_runs() -> list[tuple[str, dict]]: + """(run id, manifest) for every run dir, newest last.""" + if not BACKUP_ROOT.exists(): + return [] + out = [] + for d in sorted(BACKUP_ROOT.iterdir()): + if d.is_dir() and (d / "run.json").exists(): + out.append((d.name, load_manifest(d.name))) + return out + + +def restore_slug(slug: str, run: str, *, dry_run: bool = False) -> dict: + """Put the slug's caches back to their state before `run`: restore each + file the run snapshotted, delete each file the run created. Returns + {restored: [paths], deleted: [paths], missing: bool}.""" + m = load_manifest(run) + entry = (m.get("slugs") or {}).get(slug) + if entry is None: + return {"restored": [], "deleted": [], "missing": True} + rd = run_dir(run) + restored: list[str] = [] + deleted: list[str] = [] + + def _apply(existed: bool, backup: Path, live: Path) -> None: + rel = str(live.relative_to(ROOT)) + if existed: + if not backup.exists(): + return + if not dry_run: + live.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(backup, live) + restored.append(rel) + elif live.exists(): + if not dry_run: + live.unlink() + deleted.append(rel) + + _apply(bool(entry.get("source")), rd / "source" / f"{slug}.json", TERROIR / f"{slug}.json") + had = set(entry.get("translations") or []) + for lang in LANGS: + _apply(lang in had, rd / "translations" / lang / f"{slug}.json", TRANSLATIONS / lang / f"{slug}.json") + return {"restored": restored, "deleted": deleted, "missing": False} diff --git a/scripts/_lib/terroir_boilerplate.py b/scripts/_lib/terroir_boilerplate.py new file mode 100644 index 0000000..38c54e5 --- /dev/null +++ b/scripts/_lib/terroir_boilerplate.py @@ -0,0 +1,111 @@ +"""Boilerplate detector for terroir facts (W6 of the 2026-09-11 review). + +Some regulator texts carry a sentence that is true of any appellation — +"the uniqueness of the wines is due to the particular characteristics of +the area (soil, climate, winds)" appears verbatim in dozens of Greek PGI +specs; Alsace's shared cahier repeats "a homogeneous geo-pedological unit +with a most favourable mesoclimate"; German Landwein specs say the wines +are "shaped mainly by vintage and variety". Extracted as a fact, such a +sentence tells the reader nothing. + +The filter is deliberately conservative — two conditions, both required: + +- the fact's `cahier_quote` (normalised, ≥ `QUOTE_MIN_CHARS`) matches one + of the tautology patterns for the record's source language, AND +- that pattern is matched by at least `MIN_SHARED_RECORDS` records of the + same country. + +Shared quotes alone are never enough: several cahiers are legitimately +shared (the Anjou family, the Calabrian IGT text, the Romanian caiete, +the Alsace `produit` slice) and carry real facts. A boilerplate fact is +kept when it is the record's only fact, and a bullet that carries a +number is never treated as boilerplate — quantitative content is +information even when the quoted sentence is the tautology. + +The boilerplate sentence usually embeds the appellation's own name ("Η +μοναδικότητα των οίνων ΠΓΕ Χαλκιδική οφείλεται…") and the model quotes +spans of varying length, so "the same quote" is decided by the pattern: +records are grouped per (country, pattern), and a pattern that ≥ 3 +records of one country quote is boilerplate for all of them. +""" + +from __future__ import annotations + +import re +from collections import defaultdict + +MIN_SHARED_RECORDS = 3 +QUOTE_MIN_CHARS = 60 +_HAS_NUMBER = re.compile(r"\d") + +TAUTOLOGY_PATTERNS: dict[str, tuple[str, ...]] = { + "el": ( + r"μοναδικότητα των οίνων .{0,60}οφείλεται στα ιδιαίτερα χαρακτηριστικά", + r"οίνοι με τυπικότητα και ιδιαίτερους ποιοτικούς χαρακτήρες", + r"ευνοϊκ\S* εδαφοκλιματ\S* συνθήκ", + ), + "fr": ( + r"unité géo-pédologique associés à un mésoclimat des plus favorables", + ), + "de": ( + r"prägung überwiegend durch jahrgang und sorte", + r"jahrgangs- und sortennote", + ), + "ro": ( + r"amprenta .{0,40}soiului, solului, microclimatului", + ), + "it": ( + r"caratteristiche chimico-fisiche equilibrate", + ), + "en": ( + r"uniqueness .{0,40}attributed to", + r"directly linked to the character", + r"no uniform description", + r"defined by their typicity", + r"favourable soil and climatic conditions", + ), +} +_COMPILED = {lang: tuple(re.compile(p) for p in pats) for lang, pats in TAUTOLOGY_PATTERNS.items()} + + +def normalise_quote(q: str) -> str: + return " ".join((q or "").split()).casefold() + + +def matching_pattern(quote_norm: str, source_lang: str) -> str | None: + """The first tautology pattern (its regex source) matching the quote.""" + for rx in _COMPILED.get(source_lang, ()): + if rx.search(quote_norm): + return rx.pattern + return None + + +def is_tautology(quote_norm: str, source_lang: str) -> bool: + return matching_pattern(quote_norm, source_lang) is not None + + +def find_boilerplate(caches: dict[str, dict]) -> dict[str, list[int]]: + """slug → indices of facts to drop (never all of a record's facts).""" + hits: dict[str, list[tuple[int, str, str]]] = {} + by_pattern: dict[tuple[str, str], set[str]] = defaultdict(set) + for slug, d in caches.items(): + cc = d.get("country") or "fr" + lang = d.get("source_lang") or "fr" + for i, f in enumerate(d.get("facts") or []): + q = normalise_quote(f.get("cahier_quote")) + if len(q) < QUOTE_MIN_CHARS or _HAS_NUMBER.search(f.get("bullet") or ""): + continue + pat = matching_pattern(q, lang) + if pat is None: + continue + hits.setdefault(slug, []).append((i, cc, pat)) + by_pattern[(cc, pat)].add(slug) + out: dict[str, list[int]] = {} + for slug, rows in hits.items(): + drops = [i for i, cc, pat in rows if len(by_pattern[(cc, pat)]) >= MIN_SHARED_RECORDS] + n_facts = len(caches[slug].get("facts") or []) + if drops and len(drops) >= n_facts: + drops = drops[1:] # keep the record's only fact rather than empty it + if drops: + out[slug] = drops + return out diff --git a/scripts/_lib/terroir_cache.py b/scripts/_lib/terroir_cache.py new file mode 100644 index 0000000..aedfe08 --- /dev/null +++ b/scripts/_lib/terroir_cache.py @@ -0,0 +1,153 @@ +"""Index-aligned maintenance of the stage-02e translation caches. + +`raw/translations/terroir-facts/<lang>/<slug>.json` copies the source +cache's facts by index (`bullet`, `subsection`, `provenance`) and is keyed +on `source_facts_sha` — a hash of the source bullets. Stage 04's overlay +matches translated bullets to source facts by index (checking only the +length), and its sibling filter (commit 24bd059d) is computed on source +indices and applied after the overlay. Any post-pass that drops, reorders +or relabels source facts therefore has to apply the same change to every +aligned translation cache, or a page shows the wrong bullet under the +wrong provenance. These helpers do that; a cache that is already out of +step is left alone and reported so stage 02e re-translates it. +""" + +from __future__ import annotations + +from pathlib import Path + +from _lib import cache +from _lib.terroir_backup import snapshot_slug +from _lib.terroir_dedupe import facts_sha + +ROOT = Path(__file__).resolve().parents[2] +TERROIR = ROOT / "raw" / "terroir-facts" +TRANSLATIONS = ROOT / "raw" / "translations" / "terroir-facts" +LANGS = ("en", "fr", "es", "nl") + + +def write_source_cache(path: Path, payload: dict) -> None: + """Write a `raw/terroir-facts/<slug>.json` cache, snapshotting the slug's + current source + translation caches into the run's backup first + (`terroir_backup`). Every 02d script and post-pass writes through here + (the wiring lint in tests/test_terroir_backup.py fails otherwise).""" + snapshot_slug(path.stem) + cache.write_json(path, payload) + + +def write_translation_cache(path: Path, payload: dict) -> None: + """Write a `raw/translations/terroir-facts/<lang>/<slug>.json` cache, + snapshotting first — the 02e counterpart of `write_source_cache`.""" + snapshot_slug(path.stem) + cache.write_json(path, payload) + + +def _aligned(t: dict | None, sha: str, n: int) -> bool: + return bool(t) and t.get("mode") != "verbatim" and bool(t.get("facts")) \ + and t.get("source_facts_sha") == sha and len(t["facts"]) == n + + +def prune_translations( + slug: str, old_sha: str, n_old: int, kept_indices: list[int], new_sha: str, *, dry_run: bool, +) -> tuple[list[str], list[dict]]: + """Keep only `kept_indices` (in order) in every aligned translation cache + and re-key it on `new_sha`. Returns (pruned_langs, stale_entries).""" + pruned: list[str] = [] + stale: list[dict] = [] + for lang in LANGS: + tp = TRANSLATIONS / lang / f"{slug}.json" + if not tp.exists(): + continue + t = cache.read_json_or_none(tp) + if not t or t.get("mode") == "verbatim": + continue + if not _aligned(t, old_sha, n_old): + stale.append({"slug": slug, "lang": lang, "reason": "already-misaligned"}) + continue + t["facts"] = [t["facts"][i] for i in kept_indices] + t["source_facts_sha"] = new_sha + pruned.append(lang) + if not dry_run: + write_translation_cache(tp, t) + return pruned, stale + + +def sync_translation_provenance(slug: str, facts: list[dict], *, dry_run: bool) -> tuple[int, list[str]]: + """Copy each source fact's `provenance` into the aligned translation + caches. Returns (n_written, misaligned_langs).""" + sha = facts_sha(facts) + written = 0 + misaligned: list[str] = [] + for lang in LANGS: + tp = TRANSLATIONS / lang / f"{slug}.json" + if not tp.exists(): + continue + t = cache.read_json_or_none(tp) + if not t or t.get("mode") == "verbatim" or not t.get("facts"): + continue + if not _aligned(t, sha, len(facts)): + misaligned.append(lang) + continue + changed = False + for tf, sf in zip(t["facts"], facts): + if tf.get("provenance") != sf.get("provenance"): + tf["provenance"] = sf.get("provenance") + changed = True + if changed: + written += 1 + if not dry_run: + write_translation_cache(tp, t) + return written, misaligned + + +def sync_translation_meta(slug: str, facts: list[dict], *, dry_run: bool) -> tuple[int, list[str]]: + """Copy each source fact's `provenance` AND `subsection` into the aligned + translation caches (the gate moves misfiled facts between sub-sections + without touching the text). Returns (n_written, misaligned_langs).""" + sha = facts_sha(facts) + written = 0 + misaligned: list[str] = [] + for lang in LANGS: + tp = TRANSLATIONS / lang / f"{slug}.json" + if not tp.exists(): + continue + t = cache.read_json_or_none(tp) + if not t or t.get("mode") == "verbatim" or not t.get("facts"): + continue + if not _aligned(t, sha, len(facts)): + misaligned.append(lang) + continue + changed = False + for tf, sf in zip(t["facts"], facts): + for key in ("provenance", "subsection"): + if tf.get(key) != sf.get(key): + tf[key] = sf.get(key) + changed = True + if changed: + written += 1 + if not dry_run: + write_translation_cache(tp, t) + return written, misaligned + + +def rekey_translations(slug: str, old_sha: str, n: int, new_sha: str, *, dry_run: bool) -> tuple[list[str], list[str]]: + """After an in-place edit of source bullets that keeps count and order + (the normaliser), move every aligned translation cache to `new_sha`. + Returns (rekeyed_langs, misaligned_langs).""" + done: list[str] = [] + misaligned: list[str] = [] + for lang in LANGS: + tp = TRANSLATIONS / lang / f"{slug}.json" + if not tp.exists(): + continue + t = cache.read_json_or_none(tp) + if not t or t.get("mode") == "verbatim" or not t.get("facts"): + continue + if not _aligned(t, old_sha, n): + misaligned.append(lang) + continue + t["source_facts_sha"] = new_sha + done.append(lang) + if not dry_run: + write_translation_cache(tp, t) + return done, misaligned diff --git a/scripts/_lib/terroir_chapters.py b/scripts/_lib/terroir_chapters.py new file mode 100644 index 0000000..8ed74de --- /dev/null +++ b/scripts/_lib/terroir_chapters.py @@ -0,0 +1,65 @@ +"""Own-chapter window inside a cahier shared by several appellations. + +INAO publishes one cahier des charges for the 51 Alsace grands crus. Its +section X ("Lien au terroir") is one 339 KB text that repeats, per cru, a +chapter headed `« Alsace grand cru <Cru> »` followed by the usual +`1°- / 2°- / 3°-` sub-sections. Stage 02 does not split shared cahiers +(CLAUDE.md), so every cru record carries the whole lien, and the 02d +slicer — which keys spans by section number — used to grade every cru +against the LAST chapter (Zotzenberg, alphabetically last): 51 pages +carried Zotzenberg's geology, slope and 1992 date (2026-09-11 review). + +`own_chapter` returns the (start, end) window of the record's own +chapter so the slicer and the audit can restrict the lien to it. A +chapter heading is a guillemet-quoted `« <prefix> <name> »` on its own +line immediately followed by the `1°` anchor; names are compared +accent-folded and case-folded. +""" + +from __future__ import annotations + +import re +import unicodedata + +DEFAULT_PREFIX = "Alsace grand cru" +_ANCHOR_1 = r"\s*1°" + + +def fold(s: str) -> str: + s = unicodedata.normalize("NFKD", s or "") + s = "".join(c for c in s if not unicodedata.combining(c)) + return " ".join(s.casefold().replace("’", "'").split()) + + +def chapter_windows(lien: str, prefix: str = DEFAULT_PREFIX) -> list[tuple[str, int, int]]: + """[(chapter name, start, end)] for every chapter heading in `lien`, + in document order; `end` is the next chapter's start (or len(lien)).""" + rx = re.compile(r"«\s*" + re.escape(prefix) + r"\s+([^»\n]{1,80}?)\s*»[ \t]*\n" + _ANCHOR_1) + heads = [(m.group(1).strip(), m.start()) for m in rx.finditer(lien)] + out: list[tuple[str, int, int]] = [] + for i, (name, start) in enumerate(heads): + end = heads[i + 1][1] if i + 1 < len(heads) else len(lien) + out.append((name, start, end)) + return out + + +def is_shared(lien: str, prefix: str = DEFAULT_PREFIX) -> bool: + return len(chapter_windows(lien, prefix)) > 1 + + +def own_chapter(lien: str, record_name: str, prefix: str = DEFAULT_PREFIX) -> tuple[int, int] | None: + """Window of the chapter whose heading names `record_name` (with or + without the shared prefix), or None when the lien has several chapters + but none is this record's — the caller must not fall back to another + cru's chapter.""" + windows = chapter_windows(lien, prefix) + if len(windows) < 2: + return None + want = fold(record_name) + pfx = fold(prefix) + if want.startswith(pfx): + want = want[len(pfx):].strip() + for name, start, end in windows: + if fold(name) == want: + return start, end + return None diff --git a/scripts/_lib/terroir_coverage.py b/scripts/_lib/terroir_coverage.py new file mode 100644 index 0000000..d660ae9 --- /dev/null +++ b/scripts/_lib/terroir_coverage.py @@ -0,0 +1,163 @@ +"""Grounding coverage for terroir-fact quotes (stages 02d, the audit, and +the cache post-passes). + +A fact survives extraction only when at least one of its two verbatim +quotes (`cahier_quote`, `wiki_quote`) is found in its source text. The +test is the longest contiguous match between the normalised quote and +the normalised source, as a share of the quote's length; `FUZZY_THRESHOLD` +is the pass mark and `provenance_for` turns the two shares into the +per-fact `both` / `cahier` / `wiki` label. + +The model legitimately joins two spans of the source with an ellipsis +("[…]", "[...]", "…"). A single contiguous match then covers at most the +longer span, which capped coverage well below the threshold and flipped +hundreds of cahier-grounded facts to `wiki` (or dropped them). Such a +quote is therefore also graded span by span: the coverage of a multi-span +quote is the better of the whole-quote match and the *weakest* span's +match, so every span has to ground for the split to help, and a quote +that already passed as a whole is never demoted. Spans shorter than +`MIN_SPAN_CHARS` are too short to grade on their own and are ignored +(unless every span is that short). + +A verbatim quote can also straddle a pdftotext artefact in the source — +"gradi- giorno", a hyphenated line break, a stray footnote mark — which a +single contiguous match cannot cross: one wrong character in the middle +of a 120-character quote halved its coverage and dropped eight of nine +true facts from Montepulciano d'Abruzzo (2026-09-13). A quote is +therefore also graded by *blocks*: the longest match is taken, then the +quote's remainders on either side are matched again, recursively, and +the sizes of every block of at least `MIN_BLOCK_CHARS` are summed. The +coverage is the best of the three measures; the threshold stays 0.6, so +60 % of the quote's characters must still sit in verbatim runs of the +source — a quote stitched from scattered short phrases does not pass. + +Both sides are normalised the same way before matching, and the +normalisation folds the typography that separated verbatim quotes from +their source: curly / low-9 / guillemet quotes and apostrophes to their +straight forms, every dash to a hyphen, soft hyphens and zero-width +characters removed (in this corpus the soft hyphen is a bullet glyph, +never a hyphenation point), compatibility forms (ligatures, +superscripts, "…") decomposed (NFKC), the space a line-break hyphenation +leaves behind ("gradi- giorno", "Nieder- österreich") closed up and the +spaces inside « guillemets » dropped. On the r1 corpus 8.5 % +of the kept quotes matched better for it and 7.4 % went from +block-rescued to a single contiguous match; no quote crossed the +threshold downwards. + +Every stage-02d script and the audit import `fuzzy_coverage` from here so +the grounding rule cannot drift between countries. +""" + +from __future__ import annotations + +import re +import unicodedata +from difflib import SequenceMatcher + +FUZZY_THRESHOLD = 0.6 +MIN_SPAN_CHARS = 15 +MIN_BLOCK_CHARS = 12 +MAX_BLOCKS = 6 + +# "[…]" / "[...]" / "(…)" / "(...)" / bare "…" / bare "..." — the joins the +# model uses when it stitches two source spans into one quote. +ELLIPSIS_RE = re.compile(r"[\[(]\s*(?:…|\.{3})\s*[\])]|…|\.{3}") + + +_TYPOGRAPHY = str.maketrans({ + "\u2018": "'", "\u2019": "'", "\u201a": "'", "\u201b": "'", "\u2032": "'", "\u02bc": "'", + "\u201c": '"', "\u201d": '"', "\u201e": '"', "\u201f": '"', "\u00ab": '"', "\u00bb": '"', + "\u2010": "-", "\u2011": "-", "\u2012": "-", "\u2013": "-", "\u2014": "-", "\u2015": "-", + "\u2212": "-", + "\u00ad": "", "\u200b": "", "\u200c": "", "\u200d": "", "\ufeff": "", +}) +# "gradi- giorno": a hyphen glued to a word, then whitespace, then a letter +# — the trace of a hyphenated line break; "300 - 400 m" (space before the +# hyphen) and "2019- 2020" (digit after) are left alone. +_HYPHEN_BREAK_RE = re.compile(r"(?<=\w)-\s+(?=[^\W\d_])") +# « terroir » carries inner spaces that "terroir" does not. +_QUOTE_SPACE_RE = re.compile(r'\s*"\s*') + + +def normalize(s: str) -> str: + s = unicodedata.normalize("NFKC", s or "").translate(_TYPOGRAPHY) + s = _HYPHEN_BREAK_RE.sub("-", s) + s = _QUOTE_SPACE_RE.sub('"', s) + return " ".join(s.split()).lower() + + +def split_spans(quote_norm: str) -> list[str]: + """The non-empty pieces of a normalised quote between ellipsis joins.""" + return [p.strip() for p in ELLIPSIS_RE.split(quote_norm) if p.strip()] + + +class SourceMatcher: + """One normalised source text with its match index built once, so a + post-pass can grade every quote of a record without re-indexing a + 300 KB cahier per quote.""" + + def __init__(self, source: str) -> None: + self.source = normalize(source) + self._sm = SequenceMatcher(None, "", self.source, autojunk=False) + + def contiguous(self, quote_norm: str) -> float: + if not quote_norm: + return 0.0 + self._sm.set_seq1(quote_norm) + m = self._sm.find_longest_match(0, len(quote_norm), 0, len(self.source)) + return m.size / len(quote_norm) + + def _blocks(self, quote_norm: str, budget: int) -> int: + """Total size of the verbatim blocks (≥ MIN_BLOCK_CHARS) of + `quote_norm` in the source: the longest match, then the remainders + on either side of it, recursively, at most `budget` matches.""" + if len(quote_norm) < MIN_BLOCK_CHARS or budget <= 0: + return 0 + self._sm.set_seq1(quote_norm) + m = self._sm.find_longest_match(0, len(quote_norm), 0, len(self.source)) + if m.size < MIN_BLOCK_CHARS: + return 0 + total = m.size + left, right = quote_norm[: m.a].strip(), quote_norm[m.a + m.size:].strip() + total += self._blocks(left, budget - 1) + total += self._blocks(right, budget - 1) + return total + + def blocks(self, quote_norm: str) -> float: + if not quote_norm: + return 0.0 + return min(1.0, self._blocks(quote_norm, MAX_BLOCKS) / len(quote_norm)) + + def coverage(self, quote: str) -> float: + q = normalize(quote) + if not q: + return 0.0 + whole = self.contiguous(q) + if whole >= 1.0: + return whole + best = max(whole, self.blocks(q)) + spans = split_spans(q) + if len(spans) < 2: + return best + graded = [p for p in spans if len(p) >= MIN_SPAN_CHARS] or spans + return max(best, min(self.contiguous(p) for p in graded)) + + +def fuzzy_coverage(quote: str, source: str) -> float: + """Share of `quote` grounded in `source` (0.0–1.0); ellipsis-aware.""" + return SourceMatcher(source).coverage(quote) + + +def provenance_for( + cahier_coverage: float, wiki_coverage: float, threshold: float = FUZZY_THRESHOLD, +) -> str | None: + """`both` / `cahier` / `wiki`, or None when neither source grounds.""" + c_ok = cahier_coverage >= threshold + w_ok = wiki_coverage >= threshold + if c_ok and w_ok: + return "both" + if c_ok: + return "cahier" + if w_ok: + return "wiki" + return None diff --git a/scripts/_lib/terroir_dedupe.py b/scripts/_lib/terroir_dedupe.py new file mode 100644 index 0000000..3369648 --- /dev/null +++ b/scripts/_lib/terroir_dedupe.py @@ -0,0 +1,227 @@ +"""Intra-record de-duplication of terroir facts. + +Stage 02d extracts each record in four sub-section calls over overlapping +source text, so the same sentence is regularly restated two or three +times (a soil fact under "facteurs naturels", again under "produit" and +again under "interactions"). `dedupe_facts` collapses those restatements +inside one record; it is applied by `scripts/dedupe_terroir_facts.py` to +the existing caches and by stage 02d after its sub-section loop. + +Two facts are duplicates when + +- their bullets are near-identical (`token_set_ratio` ≥ + `BULLET_DUP_SIMILARITY`), or +- they cite the same source sentence (a normalised `cahier_quote` or + `wiki_quote` of at least `QUOTE_MIN_CHARS` that is identical, or one a + longer cut of the other) *and* their bullets overlap substantially + (≥ `QUOTE_DUP_BULLET_SIMILARITY`). A shared quote alone is not enough: + one long source sentence often yields two distinct facts (alluvial + soils / river water supply), and dropping either would lose + information. + +Neither rule fires when the two bullets carry different sets of numbers +with neither set contained in the other — different quantities are +different facts, however similar the wording. Nor when the bullets lead +with different `protected_names` (the record's sub-denominations, as +stage 04's sibling filter sees them): a parent's bullet "Rioja Alavesa: +…" must survive next to "Rioja Oriental: …" or an unlabelled twin, or a +sub-denomination page loses the one bullet that was about it. + +Duplicates are transitive: a fact restating an already-dropped fact is +dropped too (the third cut of one sentence rarely resembles the first as +closely as it resembles the second). + +Of a duplicate pair the more informative fact is kept: `both` provenance +first, then more numbers, then longer quotes; ties keep the earlier one. +The kept fact stays at the earlier fact's position, so within-sub-section +order is stable. +""" + +from __future__ import annotations + +import hashlib +import re +import unicodedata +from collections.abc import Iterable +from dataclasses import dataclass, field + +from rapidfuzz import fuzz + +QUOTE_MIN_CHARS = 30 +QUOTE_DUP_BULLET_SIMILARITY = 60 +BULLET_DUP_SIMILARITY = 85 + +_NUM_RE = re.compile(r"\d+(?:[.,]\d+)?") + + +def _norm(s: str) -> str: + return " ".join((s or "").split()).lower() + + +def _numbers(text: str) -> frozenset[str]: + return frozenset(n.replace(",", ".") for n in _NUM_RE.findall(text or "")) + + +def _numbers_compatible(a: frozenset[str], b: frozenset[str]) -> bool: + return not a or not b or a <= b or b <= a + + +def _quotes(fact: dict) -> list[str]: + out = [] + for key in ("cahier_quote", "wiki_quote"): + q = _norm(fact.get(key) or "") + if len(q) >= QUOTE_MIN_CHARS: + out.append(q) + return out + + +def _share_quote(a: dict, b: dict) -> bool: + return any(x == y or x in y or y in x for x in _quotes(a) for y in _quotes(b)) + + +def _norm_name(s: str) -> str: + s = unicodedata.normalize("NFKD", s or "") + s = "".join(c for c in s if not unicodedata.combining(c)) + return " ".join(s.casefold().split()) + + +def _leads_with_name(bullet: str, name: str) -> bool: + """Mirror of stage 04's sibling test (`_leads_with_name` in + 04_build_maps.py): the bullet starts with the name as a whole word.""" + nb, nn = _norm_name(bullet), _norm_name(name) + if not nn or not nb.startswith(nn): + return False + return len(nb) == len(nn) or not nb[len(nn)].isalnum() + + +def _lead_name(bullet: str, names: Iterable[str]) -> str | None: + return next((n for n in names if _leads_with_name(bullet, n)), None) + + +def duplicate_reason(a: dict, b: dict, protected_names: Iterable[str] = ()) -> str | None: + """`similar-bullet` / `same-quote` when `a` and `b` restate one fact.""" + bullet_a = a.get("bullet") or "" + bullet_b = b.get("bullet") or "" + if not _numbers_compatible(_numbers(bullet_a), _numbers(bullet_b)): + return None + names = tuple(protected_names) + if names and _lead_name(bullet_a, names) != _lead_name(bullet_b, names): + return None + similarity = fuzz.token_set_ratio(bullet_a, bullet_b) + if similarity >= BULLET_DUP_SIMILARITY: + return "similar-bullet" + if similarity >= QUOTE_DUP_BULLET_SIMILARITY and _share_quote(a, b): + return "same-quote" + return None + + +def _rank(fact: dict) -> tuple[bool, int, int]: + return ( + fact.get("provenance") == "both", + len(_numbers(fact.get("bullet") or "")), + len(fact.get("cahier_quote") or "") + len(fact.get("wiki_quote") or ""), + ) + + +@dataclass +class DedupeResult: + kept: list[dict] = field(default_factory=list) + kept_indices: list[int] = field(default_factory=list) + drops: list[dict] = field(default_factory=list) + + @property + def changed(self) -> bool: + return bool(self.drops) + + +def dedupe_facts(facts: list[dict], protected_names: Iterable[str] = ()) -> DedupeResult: + """Collapse restated facts; `drops` records each dropped fact with the + index and bullet of the fact it duplicated. `protected_names` are the + record's sub-denomination names (see `duplicate_reason`).""" + names = tuple(protected_names) + kept: list[tuple[int, dict]] = [] + group_of: dict[int, int] = {} # original index → position in `kept` of its group's survivor + drops: list[dict] = [] + for idx, fact in enumerate(facts): + twin_pos = None + reason = None + for j in range(idx): + reason = duplicate_reason(facts[j], fact, names) + if reason: + twin_pos = group_of[j] + break + if twin_pos is None: + group_of[idx] = len(kept) + kept.append((idx, fact)) + continue + group_of[idx] = twin_pos + twin_idx, twin = kept[twin_pos] + if _rank(fact) > _rank(twin): + kept[twin_pos] = (idx, fact) + dropped_idx, dropped, winner_idx, winner = twin_idx, twin, idx, fact + else: + dropped_idx, dropped, winner_idx, winner = idx, fact, twin_idx, twin + drops.append({ + "reason": reason, + "dropped_index": dropped_idx, + "dropped_bullet": dropped.get("bullet") or "", + "dropped_subsection": dropped.get("subsection"), + "dropped_provenance": dropped.get("provenance"), + "kept_index": winner_idx, + "kept_bullet": winner.get("bullet") or "", + "kept_subsection": winner.get("subsection"), + "kept_provenance": winner.get("provenance"), + }) + return DedupeResult( + kept=[f for _, f in kept], + kept_indices=[i for i, _ in kept], + drops=drops, + ) + + +def facts_sha(facts: list[dict]) -> str: + """sha256 of the source-language bullets joined with "\\n" — the + `source_facts_sha` every stage-02e translation cache is keyed on.""" + blob = "\n".join((f.get("bullet") or "") for f in facts) + return hashlib.sha256(blob.encode("utf-8")).hexdigest() + + +# ───────────────────────────────────────── cosmetic vs meaning change ── +# +# A rewrite this close to the original (rapidfuzz ratio, 0–100) whose +# differing words are all short is cosmetic — case, punctuation, an +# article — not a meaning change. The gate keeps the original bullet as +# supported for such a rewrite (so the rewritten cohort stays meaning +# changes only and the translations are not redone for nothing: 44 % of +# the r1 rewrites were light edits, 9 % near-cosmetic), and the feedback +# recurrence check counts a do-not-claim entry as resolved only when the +# gate's rewrite was NOT cosmetic. A hedge added to a long bullet scores +# ≈ 98 too, so the ratio alone is not the test: any differing word of +# COSMETIC_WORD_CHARS letters or more ("mainly", "esclusivamente") makes +# it a real rewrite. +NEAR_IDENTICAL_RATIO = 95 +COSMETIC_WORD_CHARS = 4 + + +def _words(s: str) -> list[str]: + folded = unicodedata.normalize("NFKD", (s or "").lower()) + folded = "".join(ch for ch in folded if not unicodedata.combining(ch)) + return re.findall(r"[^\W_]+", folded) + + +def is_cosmetic_rewrite(original: str, rewrite: str) -> bool: + """True when `rewrite` differs from `original` only cosmetically: + near-identical overall (ratio ≥ NEAR_IDENTICAL_RATIO) and every word + present in one but not the other is shorter than COSMETIC_WORD_CHARS + — so a hedge, a qualifier or a changed entity is never cosmetic.""" + o = " ".join((original or "").split()) + r = " ".join((rewrite or "").split()) + if o == r: + return True + if fuzz.ratio(o, r) < NEAR_IDENTICAL_RATIO: + return False + ow, rw = _words(o), _words(r) + if ow == rw: + return True + diff = set(ow) ^ set(rw) + return all(len(w) < COSMETIC_WORD_CHARS for w in diff) diff --git a/scripts/_lib/terroir_feedback.py b/scripts/_lib/terroir_feedback.py new file mode 100644 index 0000000..ad4d486 --- /dev/null +++ b/scripts/_lib/terroir_feedback.py @@ -0,0 +1,217 @@ +"""Per-record review feedback for the terroir-fact stages. + +`raw/terroir-facts-feedback/<slug>.json` keeps what a quality review +found about ONE record, as constraints stage 02d can read on its next +extraction — not as bullets to reproduce: + + do_not_claim verified misleading claims: the English and + source-language bullet, the failure mode, the stage + that produced it (extraction / translation / both) + and the verifier's source-quoting reason + capture_if_present what the reviewer saw the source describe + prominently and no bullet captured (hints) + record_cautions sibling text inside the source, a wrong source + binding, a source typo, a wrong Wikipedia article + history one entry per later run: bullets a gate dropped or + rewrote, so the file becomes the record's QA trail + +`with_feedback(system, slug)` appends the prompt block to a 02d system +prompt (the country scripts call it right after `EXTRACT_SYSTEM.format`); +only the `extraction` / `both` entries go to 02d — `translation` entries +are for a 02e back-check. The block phrases every negative as "only if +the source states it explicitly", so a claim that IS grounded survives. +`recurrence_findings` is the audit's regression check: a do-not-claim +entry whose source-language bullet still matches a current bullet. + +Built by `scripts/build_terroir_feedback.py` from a review's evidence +directory; never shown on the map or in the wiki. +""" + +from __future__ import annotations + +import json +from datetime import datetime, timezone +from pathlib import Path + +from rapidfuzz import fuzz + +from _lib.terroir_dedupe import is_cosmetic_rewrite +from _lib.terroir_prompts import appellation_context + +ROOT = Path(__file__).resolve().parents[2] +FEEDBACK_DIR = ROOT / "raw" / "terroir-facts-feedback" + +RECURRENCE_THRESHOLD = 85 +MAX_CLAIMS = 6 +MAX_CAPTURE = 4 +MAX_CAUTIONS = 3 +_WHY_CHARS = 320 +_HINT_CHARS = 320 +_NOTE_CHARS = 320 + +_cache: dict[str, dict | None] = {} + + +def feedback_path(slug: str) -> Path: + return FEEDBACK_DIR / f"{slug}.json" + + +def load_feedback(slug: str) -> dict | None: + """The record's feedback sidecar, or None. Cached per process; call + `clear_cache()` after writing.""" + if slug in _cache: + return _cache[slug] + path = feedback_path(slug) + fb = None + if path.exists(): + try: + fb = json.loads(path.read_text(encoding="utf-8")) + except (ValueError, OSError): + fb = None + _cache[slug] = fb + return fb + + +def clear_cache() -> None: + _cache.clear() + + +def is_stale(fb: dict, cahier_sha: str | None, wiki_revision) -> bool: + """True when the sources the review graded against have changed since + — the constraints may be obsolete, which is why the prompt block never + asserts them unconditionally.""" + graded = fb.get("graded_against") or {} + if cahier_sha and graded.get("cahier_source_sha") and graded["cahier_source_sha"] != cahier_sha: + return True + if ( + wiki_revision is not None and graded.get("wiki_source_revision") is not None + and str(graded["wiki_source_revision"]) != str(wiki_revision) + ): + return True + return False + + +def _clip(text: str, n: int) -> str: + text = " ".join((text or "").split()) + return text if len(text) <= n else text[: n - 1].rstrip() + "…" + + +def _safe(text: str) -> str: + """No `{` / `}`: a script that formats its prompt after appending the + block must not trip on feedback text.""" + return text.replace("{", "(").replace("}", ")") + + +def feedback_prompt_block( + fb: dict | None, *, stage: str = "extraction", + max_claims: int = MAX_CLAIMS, max_capture: int = MAX_CAPTURE, max_cautions: int = MAX_CAUTIONS, +) -> str: + """The English block appended to a 02d prompt, or "" when the record + has nothing for this stage. `stage` selects the do-not-claim entries: + `extraction` takes extraction + both, `translation` takes translation + + both.""" + if not fb: + return "" + wanted = {"extraction", "both"} if stage == "extraction" else {"translation", "both"} + claims = [c for c in fb.get("do_not_claim") or [] if (c.get("stage") or "extraction") in wanted] + capture = fb.get("capture_if_present") or [] + cautions = fb.get("record_cautions") or [] + if not (claims or capture or cautions): + return "" + reviews = fb.get("reviews") or [] + dates = sorted({r.get("date", "") for r in reviews if r.get("date")}) + when = f" ({', '.join(dates)})" if dates else "" + lines = [ + f"Lessons from the previous review of this record{when}. They are constraints on " + "grounding, not facts to reproduce; every bullet must still be a verbatim-quotable " + "claim of the text you are given:" + ] + for c in claims[:max_claims]: + claim = _clip(c.get("claim_en") or c.get("claim_src") or "", 200) + why = _clip(c.get("why") or "", _WHY_CHARS) + lines.append(f"- Do not assert «{claim}» unless the source states it explicitly — {why}") + if len(claims) > max_claims: + lines.append(f"- ({len(claims) - max_claims} more claims of the same kind were unsupported.)") + for h in capture[:max_capture]: + lines.append(f"- If the text describes it, capture: {_clip(h.get('hint') or '', _HINT_CHARS)}") + for n in cautions[:max_cautions]: + lines.append(f"- Caution ({n.get('kind') or 'other'}): {_clip(n.get('note') or '', _NOTE_CHARS)}") + return _safe("\n".join(lines)) + + +def with_feedback(system: str, slug: str, *, stage: str = "extraction") -> str: + """`system` + the record's per-record block: its review feedback + (do-not-claim constraints, cautions) and, for a record whose bullets + are inherited by sub-denomination pages, the instruction to name the + appellation where the source says "the appellation" + (`terroir_prompts.appellation_context`). `system` unchanged when the + record has neither.""" + parts = [system.rstrip()] + block = feedback_prompt_block(load_feedback(slug), stage=stage) + if block: + parts.append(block) + ctx = appellation_context(slug, for_translation=False) + if ctx: + parts.append(ctx) + return "\n\n".join(parts) if len(parts) > 1 else system + + +def recurrence_findings( + fb: dict | None, facts: list[dict], *, threshold: int = RECURRENCE_THRESHOLD, +) -> list[dict]: + """Do-not-claim entries (extraction / both) whose source-language bullet + still matches a current fact's bullet — the known error is still there + (before a re-run) or came back (after one). + + The match is lexical (token-set ratio), so a bullet the gate has since + rewritten still matches its own misleading original on most of its + words. Such an entry is counted as resolved, not recurring, when the + matched fact carries `support.original_bullet`, the claim matches that + original at least as well as it matches the current bullet, and the + rewrite was not cosmetic (5 of the 6 residual hits of the r1 audit were + fixed bullets).""" + if not fb: + return [] + out: list[dict] = [] + bullets = [(i, (f.get("bullet") or "")) for i, f in enumerate(facts)] + for c in fb.get("do_not_claim") or []: + if (c.get("stage") or "extraction") == "translation": + continue + probe = c.get("claim_src") or c.get("claim_en") or "" + if not probe: + continue + best_i, best = -1, 0.0 + for i, b in bullets: + if not b: + continue + score = fuzz.token_set_ratio(probe, b) + if score > best: + best_i, best = i, score + if best < threshold: + continue + current = facts[best_i].get("bullet") or "" + original = (facts[best_i].get("support") or {}).get("original_bullet") or "" + if ( + original + and fuzz.token_set_ratio(probe, original) >= best + and not is_cosmetic_rewrite(original, current) + ): + continue + out.append({ + "index": best_i, "score": round(best, 1), "mode": c.get("mode"), + "review": c.get("review"), "claim": _clip(c.get("claim_en") or probe, 160), + }) + return out + + +def append_history(slug: str, entry: dict) -> None: + """Add a run entry (`{run, kind, ...}`) to the record's `history`, + creating a minimal sidecar when none exists.""" + path = feedback_path(slug) + fb = load_feedback(slug) or {"slug": slug, "do_not_claim": [], "capture_if_present": [], + "record_cautions": [], "reviews": [], "history": []} + entry = {"at": datetime.now(timezone.utc).isoformat(timespec="seconds"), **entry} + fb.setdefault("history", []).append(entry) + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(fb, ensure_ascii=False, indent=1) + "\n", encoding="utf-8") + _cache[slug] = fb diff --git a/scripts/_lib/terroir_gate.py b/scripts/_lib/terroir_gate.py new file mode 100644 index 0000000..3aae93a --- /dev/null +++ b/scripts/_lib/terroir_gate.py @@ -0,0 +1,368 @@ +"""The claim-support gate over stage-02d terroir facts (review +2026-09-12, R2 / R3 / R7 / R8). + +Stage 02d's coverage test only checks that a bullet's *quote* exists in +the source; it never checks that the bullet's *claim* is what the quote +says. Every failure mode the review verified — an unsupported causal +link, a wrong entity, an invented qualifier, a dropped hedge, a sibling +sub-zone presented as the whole, a narrowed attribution — passes it. The +gate asks a model, per record, to grade each bullet against the full +source text (the exact text 02d graded against, via +`terroir_sources`) plus the record's review feedback, and applies the +verdicts deterministically: + + supported kept as is + rewrite the main fact holds but a qualifier / causal wrapper / + hedge goes beyond the source → the bullet is replaced by + the model's rewrite, which may only remove or soften + (guarded: no new numbers, no arrows, sane length) + drop the main claim is unsupported, contradicted, about another + appellation, a tautology, or restates another bullet + +plus an optional `subsection` when a bullet is clearly misfiled (W7), +and `restates` (the index of the bullet it duplicates — the semantic +dedupe of R8). The `interactions` sub-section is earned: a bullet there +is `supported` only when the source sentence states the causal link. + +Pure functions here (prompt building, verdict parsing, application); +the I/O, batch and cache writing live in +`scripts/02d_verify_terroir_facts.py`. +""" + +from __future__ import annotations + +import json +import re + +from _lib import llm_json +from _lib.terroir_dedupe import _numbers, dedupe_facts, is_cosmetic_rewrite +from _lib.terroir_feedback import _clip, _safe +from _lib.terroir_interactions import DEMOTION_TARGET, unearned_indices +from _lib.terroir_normalize import normalize_bullet + +SUBSECTION_KEYS = ("facteurs_naturels", "facteurs_humains", "produit", "interactions") +VERDICTS = ("supported", "rewrite", "drop") +MAX_SOURCE_CHARS = 60_000 +MAX_HINT_CHARS = 2_000 +REWRITE_MAX_CHARS = 420 +GATE_VERSION = "gate-v2" + +SYSTEM = """You are the claim-support verifier for Open Wine Map's terroir facts: short bullets an LLM extracted from a wine regulator's product specification (the "source text" — a cahier des charges, disciplinare, pliego, Einziges Dokument, …) and, secondarily, from the appellation's Wikipedia article (the "Wikipedia hints"). The bullets are in the source language; they will be translated and shown to wine enthusiasts as facts about the appellation. + +Your job is adversarial: for EACH bullet decide whether every assertion it makes — entity, number, unit, direction, spatial or temporal qualifier, hedge, causal claim — is stated by the source text or the Wikipedia hints. The quotes attached to a bullet are only where the extractor looked; judge against the whole text you are given. + +Verdicts: +- "supported": every assertion is stated by the sources. A simplification is not an error; a number written differently in the source (13°5, 22, 5, anni '60, XVIIIe) is not missing; a bullet grounded on the Wikipedia hint is legitimately grounded. +- "rewrite": the main fact is stated, but the bullet adds something the sources do not state — a causal wrapper on a mere co-occurrence ("volcanic soils give minerality" when the source only records volcanic soils and, elsewhere, minerality), a narrowed attribution (one factor credited with what the source credits to several), a spatial or temporal qualifier the source does not give, a hedge strengthened ("typically" → "exclusively", "weakly" → "moderately") or dropped, a detail from a neighbouring sub-zone applied to the whole appellation. Provide "rewrite": the bullet in the SAME language, keeping the supported fact and removing or softening only what goes beyond the source. Never add content, never add a number that is not in the bullet or the source, keep it one full sentence ending with a period. A "rewrite" verdict MUST carry a non-empty "rewrite" that differs in meaning from the bullet — if you cannot phrase the narrower sentence, choose "supported" (with the note) or "drop" instead; a rewrite that only changes wording, punctuation or word order is not a rewrite. +- "drop": the main claim itself is unsupported or contradicted, describes another appellation or a sub-zone/neighbour rather than this one, is a tautology true of any appellation ("the terroir gives the wines their typicity"), or restates a fact another bullet already gives (then set "restates" to that bullet's index; keep the more precise one and drop the other). + +Sub-section rule: a bullet filed under "interactions" (causal terroir → wine links) is "supported" only when a source sentence itself states the link with an explicit connective (because, thanks to, gives, confers, results in, explains, favours, allows, or the equivalent in the source language). If the source merely lists factors and wine traits side by side, "rewrite" the bullet into the non-causal statement the source does make (and file it under the right sub-section via "subsection"), or "drop" it when that statement is already given by another bullet. + +Misfiling (optional): set "subsection" to one of facteurs_naturels / facteurs_humains / produit / interactions ONLY when the current one is clearly wrong: a soil or climate fact under human factors; a yield rule or a history date under natural factors; a description of the wines themselves — colour, aroma, structure, ageing aptitude, versatility, suitability for blending or early drinking — under natural factors, even when the source states it inside a paragraph about a zone's climate or soils (it belongs under produit); otherwise null. + +Prior-review constraints, when given, name claims that were verified misleading before: a bullet making one of them is "drop" or "rewrite" unless the source states it explicitly. Record cautions describe known defects of the source (a section copied from another appellation, a wrong Wikipedia article): do not credit text that a caution disqualifies. + +Answer ONLY with JSON, no text before or after: +{"facts": [{"i": 0, "verdict": "supported|rewrite|drop", "note": "one precise sentence quoting the decisive source words, or the assertion the source lacks", "rewrite": "" , "restates": null, "subsection": null}, ...]} +One object per bullet, in order, with "i" equal to the bullet's index.""" + + +def needs_gate(d: dict, *, refresh: bool = False) -> bool: + """A record is due for the gate when it has facts and any of them carries + no gate verdict (`support` — a fresh 02d extraction writes none), or the + gate block predates the current GATE_VERSION or the record's source + text. Deliberately not an exact sha of the bullets: the normalise, + dedupe and boilerplate post-passes change bullets or remove facts + without invalidating the verdicts on the rest, and used to re-fire the + gate corpus-wide.""" + facts = d.get("facts") or [] + if not facts: + return False + if refresh: + return True + g = d.get("gate") or {} + if not g or g.get("version") != GATE_VERSION: + return True + if g.get("cahier_source_sha") != d.get("cahier_source_sha"): + return True + return any(not (f.get("support") or {}).get("verdict") for f in facts) + + +def _constraints_block(fb: dict | None) -> str: + if not fb: + return "" + lines: list[str] = [] + for c in fb.get("do_not_claim") or []: + if (c.get("stage") or "extraction") == "translation": + continue + claim = _clip(c.get("claim_src") or c.get("claim_en") or "", 220) + why = _clip(c.get("why") or "", 260) + if claim: + lines.append(f"- Verified misleading before: «{claim}» — {why}") + for n in fb.get("record_cautions") or []: + note = _clip(n.get("note") or "", 260) + if note: + lines.append(f"- Caution ({n.get('kind') or 'other'}): {note}") + if not lines: + return "" + return "PRIOR-REVIEW CONSTRAINTS FOR THIS RECORD\n" + "\n".join(lines) + "\n\n" + + +def build_user_message( + *, name: str, country: str, source_lang: str, cahier: str, hints: dict[str, str], + facts: list[dict], feedback: dict | None, +) -> str: + """The per-record user message: constraints, source text, per-sub-section + Wikipedia hints, then the numbered bullets with their quotes.""" + src = cahier or "" + if len(src) > MAX_SOURCE_CHARS: + src = src[:MAX_SOURCE_CHARS] + "\n[… truncated …]" + hint_lines = [] + for key in SUBSECTION_KEYS: + h = (hints or {}).get(key) or "" + if h: + hint_lines.append(f"[{key}]\n{h[:MAX_HINT_CHARS]}") + hint_block = "\n\n".join(hint_lines) or "(none)" + fact_lines = [] + for i, f in enumerate(facts): + fact_lines.append( + f"#{i} [{f.get('subsection') or 'facteurs_naturels'} · {f.get('provenance') or ''}]\n" + f"BULLET: {f.get('bullet') or ''}\n" + f"CAHIER_QUOTE: {f.get('cahier_quote') or ''}\n" + f"WIKI_QUOTE: {f.get('wiki_quote') or ''}" + ) + return _safe( + f"RECORD: {name} (country {country}, source language {source_lang}); {len(facts)} bullets.\n\n" + f"{_constraints_block(feedback)}" + f"SOURCE TEXT ({len(src)} chars)\n{src or '(no regulator text — grade against the Wikipedia hints only)'}\n\n" + f"WIKIPEDIA HINTS (per sub-section)\n{hint_block}\n\n" + f"BULLETS TO VERIFY\n" + "\n\n".join(fact_lines) + ) + + +# ─────────────────────────────────────────────────────────── parsing ── + + +_ROW_SPLIT_RE = re.compile(r"\}\s*,\s*\{") +_I_RE = re.compile(r'"i"\s*:\s*(\d+)') +_VERDICT_RE = re.compile(r'"verdict"\s*:\s*"([A-Za-z-]+)"') +_NOTE_RE = re.compile(r'"note"\s*:\s*"(.*?)"\s*,\s*"(?:rewrite|restates|subsection)"', re.S) +_REWRITE_RE = re.compile(r'"rewrite"\s*:\s*"(.*?)"\s*,\s*"(?:restates|subsection|note)"', re.S) +_REWRITE_LAST_RE = re.compile(r'"rewrite"\s*:\s*"(.*?)"\s*$', re.S) +_RESTATES_RE = re.compile(r'"restates"\s*:\s*(null|"?\d+"?)') +_SUBSECTION_RE = re.compile(r'"subsection"\s*:\s*(null|"[a-z_]+")') + + +def _recover_rows(s: str) -> list[dict]: + """Structure-anchored recovery when the reply is almost-JSON: a note + quoting the source («la "montille"») carries unescaped double quotes + that break `json.loads`. Each row's fields are located by their + key and the key that follows, the way `llm_json` recovers facts.""" + start = s.find('"facts"') + body = s[start:] if start >= 0 else s + rows: list[dict] = [] + for chunk in _ROW_SPLIT_RE.split(body): + m_i = _I_RE.search(chunk) + m_v = _VERDICT_RE.search(chunk) + if not (m_i and m_v): + continue + m_n = _NOTE_RE.search(chunk) + m_r = _REWRITE_RE.search(chunk) or _REWRITE_LAST_RE.search(chunk) + m_s = _RESTATES_RE.search(chunk) + m_sub = _SUBSECTION_RE.search(chunk) + rows.append({ + "i": int(m_i.group(1)), + "verdict": m_v.group(1), + "note": m_n.group(1).replace('\\"', '"') if m_n else "", + "rewrite": m_r.group(1).replace('\\"', '"') if m_r else "", + "restates": None if (not m_s or m_s.group(1) == "null") else m_s.group(1).strip('"'), + "subsection": None if (not m_sub or m_sub.group(1) == "null") else m_sub.group(1).strip('"'), + }) + return rows + + +def parse_verdicts(raw: str, n_facts: int) -> tuple[list[dict] | None, str | None]: + """{i → verdict row} as a list aligned on fact index, or (None, error). + Tolerant of the model omitting a bullet (treated as supported) and of + unescaped quotes inside a note (structure-anchored recovery), but not + of a reply with no gradable rows.""" + s = llm_json.strip_fences(raw or "") + rows = None + try: + data = json.loads(s) + rows = data.get("facts") if isinstance(data, dict) else None + except ValueError: + m = re.search(r"\{.*\}", s, re.S) + if not m: + return None, "no JSON object in reply" + try: + data = json.loads(m.group(0)) + rows = data.get("facts") if isinstance(data, dict) else None + except ValueError: + rows = _recover_rows(m.group(0)) or None + if rows is None: + return None, "unparseable JSON and no recoverable rows" + if not isinstance(rows, list): + return None, "no `facts` list" + out: list[dict] = [{"verdict": "supported", "note": "", "rewrite": "", "restates": None, "subsection": None} + for _ in range(n_facts)] + seen = 0 + for r in rows: + if not isinstance(r, dict): + continue + try: + i = int(r.get("i")) + except (TypeError, ValueError): + continue + if not 0 <= i < n_facts: + continue + verdict = str(r.get("verdict") or "supported").strip().lower() + if verdict not in VERDICTS: + verdict = "supported" + restates = r.get("restates") + try: + restates = int(restates) if restates is not None and str(restates) != "" else None + except (TypeError, ValueError): + restates = None + sub = r.get("subsection") + sub = sub if sub in SUBSECTION_KEYS else None + out[i] = { + "verdict": verdict, + "note": " ".join(str(r.get("note") or "").split())[:400], + "rewrite": " ".join(str(r.get("rewrite") or "").split()), + "restates": restates, + "subsection": sub, + } + seen += 1 + if seen == 0 and n_facts: + return None, "reply graded none of the bullets" + return out, None + + +# ──────────────────────────────────────────────────────── application ── + + +def rewrite_ok(original: str, rewrite: str, source: str) -> str | None: + """None when the rewrite is admissible, else the reason it is not: it + must differ, be one sane sentence, carry no arrow, and introduce no + number absent from the original bullet and the source.""" + rw = (rewrite or "").strip() + if not rw: + return "empty" + if rw == (original or "").strip(): + return "unchanged" + if "→" in rw: + return "arrow" + if len(rw) > REWRITE_MAX_CHARS or len(rw) < 20: + return "length" + new_nums = _numbers(rw) - _numbers(original) - _numbers(source or "") + if new_nums: + return f"new numbers {sorted(new_nums)}" + return None + + +def apply_verdicts( + facts: list[dict], verdicts: list[dict], *, source: str, source_lang: str = "", + run: str, model: str, +) -> dict: + """Apply the gate's verdicts to a record's facts. Returns + {facts, kept_indices, dropped, rewritten, moved, rejected_rewrites, + cosmetic_rewrites, missing_rewrites, text_changed}. Every kept fact + carries `support` ({verdict, note[, original_bullet][, moved_from]}); + dropped facts are listed with their reason so the cache and the + feedback history can record them. A `rewrite` verdict with no rewrite + text, or with a cosmetic one, keeps the original bullet as + `supported` (the note is kept, the case is listed).""" + kept: list[dict] = [] + kept_indices: list[int] = [] + dropped: list[dict] = [] + rewritten: list[dict] = [] + moved: list[dict] = [] + rejected: list[dict] = [] + cosmetic: list[dict] = [] + missing: list[dict] = [] + text_changed = False + for i, (fact, v) in enumerate(zip(facts, verdicts)): + verdict = v["verdict"] + note = v.get("note") or "" + # A `restates` pointing at a bullet that is itself dropped is not a + # duplicate any more — keep the later one unless it has its own reason. + if verdict == "drop" and v.get("restates") is not None: + target = v["restates"] + target_dropped = any(d["index"] == target for d in dropped) + if target_dropped and target != i: + verdict = "supported" + note = f"kept: it restated #{target}, which was dropped" + if verdict == "drop": + dropped.append({ + "index": i, "bullet": fact.get("bullet") or "", "subsection": fact.get("subsection"), + "note": note, "restates": v.get("restates"), + }) + continue + new = dict(fact) + support = {"verdict": verdict, "note": note, "gate": GATE_VERSION, "run": run, "model": model} + if verdict == "rewrite": + original = fact.get("bullet") or "" + proposed = (v.get("rewrite") or "").strip() + reason = rewrite_ok(original, proposed, source) + if not proposed: + support["verdict"] = "supported" + support["rewrite_missing"] = True + missing.append({"index": i, "note": note}) + elif is_cosmetic_rewrite(original, proposed): + support["verdict"] = "supported" + support["cosmetic_rewrite"] = proposed + cosmetic.append({"index": i, "from": original, "to": proposed, "note": note}) + elif reason is None: + new_bullet = normalize_bullet(v["rewrite"], source_lang) + support["original_bullet"] = fact.get("bullet") or "" + new["bullet"] = new_bullet + rewritten.append({"index": i, "from": fact.get("bullet") or "", "to": new_bullet, "note": note}) + text_changed = True + else: + support["verdict"] = "rewrite-rejected" + support["rejected_reason"] = reason + support["proposed_rewrite"] = v.get("rewrite") or "" + rejected.append({"index": i, "reason": reason, "note": note}) + sub = v.get("subsection") + if sub and sub != (fact.get("subsection") or "facteurs_naturels"): + support["moved_from"] = fact.get("subsection") or "facteurs_naturels" + new["subsection"] = sub + moved.append({"index": i, "from": support["moved_from"], "to": sub}) + new["support"] = support + kept.append(new) + kept_indices.append(i) + # The earned-interactions rule (R3), deterministic: a kept `interactions` + # bullet whose grounding quote states no link — or beyond the cap — moves + # to the natural factors; its causal wrapper, if any, was rewritten away + # by the verdict above. + for pos in unearned_indices(kept, source_lang): + f = kept[pos] + i = kept_indices[pos] + if f["support"].get("moved_from") == DEMOTION_TARGET: + # The verdict moved it INTO interactions and the earned rule sends + # it straight back: no move happened. + f["subsection"] = DEMOTION_TARGET + del f["support"]["moved_from"] + moved[:] = [m for m in moved if m["index"] != i] + continue + f["support"].setdefault("moved_from", "interactions") + f["support"]["unearned_interaction"] = True + f["subsection"] = DEMOTION_TARGET + moved.append({"index": i, "from": "interactions", "to": DEMOTION_TARGET}) + # A rewrite can make two bullets restate each other — collapse them. + dd = dedupe_facts(kept) + if dd.drops: + survivors = set(dd.kept_indices) + for pos, (orig_i, f) in enumerate(zip(kept_indices, kept)): + if pos not in survivors: + dropped.append({"index": orig_i, "bullet": f.get("bullet") or "", "subsection": f.get("subsection"), + "note": "duplicate after the gate (lexical dedupe)", "restates": None}) + kept_indices = [kept_indices[p] for p in dd.kept_indices] + kept = dd.kept + return { + "facts": kept, "kept_indices": kept_indices, "dropped": sorted(dropped, key=lambda d: d["index"]), + "rewritten": rewritten, "moved": moved, "rejected_rewrites": rejected, + "cosmetic_rewrites": cosmetic, "missing_rewrites": missing, + "text_changed": text_changed, + } diff --git a/scripts/_lib/terroir_interactions.py b/scripts/_lib/terroir_interactions.py new file mode 100644 index 0000000..28aaa78 --- /dev/null +++ b/scripts/_lib/terroir_interactions.py @@ -0,0 +1,453 @@ +"""The `interactions` sub-section is earned, not filled (review 2026-09-12, +R3). + +Every 02d prompt asks for a fourth sub-section — the causal terroir → wine +links — with a cap of one bullet, and the models fill it: the share sat +at 10.8 % of all bullets across two runs, one per record, and 7 of the +25 residual misleading bullets on the acceptance sample were causal +links the source never states. Measured on the r1 corpus, 69 % of those +quotes carry an explicit connective, 8 % carry one only in the bullet +(the manufactured link) and 23 % carry none. STYLE_RULES and the gate +both say the sub-section is admitted only when the source sentence itself +states the link; this module makes that a deterministic test on the +bullet's *grounding quote* (the source's words, never the bullet — a +bullet can add the "thanks to" the source lacks, and that is precisely +the failure mode): + + has_connective(text, lang) an explicit causal or consecutive + connective of `lang` — "grâce à", + "conferisce", "bedingt durch", "λόγω", … + earn_interactions(facts, lang) stage 02d, after its four calls: an + `interactions` fact whose quote carries + no connective is dropped (the fourth + call restates the other three when the + source states no link); at most + MAX_INTERACTIONS remain + demote_unearned(facts, lang) the gate, after its verdicts: such a + fact is moved to the natural factors + instead (its causal wrapper has been + rewritten away by then) + +The fourth call itself stays: the INAO cahier's section X.3 +"Interactions causales" and the EU single document's 8.4 are where the +regulator states the links, and dropping the call would lose them. +No bullet is ever promoted into `interactions` from the other +sub-sections — the lists below are deliberately broad (they include +"gives", "permet", "provides"), which is right for a drop / demote test +and wrong for a promotion test. A false negative drops a legitimate, +source-stated fact, the costlier error: the first table matched 69 % of +the r1 corpus's interactions quotes; the smoke's dropped bullets ("glavni +čimbenik", "fördert", "και έτσι", "hace que") and samples of the +unmatched FR / NL / RO quotes ("déterminent", "contribuant", "hetgeen +zich vertaalt in", "dă vinuri", "in quanto", "резултат от") drove it to +85 % with noun forms (factor / role / influence), word-initial verb +stems, the plain "because" conjunctions, Romanian cedilla folding and +an English fallback for Dutch records. The residual 15 % is the +manufactured link the rule exists for ("confèrent" in the bullet, +"marquée par" nowhere near it in the quote) plus non-causal +restatements. +""" + +from __future__ import annotations + +import re +from dataclasses import dataclass + +MAX_INTERACTIONS = 2 +_LANG_ALIAS = {"mt": "en", "at": "de", "ch": "fr", "lu": "fr", "gb": "en", "cz": "cs", "si": "sl", + "gr": "el", "cy": "el"} + +CONNECTIVES: dict[str, tuple[str, ...]] = { + "fr": ("grâce à", "grâce au", "grâce aux", "en raison de", "en raison du", "en raison des", + "du fait de", "du fait du", "du fait des", "à cause de", "sous l'effet", "sous l'influence", + "permet", "permettent", "permettant", "confère", "confèrent", "conférant", "favorise", + "favorisent", "favorisant", "explique", "expliquent", "entraîne", "entraînent", "induit", + "induisent", "se traduit", "se traduisent", "contribue", "contribuent", "résulte", + "résultent", "provient", "proviennent", "conduit à", "conduisent à", "génère", "génèrent", + "engendre", "engendrent", "procure", "procurent", "assure", "assurent", "garantit", + "garantissent", "à l'origine de", "d'où", "ainsi", "donc", "par conséquent", + "c'est pourquoi", "de ce fait", "influence", "influencent", "apporte", "apportent", + "donne", "donnent", "doit", "doivent", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "facteur", "facteurs", "rôle", "déterminant", "décisif", "décisive", "responsable", + "clé pour", "donnant", "apportant", "assurant", "garantissant", "expliquant", + "entraînant", "induisant", "générant", "de sorte que", "si bien que", "ce qui", + "sous l'action", "dépend", "dépendent", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "détermin", "contribu", "entraîn", "entrain", "engendr", "condui", "confèr", "confér", + "favoris", "permet", "permett", "expliqu", "indui", "génèr", "génér", "procur", "assur", + "garanti", "apport", "résult", "provien", "provenan", "tradui", "influenc", "dépend", + "issu de", "issus de", "issue de", "issues de", "lié à", "liée à", "liés à", "liées à", + "lié au", "liée au", "liés au", "liées au", "en lien", "responsab", "se reflèt", + "se traduis", "s'exprim", + # conjunctions and "result of / in combination with" forms + "marqué par", "marquée par", "marqués par", "marquées par", "parce que", "puisque", + "car ", "étant donné", "du fait", "en effet"), + "it": ("grazie a", "grazie al", "grazie alla", "grazie ai", "grazie alle", "a causa di", + "per effetto di", "per effetto del", "in virtù di", "dovuto a", "dovuta a", "dovuti a", + "dovute a", "conferisce", "conferiscono", "conferendo", "consente", "consentono", + "permette", "permettono", "favorisce", "favoriscono", "favorendo", "determina", + "determinano", "determinando", "spiega", "spiegano", "comporta", "comportano", + "contribuisce", "contribuiscono", "deriva", "derivano", "pertanto", "quindi", + "di conseguenza", "garantisce", "garantiscono", "assicura", "assicurano", "apporta", + "apportano", "influisce", "influiscono", "influenza", "influenzano", "esalta", + "esaltano", "dona", "donano", "consentendo", "responsabile di", "responsabili di", + "si traduce", "si traducono", "genera", "generano", "induce", "inducono", "ne deriva", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "fattore", "fattori", "ruolo", "determinante", "determinanti", "decisiv", "donando", + "permettendo", "garantendo", "assicurando", "esaltando", "influenzando", "contribuendo", + "generando", "inducendo", "così", "in modo che", "tale da", "tali da", "grazie", + "dipende", "dipendono", "condiziona", "condizionano", "condizionando", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "conferi", "consent", "permett", "favor", "determin", "spieg", "comport", "contribu", + "deriv", "garant", "assicur", "apport", "influ", "esalt", "gener", "induc", "condizion", + "dipend", "legat", "si rifle", "si esprim", "responsabil", + # conjunctions and "result of / in combination with" forms + "in quanto", "sostenut", "poiché", "perché", "dato che", "visto che", "siccome", + "infatti"), + "es": ("gracias a", "debido a", "a causa de", "por efecto de", "confiere", "confieren", + "permite", "permiten", "favorece", "favorecen", "determina", "determinan", "explica", + "explican", "aporta", "aportan", "contribuye", "contribuyen", "provoca", "provocan", + "da lugar", "dan lugar", "se traduce", "se traducen", "por lo que", "por tanto", + "por ello", "en consecuencia", "influye", "influyen", "garantiza", "garantizan", + "otorga", "otorgan", "propicia", "propician", "proporciona", "proporcionan", + "condiciona", "condicionan", "origina", "originan", "responsable de", "responsables de", + "se debe a", "se deben a", "consecuencia de", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "hace que", "hacen que", "haciendo que", "factor", "factores", "papel", "influencia", + "clave", "decisiv", "determinante", "determinantes", "de ahí", "así", + "por consiguiente", "generando", "permitiendo", "confiriendo", "aportando", + "dando lugar", "favoreciendo", "contribuyendo", "depende", "dependen", "gracias", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "confier", "permit", "favorec", "determin", "explic", "aport", "contribu", "provoc", + "influ", "garantiz", "otorg", "propici", "proporcion", "condicion", "origin", "depend", + "vincul", "ligad", "se refle", "se expres", "se manifiest", "responsab", + # conjunctions and "result of / in combination with" forms + "asegur", "result", "relación", "puesto que", "ya que", "porque", "dado que", + "de hecho"), + "pt": ("graças a", "graças à", "graças ao", "devido a", "devido à", "devido ao", "por causa de", + "confere", "conferem", "permite", "permitem", "favorece", "favorecem", "determina", + "determinam", "explica", "explicam", "contribui", "contribuem", "proporciona", + "proporcionam", "resulta", "resultam", "traduz-se", "traduzem-se", "por isso", + "consequentemente", "influencia", "influenciam", "origina", "originam", "garante", + "garantem", "conduz", "conduzem", "dá origem", "dão origem", "deve-se", "devem-se", + "responsável por", "responsáveis por", "condiciona", "condicionam", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "fator", "fatores", "factor", "factores", "papel", "influência", "determinante", + "decisiv", "chave", "conferindo", "permitindo", "favorecendo", "contribuindo", + "proporcionando", "originando", "resultando", "assim", "de modo que", "faz com que", + "fazem com que", "depende", "dependem", "graças", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "confer", "permit", "favorec", "determin", "explic", "contribu", "proporcion", "result", + "influenc", "origin", "garant", "conduz", "condicion", "depend", "ligad", "associad", + "reflet", "se express", "se manifest", "responsáv", + # conjunctions and "result of / in combination with" forms + "assegur", "relação", "porque", "pois", "já que", "uma vez que", "visto que", + "de facto", "de fato"), + "de": ("dank", "aufgrund", "auf grund", "wegen", "infolge", "bedingt durch", "bedingt", + "bedingen", "bewirkt", "bewirken", "verleiht", "verleihen", "ermöglicht", "ermöglichen", + "begünstigt", "begünstigen", "führt zu", "führen zu", "sorgt für", "sorgen für", + "trägt bei", "tragen bei", "beitragen", "prägt", "prägen", "erklärt", "resultiert", + "resultieren", "ergibt sich", "ergeben sich", "daher", "deshalb", "somit", "dadurch", + "hierdurch", "wodurch", "folglich", "verantwortlich für", "beeinflusst", "beeinflussen", + "zurückzuführen", "verdankt", "verdanken", "hervorbringt", "hervorbringen", + "mit sich bringt", "zur folge", "bestimmt", "bestimmen", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "fördert", "fördern", "fördernd", "wirkt sich", "wirken sich", "auswirkung", + "auswirkungen", "einfluss", "einflüsse", "faktor", "faktoren", "bedeutung für", + "entscheidend", "maßgeblich", "prägend", "ursache", "verursacht", "verursachen", + "erlaubt", "erlauben", "unterstützt", "unterstützen", "abhängig", "hängt", "hängen", + "grundlage für", "voraussetzung", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "bewirk", "verleih", "ermöglich", "begünstig", "beeinfluss", "bestimm", "präg", + "förder", "verursach", "unterstütz", "bedeut", "resultier", "beding", "abhäng", "beruh", + "zurückzuführ", "verdank", "beitr", "widerspiegel", "spiegelt", "spiegeln", + "äußert sich", "äußern sich", "zeigt sich", "zeigen sich", "verantwortlich", + # conjunctions and "result of / in combination with" forms + "weil", "weshalb", "sodass", "so dass", "damit", "somit", "zumal", "nämlich"), + "nl": ("dankzij", "vanwege", "als gevolg van", "ten gevolge van", "zorgt voor", "zorgen voor", + "zorgt ervoor", "zorgen ervoor", "verleent", "verlenen", "leidt tot", "leiden tot", + "resulteert in", "resulteren in", "draagt bij", "dragen bij", "waardoor", "daardoor", + "hierdoor", "bevordert", "bevorderen", "verklaart", "verklaren", "bepaalt", "bepalen", + "beïnvloedt", "beïnvloeden", "te danken aan", "toe te schrijven aan", "daarom", + "mogelijk maakt", "mogelijk maken", "geeft", "geven", "veroorzaakt", "veroorzaken", + "verantwoordelijk voor", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "factor", "factoren", "rol", "invloed", "bepalend", "doorslaggevend", "cruciaal", + "essentieel", "zodat", "maakt dat", "maken dat", "hangt af", "hangen af", "afhankelijk", + "basis voor", "voorwaarde", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "ondersteun", "vertaal", "oplever", "lever", "maakt het mogelijk", "maken het mogelijk", + "uit zich", "uiten zich", "tot uiting", "zo kan", "zo kunnen", "zo ontsta", "bijdra", + "beïnvloed", "bepal", "leid", "resulteer", "afhankelijk", "hangt", "hangen", + "te danken", "toe te schrijven", "verklaar", "verleen", "weerspiegel", "gevolg", + "verantwoordelijk", + # conjunctions and "result of / in combination with" forms + "omdat", "doordat", "want ", "aangezien", "immers"), + "en": ("thanks to", "due to", "because", "owing to", "as a result", "results in", "result in", + "resulting in", "gives", "give", "giving", "confers", "confer", "conferring", "allows", + "allow", "allowing", "enables", "enable", "enabling", "favours", "favors", "favour", + "favor", "explains", "explain", "leads to", "lead to", "leading to", "contributes", + "contribute", "contributing", "therefore", "hence", "thus", "consequently", + "influences", "influence", "responsible for", "attributable to", "imparts", "impart", + "ensures", "ensure", "promotes", "promote", "determines", "determine", "provides", + "provide", "creates", "create", "brings", "bring", "helps", "help", "means that", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "factor", "factors", "role", "key to", "shapes", "shape", "shaping", "drives", "drive", + "driving", "owes", "owe", "owing", "makes", "make", "making", "so that", "affects", + "affect", "affecting", "decisive", "determining", "crucial", "essential", "underpins", + "underpin", "depends", "depend", "dependent", "basis for", "conducive", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "confer", "allow", "enabl", "favour", "favor", "explain", "contribut", "influenc", + "impart", "ensur", "promot", "determin", "provid", "creat", "help", "shap", "driv", + "affect", "depend", "underpin", "result", "lead", "led to", "yield", "support", + "linked to", "link between", "reflect", "express", "translat", "manifest", "responsib", + "attributable", "owing", "thanks", "because", "due to", "conducive", + # conjunctions and "result of / in combination with" forms + "as a consequence", "for this reason", "that is why", "indeed"), + "el": ("χάρη σ", "λόγω", "εξαιτίας", "οφείλεται", "οφείλονται", "προσδίδει", "προσδίδουν", + "επιτρέπει", "επιτρέπουν", "ευνοεί", "ευνοούν", "συμβάλλει", "συμβάλλουν", "οδηγεί", + "οδηγούν", "καθορίζει", "καθορίζουν", "εξηγεί", "εξηγούν", "με αποτέλεσμα", + "ως αποτέλεσμα", "κατά συνέπεια", "συνεπώς", "επομένως", "επηρεάζει", "επηρεάζουν", + "εξασφαλίζει", "εξασφαλίζουν", "δίνει", "δίνουν", "χαρίζει", "χαρίζουν", "αποδίδει", + "αποδίδουν", "διαμορφώνει", "διαμορφώνουν", "προσφέρει", "προσφέρουν", "υπεύθυν", + "δημιουργεί", "δημιουργούν", "επιδρ", "συντελεί", "συντελούν", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "έτσι", "παράγοντ", "επίδρασ", "ρόλο", "αιτία", "χάρις", "ώστε", "καθοριστικ", + "αποφασιστικ", "εξαρτ", "βασίζ", "προϋπόθεση", "με τη σειρά", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "προσδίδ", "επιτρέπ", "ευνο", "συμβάλλ", "οδηγ", "καθορίζ", "εξηγ", "επηρε", + "εξασφαλίζ", "χαρίζ", "αποδίδ", "διαμορφών", "προσφέρ", "δημιουργ", "συντελ", "οφείλ", + "εξαρτ", "ευθύν", "συνδέ", "αντανακλ", "εκφράζ", "μεταφράζ", "καθιστ", + # conjunctions and "result of / in combination with" forms + "συνδυασμ", "ανάλογα", "επειδή", "διότι", "καθώς", "αφού", "γι' αυτό", "γι’ αυτό"), + "bg": ("благодарение на", "поради", "вследствие", "в резултат", "се дължи", "се дължат", + "придава", "придават", "позволява", "позволяват", "благоприятства", "благоприятстват", + "допринася", "допринасят", "води до", "водят до", "определя", "определят", "обуславя", + "обуславят", "затова", "следователно", "влияе", "влияят", "осигурява", "осигуряват", + "обяснява", "обясняват", "формира", "формират", "създава", "създават", "спомага", + "спомагат", "отговорн", "предпоставка", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "фактор", "фактори", "роля", "влияние", "определящ", "решаващ", "ключов", "така", + "по този начин", "предопредел", "зависи", "зависят", "основа за", "условие за", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "придав", "позволяв", "благоприят", "допринас", "определ", "обуслав", "влия", + "осигуряв", "обясняв", "формир", "създав", "спомаг", "завис", "дълж", "свързан", + "отраз", "изразяв", "предпостав", "благодарение", "поради", + # conjunctions and "result of / in combination with" forms + "резултат", "съчетани", "защото", "тъй като", "понеже", "ето защо"), + "hu": ("köszönhető", "köszönhetően", "miatt", "következtében", "eredményeként", + "eredményeképpen", "hatására", "biztosít", "biztosítja", "biztosítják", "lehetővé tesz", + "lehetővé teszi", "lehetővé teszik", "kedvez", "kedveznek", "hozzájárul", + "hozzájárulnak", "eredményez", "eredményezi", "eredményezik", "meghatároz", + "meghatározza", "meghatározzák", "magyaráz", "magyarázza", "ezért", "így", "tehát", + "ennélfogva", "befolyásol", "befolyásolja", "befolyásolják", "kölcsönöz", "kölcsönöznek", + "okoz", "okozza", "okozzák", "alakít", "alakítja", "alakítják", "teszi lehetővé", + "felelős", "elősegít", "elősegíti", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "tényező", "tényezők", "szerep", "hatás", "döntő", "kulcs", "ezáltal", "függ", + "függenek", "alapja", "feltétele", "meghatározó", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "biztosít", "lehetővé", "kedvez", "hozzájárul", "eredményez", "meghatároz", "magyaráz", + "befolyásol", "kölcsönöz", "okoz", "alakít", "elősegít", "függ", "köszönhet", "tükröz", + "megnyilvánul", "összefügg", "kapcsolat", "hatás", "adja", "adják", "ad a", + # conjunctions and "result of / in combination with" forms + "garantál", "adódó", "jelenti", "alapj", "megteremt", "mivel", "mert", "hiszen", + "ugyanis"), + "cs": ("díky", "kvůli", "v důsledku", "vlivem", "následkem", "dodává", "dodávají", "umožňuje", + "umožňují", "podporuje", "podporují", "přispívá", "přispívají", "vede k", "vedou k", + "určuje", "určují", "způsobuje", "způsobují", "vysvětluje", "proto", "tudíž", "tedy", + "takže", "ovlivňuje", "ovlivňují", "zajišťuje", "zajišťují", "propůjčuje", "propůjčují", + "má za následek", "mají za následek", "podmiňuje", "podmiňují", "utváří", "vytváří", + "zodpovědn", "zodpovídá", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "faktor", "faktory", "role", "vliv", "určující", "rozhodující", "klíčov", "závisí", + "závislý", "závislé", "základ pro", "předpoklad", "podmínk", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "dodáv", "umožň", "podporuj", "přispív", "vede k", "vedou k", "určuj", "způsobuj", + "vysvětl", "ovlivň", "zajišť", "propůjč", "podmiň", "utvář", "vytvář", "závis", "odráž", + "projevuj", "souvis", "spojen", "díky", "dává", "dávají", + # conjunctions and "result of / in combination with" forms + "protože", "neboť", "jelikož", "výsledk", "kombinac"), + "sk": ("vďaka", "kvôli", "v dôsledku", "vplyvom", "následkom", "dodáva", "dodávajú", "umožňuje", + "umožňujú", "podporuje", "podporujú", "prispieva", "prispievajú", "vedie k", "vedú k", + "určuje", "určujú", "spôsobuje", "spôsobujú", "vysvetľuje", "preto", "teda", "takže", + "ovplyvňuje", "ovplyvňujú", "zabezpečuje", "zabezpečujú", "prepožičiava", "prepožičiavajú", + "má za následok", "podmieňuje", "podmieňujú", "vytvára", "vytvárajú", "utvára", + "zodpovedn", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "faktor", "faktory", "rola", "úloha", "vplyv", "určujúc", "rozhodujúc", "kľúčov", + "závisí", "závislý", "závislé", "základ pre", "predpoklad", "podmienk", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "dodáv", "umožň", "podporuj", "prispiev", "vedie k", "vedú k", "určuj", "spôsobuj", + "vysvetľ", "ovplyvň", "zabezpeč", "prepožič", "podmieň", "utvár", "vytvár", "závis", + "odráž", "prejavuj", "súvis", "spojen", "vďaka", "dáva", "dávajú", + # conjunctions and "result of / in combination with" forms + "pretože", "lebo", "keďže", "výsledk", "kombináci"), + "sl": ("zaradi", "po zaslugi", "zahvaljujoč", "kot posledica", "posledično", "daje", "dajejo", + "omogoča", "omogočajo", "spodbuja", "spodbujajo", "prispeva", "prispevajo", "vodi do", + "vodijo do", "določa", "določajo", "povzroča", "povzročajo", "pojasnjuje", "zato", + "torej", "vpliva", "vplivajo", "zagotavlja", "zagotavljajo", "oblikuje", "oblikujejo", + "ustvarja", "ustvarjajo", "odgovorn", "pogojuje", "pogojujejo", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "dejavnik", "dejavniki", "vloga", "vpliv", "odločiln", "ključn", "odvisn", "osnova za", + "pogoj", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "omogoč", "spodbuj", "prispev", "vodi", "določ", "povzroč", "pojasn", "vpliv", + "zagotav", "oblikuj", "ustvarj", "odvisn", "odraž", "kaže se", "kažejo se", "izraž", + "povez", "zaradi", "daj", + # conjunctions and "result of / in combination with" forms + "ker ", "saj ", "kajti", "rezultat", "kombinacij"), + "hr": ("zahvaljujući", "zbog", "uslijed", "kao posljedica", "posljedično", "daje", "daju", + "omogućuje", "omogućuju", "omogućava", "omogućavaju", "pogoduje", "pogoduju", "pridonosi", + "pridonose", "doprinosi", "doprinose", "dovodi do", "dovode do", "određuje", "određuju", + "uzrokuje", "uzrokuju", "objašnjava", "stoga", "zato", "dakle", "utječe", "utječu", + "osigurava", "osiguravaju", "oblikuje", "oblikuju", "stvara", "stvaraju", "rezultira", + "rezultiraju", "odgovorn", "uvjetuje", "uvjetuju", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "čimben", "faktor", "faktori", "utjecaj", "ulog", "uzrok", "razlog", "zaslug", + "preduvjet", "ključn", "presudn", "odlučuj", "ovisi", "ovise", "temelj", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "omoguć", "pogod", "pridonos", "doprinos", "dovod", "određ", "uzrok", "objašnj", "utje", + "osigur", "oblikuj", "stvar", "rezultir", "uvjet", "ovis", "odraž", "očituj", "povez", + "zahvaljuj", "zaslug", "daj", + # conjunctions and "result of / in combination with" forms + "jer ", "budući", "pošto", "s obzirom", "rezultat", "kombinacij"), + "ro": ("datorită", "din cauza", "ca urmare", "drept urmare", "ca rezultat", "conferă", + "permite", "permit", "favorizează", "contribuie", "duce la", "duc la", "conduce la", + "conduc la", "determină", "explică", "generează", "asigură", "de aceea", "prin urmare", + "astfel", "influențează", "imprimă", "oferă", "rezultă", "se datorează", "se datorează", + "responsabil", "condiționează", "face posibil", "fac posibil", + # noun, gerund and consecutive forms (2026-09-14 smoke: "glavni čimbenik", + # "fördert", "και έτσι", "hace que" were dropped as unearned) + "factor", "factori", "rol", "influenț", "determinant", "decisiv", "cheie", "depinde", + "depind", "face ca", "fac ca", "conferind", "permițând", "favorizând", "contribuind", + "asigurând", "generând", "baza", + # word-initial stems (the inflections and gerunds the smoke and the FR / NL / RO + # samples showed missing: "déterminent", "contribuant", "vertaalt zich") + "confer", "permit", "favoriz", "contribu", "duce la", "duc la", "conduc", "determin", + "explic", "gener", "asigur", "influenţ", "influenț", "imprim", "ofer", "rezult", + "datorit", "depind", "facilit", "înseamnă", "face din", "fac din", "dă", "dau", + "condiţii în care", "condiții în care", "reflect", "se regăs", "legat de", "legătur", + "condiţion", "condiț", + # conjunctions and "result of / in combination with" forms + "deoarece", "pentru că", "întrucât", "fiindcă", "relaţi", "relați", "combinaţi", + "combinați", "asigur"), +} + +_COMPILED: dict[str, re.Pattern[str]] = {} + + +def _pattern(lang: str) -> re.Pattern[str] | None: + lang = _LANG_ALIAS.get(lang, lang) + if lang in _COMPILED: + return _COMPILED[lang] + terms = CONNECTIVES.get(lang) + if not terms: + return None + # Word-initial anchoring only: many entries are stems ("επιδρ", "odgovorn"). + alts = "|".join(re.escape(t) for t in sorted(terms, key=len, reverse=True)) + pat = re.compile(rf"(?<![^\W\d_])(?:{alts})", re.IGNORECASE) + _COMPILED[lang] = pat + return pat + + +_RO_CEDILLA = str.maketrans({"ţ": "ț", "ş": "ș", "Ţ": "Ț", "Ş": "Ș"}) + + +def has_connective(text: str, lang: str) -> bool: + """True when `text` (a source quote) carries an explicit causal or + consecutive connective of `lang`. Unknown language → False. Romanian + text is folded to comma-below (ț / ș) first — the sources mix both + spellings; a Dutch record is also tried against the English table + (Ambt Delden's single document exists only in English).""" + if not text: + return False + pat = _pattern(lang) + if not pat: + return False + if lang == "ro": + text = text.translate(_RO_CEDILLA) + if pat.search(text): + return True + return lang == "nl" and bool(_pattern("en").search(text)) + + +def quote_has_connective(fact: dict, lang: str) -> bool: + """The grounding quote states a link: the cahier quote when the fact + is cahier-grounded, the Wikipedia quote when wiki-grounded.""" + prov = fact.get("provenance") or "cahier" + if prov in ("cahier", "both") and has_connective(fact.get("cahier_quote") or "", lang): + return True + return prov in ("wiki", "both") and has_connective(fact.get("wiki_quote") or "", lang) + + +DEMOTION_TARGET = "facteurs_naturels" + + +@dataclass +class EarnResult: + kept: list[dict] + dropped: list[dict] + + +def earn_interactions(facts: list[dict], lang: str, *, max_interactions: int = MAX_INTERACTIONS) -> EarnResult: + """Stage 02d: keep an `interactions` fact only when its quote carries + a connective, at most `max_interactions` of them (in order); the rest + are dropped. Other sub-sections pass through untouched.""" + kept: list[dict] = [] + dropped: list[dict] = [] + n = 0 + for f in facts: + if (f.get("subsection") or DEMOTION_TARGET) != "interactions": + kept.append(f) + continue + if quote_has_connective(f, lang) and n < max_interactions: + n += 1 + kept.append(f) + else: + dropped.append(f) + return EarnResult(kept, dropped) + + +def unearned_indices(facts: list[dict], lang: str, *, max_interactions: int = MAX_INTERACTIONS) -> list[int]: + """Indices of the `interactions` facts the earned rule rejects: quote + without a connective, or beyond the cap.""" + out: list[int] = [] + n = 0 + for i, f in enumerate(facts): + if (f.get("subsection") or DEMOTION_TARGET) != "interactions": + continue + if quote_has_connective(f, lang) and n < max_interactions: + n += 1 + else: + out.append(i) + return out diff --git a/scripts/_lib/terroir_normalize.py b/scripts/_lib/terroir_normalize.py new file mode 100644 index 0000000..6d6760f --- /dev/null +++ b/scripts/_lib/terroir_normalize.py @@ -0,0 +1,215 @@ +"""Deterministic clean-up of terroir-fact bullets (stage-04 render + the +`normalize_terroir_facts.py` cache post-pass). + +Three mechanical defects the 2026-09-11 review found in 8–12 % of bullets, +none of which needs a model to fix: + +- regulatory colour codes copied from the cahier's grape roster + ("Pinot noir N", "Riesling B", "Gewurztraminer Rs") — stripped when the + preceding words resolve to a grape in the lexicon, so a stray capital + after a place name is left alone; +- the INAO mentions "VT" / "SGN" left unexpanded — spelled out; +- no terminal punctuation — a period is appended; +- residual Greek / Cyrillic script in a Latin-target translation: a + homoglyph inside a Latin word ("Thermoheliоhydric" with a Cyrillic о, + "Piniatorοs" with a Greek ο) is mapped to its Latin lookalike, and a + whole non-Latin token left as a gloss ("(ξερολιθιές)", "(смолница)", + "Мискет врачански") is transliterated with unidecode. Only applied to + the four target locales, only when the bullet is predominantly Latin + (a Greek source bullet rendered untranslated is left alone), and never + to a Greek-letter chemical prefix ("α-terpineol"). + +Arrows, label prefixes and "according to the document" are left to a +targeted re-extraction under the prompt rules in `terroir_prompts`: they +need rewording, not a regex. +""" + +from __future__ import annotations + +import re + +from unidecode import unidecode + +from _lib import grape_entity, grape_lexicon + +COLOUR_CODES = ("N", "B", "G", "Rs", "Rg") +_CODE_RE = re.compile(r"(?:\s+|\s*\()(N|B|G|Rs|Rg)\)?(?=[\s,;:.)\]/]|$)") +# apostrophes split ("l'ugni blanc" → l / ugni / blanc) so the elided article never hides a name +_WORD_RE = re.compile(r"[\w\-]+", re.UNICODE) +_MAX_NAME_WORDS = 4 + +_EXPANSIONS = [ + (re.compile(r"\bVT\s*/\s*SGN\b"), "Vendanges Tardives / Sélection de Grains Nobles"), + (re.compile(r"\bVT\b"), "Vendanges Tardives"), + (re.compile(r"\bSGN\b"), "Sélection de Grains Nobles"), +] +_TERMINAL = ".!?…" +_TARGET_LOCALES = ("en", "fr", "es", "nl") +_NON_LATIN = re.compile(r"[Ѐ-ӿͰ-Ͽ]") +_LETTER = re.compile(r"[^\W\d_]", re.UNICODE) +_CHEM_PREFIX = re.compile(r"^[αβγδ]-\w") +_HOMOGLYPH = str.maketrans({ + "а": "a", "е": "e", "о": "o", "р": "p", "с": "c", "у": "y", "х": "x", "і": "i", "ј": "j", + "к": "k", "м": "m", "т": "t", "н": "n", "в": "v", "ѕ": "s", "ԁ": "d", "ԛ": "q", + "А": "A", "Е": "E", "О": "O", "Р": "P", "С": "C", "У": "Y", "Х": "X", "І": "I", "Ј": "J", + "К": "K", "М": "M", "Т": "T", "Н": "H", "В": "B", "Ѕ": "S", + "ο": "o", "ς": "s", "σ": "s", "α": "a", "ε": "e", "ι": "i", "κ": "k", "ν": "n", "ρ": "p", + "τ": "t", "υ": "u", "χ": "x", "Α": "A", "Β": "B", "Ε": "E", "Ζ": "Z", "Η": "H", "Ι": "I", + "Κ": "K", "Μ": "M", "Ν": "N", "Ο": "O", "Ρ": "P", "Τ": "T", "Υ": "Y", "Χ": "X", +}) +_HOMOGLYPH_CHARS = {chr(k) for k in _HOMOGLYPH} + + +_VOCAB = None + + +def _is_grape(name: str) -> bool: + """True when `name` is a known variety surface: the grape matcher's exact + index (VIVC prime names + synonyms + the lexicon) first, then the + lexicon tables for surfaces the index does not carry.""" + global _VOCAB + if _VOCAB is None: + _VOCAB = grape_entity._load_vocabulary() + if grape_entity._normalise(name) in _VOCAB.exact_index: + return True + slug = grape_lexicon._canonical_slug(name) + return bool(slug) and (slug in grape_lexicon.DEFAULT_COLOUR or slug in grape_lexicon.GRAPE_ALIAS) + + +def strip_colour_codes(bullet: str) -> str: + """Remove ` N` / ` B` / ` G` / ` Rs` / ` Rg` (also `(B)`) after a grape name.""" + out = bullet + pos = 0 + while True: + m = _CODE_RE.search(out, pos) + if not m: + return out + before = out[: m.start()] + words = _WORD_RE.findall(before)[-_MAX_NAME_WORDS:] + hit = any(_is_grape(" ".join(words[i:])) for i in range(len(words))) + if hit: + out = before + out[m.end():] + pos = m.start() + else: + pos = m.end() + + +def expand_mentions(bullet: str) -> str: + for rx, full in _EXPANSIONS: + bullet = rx.sub(full, bullet) + return bullet + + +# A bullet that ends by citing its own source — ", secondo il disciplinare." +# — despite the prompt's rule; the clause is dropped, the fact stays. +_TRAILING_META_RE = re.compile( + r"[,;]?\s+(?:secondo (?:il|quanto (?:previsto|indicato|riportato) (?:dal|nel)) disciplinare" + r"|come (?:indicato|previsto|riportato) (?:dal|nel) disciplinare" + r"|selon le cahier des charges|comme (?:l'indique|le précise|le prévoit) le cahier des charges" + r"|según (?:el|lo (?:establecido|indicado) en el) pliego(?: de condiciones)?" + r"|laut (?:der |dem )?(?:produktspezifikation|einzige[nm] dokument)" + r"|gemäß (?:der |dem )?(?:produktspezifikation|einzige[nm] dokument)" + r"|volgens het (?:productdossier|enig document)|conform het (?:productdossier|enig document)" + r"|segundo o caderno(?: de especificações)?|de acordo com o caderno(?: de especificações)?" + r"|(?:according to|as (?:stated|set out|indicated) in) the (?:production |product )?(?:specification|document|cahier|disciplinare|pliego)" + r"|selon le (?:document unique|pliego)|según el documento único|laut (?:dem )?(?:einzigen dokument|produktdossier))" + r"\s*(?=[.!?…]?\s*$)", + re.IGNORECASE, +) + + +def strip_trailing_meta(bullet: str) -> str: + return _TRAILING_META_RE.sub("", bullet or "") + + +def ensure_terminal_period(bullet: str) -> str: + b = bullet.rstrip() + if not b or b[-1] in _TERMINAL: + return b + return b + "." + + +def _latinize_token(token: str) -> str: + if _CHEM_PREFIX.match(token): + return token + letters = _LETTER.findall(token) + non_latin = [c for c in letters if _NON_LATIN.match(c)] + if not non_latin: + return token + if len(non_latin) * 2 < len(letters) and all(c in _HOMOGLYPH_CHARS for c in non_latin): + return token.translate(_HOMOGLYPH) # a stray lookalike inside a Latin word + return "".join(unidecode(c) if _NON_LATIN.match(c) else c for c in token) + + +def latinize_residual_script(bullet: str) -> str: + """Latin-script a mostly-Latin bullet: homoglyphs mapped, whole + non-Latin tokens transliterated. A bullet whose letters are ≥ 80 % + non-Latin is left untouched — it is an untranslated source bullet, not + a residue.""" + letters = _LETTER.findall(bullet) + if not letters: + return bullet + share = sum(1 for c in letters if _NON_LATIN.match(c)) / len(letters) + if share == 0 or share >= 0.8: + return bullet + return re.sub(r"\S+", lambda m: _latinize_token(m.group(0)), bullet) + + +# Dutch: the common noun is "appellatie" (the UI's own word — "Appellatie +# zoeken…", "{p} appellaties"); the French loanword stayed in 260 NL bullets +# ("De appellation is gelegen in de streek Revermont", 2026-09-15). The +# registered term "appellation d'origine contrôlée / protégée" is a name +# and stays. +_NL_APPELLATION_RE = re.compile(r"\b([Aa])ppellation(s?)\b(?!\s+d[’']origine)") + +_LOCALE_FIXES = { + "nl": lambda b: _NL_APPELLATION_RE.sub(r"\1ppellatie\2", b), +} + + +def locale_fixes(bullet: str, lang: str) -> str: + fix = _LOCALE_FIXES.get(lang) + return fix(bullet) if fix else bullet + + +def normalize_bullet(bullet: str, lang: str = "") -> str: + """Apply every deterministic fix. `lang` is the locale the bullet is + rendered in: for the four target locales residual Greek / Cyrillic + script is also Latinised.""" + if not bullet: + return bullet + out = strip_trailing_meta(expand_mentions(strip_colour_codes(bullet))) + if lang in _TARGET_LOCALES: + out = latinize_residual_script(locale_fixes(out, lang)) + return ensure_terminal_period(out) + + +def normalize_facts(facts: list[dict], lang: str = "") -> int: + """Normalise `bullet` on every fact in place; returns how many changed.""" + n = 0 + for f in facts: + b = f.get("bullet") or "" + nb = normalize_bullet(b, lang) + if nb != b: + f["bullet"] = nb + n += 1 + return n + + +def normalize_aocs(aocs: dict[str, dict], lang: str = "") -> dict[str, dict]: + """Stage-04 hook: return a new aocs dict whose `terroir_facts.facts[].bullet` + are normalised. Records are copied, never mutated — the per-locale dicts + share record objects with the source `aocs`.""" + out: dict[str, dict] = {} + for slug, rec in aocs.items(): + tf = rec.get("terroir_facts") + facts = (tf or {}).get("facts") if isinstance(tf, dict) else None + if not facts: + out[slug] = rec + continue + new_facts = [{**f, "bullet": normalize_bullet(f.get("bullet") or "", lang)} for f in facts] + if all(a.get("bullet") == b.get("bullet") for a, b in zip(facts, new_facts)): + out[slug] = rec + continue + out[slug] = {**rec, "terroir_facts": {**tf, "facts": new_facts}} + return out diff --git a/scripts/_lib/terroir_prompts.py b/scripts/_lib/terroir_prompts.py new file mode 100644 index 0000000..c3880a9 --- /dev/null +++ b/scripts/_lib/terroir_prompts.py @@ -0,0 +1,222 @@ +"""Prompt fragments shared by the 21 stage-02d extraction scripts. + +Every country's `EXTRACT_SYSTEM` is written in its own source language, +which kept the *style* rules from being maintained in one place: arrows, +regulatory colour codes ("Pinot noir N"), unexpanded VT / SGN, bullets +that mention "the document", hedges strengthened to "exclusively", and +one sentence restated by three sub-section calls all slipped through +(2026-09-11 review). `STYLE_RULES` is one English block — the models read +it fine inside a Greek or Bulgarian prompt — and `with_style_rules` +splices it in front of the prompt's final paragraph (the JSON-only +instruction), so every script carries the same rules. + +The block opens with the claim-support rule — the gate's own definition +of an over-claim (a causal wrapper on a co-occurrence, a narrowed +attribution, an invented qualifier, a sibling's statement, a +strengthened hedge) — so the extractor does the gate's job first: the +gate rewrote 27 % of the r1 bullets, 44 % of them light edits. +""" + +from __future__ import annotations + +STYLE_RULES = """\ +Claim-support rule — it comes first and every bullet is checked against it afterwards by a verifier that reads the whole source: +- A bullet asserts only what its quoted sentence itself states — every entity, number, unit, direction, spatial or temporal qualifier, hedge and causal link in the bullet must be in the quote. Concretely: never turn the mere presence of a factor into a cause ("volcanic soils give the wine minerality" is admissible only when the source says the soils give it — otherwise write "the soils are volcanic"); never credit one factor with what the source credits to several, and never narrow an attribution the source makes en bloc ("soils, climate and exposure" stays all three); never add a spatial or temporal qualifier ("on the upper slopes", "since the 1960s") the source does not give; never move a statement about a sub-zone, a single site or a neighbouring appellation onto the appellation as a whole; never strengthen or drop a hedge. When the sentence you would like to write goes beyond what the quote supports, write the narrower sentence the quote does support — or leave the fact out. + +Style rules (they apply whatever the language of the bullets): +- One fact per bullet, written as one full sentence that ends with a period. No arrows (→), no label prefixes such as "Colour:", "Climate:", "Soils:". +- A bullet is a complete sentence of roughly 120–220 characters. Never compress it into a telegraphic fragment ("Roero (syn. Tanaro)", "Climate: avg 9.8 °C"); a qualifier, a hedge or a spatial attribution is never dropped to save space — write the longer precise sentence instead. +- Prefer bullets that carry a specific named entity or figure — a named wind, lake, river, mountain, geological formation, an altitude, area or rainfall figure, a dated event, a named practice — over general statements. When the text you are given is long (over ~4,000 characters) and describes more such specifics than the maximum allows, you may return up to two extra bullets. +- The causal-interactions sub-section is earned, not filled: a bullet there is admitted only when the quoted source sentence itself states the link between a terroir factor and a wine trait with an explicit connective (because, thanks to, gives, confers, results in, explains, favours, allows — or its equivalent in the source language). If the text only lists factors and wine traits side by side, return an empty list for that sub-section rather than composing the link yourself, and never restate under it a fact already given under the natural factors or the product. +- Grape names without regulatory colour codes: write "Pinot noir", never "Pinot noir N", "Chardonnay B", "Grenache G", "Gewurztraminer Rs", "Pinot gris G". +- Spell out abbreviations on first use: VT = Vendanges Tardives, SGN = Sélection de Grains Nobles, TBA = Trockenbeerenauslese. +- Never refer to the document or its sources inside a bullet: no "the specification", "the cahier", "the disciplinare", "Wikipedia", "according to the document", "confirmed by". +- Keep the source's hedges ("mainly", "mostly", "essentially", "sometimes", "often", "generally"); never strengthen them to "exclusively", "only", "always" or "100 %". +- Do not restate a fact already given in another bullet or covered by another sub-section. +- Skip statements that would be true of any appellation ("the wine's uniqueness stems from soil, climate and grape varieties", "the terroir gives the wines their typicity", "favourable soil and climatic conditions").""" + + +def with_style_rules(prompt): + """Return `prompt` (a str, or a {lang: str} dict) with `STYLE_RULES` + inserted before its final paragraph — the JSON-only instruction that + every extraction prompt ends with.""" + if isinstance(prompt, dict): + return {k: with_style_rules(v) for k, v in prompt.items()} + head, sep, tail = prompt.rpartition("\n\n") + if not sep: + return prompt + "\n\n" + STYLE_RULES + return f"{head}\n\n{STYLE_RULES}\n\n{tail}" + + +# ───────────────────────────────────── stage-02e translation rules (W1) ── +# +# The 21 stage-02e translation scripts each carried one long +# "Preserve <Language> proper nouns verbatim: …" line whose roster mixed +# real names (Barolo, Xinomavro, Muschelkalk) with common nouns (argille, +# lösz, ЗНП, dűlő, kasna berba) — so the common nouns leaked untranslated +# into every target locale (16 % of all bullets, 37–51 % for GR/BG/HU/CZ/ +# HR/LU in the 2026-09-11 review). `translation_rules` is the one shared +# two-bucket rule block; each script keeps only its *names* roster and +# passes it in as `proper_nouns`. Plain text, no `{` / `}` anywhere, so a +# script that still pushes its prompt through `str.format` cannot break. + +from _lib.terroir_roster import children_names, record_name # noqa: E402 +from _lib.translation_glossary import glossary_for # noqa: E402 + +_LANG_NAME = { + "en": "English", "fr": "French", "es": "Spanish", "nl": "Dutch", "de": "German", + "it": "Italian", "pt": "Portuguese", "el": "Greek", "bg": "Bulgarian", "hu": "Hungarian", + "cs": "Czech", "sk": "Slovak", "sl": "Slovenian", "hr": "Croatian", "ro": "Romanian", + "mt": "Maltese", +} + +_NON_LATIN_SOURCES = ("el", "bg") + +_KEEP_VERBATIM = ( + "- Keep verbatim (these are names, not vocabulary): appellation names exactly as " + "registered, even when they coincide with a region (Toscana IGT, Bourgogne, Alsace grand " + "cru Rangen, Wien DAC); commune and vineyard-site names; institutions (Junta de " + "Andalucía, Consorzio); grape variety names, including those built on a place (Melon de " + "Bourgogne), transliterated to the EU-official Latin form when the source script is not " + "Latin; NAMED geological formations and named winds " + "(Marnes à exogyra virgula, Flysch di Cormons, llicorella, albariza, tuffeau, Muschelkalk, " + "Rotliegend, Mistral, Bora, Meltemi); registered traditional terms and Prädikat tiers " + "(Aszú, Szamorodni, Vinsanto, Nychteri, tokajský výber, Trockenbeerenauslese, " + "Vendanges Tardives, Sélection de Grains Nobles)." +) + +_GEOGRAPHY = ( + "- Geography takes the established TARGET name whenever one exists — countries, regions " + "used as places, mountain ranges, rivers, lakes and seas are vocabulary, not labels: " + "Vosges → Vogezen (Dutch) / Vosgos (Spanish); Rhin, Rhein → Rijn / Rhine / Rin; Donau → " + "Danube / Donau; Appennino → Apennines / Apennijnen / Apeninos; Méditerranée, Mediterraneo → " + "Mediterranean / Middellandse Zee / Mediterráneo; Atlantique → Atlantic / Atlantische Oceaan; " + "Toscana → Tuscany / Toscane when it names the region (verbatim only as the appellation); " + "Piemonte → Piedmont / Piëmont / Piamonte; Sicilia → Sicily / Sicilië / Sicile; Sardegna → " + "Sardinia / Sardinië / Sardaigne / Cerdeña; Bourgogne → Burgundy / Bourgondië / Borgoña as " + "the region; Alsace → Elzas / Alsacia as the region; Wien → Vienna / Wenen / Vienne / Viena " + "as the city; Steiermark → Styria / Stiermarken / Styrie / Estiria; Danube Plain → Danubian " + "Plain / Donauvlakte / Llanura del Danubio / plaine du Danube. When the target language " + "has no established form, keep the source form." +) + +_TRANSLATE_INTO = ( + "- Translate into TARGET everything else — in particular these common nouns, which are " + "NOT names and must never be left in the source language: generic soil and rock words " + "(argille, calcare, marne, arenaria, scisti, calcareniti, argilliti; lösz, mészkő, homokkő, " + "agyagpala, barna erdőtalaj, csernozjom, vulkáni talaj; льос, чернозем, канелена горска " + "почва, смолница; vapnenac, crvenica, fliš, apnenec, ilovica, laporovec, lapor; xisto; " + "Lehm, Quarzit, Gneis, Granit, Urgestein, Vulkangestein, Steillage, Lagenwein; leem, klei, " + "zandleem, mergel; gypse, marnes keupériennes, calcaire conchylien); climate phrases " + "(умереноконтинентален климат, μεσογειακό κλίμα, ηπειρωτικό κλίμα, kontinentalna klima, " + "kontinentální podnebí, pannonisches Klima); generic harvest and wine-law categories " + "(kasna berba, desertno vino, predikatno vino, pozna trgatev, ledeno vino, suhi jagodni " + "izbor, pozdní sběr, slámové víno, αφρώδεις οίνοι, λιαστοί οίνοι, pezsgő, gyöngyözőbor, " + "vendemmia, fruttaia); site words (lege, dűlő, viniční trať, podgorie, borvidék, vinorodni " + "okoliš, ribera, páramo, gromače, emparrado); scheme abbreviations (ΠΓΕ / ΠΟΠ / ЗНП / ЗГУ / " + "OEM / OFJ / CHOP / CHZO / ZOI become PDO / PGI in their TARGET form)." +) + +_GLOSS_ONCE = ( + "- A one-time gloss of a genuinely technical local term is allowed — the TARGET word first, " + "the local word once in parentheses, the way \"boulder clay (keileem)\" or \"dry-stone walls " + "(prizidi)\" does it. Never the reverse (\"Continental (ηπειρωτικό κλίμα)\")." +) + +_LATIN_SCRIPT = ( + "- The output must be entirely in Latin script. Transliterate Greek and Cyrillic proper " + "nouns to the EU-official Latin form (Ξινόμαυρο → Xinomavro, Стара планина → Stara Planina, " + "Гъмза → Gamza, Σαντορίνη → Santorini); the eAmbrosia transcription and the grape lexicon's " + "Latin slugs are the reference spellings." +) + +_STYLE = ( + "- Drop regulatory colour-code suffixes after grape names (N / B / G / Rs / Rg): write " + "\"Pinot noir\", never \"Pinot noir N\".\n" + "- End every bullet with a period.\n" + "- Keep the source's hedges (mainly / mostly / sometimes / often / generally); never " + "strengthen them to exclusively / only / always / 100 %." +) + + +def translation_rules(source_lang: str, target_lang: str, *, proper_nouns: str) -> str: + """The shared two-bucket terminology block for a 02e system prompt. + + `proper_nouns` is the country's own *names* roster (appellations, + regions, communes, grapes, named formations / winds, registered + terms) — may be empty. The returned text contains no `{` / `}`.""" + source = _LANG_NAME.get(source_lang, source_lang) + target = _LANG_NAME.get(target_lang, target_lang) + keep = _KEEP_VERBATIM + roster = (proper_nouns or "").strip().replace("{", "(").replace("}", ")") + if roster: + keep += " In this corpus: " + roster.rstrip(".") + "." + lines = [ + f"Terminology rules ({source} → {target}):", + keep, + _GEOGRAPHY.replace("TARGET", target), + _TRANSLATE_INTO.replace("TARGET", target), + _GLOSS_ONCE.replace("TARGET", target), + ] + if source_lang in _NON_LATIN_SOURCES: + lines.append(_LATIN_SCRIPT) + lines.append(_STYLE) + return "\n".join(lines) + + +def translation_system_prompt( + base: str, *, source_lang: str, target_lang: str, proper_nouns: str, +) -> str: + """`base` (already formatted) + the shared rules + the target-locale + glossary (`translation_glossary.glossary_for`, when non-empty).""" + parts = [base, translation_rules(source_lang, target_lang, proper_nouns=proper_nouns)] + glossary = glossary_for(target_lang) + if glossary: + parts.append(glossary) + return "\n\n".join(parts) + + +# ───────────────────────────── appellation context on sub-denomination pages ── +# +# A parent's bullets are inherited by its sub-denominations' pages (FR DGCs, +# ES subzonas, IT sottozone, …). "The clay-limestone soils in the northernmost +# part of the appellation, straddling Rioja Alavesa and Rioja Alta …" is +# Rioja's own sentence, but on the Rioja Alavesa page "the appellation" reads +# as Alavesa (2026-09-15). For a record that has sub-denominations the +# extraction and translation prompts therefore ask for the appellation's +# name wherever the source refers to it only generically. Per record, so it +# lives in the per-record parts of the prompts (the 02e user message, the +# 02d per-record block) — the shared, cached system prompts stay identical. + +MAX_LISTED_CHILDREN = 6 + + +def _listed(names: list[str]) -> str: + head = ", ".join(names[:MAX_LISTED_CHILDREN]) + rest = len(names) - MAX_LISTED_CHILDREN + return f"{head} and {rest} more" if rest > 0 else head + + +def appellation_context(slug: str, *, for_translation: bool) -> str: + """The per-record instruction for a record whose bullets are also + shown on its sub-denominations' pages; "" for a record without any.""" + kids = children_names(slug) + if not kids: + return "" + name = record_name(slug) or slug + where = "in the translation" if for_translation else "in the bullet" + return ( + f"CONTEXT: these bullets describe «{name}» and are also shown on the pages of its " + f"{len(kids)} sub-denominations ({_listed(kids)}). Where the source refers to the appellation " + f"as a whole only generically — \"the appellation\", \"the denomination\", \"the DOC / DOCa / DOP\", " + f"\"the geographical area\", \"the zone\", \"the vineyard\" — name it {where} (\"the {name} appellation\" " + f"or \"{name}\"), so the sentence stays unambiguous on a sub-denomination page. Keep every " + f"sub-denomination's own name exactly as it is, and never attach the parent's name to a statement " + f"that the source makes about one sub-denomination only." + ) + + +def with_appellation_context(user: str, slug: str) -> str: + """A 02e user message + the record's context (translation wording).""" + ctx = appellation_context(slug, for_translation=True) + return f"{user.rstrip()}\n\n{ctx}" if ctx else user diff --git a/scripts/_lib/terroir_roster.py b/scripts/_lib/terroir_roster.py new file mode 100644 index 0000000..b82f3d4 --- /dev/null +++ b/scripts/_lib/terroir_roster.py @@ -0,0 +1,72 @@ +"""The sub-denomination roster as stage 04 renders it: parent slug → the +names of the sub-denominations whose pages inherit the parent's terroir +facts (FR DGCs, ES subzonas, IT sottozone, PT sub-regiões, DE +Einzellagen, CH régionale / locale tiers, LU communes, …). + +`wiki/_index.json` (stage 03) carries `parent_slug` for every on-disk +sub-denomination; the sottozone stage 04 synthesises from the MASAF +sidecars exist only in the startup blob, where the parent is the longest +parent slug prefixing the sottozona slug (chianti-rufina → chianti). +Used by the dedupe post-pass (never collapse two bullets leading with +different sub-denomination names) and by the extraction / translation +prompts (name the appellation where the source says "the appellation", +because the bullet is also shown on the sub-denominations' pages). +""" + +from __future__ import annotations + +import json +from collections import defaultdict +from functools import lru_cache +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] + + +@lru_cache(maxsize=1) +def _load() -> tuple[dict[str, list[str]], dict[str, str], dict[str, int]]: + children: dict[str, list[str]] = defaultdict(list) + names: dict[str, str] = {} + index_path = ROOT / "wiki" / "_index.json" + index = json.loads(index_path.read_text(encoding="utf-8")) if index_path.exists() else {} + for slug, rec in index.items(): + if rec.get("name"): + names[slug] = rec["name"] + if rec.get("parent_slug") and rec.get("name"): + children[rec["parent_slug"]].append(rec["name"]) + n_index = sum(len(v) for v in children.values()) + blobs = sorted((ROOT / "wiki" / "data").glob("aocs.en.*.js")) + n_blob = 0 + if blobs: + text = blobs[-1].read_text(encoding="utf-8") + aocs = json.loads(text[text.index("=") + 1:].strip().rstrip(";")).get("aocs") or {} + parents = sorted((s for s, r in aocs.items() if not r.get("is_sub_denomination")), key=len, reverse=True) + for slug, rec in aocs.items(): + if rec.get("name") and slug not in names: + names[slug] = rec["name"] + if not rec.get("is_sub_denomination") or not rec.get("name"): + continue + if slug in index and index[slug].get("parent_slug"): + continue + parent = next((p for p in parents if slug.startswith(p + "-")), None) + if parent: + children[parent].append(rec["name"]) + n_blob += 1 + return dict(children), names, {"index": n_index, "blob": n_blob} + + +def children_names(slug: str) -> list[str]: + """Names of the sub-denominations of `slug` (empty for a leaf).""" + return list(_load()[0].get(slug) or []) + + +def load_children_names() -> dict[str, list[str]]: + return dict(_load()[0]) + + +def record_name(slug: str) -> str: + return _load()[1].get(slug, "") + + +def roster_stats() -> dict[str, int]: + return dict(_load()[2]) diff --git a/scripts/_lib/terroir_sources.py b/scripts/_lib/terroir_sources.py new file mode 100644 index 0000000..ff5a3e7 --- /dev/null +++ b/scripts/_lib/terroir_sources.py @@ -0,0 +1,112 @@ +"""The exact source text stage 02d graded each country's terroir facts +against, resolved through the country's own 02d module. + +Every `scripts/<cc>/02d_extract_terroir_facts.py` owns its source +resolution — the lien (or, for CH/MT/GB, the règlement / spec context +block), the national-spec sidecar fallbacks, and the per-sub-section +Wikipedia hint with its country-specific heading table and character +cap. Anything that wants to re-grade or audit the cached facts (the +provenance post-pass, `audit_terroir_facts.py`) must use precisely that +text, or a perfectly grounded quote reports as eroded. Rather than +mirroring 21 heading tables, this module loads each stage module with +importlib and calls its `enumerate_aocs()` / `collect_targets()` and +`_wiki_hint_for_subsection()`. +""" + +from __future__ import annotations + +import hashlib +import importlib.util +import inspect +import json +from dataclasses import dataclass +from pathlib import Path + +from _lib.terroir_coverage import SourceMatcher + +ROOT = Path(__file__).resolve().parents[2] +COUNTRIES = ("fr", "at", "be", "bg", "ch", "cy", "cz", "de", "es", "gb", "gr", "hr", "hu", + "it", "lu", "mt", "nl", "pt", "ro", "si", "sk") + + +def _sha(text: str) -> str: + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def stage_path(country: str) -> Path: + if country == "fr": + return ROOT / "scripts" / "02d_extract_terroir_facts.py" + return ROOT / "scripts" / country / "02d_extract_terroir_facts.py" + + +def load_stage(country: str): + path = stage_path(country) + spec = importlib.util.spec_from_file_location(f"owm_02d_{country}", path) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +@dataclass +class Sources: + cahier: str + cahier_sha: str + wiki_revision: object + hints: dict[str, str] + _matcher: SourceMatcher | None = None + + @property + def matcher(self) -> SourceMatcher: + if self._matcher is None: + self._matcher = SourceMatcher(self.cahier) + return self._matcher + + +def _wiki_record(mod, rec: dict, lang: str) -> dict: + slug = rec["slug"] + if hasattr(mod, "_wiki_record_for"): + params = inspect.signature(mod._wiki_record_for).parameters + return mod._wiki_record_for(slug, lang) if "lang" in params else mod._wiki_record_for(slug) + wiki_path = mod.WIKI_AOCS / f"{slug}.json" + if not wiki_path.exists(): + return {} + try: + return json.loads(wiki_path.read_text(encoding="utf-8")) + except (ValueError, OSError): + return {} + + +def _hints(mod, wiki: dict, lang: str, sub_keys: list[str]) -> dict[str, str]: + fn = mod._wiki_hint_for_subsection + if "lang" in inspect.signature(fn).parameters: + return {k: fn(wiki, lang, k) for k in sub_keys} + return {k: fn(wiki, k) for k in sub_keys} + + +def resolve_sources(country: str) -> dict[str, Sources]: + """slug → the exact cahier text + per-sub-section wiki hints stage 02d + grades against for this country, built through the stage's own code.""" + mod = load_stage(country) + out: dict[str, Sources] = {} + if country == "fr": + for job in mod.enumerate_aocs(): + out[job["slug"]] = Sources( + job["lien"], job["lien_sha"], + job["wiki_meta"]["wiki_source_revision"], dict(job["wiki_hints"]), + ) + return out + sub_keys = [s["key"] for s in mod.SUBSECTIONS] + default_lang = getattr(mod, "SOURCE_LANG", None) or "fr" + for rec in mod.collect_targets(): + lang = rec.get("source_lang") or default_lang + if "_cahier_ctx" in rec: + cahier = rec["_cahier_ctx"] or "" + wiki = rec.get("_wiki_record") or {} + else: + cahier = rec.get("link_to_terroir") or "" + wiki = _wiki_record(mod, rec, lang) + out[rec["slug"]] = Sources( + cahier, _sha(cahier), wiki.get("revision") if wiki else None, + _hints(mod, wiki, lang, sub_keys), + ) + return out diff --git a/scripts/_lib/terroir_verbatim.py b/scripts/_lib/terroir_verbatim.py index 62cac31..ab91a56 100644 --- a/scripts/_lib/terroir_verbatim.py +++ b/scripts/_lib/terroir_verbatim.py @@ -120,6 +120,9 @@ def is_verbatim_cache_valid(cache_path: Path, country: str, lien: str) -> bool: def write_verbatim_record(cache_path: Path, payload: dict) -> None: + from _lib.terroir_backup import snapshot_slug # local import: no cycle with terroir_cache + + snapshot_slug(cache_path.stem) cache_path.parent.mkdir(parents=True, exist_ok=True) cache_path.write_text( json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True), diff --git a/scripts/_lib/traditional_terms.json b/scripts/_lib/traditional_terms.json new file mode 100644 index 0000000..d842990 --- /dev/null +++ b/scripts/_lib/traditional_terms.json @@ -0,0 +1,1369 @@ +{ + "__doc__": "Curated ruling + definition table for the two naming axes stage 04 derives on every appellation record: eu_scheme (pdo / pgi / spirit-gi / uk-pdo / uk-pgi / none) and national_term — the traditional term (Reg. (EU) 1308/2013 Art. 112(a); registered under Reg. (EC) 607/2009 Annex XII, now kept in the EU register per Reg. (EU) 2019/33) that the member state's regulator attaches to the geographical indication as a whole. Admission rule for a term: (1) it is registered as a traditional term in Annex XII / the EU register (or, for Switzerland, defined in the federal OVin); (2) it is GI-wide — constant for every wine of the GI, never a lot-level quality grade (Qualitätswein, Prädikatswein, kakovostno, jakostní, akostné are excluded); (3) it is attached to the GI by a regulator document (national law, decree, DAC-Verordnung, the register itself) or by a cited curator pin. Scheme abbreviations and their translations (PDO/PGI, AOP/IGP, OEM/OFJ, ZOP/ZGO, ΠΟΠ/ΠΓΕ, CHOP/CHZO, ЗНП/ЗГУ, BOB/BGA, g.U./g.g.A.) are NOT terms. Sections: schemes (the six eu_scheme values, with UI copy), constants (per country, keyed by the record's kind token — FR: AOC/IGP/EDV, others: DOP/IGP; \"\" = no term and a missing (country, kind) also means no term; countries with a per-record roster — it DOP, es DOP+IGP, at DOP — omit that kind here), rulings (the why per country, with sources), pins (per-country file_number → term overrides, e.g. the Austrian DACs), terms (tooltip definitions keyed '<cc>:<term>' exactly as spelled in constants / pins / rosters). Rendering is 'TERM (SCHEME)'; term only when scheme is none; scheme only when the country has no term. To extend: add a constant or a pin with at least one public source URL, add the matching '<cc>:<term>' definition with all four locales, and record the ruling; tests/test_traditional_terms.py validates the shape. Every fact here cites a public source; notes are hand-authored UI copy (en/fr/es/nl), never machine-translated.", + "constants": { + "at": { + "IGP": "Landwein" + }, + "be": { + "DOP": "", + "IGP": "" + }, + "bg": { + "DOP": "", + "IGP": "" + }, + "ch": { + "AOC": "AOC" + }, + "cy": { + "DOP": "", + "IGP": "" + }, + "cz": { + "DOP": "", + "IGP": "" + }, + "de": { + "DOP": "", + "IGP": "Landwein" + }, + "es": {}, + "fr": { + "AOC": "AOC", + "EDV": "AOC", + "IGP": "" + }, + "gb": { + "DOP": "", + "IGP": "" + }, + "gr": { + "DOP": "", + "IGP": "" + }, + "hr": { + "DOP": "", + "IGP": "" + }, + "hu": { + "DOP": "", + "IGP": "" + }, + "it": { + "IGP": "IGT" + }, + "lu": { + "DOP": "", + "IGP": "" + }, + "mt": { + "DOP": "DOK", + "IGP": "IĠT" + }, + "nl": { + "DOP": "", + "IGP": "" + }, + "pt": { + "DOP": "DOC", + "IGP": "Vinho Regional" + }, + "ro": { + "DOP": "DOC", + "IGP": "IG" + }, + "si": { + "DOP": "", + "IGP": "" + }, + "sk": { + "DOP": "", + "IGP": "" + } + }, + "pins": { + "at": { + "PDO-AT-02593": { + "since_vintage": "2013", + "sources": [ + { + "label": "DAC-Verordnung „Wiener Gemischter Satz“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_wiener_gemischter_satz.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Wien (Wiener Gemischter Satz DAC seit Jahrgang 2013)", + "url": "https://www.oesterreichwein.at/unser-wein/weinbaugebiete/wien" + } + ], + "term": "DAC" + }, + "PDO-AT-02594": { + "since_vintage": "2017", + "sources": [ + { + "label": "DAC-Verordnung „Rosalia“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_rosalia.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + } + ], + "term": "DAC" + }, + "PDO-AT-02769": { + "since_vintage": "2020", + "sources": [ + { + "label": "DAC-Verordnung „Ruster Ausbruch“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_ruster_ausbruch.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Gebietstypischer Qualitätswein (DAC)", + "url": "https://www.oesterreichwein.at/unser-wein/strategie-des-herkunftsmarketings/gebietstypischer-qualitaetswein-dac" + } + ], + "term": "DAC" + }, + "PDO-AT-A0205": { + "since_vintage": "2020", + "sources": [ + { + "label": "DAC-Verordnung „Wachau“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_wachau.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Gebietstypischer Qualitätswein (DAC)", + "url": "https://www.oesterreichwein.at/unser-wein/strategie-des-herkunftsmarketings/gebietstypischer-qualitaetswein-dac" + } + ], + "term": "DAC" + }, + "PDO-AT-A0206": { + "since_vintage": "2002", + "sources": [ + { + "label": "DAC-Verordnung „Weinviertel“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_weinviertel.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Weinviertel (DAC ab Jahrgang 2002)", + "url": "https://www.oesterreichwein.at/unser-wein/weinbaugebiete/niederoesterreich/weinviertel" + } + ], + "term": "DAC" + }, + "PDO-AT-A0208": { + "since_vintage": "2007", + "sources": [ + { + "label": "DAC-Verordnung „Kremstal“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_kremstal.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Kremstal (DAC seit 2007)", + "url": "https://www.oesterreichwein.at/unser-wein/weinbaugebiete/niederoesterreich/kremstal" + } + ], + "term": "DAC" + }, + "PDO-AT-A0209": { + "since_vintage": "2008", + "sources": [ + { + "label": "DAC-Verordnung „Kamptal“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_kamptal.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Kamptal (DAC seit Jahrgang 2008)", + "url": "https://www.oesterreichwein.at/unser-wein/weinbaugebiete/niederoesterreich/kamptal" + } + ], + "term": "DAC" + }, + "PDO-AT-A0210": { + "since_vintage": "2006", + "sources": [ + { + "label": "DAC-Verordnung „Traisental“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_traisental.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Traisental (DAC seit Jahrgang 2006)", + "url": "https://www.oesterreichwein.at/unser-wein/weinbaugebiete/niederoesterreich/traisental" + } + ], + "term": "DAC" + }, + "PDO-AT-A0214": { + "since_vintage": "2005", + "sources": [ + { + "label": "DAC-Verordnung „Mittelburgenland“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_mittelburgenland.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + } + ], + "term": "DAC" + }, + "PDO-AT-A0215": { + "since_vintage": "2009", + "sources": [ + { + "label": "DAC-Verordnung „Eisenberg“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_eisenberg.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Eisenberg (DAC seit Jahrgang 2009)", + "url": "https://www.oesterreichwein.at/unser-wein/weinbaugebiete/burgenland/eisenberg" + } + ], + "term": "DAC" + }, + "PDO-AT-A0216": { + "since_vintage": "2008", + "sources": [ + { + "label": "DAC-Verordnung „Leithaberg“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_leithaberg.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + } + ], + "term": "DAC" + }, + "PDO-AT-A0217": { + "since_vintage": "2019", + "sources": [ + { + "label": "DAC-Verordnung „Carnuntum“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_carnuntum.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Gebietstypischer Qualitätswein (DAC)", + "url": "https://www.oesterreichwein.at/unser-wein/strategie-des-herkunftsmarketings/gebietstypischer-qualitaetswein-dac" + } + ], + "term": "DAC" + }, + "PDO-AT-A0219": { + "since_vintage": "2011", + "sources": [ + { + "label": "DAC-Verordnung „Neusiedlersee“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_neusiedlersee.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Neusiedlersee (DAC trocken seit Jahrgang 2011)", + "url": "https://www.oesterreichwein.at/unser-wein/weinbaugebiete/burgenland/neusiedlersee" + } + ], + "term": "DAC" + }, + "PDO-AT-A0226": { + "since_vintage": "2018", + "sources": [ + { + "label": "DAC-Verordnung „Vulkanland Steiermark“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_vulkanland.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Gebietstypischer Qualitätswein (DAC)", + "url": "https://www.oesterreichwein.at/unser-wein/strategie-des-herkunftsmarketings/gebietstypischer-qualitaetswein-dac" + } + ], + "term": "DAC" + }, + "PDO-AT-A0228": { + "since_vintage": "2018", + "sources": [ + { + "label": "DAC-Verordnung „Südsteiermark“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_suedsteiermark.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Gebietstypischer Qualitätswein (DAC)", + "url": "https://www.oesterreichwein.at/unser-wein/strategie-des-herkunftsmarketings/gebietstypischer-qualitaetswein-dac" + } + ], + "term": "DAC" + }, + "PDO-AT-A0229": { + "since_vintage": "2023", + "sources": [ + { + "label": "DAC-Verordnung „Thermenregion“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_thermenregion.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Gebietstypischer Qualitätswein (DAC)", + "url": "https://www.oesterreichwein.at/unser-wein/strategie-des-herkunftsmarketings/gebietstypischer-qualitaetswein-dac" + } + ], + "term": "DAC" + }, + "PDO-AT-A0233": { + "since_vintage": "2021", + "sources": [ + { + "label": "DAC-Verordnung „Wagram“, BGBl. II Nr. 30/2022 (RIS, geltende Fassung) — § 9: gilt für Wein ab dem Jahrgang 2021", + "url": "https://www.ris.bka.gv.at/GeltendeFassung.wxe?Abfrage=Bundesnormen&Gesetzesnummer=20011803" + }, + { + "label": "DAC-Verordnung „Wagram“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_wagram.pdf" + }, + { + "label": "Österreich Wein Marketing — Gebietstypischer Qualitätswein (DAC)", + "url": "https://www.oesterreichwein.at/unser-wein/strategie-des-herkunftsmarketings/gebietstypischer-qualitaetswein-dac" + } + ], + "term": "DAC" + }, + "PDO-AT-A0234": { + "since_vintage": "2018", + "sources": [ + { + "label": "DAC-Verordnung „Weststeiermark“ (konsolidiert, Bundeskellereiinspektion)", + "url": "http://www.bundeskellereiinspektion.at/downloads/dac_vo/dac_vo_weststeiermark.pdf" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Gebietstypischer Qualitätswein (DAC)", + "url": "https://www.oesterreichwein.at/unser-wein/strategie-des-herkunftsmarketings/gebietstypischer-qualitaetswein-dac" + } + ], + "term": "DAC" + } + }, + "fr": { + "marc-d-alsace-gewurztraminer": { + "note": "SIQO row with an empty categorie (see CLAUDE.md, FR register pins): a marc (grape-pomace spirit), so a spirit-drink GI carrying the AOC term, not a wine PDO. Pinned by slug because signe_fr/signe_ue are both empty on the row.", + "scheme": "spirit-gi", + "sources": [ + { + "label": "eAmbrosia — Marc d'Alsace Gewurztraminer (spirit drink GI)", + "url": "https://ec.europa.eu/agriculture/eambrosia/geographical-indications-register/" + }, + { + "label": "INAO — Marc d'Alsace Gewurztraminer (AOC)", + "url": "https://www.inao.gouv.fr/produit/" + } + ], + "term": "AOC" + } + } + }, + "rulings": { + "at": { + "reason": "DOP: 'Districtus Austriae Controllatus (DAC)' is the Annex XII term, but it attaches only to the Weinbaugebiete for which a DAC-Verordnung under § 34 Weingesetz 2009 exists (BML lists 17; Wagram is the 18th with its own VO, BGBl. II Nr. 30/2022) — so the DOP kind has no constant and is served by 'pins'. The generic Bundesland g.U.s (Burgenland, Niederösterreich, Steiermark, Wien …) carry no term; 'Prädikatswein' is a lot-level grade, excluded. IGP: 'Landwein' is the Annex XII PGI term and every Austrian PGI (Bergland, Steirerland, Weinland) is a Landwein.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "Weingesetz 2009, BGBl. I Nr. 111/2009 (RIS, geltende Fassung) — § 34 DAC-Verordnungen, Landwein", + "url": "https://www.ris.bka.gv.at/GeltendeFassung.wxe?Abfrage=Bundesnormen&Gesetzesnummer=20006470" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Gebietstypischer Qualitätswein (DAC)", + "url": "https://www.oesterreichwein.at/unser-wein/strategie-des-herkunftsmarketings/gebietstypischer-qualitaetswein-dac" + } + ] + }, + "be": { + "reason": "Annex XII registers 'Appellation d'origine contrôlée' / 'Gecontroleerde oorsprongsbenaming' (PDO) and 'Vin de pays' / 'Landwijn' (PGI) for Belgium, but by its own definition column they are 'traditional terms replacing' the scheme wording, and the Flemish and Walloon regulator documents label the wines BOB/BGA and AOP/IGP — scheme translations, not a distinct national tier. No GI-wide term; a curator pin pass could attach the language-community term per wine if a regional decree is found to require it.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "Vlaamse overheid — geconsolideerd enig document « Vlaamse mousserende kwaliteitswijn » (BOB)", + "url": "https://lv.vlaanderen.be/media/9156/download" + } + ] + }, + "bg": { + "reason": "Annex XII terms ГНП / ГКНП (PDO) and 'Регионално вино' (PGI) date from the pre-2012 wine law; the current IAVV product specifications label wines ЗНП / ЗГУ, which are scheme translations, not terms. No GI-wide term in v1; 'Регионално вино' for the two macro PGIs is a candidate for a curator pin pass citing the ЗВСН.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "ИАЛВ / IAVV — Спецификации на вината със ЗНП и ЗГУ", + "url": "https://eavw.com/wps/portal/eavw/legislation/wines.with.pdo.and.pgi/specifications.of.wines.with.pdo.and.pgi" + } + ] + }, + "ch": { + "reason": "Outside the EU scheme (eu_scheme = none). Every Swiss wine appellation is a cantonal AOC under OVin art. 21 ss.; the OFAG répertoire is the register, so the AOC kind carries the constant 'AOC' and it is rendered alone (no scheme suffix).", + "sources": [ + { + "label": "Ordonnance sur la viticulture et l'importation de vin (OVin, RS 916.140), art. 21 ss. — AOC", + "url": "https://www.fedlex.admin.ch/eli/cc/2007/833/fr" + }, + { + "label": "OFAG / BLW — Répertoire suisse des AOC (état au 1er janvier 2026)", + "url": "https://www.blw.admin.ch/dam/fr/sd-web/bQQd5jj01Ivw/AOC_KUB-AOC_DOC_1er%20janvier%202026.pdf" + } + ] + }, + "cy": { + "reason": "Annex XII registers 'Οίνος Ελεγχόμενης Ονομασίας Προέλευσης (ΟΕΟΠ)' (PDO) and 'Τοπικός Οίνος' (PGI) for Cyprus; the technical files served by the Department of Agriculture label wines ΠΟΠ / ΠΓΕ (scheme translations). Both registered terms are plausibly GI-wide but are deferred to a curator pin pass citing the founding Κ.Δ.Π. regulations; empty in v1.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "Τμήμα Γεωργίας (CY) — Οίνοι ΠΟΠ / ΠΓΕ, τεχνικοί φάκελοι (moa.gov.cy)", + "url": "https://www.moa.gov.cy/moa/da/da.nsf/" + } + ] + }, + "cz": { + "reason": "CHOP / CHZO are scheme translations. 'Jakostní víno', 'jakostní víno s přívlastkem' and 'zemské víno' are lot-level categories under zákon 321/2004 Sb., excluded. 'Víno originální certifikace (VOC)' is a registered term that attaches to specific certifications (VOC Znojmo, VOC Mikulov …) — deferred to a curator pin pass citing the founding decrees; empty in v1.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "Zákon č. 321/2004 Sb., o vinohradnictví a vinařství — jakostní víno, VOC, zemské víno", + "url": "https://www.zakonyprolidi.cz/cs/2004-321" + }, + { + "label": "SZPI — Specifikace CHZO „moravské“", + "url": "https://www.szpi.gov.cz/soubor/specifikace-chzo-moravske.aspx" + } + ] + }, + "de": { + "reason": "DOP: Germany's Annex XII PDO terms (Qualitätswein, Prädikatswein, Sekt b.A., Winzersekt …) are lot-level quality categories that vary per bottle within one Anbaugebiet (Weingesetz § 3), so no GI-wide term exists — Mosel renders as scheme only. IGP: 'Landwein' is the Annex XII PGI term and every German PGI is a Landwein g.g.A.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "Weingesetz (DE), § 3 — Landwein (g.g.A.), Qualitätswein und Prädikatswein (g.U.)", + "url": "https://www.gesetze-im-internet.de/weing_1994/__3.html" + } + ] + }, + "es": { + "reason": "Ley 6/2015 (DA 3.ª) fixes five traditional terms — Denominación de Origen, Denominación de Origen Calificada, Vino de Pago, Vino de Calidad con Indicación Geográfica (PDO) and Vino de la Tierra (PGI) — but which one applies is a per-GI decision (DOCa Rioja vs DO Rueda; DOQ Priorat in its Catalan form), so both DOP and IGP are served by a per-record roster rather than a constant.", + "sources": [ + { + "label": "Ley 6/2015, disposición adicional tercera — términos tradicionales de los vinos DOP/IGP (BOE)", + "url": "https://www.boe.es/buscar/act.php?id=BOE-A-2015-5288" + }, + { + "label": "MAPA — Indicaciones geográficas de vinos (FAQ, términos tradicionales)", + "url": "https://www.mapa.gob.es/dam/mapa/contenido/alimentacion/temas/calidad-agroalimentaria/2017-calidad-diferenciada/4-informacion-de-interes/especificaciones-faq-agroalim-vinos-bbee-v-aromatiz/faqvinosbbeevaromatizados.pdf" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "INCAVI (Generalitat de Catalunya) — DOQ Priorat", + "url": "https://incavi.gencat.cat/ca/coneix-vi-catala/denominacions-origen-catalanes/priorat/index.html" + } + ] + }, + "fr": { + "reason": "AOC: 'Appellation d'origine contrôlée' is the Annex XII PDO term and the mandatory national recognition stage (Code rural L641-5) — every French wine AOP is an AOC. EDV: French spirit-drink GIs (Cognac, Armagnac, Calvados …) are AOCs under the same Code rural article, protected at EU level as spirit-drink GIs (Reg. 2019/787), so the term is AOC with eu_scheme spirit-gi. IGP: 'Vin de pays' is registered but INAO's PGI wines are labelled IGP since 2009, so no term.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "Code rural et de la pêche maritime, art. L641-5 (appellation d'origine contrôlée)", + "url": "https://www.legifrance.gouv.fr/codes/article_lc/LEGIARTI000022190244" + }, + { + "label": "INAO — Appellation d'origine protégée / contrôlée (AOP / AOC)", + "url": "https://www.inao.gouv.fr/en/aop-appellation-origine-protegee" + }, + { + "label": "INAO — Indication géographique protégée (IGP)", + "url": "https://www.inao.gouv.fr/en/igp-indication-geographique-protegee" + }, + { + "label": "INAO — Document méthodologique « Boissons spiritueuses AOC, IG, AOR et AO » (DM/4C/ENQ/003)", + "url": "https://extranet.inao.gouv.fr/fichier/COMNAT-EDV-15122015-DM-4C-ENQ-003-V01.pdf" + } + ] + }, + "gb": { + "reason": "The UK GI scheme uses the words PDO / PGI themselves (eu_scheme uk-pdo / uk-pgi), so there is no separate national term; the legacy Annex XII terms 'quality (sparkling) wine' / 'Regional (sparkling) wine' are lot-level categories, excluded.", + "sources": [ + { + "label": "GOV.UK — Protected geographical food and drink names: UK GI schemes", + "url": "https://www.gov.uk/guidance/protected-geographical-food-and-drink-names-uk-gi-schemes" + }, + { + "label": "GOV.UK — Protected food and drink names register", + "url": "https://www.gov.uk/protected-food-drink-names" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "gr": { + "reason": "ΠΟΠ / ΠΓΕ are scheme translations. Annex XII registers 'Ονομασία Προέλευσης Ανωτέρας Ποιότητας (ΟΠΑΠ)' and 'Ονομασία Προέλευσης Ελεγχόμενη (ΟΠΕ)' (PDO) and 'τοπικός οίνος' (PGI); ΟΠΑΠ / ΟΠΕ are GI-wide (each PDO is one or the other) but are deferred to a curator pin pass citing the founding decrees; empty in v1.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "ΥΠΑΑΤ — Προδιαγραφές προϊόντος ΠΟΠ / ΠΓΕ οίνων (minagric.gr)", + "url": "http://wwww.minagric.gr/greek/data/pop-pge/" + } + ] + }, + "hr": { + "reason": "Annex XII predates Croatia's accession; the Croatian traditional terms later registered (kvalitetno vino KZP, vrhunsko vino KZP …) are lot-level quality categories, and the Ministry's product specifications label wines ZOI (scheme translation). No GI-wide term.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "Ministarstvo poljoprivrede (HR) — Oznake izvornosti vina (specifikacije proizvoda)", + "url": "https://poljoprivreda.gov.hr/istaknute-teme/hrana-111/oznake-kvalitete/oznake-izvornosti-vina/229" + }, + { + "label": "eAmbrosia — EU geographical indications register", + "url": "https://ec.europa.eu/agriculture/eambrosia/geographical-indications-register/" + } + ] + }, + "hu": { + "reason": "OEM / OFJ are scheme translations. 'Minőségi bor' and 'Védett eredetű bor' (Annex XII, PDO) are quality categories rather than a GI-wide term; 'Tájbor' (PGI) is optional on labels and the termékleírás use OFJ. No term in v1; 'Tájbor' is a candidate for a curator pin pass.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "boraszat.kormany.hu — Termékleírások (HU national product specifications)", + "url": "https://boraszat.kormany.hu/termekleirasok2" + } + ] + }, + "it": { + "reason": "DOP: Legge 238/2016 art. 28 makes DOCG and DOC the traditional terms for Italian PDO wines, but which one applies is a per-GI decision (Barolo DOCG vs Bardolino DOC), so DOP is served by a per-record roster. IGP: 'Indicazione geografica tipica (IGT)' is the single term for every Italian PGI.", + "sources": [ + { + "label": "Legge 12 dicembre 2016, n. 238 (Testo unico del vino), art. 28 — menzioni specifiche tradizionali DOCG, DOC, IGT", + "url": "https://www.normattiva.it/uri-res/N2Ls?urn:nir:stato:legge:2016-12-12;238" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "MASAF — Disciplinari di produzione dei vini DOP e IGP", + "url": "https://www.masaf.gov.it/flex/cm/pages/ServeBLOB.php/L/IT/IDPagina/4625" + } + ] + }, + "lu": { + "reason": "Luxembourg's Annex XII terms are 'Marque nationale' (a national quality seal awarded per wine) and 'Crémant de Luxembourg' (a product designation), neither a GI-wide term; the RGD of 17 December 2015 names the single GI 'AOP Moselle luxembourgeoise'. No term.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "Règlement grand-ducal du 17 décembre 2015 — AOP « Moselle luxembourgeoise » (Legilux)", + "url": "https://legilux.public.lu/eli/etat/leg/rgd/2015/12/17" + } + ] + }, + "mt": { + "reason": "Annex XII registers 'Denominazzjoni ta' Origini Kontrollata (D.O.K.)' (PDO) and 'Indikazzjoni Ġeografika Tipika (I.G.T.)' (PGI); Malta's DOK/IĠT scheme attaches them GI-wide — DOK Malta, DOK Gozo, IĠT Maltese Islands — so both kinds carry a constant.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "Malta — DOK and IĠT Wine Quality Scheme (Department for Rural Affairs; L.N. 167/2007, L.N. 416/2007)", + "url": "https://ruralaffairs.gov.mt/en/agrikwalita-services/dok-igt-wine-quality-scheme/" + } + ] + }, + "nl": { + "reason": "'Landwijn' is the Annex XII PGI term for the Netherlands, but the twelve provincie PGIs are registered under the bare province name and their enig document labels them BGA (scheme translation); no PDO term exists. No GI-wide term in v1; 'Landwijn' is a candidate for a curator pin pass.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "eAmbrosia — Dutch wine PGIs (the twelve provincies) and PDOs", + "url": "https://ec.europa.eu/agriculture/eambrosia/geographical-indications-register/" + } + ] + }, + "pt": { + "reason": "Decreto-Lei 212/2004 fixes the menções tradicionais 'Denominação de Origem Controlada' / DOC for DO wines and 'Vinho Regional' for IG wines; IVV applies them GI-wide (every Portuguese DOP is a DOC, every IGP a Vinho Regional), so both kinds carry a constant. The older Annex XII terms D.O. and I.P.R. are no longer attached to any GI.", + "sources": [ + { + "label": "Decreto-Lei n.º 212/2004 — menções tradicionais «Denominação de Origem Controlada» / «Vinho Regional» (Diário da República)", + "url": "https://diariodarepublica.pt/dr/detalhe/decreto-lei/212-2004-479875" + }, + { + "label": "IVV — Cadernos de especificações das DOP (Instituto da Vinha e do Vinho)", + "url": "https://www.ivv.gov.pt/np4/8617.html" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "ro": { + "reason": "Legea 164/2015 (art. 32, art. 35) and the ONVPV caiete de sarcini attach 'vin cu denumire de origine controlată (D.O.C.)' to every Romanian PDO and 'vin cu indicație geografică (I.G.)' to every PGI; both are Annex XII terms, so both kinds carry a constant.", + "sources": [ + { + "label": "Legea nr. 164/2015 a viei și vinului, art. 32 (D.O.C.) și art. 35 (I.G.) (Portal Legislativ)", + "url": "https://legislatie.just.ro/Public/DetaliiDocument/169280" + }, + { + "label": "ONVPV — Legislație națională în domeniul vitivinicol", + "url": "https://www.onvpv.ro/ro/node/23" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "si": { + "reason": "ZOP / ZGO are scheme translations. 'Kakovostno' / 'vrhunsko vino ZGP' and 'deželno vino PGO' (Annex XII) are lot-level quality categories, excluded. 'Vino s priznanim tradicionalnim poimenovanjem (vino PTP)' attaches GI-wide to specific PDOs (Cviček, Teran, Metliška črnina, Belokranjec, Bizeljčan) but is deferred to a curator pin pass citing the founding pravilniki; empty in v1.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "Uradni list RS 49/2007 — Pravilnik o seznamu geografskih označb za vina in trsnem izboru", + "url": "https://www.uradni-list.si/glasilo-uradni-list-rs/vsebina/2007-01-2634?sop=2007-01-2634" + } + ] + }, + "sk": { + "reason": "CHOP / CHZO are scheme translations. Slovakia's Annex XII terms ('akostné víno', 'akostné víno s prívlastkom', the Tokaj predicates) are lot-level quality categories, and the ÚPV SR specifications label wines CHOP. No GI-wide term.", + "sources": [ + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "ÚPV SR — Špecifikácia CHOP „Nitrianska“ (Slovak national product specification)", + "url": "https://www.indprop.gov.sk/swift_data/source/pdf/specifikacie_op_oz/nitrianska.pdf" + } + ] + } + }, + "schemes": { + "none": { + "full": { + "en": "Outside the EU scheme (Swiss cantonal AOC)", + "es": "Fuera del régimen de la UE (AOC cantonal suiza)", + "fr": "Hors régime de l'UE (AOC cantonale suisse)", + "nl": "Buiten de EU-regeling (Zwitserse kantonnale AOC)" + }, + "note": { + "en": "Switzerland is not part of the EU GI scheme; its wine appellations are cantonal AOCs under the federal Ordonnance sur le vin (OVin), each canton setting the production rules. The OFAG/BLW répertoire suisse des AOC is the register.", + "es": "Suiza no forma parte del régimen de indicaciones geográficas de la UE; sus denominaciones vinícolas son AOC cantonales según la Ordenanza federal sobre el vino (OVin), y cada cantón fija las reglas de producción. El repertorio suizo de AOC de la OFAG/BLW es el registro.", + "fr": "La Suisse ne relève pas du régime d'indications géographiques de l'UE ; ses appellations viticoles sont des AOC cantonales au titre de l'Ordonnance fédérale sur le vin (OVin), chaque canton fixant les conditions de production. Le répertoire suisse des AOC de l'OFAG en est le registre.", + "nl": "Zwitserland valt buiten de EU-GI-regeling; zijn wijnappellaties zijn kantonnale AOC's onder de federale Ordonnance sur le vin (OVin), waarbij elk kanton de productieregels vastlegt. Het répertoire suisse des AOC van OFAG/BLW is het register." + }, + "sources": [ + { + "label": "Ordonnance sur la viticulture et l'importation de vin (OVin, RS 916.140), art. 21 ss. — AOC", + "url": "https://www.fedlex.admin.ch/eli/cc/2007/833/fr" + }, + { + "label": "OFAG / BLW — Répertoire suisse des AOC (état au 1er janvier 2026)", + "url": "https://www.blw.admin.ch/dam/fr/sd-web/bQQd5jj01Ivw/AOC_KUB-AOC_DOC_1er%20janvier%202026.pdf" + } + ] + }, + "pdo": { + "full": { + "en": "Protected Designation of Origin", + "es": "Denominación de Origen Protegida", + "fr": "Appellation d'origine protégée", + "nl": "Beschermde oorsprongsbenaming" + }, + "note": { + "en": "EU quality scheme for wines whose quality or characteristics are essentially due to their geographical origin, with grapes grown and wine made exclusively in the area. Member states may register a traditional term of their own that stands in for it on the label.", + "es": "Figura de calidad de la UE para vinos cuya calidad o características se deben esencialmente a su origen geográfico, con uva y elaboración exclusivamente en la zona. Cada Estado miembro puede registrar un término tradicional propio que la sustituya en la etiqueta.", + "fr": "Signe européen pour les vins dont la qualité ou les caractéristiques sont dues essentiellement à l'origine géographique, raisins et vinification restant dans l'aire. Chaque État membre peut faire enregistrer sa propre mention traditionnelle, qui le remplace sur l'étiquette.", + "nl": "EU-kwaliteitsregeling voor wijnen waarvan de kwaliteit of kenmerken hoofdzakelijk aan de geografische oorsprong te danken zijn, met druiven en vinificatie uitsluitend in het gebied. Elke lidstaat kan een eigen traditionele aanduiding laten registreren die haar op het etiket vervangt." + }, + "sources": [ + { + "label": "Reg. (EU) 1308/2013, Art. 112 — traditional terms (EUR-Lex)", + "url": "https://eur-lex.europa.eu/eli/reg/2013/1308/oj" + }, + { + "label": "Reg. (EU) 2024/1143 on geographical indications for wine, spirit drinks and agricultural products", + "url": "https://eur-lex.europa.eu/eli/reg/2024/1143/oj" + }, + { + "label": "eAmbrosia — EU geographical indications register", + "url": "https://ec.europa.eu/agriculture/eambrosia/geographical-indications-register/" + } + ] + }, + "pgi": { + "full": { + "en": "Protected Geographical Indication", + "es": "Indicación Geográfica Protegida", + "fr": "Indication géographique protégée", + "nl": "Beschermde geografische aanduiding" + }, + "note": { + "en": "EU quality scheme for wines with a quality, reputation or characteristic attributable to their geographical origin, where at least 85 % of the grapes come from the area. Member states may register a traditional term of their own that stands in for it on the label.", + "es": "Figura de calidad de la UE para vinos con una cualidad, reputación o característica atribuible a su origen geográfico, con al menos el 85 % de la uva procedente de la zona. Cada Estado miembro puede registrar un término tradicional propio que la sustituya en la etiqueta.", + "fr": "Signe européen pour les vins dont une qualité, la réputation ou une caractéristique est attribuable à l'origine géographique, avec au moins 85 % des raisins issus de l'aire. Chaque État membre peut faire enregistrer sa propre mention traditionnelle, qui le remplace sur l'étiquette.", + "nl": "EU-kwaliteitsregeling voor wijnen met een kwaliteit, reputatie of kenmerk dat aan de geografische oorsprong is toe te schrijven, waarbij minstens 85 % van de druiven uit het gebied komt. Elke lidstaat kan een eigen traditionele aanduiding laten registreren die haar op het etiket vervangt." + }, + "sources": [ + { + "label": "Reg. (EU) 1308/2013, Art. 112 — traditional terms (EUR-Lex)", + "url": "https://eur-lex.europa.eu/eli/reg/2013/1308/oj" + }, + { + "label": "Reg. (EU) 2024/1143 on geographical indications for wine, spirit drinks and agricultural products", + "url": "https://eur-lex.europa.eu/eli/reg/2024/1143/oj" + }, + { + "label": "eAmbrosia — EU geographical indications register", + "url": "https://ec.europa.eu/agriculture/eambrosia/geographical-indications-register/" + } + ] + }, + "spirit-gi": { + "full": { + "en": "Geographical indication for spirit drinks", + "es": "Indicación geográfica de bebida espirituosa", + "fr": "Indication géographique de boisson spiritueuse", + "nl": "Geografische aanduiding voor gedistilleerde dranken" + }, + "note": { + "en": "EU protection for spirit drinks under Regulation (EU) 2019/787 — a single geographical-indication tier, without the wine PDO/PGI split. French eaux-de-vie (Cognac, Armagnac, Calvados) keep AOC as their national term under the Code rural.", + "es": "Protección de la UE para bebidas espirituosas conforme al Reglamento (UE) 2019/787: un único nivel de indicación geográfica, sin la distinción DOP/IGP del vino. Los aguardientes franceses (Cognac, Armagnac, Calvados) conservan la AOC como término nacional en virtud del Code rural.", + "fr": "Protection européenne des boissons spiritueuses au titre du règlement (UE) 2019/787 — un seul niveau d'indication géographique, sans la distinction AOP/IGP des vins. Les eaux-de-vie françaises (Cognac, Armagnac, Calvados) conservent l'AOC comme mention nationale au titre du Code rural.", + "nl": "EU-bescherming voor gedistilleerde dranken onder Verordening (EU) 2019/787 — één niveau van geografische aanduiding, zonder de BOB/BGA-splitsing van wijn. Franse eaux-de-vie (Cognac, Armagnac, Calvados) behouden AOC als nationale term krachtens de Code rural." + }, + "sources": [ + { + "label": "Reg. (EU) 2019/787 on the definition, description, presentation and labelling of spirit drinks", + "url": "https://eur-lex.europa.eu/eli/reg/2019/787/oj" + }, + { + "label": "INAO — Document méthodologique « Boissons spiritueuses AOC, IG, AOR et AO » (DM/4C/ENQ/003)", + "url": "https://extranet.inao.gouv.fr/fichier/COMNAT-EDV-15122015-DM-4C-ENQ-003-V01.pdf" + }, + { + "label": "Code rural et de la pêche maritime, art. L641-5 (appellation d'origine contrôlée)", + "url": "https://www.legifrance.gouv.fr/codes/article_lc/LEGIARTI000022190244" + } + ] + }, + "uk-pdo": { + "full": { + "en": "UK Protected Designation of Origin", + "es": "Denominación de Origen Protegida (Reino Unido)", + "fr": "Appellation d'origine protégée (Royaume-Uni)", + "nl": "Beschermde oorsprongsbenaming (Verenigd Koninkrijk)" + }, + "note": { + "en": "The post-Brexit UK GI scheme, administered by DEFRA on the GOV.UK register since 1 January 2021, keeps the words PDO and PGI as its own designations. The six UK wines are also listed in the EU register under the Withdrawal Agreement and the UK–EU agreement.", + "es": "El régimen británico de indicaciones geográficas posterior al Brexit, gestionado por DEFRA en el registro GOV.UK desde el 1 de enero de 2021, conserva los términos PDO y PGI como designaciones propias. Los seis vinos británicos figuran también en el registro de la UE en virtud del Acuerdo de Retirada y del acuerdo UE–Reino Unido.", + "fr": "Le régime britannique d'indications géographiques d'après le Brexit, tenu par le DEFRA sur le registre GOV.UK depuis le 1er janvier 2021, conserve les termes PDO et PGI comme désignations propres. Les six vins britanniques figurent aussi au registre de l'UE au titre de l'accord de retrait et de l'accord UE–Royaume-Uni.", + "nl": "De Britse GI-regeling van na de Brexit, beheerd door DEFRA in het GOV.UK-register sinds 1 januari 2021, behoudt de woorden PDO en PGI als eigen aanduidingen. De zes Britse wijnen staan ook in het EU-register op grond van het terugtrekkingsakkoord en het akkoord EU–VK." + }, + "sources": [ + { + "label": "GOV.UK — Protected geographical food and drink names: UK GI schemes", + "url": "https://www.gov.uk/guidance/protected-geographical-food-and-drink-names-uk-gi-schemes" + }, + { + "label": "GOV.UK — Protected food and drink names register", + "url": "https://www.gov.uk/protected-food-drink-names" + }, + { + "label": "eAmbrosia — EU geographical indications register", + "url": "https://ec.europa.eu/agriculture/eambrosia/geographical-indications-register/" + } + ] + }, + "uk-pgi": { + "full": { + "en": "UK Protected Geographical Indication", + "es": "Indicación Geográfica Protegida (Reino Unido)", + "fr": "Indication géographique protégée (Royaume-Uni)", + "nl": "Beschermde geografische aanduiding (Verenigd Koninkrijk)" + }, + "note": { + "en": "The PGI tier of the post-Brexit UK GI scheme (GOV.UK register, since 1 January 2021); the two UK wine PGIs are English Regional and Welsh Regional. The UK scheme uses the words PDO/PGI themselves, so no separate national term applies.", + "es": "El nivel IGP del régimen británico de indicaciones geográficas posterior al Brexit (registro GOV.UK, desde el 1 de enero de 2021); las dos IGP vinícolas británicas son English Regional y Welsh Regional. El régimen usa los propios términos PDO/PGI, sin término nacional aparte.", + "fr": "Le niveau IGP du régime britannique d'indications géographiques d'après le Brexit (registre GOV.UK, depuis le 1er janvier 2021) ; les deux IGP viticoles britanniques sont English Regional et Welsh Regional. Le régime emploie lui-même les termes PDO/PGI, sans mention nationale distincte.", + "nl": "Het BGA-niveau van de Britse GI-regeling van na de Brexit (GOV.UK-register, sinds 1 januari 2021); de twee Britse wijn-BGA's zijn English Regional en Welsh Regional. De regeling gebruikt zelf de woorden PDO/PGI, dus er is geen aparte nationale term." + }, + "sources": [ + { + "label": "GOV.UK — Protected geographical food and drink names: UK GI schemes", + "url": "https://www.gov.uk/guidance/protected-geographical-food-and-drink-names-uk-gi-schemes" + }, + { + "label": "GOV.UK — Protected food and drink names register", + "url": "https://www.gov.uk/protected-food-drink-names" + } + ] + } + }, + "terms": { + "at:DAC": { + "full": "Districtus Austriae Controllatus", + "note": { + "en": "Austria's traditional term for regionally typical quality wine: a Weinbaugebiet becomes a DAC when a DAC-Verordnung under § 34 Weingesetz 2009 fixes its permitted varieties and style, on the proposal of the regional committee. Attached per region — 18 DACs from Weinviertel (2002 vintage) to Thermenregion (2023).", + "es": "Término tradicional austriaco para el vino de calidad típico de su región: un Weinbaugebiet pasa a ser DAC cuando un DAC-Verordnung dictado según el § 34 Weingesetz 2009 fija variedades y estilo permitidos, a propuesta del comité regional. Se atribuye por región: 18 DAC, de Weinviertel (añada 2002) a Thermenregion (2023).", + "fr": "Mention traditionnelle autrichienne pour le vin de qualité typique de sa région : un Weinbaugebiet devient DAC lorsqu'un DAC-Verordnung pris au titre du § 34 Weingesetz 2009 fixe cépages et style admis, sur proposition du comité régional. Attribuée par région — 18 DAC, du Weinviertel (millésime 2002) à la Thermenregion (2023).", + "nl": "Oostenrijks traditionele aanduiding voor gebiedstypische kwaliteitswijn: een Weinbaugebiet wordt DAC wanneer een DAC-Verordnung krachtens § 34 Weingesetz 2009 de toegestane rassen en stijl vastlegt, op voorstel van het regionale comité. Per gebied toegekend — 18 DAC's, van Weinviertel (jaargang 2002) tot Thermenregion (2023)." + }, + "scheme": "pdo", + "sources": [ + { + "label": "Weingesetz 2009, BGBl. I Nr. 111/2009 (RIS, geltende Fassung) — § 34 DAC-Verordnungen, Landwein", + "url": "https://www.ris.bka.gv.at/GeltendeFassung.wxe?Abfrage=Bundesnormen&Gesetzesnummer=20006470" + }, + { + "label": "BML — DAC-Weinbaugebiete (Districtus Austriae Controllatus)", + "url": "https://www.bmluk.gv.at/themen/landwirtschaft/gemeinsame-agrarpolitik-foerderungen/gmo-rechtsinfo/gmo-wein/DAC.html" + }, + { + "label": "Österreich Wein Marketing — Gebietstypischer Qualitätswein (DAC)", + "url": "https://www.oesterreichwein.at/unser-wein/strategie-des-herkunftsmarketings/gebietstypischer-qualitaetswein-dac" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "at:Landwein": { + "full": "Landwein (Wein mit geschützter geografischer Angabe)", + "note": { + "en": "Austria's traditional term for PGI wine: the three Landwein regions Weinland, Steirerland and Bergland, with a minimum must weight of 14° KMW and lighter rules than Qualitätswein. It stands in for g.g.A. on the label.", + "es": "Término tradicional austriaco para vino IGP: las tres regiones de Landwein — Weinland, Steirerland y Bergland — con un peso de mosto mínimo de 14° KMW y reglas más ligeras que el Qualitätswein. Sustituye a g.g.A. en la etiqueta.", + "fr": "Mention traditionnelle autrichienne pour le vin IGP : les trois régions de Landwein — Weinland, Steirerland et Bergland — avec un poids de moût minimal de 14° KMW et des règles plus légères que le Qualitätswein. Elle tient lieu de g.g.A. sur l'étiquette.", + "nl": "Oostenrijks traditionele aanduiding voor BGA-wijn: de drie Landwein-regio's Weinland, Steirerland en Bergland, met een minimaal mostgewicht van 14° KMW en lichtere regels dan Qualitätswein. Vervangt g.g.A. op het etiket." + }, + "scheme": "pgi", + "sources": [ + { + "label": "Weingesetz 2009, BGBl. I Nr. 111/2009 (RIS, geltende Fassung) — § 34 DAC-Verordnungen, Landwein", + "url": "https://www.ris.bka.gv.at/GeltendeFassung.wxe?Abfrage=Bundesnormen&Gesetzesnummer=20006470" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "ch:AOC": { + "full": "Appellation d'origine contrôlée (Swiss cantonal AOC)", + "note": { + "en": "Swiss cantonal designation of origin under the federal OVin: the canton delimits the area and fixes varieties, yields and minimum must weights. Outside the EU scheme; the OFAG répertoire suisse des AOC is the register.", + "es": "Denominación de origen cantonal suiza según la OVin federal: el cantón delimita la zona y fija variedades, rendimientos y grados mínimos del mosto. Fuera del régimen de la UE; el repertorio suizo de AOC de la OFAG es el registro.", + "fr": "Appellation d'origine cantonale suisse au titre de l'OVin fédérale : le canton délimite l'aire et fixe cépages, rendements et teneurs minimales en sucre. Hors régime de l'UE ; le répertoire suisse des AOC de l'OFAG en est le registre.", + "nl": "Zwitserse kantonnale oorsprongsbenaming onder de federale OVin: het kanton bakent het gebied af en legt rassen, opbrengsten en minimale mostgewichten vast. Buiten de EU-regeling; het répertoire suisse des AOC van OFAG is het register." + }, + "scheme": "none", + "sources": [ + { + "label": "Ordonnance sur la viticulture et l'importation de vin (OVin, RS 916.140), art. 21 ss. — AOC", + "url": "https://www.fedlex.admin.ch/eli/cc/2007/833/fr" + }, + { + "label": "OFAG / BLW — Répertoire suisse des AOC (état au 1er janvier 2026)", + "url": "https://www.blw.admin.ch/dam/fr/sd-web/bQQd5jj01Ivw/AOC_KUB-AOC_DOC_1er%20janvier%202026.pdf" + } + ] + }, + "de:Landwein": { + "full": "Landwein (Wein mit geschützter geografischer Angabe)", + "note": { + "en": "Germany's traditional term for PGI wine (Weingesetz § 3): wine from one of the Landwein areas, with a must weight slightly above table-wine level and looser rules than Qualitätswein. It stands in for g.g.A. on the label.", + "es": "Término tradicional alemán para vino IGP (Weingesetz, § 3): vino de una de las zonas de Landwein, con un peso de mosto ligeramente superior al del vino de mesa y reglas más flexibles que el Qualitätswein. Sustituye a g.g.A. en la etiqueta.", + "fr": "Mention traditionnelle allemande pour le vin IGP (Weingesetz, § 3) : vin d'une des régions de Landwein, avec un poids de moût légèrement supérieur au vin de table et des règles plus souples que le Qualitätswein. Elle tient lieu de g.g.A. sur l'étiquette.", + "nl": "Duitslands traditionele aanduiding voor BGA-wijn (Weingesetz § 3): wijn uit een van de Landwein-gebieden, met een mostgewicht iets boven tafelwijnniveau en soepelere regels dan Qualitätswein. Vervangt g.g.A. op het etiket." + }, + "scheme": "pgi", + "sources": [ + { + "label": "Weingesetz (DE), § 3 — Landwein (g.g.A.), Qualitätswein und Prädikatswein (g.U.)", + "url": "https://www.gesetze-im-internet.de/weing_1994/__3.html" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "es:DO": { + "full": "Denominación de Origen", + "note": { + "en": "Spain's standard traditional term for PDO wines (Ley 6/2015): the name of a region or place of established trade prestige whose wines owe their quality to that origin. It stands in for DOP on the label.", + "es": "Término tradicional español estándar para vinos DOP (Ley 6/2015): nombre de una región o lugar de prestigio comercial reconocido cuyos vinos deben su calidad a ese origen. Sustituye a DOP en la etiqueta.", + "fr": "Mention traditionnelle espagnole courante pour les vins AOP (loi 6/2015) : nom d'une région ou d'un lieu de prestige commercial établi dont les vins doivent leur qualité à cette origine. Elle tient lieu de DOP sur l'étiquette.", + "nl": "Spanjes gangbare traditionele aanduiding voor BOB-wijnen (wet 6/2015): de naam van een streek of plaats met gevestigd handelsprestige waarvan de wijnen hun kwaliteit aan die oorsprong danken. Vervangt DOP op het etiket." + }, + "scheme": "pdo", + "sources": [ + { + "label": "Ley 6/2015, disposición adicional tercera — términos tradicionales de los vinos DOP/IGP (BOE)", + "url": "https://www.boe.es/buscar/act.php?id=BOE-A-2015-5288" + }, + { + "label": "MAPA — Indicaciones geográficas de vinos (FAQ, términos tradicionales)", + "url": "https://www.mapa.gob.es/dam/mapa/contenido/alimentacion/temas/calidad-agroalimentaria/2017-calidad-diferenciada/4-informacion-de-interes/especificaciones-faq-agroalim-vinos-bbee-v-aromatiz/faqvinosbbeevaromatizados.pdf" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "es:DOCa": { + "full": "Denominación de Origen Calificada", + "note": { + "en": "Spain's highest traditional term for PDO wines (Ley 6/2015): a DO of at least ten years' standing with stricter controls and bottling in the area. Held by Rioja and, in its Catalan form DOQ, by Priorat.", + "es": "El término tradicional español más alto para vinos DOP (Ley 6/2015): una DO con al menos diez años de reconocimiento, controles más estrictos y embotellado en la zona. La ostentan Rioja y, en su forma catalana DOQ, Priorat.", + "fr": "Mention traditionnelle espagnole la plus élevée pour les vins AOP (loi 6/2015) : une DO reconnue depuis dix ans au moins, avec des contrôles renforcés et une mise en bouteille dans l'aire. Portée par la Rioja et, sous sa forme catalane DOQ, par le Priorat.", + "nl": "Spanjes hoogste traditionele aanduiding voor BOB-wijnen (wet 6/2015): een DO van minstens tien jaar oud met strengere controles en botteling in het gebied. Gevoerd door Rioja en, in de Catalaanse vorm DOQ, door Priorat." + }, + "scheme": "pdo", + "sources": [ + { + "label": "Ley 6/2015, disposición adicional tercera — términos tradicionales de los vinos DOP/IGP (BOE)", + "url": "https://www.boe.es/buscar/act.php?id=BOE-A-2015-5288" + }, + { + "label": "MAPA — Indicaciones geográficas de vinos (FAQ, términos tradicionales)", + "url": "https://www.mapa.gob.es/dam/mapa/contenido/alimentacion/temas/calidad-agroalimentaria/2017-calidad-diferenciada/4-informacion-de-interes/especificaciones-faq-agroalim-vinos-bbee-v-aromatiz/faqvinosbbeevaromatizados.pdf" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "es:DOQ": { + "castilian_form": "DOCa", + "full": "Denominació d'Origen Qualificada", + "note": { + "en": "Catalan form of Denominación de Origen Calificada, Spain's highest traditional term for PDO wines, used by Priorat — the only Catalan DO to hold it (recognised 2000). Same legal tier as DOCa Rioja.", + "es": "Forma catalana de Denominación de Origen Calificada, el término tradicional español más alto para vinos DOP, que ostenta el Priorat, única DO catalana en tenerlo (reconocida en 2000). Mismo nivel legal que la DOCa Rioja.", + "fr": "Forme catalane de Denominación de Origen Calificada, mention traditionnelle espagnole la plus élevée pour les vins AOP, portée par le Priorat — seule DO catalane à la détenir (reconnue en 2000). Même niveau légal que la DOCa Rioja.", + "nl": "Catalaanse vorm van Denominación de Origen Calificada, Spanjes hoogste traditionele aanduiding voor BOB-wijnen, gevoerd door Priorat — de enige Catalaanse DO die ze draagt (erkend in 2000). Zelfde wettelijke rang als DOCa Rioja." + }, + "scheme": "pdo", + "sources": [ + { + "label": "INCAVI (Generalitat de Catalunya) — DOQ Priorat", + "url": "https://incavi.gencat.cat/ca/coneix-vi-catala/denominacions-origen-catalanes/priorat/index.html" + }, + { + "label": "Ley 6/2015, disposición adicional tercera — términos tradicionales de los vinos DOP/IGP (BOE)", + "url": "https://www.boe.es/buscar/act.php?id=BOE-A-2015-5288" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "es:Vino de Calidad": { + "full": "Vino de calidad con indicación geográfica", + "note": { + "en": "Spanish traditional term for the entry PDO tier (Ley 6/2015), labelled 'Vino de calidad de …': a delimited area with rules lighter than a DO, often a step towards it. Despite the wording it is a PDO, not a PGI.", + "es": "Término tradicional español del nivel DOP de entrada (Ley 6/2015), etiquetado «Vino de calidad de …»: zona delimitada con reglas más ligeras que una DO, a menudo un paso previo a ella. Pese a su nombre es una DOP, no una IGP.", + "fr": "Mention traditionnelle espagnole du premier niveau AOP (loi 6/2015), étiquetée « Vino de calidad de … » : aire délimitée aux règles plus légères qu'une DO, souvent une étape vers celle-ci. Malgré son libellé, c'est une AOP, non une IGP.", + "nl": "Spaanse traditionele aanduiding voor het instap-BOB-niveau (wet 6/2015), geëtiketteerd als 'Vino de calidad de …': een afgebakend gebied met lichtere regels dan een DO, vaak een opstap ernaartoe. Ondanks de naam een BOB, geen BGA." + }, + "scheme": "pdo", + "sources": [ + { + "label": "Ley 6/2015, disposición adicional tercera — términos tradicionales de los vinos DOP/IGP (BOE)", + "url": "https://www.boe.es/buscar/act.php?id=BOE-A-2015-5288" + }, + { + "label": "MAPA — Indicaciones geográficas de vinos (FAQ, términos tradicionales)", + "url": "https://www.mapa.gob.es/dam/mapa/contenido/alimentacion/temas/calidad-agroalimentaria/2017-calidad-diferenciada/4-informacion-de-interes/especificaciones-faq-agroalim-vinos-bbee-v-aromatiz/faqvinosbbeevaromatizados.pdf" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "es:Vino de Pago": { + "full": "Vino de Pago", + "note": { + "en": "Spanish traditional term for a single-estate PDO (Ley 6/2015): a pago is a rural site with its own soil and microclimate, and the wine is made and bottled on the estate. A distinct PDO tier, not a sub-zone of a DO.", + "es": "Término tradicional español para una DOP de finca única (Ley 6/2015): el pago es un paraje rural con suelo y microclima propios, y el vino se elabora y embotella en la finca. Un nivel DOP distinto, no una subzona de una DO.", + "fr": "Mention traditionnelle espagnole pour une AOP de domaine unique (loi 6/2015) : le pago est un lieu-dit rural doté de son propre sol et microclimat, le vin étant vinifié et embouteillé au domaine. Niveau AOP distinct, non une sous-zone de DO.", + "nl": "Spaanse traditionele aanduiding voor een BOB van één landgoed (wet 6/2015): een pago is een landelijke plek met eigen bodem en microklimaat, en de wijn wordt op het landgoed gemaakt en gebotteld. Een apart BOB-niveau, geen subzone van een DO." + }, + "scheme": "pdo", + "sources": [ + { + "label": "Ley 6/2015, disposición adicional tercera — términos tradicionales de los vinos DOP/IGP (BOE)", + "url": "https://www.boe.es/buscar/act.php?id=BOE-A-2015-5288" + }, + { + "label": "MAPA — Indicaciones geográficas de vinos (FAQ, términos tradicionales)", + "url": "https://www.mapa.gob.es/dam/mapa/contenido/alimentacion/temas/calidad-agroalimentaria/2017-calidad-diferenciada/4-informacion-de-interes/especificaciones-faq-agroalim-vinos-bbee-v-aromatiz/faqvinosbbeevaromatizados.pdf" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "es:Vino de la Tierra": { + "full": "Vino de la Tierra", + "note": { + "en": "Spain's traditional term for PGI wines (Ley 6/2015): a regional indication with at least 85 % of the grapes from the area and looser rules than a DO. It stands in for IGP on the label.", + "es": "Término tradicional español para vinos IGP (Ley 6/2015): indicación regional con al menos el 85 % de la uva de la zona y reglas más flexibles que una DO. Sustituye a IGP en la etiqueta.", + "fr": "Mention traditionnelle espagnole pour les vins IGP (loi 6/2015) : indication régionale avec au moins 85 % des raisins issus de l'aire et des règles plus souples qu'une DO. Elle tient lieu d'IGP sur l'étiquette.", + "nl": "Spanjes traditionele aanduiding voor BGA-wijnen (wet 6/2015): een regionale aanduiding met minstens 85 % van de druiven uit het gebied en soepelere regels dan een DO. Vervangt IGP op het etiket." + }, + "scheme": "pgi", + "sources": [ + { + "label": "Ley 6/2015, disposición adicional tercera — términos tradicionales de los vinos DOP/IGP (BOE)", + "url": "https://www.boe.es/buscar/act.php?id=BOE-A-2015-5288" + }, + { + "label": "MAPA — Indicaciones geográficas de vinos (FAQ, términos tradicionales)", + "url": "https://www.mapa.gob.es/dam/mapa/contenido/alimentacion/temas/calidad-agroalimentaria/2017-calidad-diferenciada/4-informacion-de-interes/especificaciones-faq-agroalim-vinos-bbee-v-aromatiz/faqvinosbbeevaromatizados.pdf" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "fr:AOC": { + "full": "Appellation d'origine contrôlée", + "note": { + "en": "France's national designation of origin, the mandatory recognition stage before EU registration; wine is the only sector that may keep AOC on the label instead of AOP. Registered as a traditional term standing in for PDO.", + "es": "Denominación de origen nacional francesa, etapa de reconocimiento obligatoria antes del registro europeo; el vino es el único sector que puede mantener AOC en la etiqueta en lugar de AOP. Término tradicional registrado que sustituye a la DOP.", + "fr": "Appellation d'origine nationale française, étape de reconnaissance obligatoire avant l'enregistrement européen ; le vin est le seul secteur autorisé à conserver AOC sur l'étiquette au lieu d'AOP. Mention traditionnelle enregistrée tenant lieu d'AOP.", + "nl": "De Franse nationale oorsprongsbenaming, verplichte erkenningsstap vóór de EU-registratie; wijn is de enige sector die AOC in plaats van AOP op het etiket mag houden. Geregistreerde traditionele aanduiding die BOB vervangt." + }, + "scheme": "pdo", + "sources": [ + { + "label": "Code rural et de la pêche maritime, art. L641-5 (appellation d'origine contrôlée)", + "url": "https://www.legifrance.gouv.fr/codes/article_lc/LEGIARTI000022190244" + }, + { + "label": "INAO — Appellation d'origine protégée / contrôlée (AOP / AOC)", + "url": "https://www.inao.gouv.fr/en/aop-appellation-origine-protegee" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "it:DOC": { + "full": "Denominazione di origine controllata", + "note": { + "en": "Italy's traditional term for PDO wines (Legge 238/2016 art. 28): a geographical name for a wine whose quality and character derive from the environment and the disciplinare. It may appear alone or together with DOP on the label.", + "es": "Término tradicional italiano para vinos DOP (Ley 238/2016, art. 28): nombre geográfico de un vino cuya calidad y carácter derivan del entorno y del disciplinare. Puede figurar solo o junto a DOP en la etiqueta.", + "fr": "Mention traditionnelle italienne pour les vins AOP (loi 238/2016, art. 28) : nom géographique d'un vin dont la qualité et les caractères découlent du milieu et du disciplinare. Elle peut figurer seule ou avec DOP sur l'étiquette.", + "nl": "Italiës traditionele aanduiding voor BOB-wijnen (wet 238/2016, art. 28): een geografische naam voor een wijn waarvan kwaliteit en karakter uit de omgeving en het disciplinare voortkomen. Mag alleen of samen met DOP op het etiket staan." + }, + "scheme": "pdo", + "sources": [ + { + "label": "Legge 12 dicembre 2016, n. 238 (Testo unico del vino), art. 28 — menzioni specifiche tradizionali DOCG, DOC, IGT", + "url": "https://www.normattiva.it/uri-res/N2Ls?urn:nir:stato:legge:2016-12-12;238" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "MASAF — Disciplinari di produzione dei vini DOP e IGP", + "url": "https://www.masaf.gov.it/flex/cm/pages/ServeBLOB.php/L/IT/IDPagina/4625" + } + ] + }, + "it:DOCG": { + "full": "Denominazione di origine controllata e garantita", + "note": { + "en": "Italy's top traditional term for PDO wines (Legge 238/2016 art. 28): a DOC of particular repute, promoted after at least ten years, with a numbered state seal on every bottle. It may appear alone or together with DOP on the label.", + "es": "El término tradicional italiano más alto para vinos DOP (Ley 238/2016, art. 28): una DOC de especial prestigio, promovida tras al menos diez años, con precinto estatal numerado en cada botella. Puede figurar sola o junto a DOP en la etiqueta.", + "fr": "Mention traditionnelle italienne la plus élevée pour les vins AOP (loi 238/2016, art. 28) : une DOC de réputation particulière, promue après dix ans au moins, avec un sceau d'État numéroté sur chaque bouteille. Elle peut figurer seule ou avec DOP sur l'étiquette.", + "nl": "Italiës hoogste traditionele aanduiding voor BOB-wijnen (wet 238/2016, art. 28): een DOC van bijzondere faam, na minstens tien jaar bevorderd, met een genummerd staatszegel op elke fles. Mag alleen of samen met DOP op het etiket staan." + }, + "scheme": "pdo", + "sources": [ + { + "label": "Legge 12 dicembre 2016, n. 238 (Testo unico del vino), art. 28 — menzioni specifiche tradizionali DOCG, DOC, IGT", + "url": "https://www.normattiva.it/uri-res/N2Ls?urn:nir:stato:legge:2016-12-12;238" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + }, + { + "label": "MASAF — Disciplinari di produzione dei vini DOP e IGP", + "url": "https://www.masaf.gov.it/flex/cm/pages/ServeBLOB.php/L/IT/IDPagina/4625" + } + ] + }, + "it:IGT": { + "full": "Indicazione geografica tipica", + "note": { + "en": "Italy's traditional term for PGI wines (Legge 238/2016 art. 28): a regional indication with at least 85 % of the grapes from the area and looser rules than DOC. It may appear alone or together with IGP on the label.", + "es": "Término tradicional italiano para vinos IGP (Ley 238/2016, art. 28): indicación regional con al menos el 85 % de la uva de la zona y reglas más flexibles que la DOC. Puede figurar solo o junto a IGP en la etiqueta.", + "fr": "Mention traditionnelle italienne pour les vins IGP (loi 238/2016, art. 28) : indication régionale avec au moins 85 % des raisins issus de l'aire et des règles plus souples que la DOC. Elle peut figurer seule ou avec IGP sur l'étiquette.", + "nl": "Italiës traditionele aanduiding voor BGA-wijnen (wet 238/2016, art. 28): een regionale aanduiding met minstens 85 % van de druiven uit het gebied en soepelere regels dan DOC. Mag alleen of samen met IGP op het etiket staan." + }, + "scheme": "pgi", + "sources": [ + { + "label": "Legge 12 dicembre 2016, n. 238 (Testo unico del vino), art. 28 — menzioni specifiche tradizionali DOCG, DOC, IGT", + "url": "https://www.normattiva.it/uri-res/N2Ls?urn:nir:stato:legge:2016-12-12;238" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "mt:DOK": { + "full": "Denominazzjoni ta' Oriġini Kontrollata", + "note": { + "en": "Malta's traditional term for PDO wines (L.N. 416/2007 under the Wine Act): DOK Malta and DOK Gozo, with stricter yields, cultivation rules and a tasting panel before certification. It stands in for PDO on the label.", + "es": "Término tradicional maltés para vinos DOP (L.N. 416/2007 según la Wine Act): DOK Malta y DOK Gozo, con rendimientos y normas de cultivo más estrictos y un panel de cata previo a la certificación. Sustituye a DOP en la etiqueta.", + "fr": "Mention traditionnelle maltaise pour les vins AOP (L.N. 416/2007 au titre du Wine Act) : DOK Malta et DOK Gozo, avec rendements et règles de culture plus stricts et un jury de dégustation avant certification. Elle tient lieu d'AOP sur l'étiquette.", + "nl": "Maltas traditionele aanduiding voor BOB-wijnen (L.N. 416/2007 onder de Wine Act): DOK Malta en DOK Gozo, met strengere opbrengsten, teeltregels en een proefpanel vóór certificering. Vervangt BOB op het etiket." + }, + "scheme": "pdo", + "sources": [ + { + "label": "Malta — DOK and IĠT Wine Quality Scheme (Department for Rural Affairs; L.N. 167/2007, L.N. 416/2007)", + "url": "https://ruralaffairs.gov.mt/en/agrikwalita-services/dok-igt-wine-quality-scheme/" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "mt:IĠT": { + "full": "Indikazzjoni Ġeografika Tipika", + "note": { + "en": "Malta's traditional term for PGI wine (L.N. 167/2007 under the Wine Act): IĠT Maltese Islands, made from grapes grown on the islands under a yield-limited production protocol. It stands in for PGI on the label.", + "es": "Término tradicional maltés para vino IGP (L.N. 167/2007 según la Wine Act): IĠT Maltese Islands, elaborado con uva cultivada en las islas bajo un protocolo de producción con rendimiento limitado. Sustituye a IGP en la etiqueta.", + "fr": "Mention traditionnelle maltaise pour le vin IGP (L.N. 167/2007 au titre du Wine Act) : IĠT Maltese Islands, issu de raisins cultivés sur les îles selon un protocole de production à rendement limité. Elle tient lieu d'IGP sur l'étiquette.", + "nl": "Maltas traditionele aanduiding voor BGA-wijn (L.N. 167/2007 onder de Wine Act): IĠT Maltese Islands, gemaakt van op de eilanden geteelde druiven volgens een productieprotocol met opbrengstlimiet. Vervangt BGA op het etiket." + }, + "scheme": "pgi", + "sources": [ + { + "label": "Malta — DOK and IĠT Wine Quality Scheme (Department for Rural Affairs; L.N. 167/2007, L.N. 416/2007)", + "url": "https://ruralaffairs.gov.mt/en/agrikwalita-services/dok-igt-wine-quality-scheme/" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "pt:DOC": { + "full": "Denominação de Origem Controlada", + "note": { + "en": "Portugal's traditional term for PDO wines (Decreto-Lei 212/2004): every Portuguese DOP is a DOC, certified by its regional commission or by IVV. It stands in for DOP on the label.", + "es": "Término tradicional portugués para vinos DOP (Decreto-Lei 212/2004): cada DOP portuguesa es una DOC, certificada por su comisión regional o por el IVV. Sustituye a DOP en la etiqueta.", + "fr": "Mention traditionnelle portugaise pour les vins AOP (décret-loi 212/2004) : chaque DOP portugaise est une DOC, certifiée par sa commission régionale ou par l'IVV. Elle tient lieu de DOP sur l'étiquette.", + "nl": "Portugals traditionele aanduiding voor BOB-wijnen (Decreto-Lei 212/2004): elke Portugese DOP is een DOC, gecertificeerd door haar regionale commissie of door het IVV. Vervangt DOP op het etiket." + }, + "scheme": "pdo", + "sources": [ + { + "label": "Decreto-Lei n.º 212/2004 — menções tradicionais «Denominação de Origem Controlada» / «Vinho Regional» (Diário da República)", + "url": "https://diariodarepublica.pt/dr/detalhe/decreto-lei/212-2004-479875" + }, + { + "label": "IVV — Cadernos de especificações das DOP (Instituto da Vinha e do Vinho)", + "url": "https://www.ivv.gov.pt/np4/8617.html" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "pt:Vinho Regional": { + "full": "Vinho Regional", + "note": { + "en": "Portugal's traditional term for PGI wines (Decreto-Lei 212/2004): a regional indication (Alentejano, Lisboa, Minho …) with at least 85 % of the grapes from the area and a wider variety list than the DOCs. It stands in for IGP on the label.", + "es": "Término tradicional portugués para vinos IGP (Decreto-Lei 212/2004): indicación regional (Alentejano, Lisboa, Minho …) con al menos el 85 % de la uva de la zona y una lista de variedades más amplia que las DOC. Sustituye a IGP en la etiqueta.", + "fr": "Mention traditionnelle portugaise pour les vins IGP (décret-loi 212/2004) : indication régionale (Alentejano, Lisboa, Minho …) avec au moins 85 % des raisins issus de l'aire et un encépagement plus large que les DOC. Elle tient lieu d'IGP sur l'étiquette.", + "nl": "Portugals traditionele aanduiding voor BGA-wijnen (Decreto-Lei 212/2004): een regionale aanduiding (Alentejano, Lisboa, Minho …) met minstens 85 % van de druiven uit het gebied en een ruimere rassenlijst dan de DOC's. Vervangt IGP op het etiket." + }, + "scheme": "pgi", + "sources": [ + { + "label": "Decreto-Lei n.º 212/2004 — menções tradicionais «Denominação de Origem Controlada» / «Vinho Regional» (Diário da República)", + "url": "https://diariodarepublica.pt/dr/detalhe/decreto-lei/212-2004-479875" + }, + { + "label": "IVV — Cadernos de especificações das DOP (Instituto da Vinha e do Vinho)", + "url": "https://www.ivv.gov.pt/np4/8617.html" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "ro:DOC": { + "full": "Vin cu denumire de origine controlată", + "note": { + "en": "Romania's traditional term for PDO wines (Legea 164/2015 art. 32): a delimited vineyard area whose wine is made entirely from its own Vitis vinifera grapes under an ONVPV-approved caiet de sarcini. It stands in for DOP on the label.", + "es": "Término tradicional rumano para vinos DOP (Ley 164/2015, art. 32): zona vitícola delimitada cuyo vino procede íntegramente de su propia uva Vitis vinifera según un pliego aprobado por la ONVPV. Sustituye a DOP en la etiqueta.", + "fr": "Mention traditionnelle roumaine pour les vins AOP (loi 164/2015, art. 32) : aire viticole délimitée dont le vin est issu exclusivement de ses propres raisins de Vitis vinifera selon un cahier des charges approuvé par l'ONVPV. Elle tient lieu de DOP sur l'étiquette.", + "nl": "Roemeniës traditionele aanduiding voor BOB-wijnen (wet 164/2015, art. 32): een afgebakend wijngebied waarvan de wijn volledig uit eigen Vitis vinifera-druiven wordt gemaakt volgens een door ONVPV goedgekeurd caiet de sarcini. Vervangt DOP op het etiket." + }, + "scheme": "pdo", + "sources": [ + { + "label": "Legea nr. 164/2015 a viei și vinului, art. 32 (D.O.C.) și art. 35 (I.G.) (Portal Legislativ)", + "url": "https://legislatie.just.ro/Public/DetaliiDocument/169280" + }, + { + "label": "ONVPV — Legislație națională în domeniul vitivinicol", + "url": "https://www.onvpv.ro/ro/node/23" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + }, + "ro:IG": { + "full": "Vin cu indicație geografică", + "note": { + "en": "Romania's traditional term for PGI wines (Legea 164/2015 art. 35): a delimited area with at least 85 % of the grapes from it, Vitis vinifera or interspecific crosses allowed. It stands in for IGP on the label.", + "es": "Término tradicional rumano para vinos IGP (Ley 164/2015, art. 35): zona delimitada de la que procede al menos el 85 % de la uva, admitiéndose Vitis vinifera o cruces interespecíficos. Sustituye a IGP en la etiqueta.", + "fr": "Mention traditionnelle roumaine pour les vins IGP (loi 164/2015, art. 35) : aire délimitée dont proviennent au moins 85 % des raisins, Vitis vinifera ou hybrides interspécifiques admis. Elle tient lieu d'IGP sur l'étiquette.", + "nl": "Roemeniës traditionele aanduiding voor BGA-wijnen (wet 164/2015, art. 35): een afgebakend gebied waaruit minstens 85 % van de druiven komt, Vitis vinifera of interspecifieke kruisingen toegestaan. Vervangt IGP op het etiket." + }, + "scheme": "pgi", + "sources": [ + { + "label": "Legea nr. 164/2015 a viei și vinului, art. 32 (D.O.C.) și art. 35 (I.G.) (Portal Legislativ)", + "url": "https://legislatie.just.ro/Public/DetaliiDocument/169280" + }, + { + "label": "ONVPV — Legislație națională în domeniul vitivinicol", + "url": "https://www.onvpv.ro/ro/node/23" + }, + { + "label": "Reg. (EC) 607/2009, Annex XII — List of traditional terms (legislation.gov.uk copy)", + "url": "https://www.legislation.gov.uk/eur/2009/607/annex/XII/adopted/data.xht" + } + ] + } + } +} diff --git a/scripts/_lib/translation_glossary.py b/scripts/_lib/translation_glossary.py index c1ccb87..6b65b30 100644 --- a/scripts/_lib/translation_glossary.py +++ b/scripts/_lib/translation_glossary.py @@ -7,19 +7,23 @@ SYSTEM_PROMPT under a blank-line separator; an empty return makes the append a no-op for locales without curated guidance. -Curated for NL and EN based on a sweep of the existing FR + ES → target -corpus (~6850 terroir-fact bullets and ~340 ollama-translated summaries -per target). ES and FR targets were probed and showed no recurring -issues worth a rule today — Mistral handles FR↔ES cleanly, and the few -suspicious-looking FR phrases ("phase visuelle", "vins francs") turn out -to be legitimate French wine vocabulary. +Curated for NL and EN. The first pass came from a sweep of the FR + ES → +target corpus (~6850 terroir-fact bullets and ~340 ollama-translated +summaries per target); the 2026-09-11 corpus-wide review (plan W1) added +the EN calques that the IT / CZ / SI / HU / PT / FR sources produce. Since +that pass the glossary is appended by `terroir_prompts.translation_system_ +prompt` for every 02e script (all 21 source corpora), not only the FR one, +so entries must hold whatever the source language. ES and FR targets were +probed and showed no recurring issues worth a rule today — Mistral handles +FR↔ES cleanly, and the few suspicious-looking FR phrases ("phase visuelle", +"vins francs") turn out to be legitimate French wine vocabulary. """ from __future__ import annotations _NL_GLOSSARY = """\ -Dutch (NL) sommelier-register vocabulary — applies to translations from \ -both French and Spanish source. Prefer the LEFT term over the RIGHT: +Dutch (NL) sommelier-register vocabulary — applies whatever the source \ +language. Prefer the LEFT term over the RIGHT: - "stille wijn(en)" NOT "rustige wijn(en)" — for FR "tranquille" / ES "tranquilo"; "rustige" reads as "calm/peaceful". - "mousserende wijn(en)" NOT "schuimwijn" — for FR "mousseux" / ES "espumoso". - "aroma's" NOT "aromen" — plural of aroma in modern NL wine writing. @@ -32,6 +36,8 @@ - "vlezig" (one E) NOT "vleesig" — fleshy mouthfeel. - "smeuïg" NOT "smeerend" — smooth, unctuous palate. - "geconfijte vruchten" NOT "kandijvruchten" or "confiteraadjes" — for FR "fruits confits" / ES "frutas confitadas". +- "appellatie(s)" NOT "appellation(s)" as the common noun — the site's own NL wording ("de appellatie strekt zich uit over…"); the registered French term "appellation d'origine contrôlée / protégée" is a name and stays as it is. +- "de Marnevallei" / "het Marnedal", "de Audevallei", "de Rhônevallei" NOT a blend such as "Marnedallei" — for FR "la vallée de la Marne / de l'Aude / du Rhône", IT "la valle del …", ES "el valle del …": a river valley is "X-vallei" or "het X-dal"; "-dallei" is not a Dutch word (it appeared in the 2026-09-14 Champagne translation). - "polyfenolen" NOT "polyphenolen" — Dutch spells with f. - "tanninerijk" or "rijk aan tannines" NOT "tannisch" — tannic; "tannisch" is a French borrowing not idiomatic in NL wine writing. - "bottelen" or "flesrijping" NOT "flessenwijze" — for FR "mise en bouteille / élevage en bouteille" / ES "embotellado / crianza en botella". @@ -39,12 +45,30 @@ _EN_GLOSSARY = """\ -English (EN) sommelier-register vocabulary — applies to translations from \ -both French and Spanish source. Prefer the LEFT term over the RIGHT: +English (EN) sommelier-register vocabulary — applies whatever the source \ +language. Prefer the LEFT term over the RIGHT: - "appearance / nose / palate" NOT "visual phase / olfactory phase / gustatory phase" — for FR "phase visuelle/olfactive/gustative" or ES "fase visual/olfativa/gustativa"; these are the standard English tasting-note phase names. - "clean" or "fault-free" (of aromas or wines) NOT "frank" — for ES "vinos francos / aromas francos" (or FR "vins francs"); "frank" carries no oenological meaning in English. - "brick(-red)" or "tile(-red)" NOT "brick tone" or "tile tone" — for FR "tuile" / ES "teja" (the colour of aged wine). -- Render Spanish-pliego header fragments like "Wine product", "Wine product VINO", "Wine product VINO Whites and rosés" as plain "Wines" or just drop them — they are pliego template scaffolding, not titles to preserve.""" +- Render Spanish-pliego header fragments like "Wine product", "Wine product VINO", "Wine product VINO Whites and rosés" as plain "Wines" or just drop them — they are pliego template scaffolding, not titles to preserve. +- "minerality" NOT "mineralité" — for FR "minéralité"; the English word exists. +- "wine-grape varieties" or "grape varieties" NOT "must varieties" — for CZ "moštové odrůdy" / SK "muštové odrody" (the legal class of Vitis vinifera varieties authorised for winemaking). +- "style", "version" or "wine type" NOT "typology" — for IT "tipologia" (the wine types a DOC defines: Riserva, Superiore, Spumante, …). +- "actual alcohol" or "actual alcoholic strength" NOT "developed alcohol" — for IT "gradi svolti" / "alcol svolto" (the alcohol actually present, as opposed to potential). +- "carbonate soils" or "calcareous soils" NOT "carbonated soils" — for FR "sols carbonatés"; "carbonated" means fizzy. +- "vineyard sites" or "named vineyards" NOT "lege" / "dűlők" / "tratě" — for SI "lege", HU "dűlő(k)", CZ "viniční tratě" (the site word is a common noun; a site's own name stays verbatim). +- "para-barros" is a Portuguese soil-classification class (the lighter relatives of the "barros" clay soils) — keep it verbatim; never invent "proto-barros". +- "submerged cap" NOT "grillage" — for FR "grillage" (the grid that holds the cap under the surface in Beaujolais / Burgundy vats); "punch-down" for "pigeage"; "pump-over" for "remontage". +- "climat" (kept, italic-free) NOT "climate" — for the Burgundian FR "climat" meaning a named, delimited vineyard site (Les Clos, La Grande Rue); "climate" only for "climat" in its weather sense. +- "tirage" or "bottling for the second fermentation" NOT "disgorgement" — for FR "tirage" (sparkling: the bottling with liqueur de tirage; "dégorgement" is disgorgement; "à compter du tirage" = "from the date of tirage"). +- "fortified wine(s)" or "generoso wine(s)" NOT "generous wine(s)" — for ES "vino generoso / vinos generosos" (the regulatory category of fortified, oxidatively aged wines: fino, amontillado, oloroso). +- "loam" NOT "clay" — for DE "Lehm" (clay is "Ton"); "loess" for "Löss"; "primary rock" or "crystalline basement" NOT "primeval rock" for "Urgestein". +- "slight sparkle" or "light spritz" NOT "spiciness" — for DE "Spritzigkeit / spritzig". +- "depth of colour" / "deep-coloured" NOT "layer" — for ES "capa" as in "capa alta / media" (colour intensity). +- "Riesling" for BG "Немски ризлинг / Рейнски ризлинг" and "Welschriesling" for "Италиански ризлинг" — never swap the two. +- "Rhodopes" (EN) for BG "Родопи" / "Rodopi"; keep "Stara Planina" as such (a one-time gloss "(Balkan Mountains)" is fine); "Bavaria" for DE "Bayern". +- A grape name is never translated by sound-alike: HU "Pintes" is the variety Pintes, not "Pinot"; HR "Plavac" stays Plavac. +- A demonym is rendered as its town, never back-formed into a place: IT "caiatino" is "of Caiazzo" (not "the Caiata area"), "aversano" is "of Aversa".""" _GLOSSARIES: dict[str, str] = { diff --git a/scripts/at/02d_extract_terroir_facts.py b/scripts/at/02d_extract_terroir_facts.py index c018d5e..4be2436 100644 --- a/scripts/at/02d_extract_terroir_facts.py +++ b/scripts/at/02d_extract_terroir_facts.py @@ -28,7 +28,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -37,6 +36,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "at" / "dokumente-extracted" WIKI_AOCS = ROOT / "raw" / "wikipedia" / "aocs" / "de" @@ -131,37 +138,22 @@ - Zitate sind WÖRTLICH (kopiert und eingefügt) aus der jeweiligen Quelle. Schreibe NIEMALS einer Quelle einen Text zu, der dort nicht vorkommt. - Keine Werturteile ("außergewöhnlich", "prestigeträchtig"...). - Keine externen Schlussfolgerungen. Keine Zahlen, die in keiner der beiden Quellen stehen. -- Maximal {max_bullets} Einträge, je ≤ 140 Zeichen. +- Maximal {max_bullets} Einträge; jeder Eintrag ist ein vollständiger Satz von etwa 120–220 Zeichen — nie ein telegrafisches Fragment. - Wenn weder das Einzige Dokument noch Wikipedia einen konkreten bemerkenswerten Fakt für diesen Unterabschnitt enthält, gib eine leere Liste zurück. Antworte NUR in JSON, ohne Text davor oder danach: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Verwende einen leeren String "" für das fehlende Zitat.""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - """Longest-contiguous-match coverage of `quote` in `source` (0–1).""" - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -274,13 +266,12 @@ def collect_targets() -> list[dict]: return out +def _ask_line(label: str) -> str: + return f"Zu behandelnder Unterabschnitt: {label}" -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Zu behandelnder Unterabschnitt: {label}\n\n" - f"Text des Einzigen Dokuments (Beschreibung des Zusammenhangs):\n\n{lien_text}" - ) +def _document_block(lien_text: str) -> str: + return f"Text des Einzigen Dokuments (Beschreibung des Zusammenhangs):\n\n{lien_text}" def _process_subsection( @@ -296,9 +287,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -333,6 +328,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "de") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "at", "source_lang": "de", @@ -340,6 +341,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -350,7 +353,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -419,12 +422,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(kein Wikipedia-Auszug verfügbar)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -465,7 +468,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts(slug_items: list[dict], lien: str) -> list[dict]: @@ -603,7 +606,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -636,7 +639,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-at.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -686,7 +689,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/at/02e_translate_terroir_facts.py b/scripts/at/02e_translate_terroir_facts.py index 4a68ad7..e655bba 100644 --- a/scripts/at/02e_translate_terroir_facts.py +++ b/scripts/at/02e_translate_terroir_facts.py @@ -27,6 +27,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -39,12 +42,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve German proper nouns verbatim: appellation names ("Wachau", "Kamptal", "Kremstal", "Weinviertel", "Wagram", "Traisental", "Carnuntum", "Thermenregion", "Leithaberg", "Eisenberg", "Neusiedlersee", "Mittelburgenland", "Südsteiermark", "Vulkanland Steiermark", "Weststeiermark", "Ruster Ausbruch", "Wiener Gemischter Satz"), Bundesland names ("Niederösterreich", "Burgenland", "Steiermark", "Wien", "Kärnten", "Oberösterreich", "Salzburg", "Tirol", "Vorarlberg"), commune and single-vineyard ("Ried") names, grape variety names ("Grüner Veltliner", "Zweigelt", "Blaufränkisch", "Welschriesling", "Riesling", "Weißburgunder", "Neuburger", "Sankt Laurent", "Sauvignon Blanc", "Muskateller", "Rotgipfler", "Zierfandler", "Blauer Wildbacher"), named geological formations and soil types ("Löss", "Urgestein", "Gneis", "Glimmerschiefer", "Schiefer", "Kalk", "Konglomerat", "Schwemmland", "Verwitterungsböden", "Vulkangestein"), named climatic features ("pannonisches Klima", "illyrisches Klima", "Donau-Einfluss"), and Austrian wine-law / Prädikat terms ("Steinfeder", "Federspiel", "Smaragd", "Ried", "Riedenwein", "DAC", "Gemischter Satz", "Spätlese", "Auslese", "Beerenauslese", "Trockenbeerenauslese", "Ausbruch", "Eiswein", "Strohwein"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the German form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "de" +PROPER_NOUNS = """appellation names (Wachau, Kamptal, Kremstal, Weinviertel, Wagram, Traisental, Carnuntum, Thermenregion, Leithaberg, Eisenberg, Neusiedlersee, Mittelburgenland, Südsteiermark, Vulkanland Steiermark, Weststeiermark, Ruster Ausbruch, Wiener Gemischter Satz); Bundesland names (Niederösterreich, Burgenland, Steiermark, Wien, Kärnten, Oberösterreich, Salzburg, Tirol, Vorarlberg); commune and single-vineyard names; grape names (Grüner Veltliner, Zweigelt, Blaufränkisch, Welschriesling, Riesling, Weißburgunder, Neuburger, Sankt Laurent, Sauvignon Blanc, Muskateller, Rotgipfler, Zierfandler, Blauer Wildbacher); registered Austrian terms and Prädikat tiers (DAC, Steinfeder, Federspiel, Smaragd, Gemischter Satz, Spätlese, Auslese, Beerenauslese, Trockenbeerenauslese, Ausbruch, Eiswein, Strohwein)""" + # ─────────────────────────────────────────────────────────────── helpers ── @@ -102,7 +107,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -157,10 +162,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -291,6 +301,7 @@ def _build_argparser() -> argparse.ArgumentParser: "--workers", type=int, default=1, help="concurrent (lang, slug) pairs (default 1, keep 1 for Ollama)", ) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true", help="re-translate even if cached") ap.add_argument( "--batch", action="store_true", @@ -378,8 +389,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -411,6 +424,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] @@ -420,7 +435,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: for j in jobs: diff --git a/scripts/audit_gi_terms.py b/scripts/audit_gi_terms.py new file mode 100644 index 0000000..e480b32 --- /dev/null +++ b/scripts/audit_gi_terms.py @@ -0,0 +1,125 @@ +#!/usr/bin/env python3 +"""Audit the two naming axes on the built corpus: legal scheme + traditional term. + +Reads the EN startup blob (wiki/data/aocs.en.<hash>.js) and reports, per +country, the eu_scheme / national_term distribution (parents, wines only), the +records with no class_label, sub-denominations whose axes differ from their +parent, and terms whose registered scheme (traditional_terms.json) contradicts +the record's scheme. IT and ES counts are compared with the rosters +(MASAF elenco / MAPA listado) that feed them. + + .venv/bin/python scripts/audit_gi_terms.py [--strict] + +--strict exits non-zero on any mismatch or empty class_label. Read-only. +""" +from __future__ import annotations + +import argparse +import collections +import glob +import json +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib.gi_terms import load_terms_table, term_key # noqa: E402 + + +def load_blob() -> dict: + paths = sorted(glob.glob(str(ROOT / "wiki" / "data" / "aocs.en.*.js"))) + if not paths: + sys.exit("no wiki/data/aocs.en.*.js — run scripts/04_build_maps.py first") + src = Path(paths[-1]).read_text(encoding="utf-8") + return json.loads(src[src.index("{"): src.rindex("}") + 1])["aocs"] + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) + ap.add_argument("--strict", action="store_true") + args = ap.parse_args() + + aocs = load_blob() + table = load_terms_table() + term_scheme = {k: v.get("scheme") for k, v in (table.get("terms") or {}).items()} + problems: list[str] = [] + + by_country: dict[str, collections.Counter] = collections.defaultdict(collections.Counter) + for slug, r in aocs.items(): + if r.get("is_sub_denomination") or r.get("is_wine") is False: + continue + by_country[r.get("country") or "?"][(r.get("eu_scheme") or "-", r.get("national_term") or "")] += 1 + print("country scheme term n") + for cc in sorted(by_country): + for (scheme, term), n in sorted(by_country[cc].items(), key=lambda kv: (-kv[1], kv[0])): + print(f"{cc:<8} {scheme:<11} {term or '(none)':<26} {n:>5}") + + empty = [s for s, r in aocs.items() if not r.get("class_label")] + if empty: + problems.append(f"{len(empty)} records with empty class_label: {empty[:10]}") + + for slug, r in aocs.items(): + if not r.get("is_sub_denomination"): + continue + parent = aocs.get(r.get("parent_slug") or "") + if not parent: + continue + mine = (r.get("eu_scheme"), r.get("national_term")) + theirs = (parent.get("eu_scheme"), parent.get("national_term")) + if mine != theirs: + problems.append(f"sub {slug} {mine} != parent {r.get('parent_slug')} {theirs}") + + for slug, r in aocs.items(): + term = r.get("national_term") or "" + if not term: + continue + registered = term_scheme.get(f"{r.get('country')}:{term}") + if registered is None: + problems.append(f"{slug}: term {term!r} has no entry in traditional_terms.json terms") + continue + actual = (r.get("eu_scheme") or "").replace("uk-", "") + if registered != actual and not (registered == "pdo" and actual == "spirit-gi"): + problems.append(f"{slug}: term {term!r} is registered {registered} but record is {actual}") + + it_parents = collections.Counter( + r.get("national_term") or "" for r in aocs.values() + if r.get("country") == "it" and not r.get("is_sub_denomination") + ) + try: + from _lib.it.national_term import load_it_terms + roster = collections.Counter(load_it_terms().values()) + print(f"\nIT parents: {dict(it_parents)} | MASAF elenco: {dict(roster)}") + except Exception as exc: # roster optional (no raw/ in CI) + print(f"\nIT parents: {dict(it_parents)} | roster unavailable: {exc}") + es_parents = collections.Counter( + r.get("national_term") or "" for r in aocs.values() + if r.get("country") == "es" and not r.get("is_sub_denomination") + ) + print(f"ES parents: {dict(es_parents)}") + + facet_keys = collections.Counter() + for r in aocs.values(): + if r.get("is_sub_denomination") or r.get("is_wine") is False: + continue + for tok in (r.get("class_key") or "").strip(";").split(";"): + if tok: + facet_keys[tok] += 1 + unknown = [t for t in facet_keys if ":" in t and t not in { + term_key(k.split(":")[0], k.split(":", 1)[1]) for k in term_scheme + }] + if unknown: + problems.append(f"facet term keys without a tooltip entry: {unknown}") + + print() + if problems: + print(f"{len(problems)} problem(s):") + for p in problems: + print(" -", p) + return 1 if args.strict else 0 + print("OK — no mismatches") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/audit_terroir_facts.py b/scripts/audit_terroir_facts.py index da1096c..8e97420 100644 --- a/scripts/audit_terroir_facts.py +++ b/scripts/audit_terroir_facts.py @@ -1,335 +1,579 @@ -"""Health check for stage 02d terroir-fact caches. - -Not a pipeline stage. Run after 02d (or on a schedule) to detect: - - Drift: cached `cahier_source_sha` or `wiki_source_revision` no longer - matches the current cahier / Wikipedia cache (a source was refreshed - upstream; stage 02d should be re-run for affected AOCs). - - Coverage erosion: per bullet, recompute fuzzy_coverage against the - CURRENT cahier text and Wikipedia hint. If a bullet's coverage has - dropped below the threshold, flag it (the source moved out from under - the bullet). - - Length cap: bullets whose text exceeds 140 chars (soft cap). - - Provenance / sub-section / translator-kind distributions. - -Outputs: - - Per-AOC report on stderr (one line per AOC unless --verbose). - - Aggregate summary (counts + medians) at the end. - - Optional `--report PATH` writes the full per-AOC + per-bullet audit - as JSON for follow-up tooling. - -Exit code is non-zero only on internal errors (file I/O); audit findings -are reported but don't fail the run. +"""Health check for the stage-02d terroir-fact caches and their stage-02e +translations. Not a pipeline stage: run it after 02d / 02e (or on a +schedule), and in `--strict` mode as the acceptance gate before a build. + +Sources are resolved through each country's own stage-02d module +(`_lib.terroir_sources.resolve_sources`) — lazily, once per country, and +only for the countries present in the selected caches — so every record +is graded against precisely the text 02d graded it against: the lien (or +the CH / MT / GB context block), the national-spec sidecar fallbacks and +the per-sub-section Wikipedia hint with its country-specific cap. + +Checks. S = strict (counts towards the `--strict` exit code), R = report +only (a count and the offending rows, never a failure): + + cahier / wiki drift R cached `cahier_source_sha` / + `wiki_source_revision` no longer match the + current sources — 02d should be re-run for + the record. + erosion R per bullet, the coverage of `cahier_quote` + / `wiki_quote` is recomputed against the + CURRENT sources; a bullet none of whose + quotes still reaches FUZZY_THRESHOLD has + eroded (the source moved out from under it). + length cap R bullets over 240 chars (soft cap; the + prompts ask for 120–220). + non_latin S Cyrillic or Greek characters in a + translated bullet — every locale under + raw/translations/terroir-facts/ is + Latin-script. + colour_code S a regulatory grape colour code left in a + bullet ("pinot noir N", "riesling B"), + source and translated; only fires after a + real grape name (`strip_colour_codes`). + no_terminal_punct S the bullet does not end in . ! ? … — + source and translated. + arrow R "→" in a bullet, source and translated. + label_prefix R the bullet opens with "Label: " — a + sub-zone name ("Rioja Oriental: …") is a + legitimate lead, so report only. + meta_text R the bullet talks about the document or + Wikipedia instead of the terroir + ("confirmed by Wikipedia"). + intra_record_duplicates S two facts of one record that + `duplicate_reason` deems restatements + (near-identical bullets, or the same + source quote with overlapping bullets). + cross_record_shared_quotes R one normalised `cahier_quote` (≥ 60 chars) + shared by ≥ 3 records of a country — a + shared cahier or boilerplate. The count is + the number of such groups; the report lists + the 30 largest with their slugs. + name_guard S an FR record whose current lien (≥ 800 + chars) never names the appellation — the + signature of the wrong BO Agri PDF bound + to the record (plan W2b). The name is + tokenised (accents and case folded, stop + words and tokens under 4 letters dropped) + and one token must occur in the lien. + `NAME_GUARD_WHITELIST` lists the verified + exceptions (Saône-et-Loire: correct lien + that never names it). + no_own_chapter S an FR record on a shared cahier (the 51 + Alsace grands crus share one lien) whose + own `« Alsace grand cru <Cru> »` chapter + cannot be located (plan W2a). + quote_outside_own_chapter S a shared-cahier fact whose `cahier_quote` + does not ground (FUZZY_THRESHOLD) inside the + record's own chapter — it was extracted + from another cru's chapter (plan W2a). + feedback_recurrence R a do-not-claim entry of the record's review + feedback (`raw/terroir-facts-feedback/`, + `_lib/terroir_feedback.py`) whose + source-language bullet still matches a + current bullet: the known misleading claim + is still there (before a 02d re-run) or + came back (after one). Only extraction / + both entries count. + wiki_with_cahier_quote R provenance `wiki` although a `cahier_quote` + is present: the quote grounds below the + threshold (paraphrase or typography, plan + W4). + multi_sentence R a source bullet holding two or more + sentences (review 2026-09-12, R9). + en_equals_src R a translated bullet identical to its + source bullet — the translation did not + happen. + cross_record_identical_en R one EN bullet shared verbatim by ≥ 2 + records (the Alsace produit slice, the + retsina cluster); count = groups. + foreign_name R the record's source text names ANOTHER + appellation of the same country ≥ 3 times + and its own 0 times — a pasted section + (ΥΠΑΑΤ), a mis-bound PDF, an annex (R4). + wiki_binding R the bound Wikipedia article's title shares + no token with the record name (Toro for + Tirol, a village for an IGT); curator + pins are trusted. + masaf_sidecar_stale R an IT record whose MASAF sidecar predates + the annex-aware article slicer (R4). + gate_pending R a record never passed through the + claim-support gate, or whose facts changed + since (`scripts/02d_verify_terroir_facts.py`). + rewrite_rejected R a gate rewrite the guards refused (kept the + original bullet) — for a human look. + translation_stale R a translation cache keyed to a source facts + sha that is no longer the record's, and not + re-keyed `pending:` — 02e will redo it on its + next corpus-wide pass; listed so it is not a + surprise there. + rewrite_missing R a gate `rewrite` verdict that came back with + no rewrite text (kept the original bullet + as supported, note kept) — for a human look. + + name_guard_other R the name guard for the 20 non-FR countries + (the record's source text through + `terroir_sources`). Report only: CZ + records ground on a region-wide CHZO or a + per-wine-type fiche text that never names + the wine by design, and inflected names + (Greek, Slavic, Hungarian) are matched on + a crude stem. + +The name guard requires the whole folded name minus stop words — or the +stem of its longest token of ≥ 6 letters — to occur, not any 4-letter +token ("tout" and "grains" occur in any French cahier). `--strict-labels` +promotes `label_prefix` to strict once the pre-style-block corpus has +been re-extracted. + +The three FR checks read the extracted record's full `lien_au_terroir` +(raw/inao/cahier-extracted/), not the text 02d grades against: stage 02d +now pre-slices a shared cahier to the record's own chapter, and the +checks must still catch a cache extracted before it did. + +The drift / erosion / name-guard / chapter checks need the record's +current sources; a record whose slug is no longer a 02d target is graded +against empty sources (so its bullets report as eroded) and listed under +`no_source_today_aocs`, and a country whose 02d module fails to load is +listed under `source_unresolved_countries` with its records exempt from +those checks (the pure bullet checks still run). + +Outputs: one line per record on stderr (unless --quiet; --verbose adds +the bullets), the aggregate summary as JSON on stderr, and with +`--report PATH` the full JSON: `summary`, per-record `aocs` (with +per-bullet coverage and style findings) and `findings` — the offending +(slug[, lang, index]) rows per check. + +Exit code: 0, or 1 on an internal error (missing cache directory). With +`--strict`: 1 when any strict check has a count > 0. + +Usage: + .venv/bin/python scripts/audit_terroir_facts.py --quiet --report tmp/terroir-facts-review/audit.json + .venv/bin/python scripts/audit_terroir_facts.py --country fr --slug chablis --verbose + .venv/bin/python scripts/audit_terroir_facts.py --strict --quiet """ from __future__ import annotations import argparse -import hashlib import json +import random import re import statistics import sys -from collections import Counter +import time +import unicodedata +from collections import Counter, defaultdict from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path +from urllib.parse import unquote ROOT = Path(__file__).resolve().parent.parent -TERROIR_FACTS = ROOT / "raw" / "terroir-facts" - -# Per-country source dispatch. Each entry maps the country code carried in -# the terroir-facts cache to (extracted records dir, wiki cache dir, -# record-key for the lien text). Keep in sync with stage 02d's two -# variants — `scripts/02d_extract_terroir_facts.py` (FR) and -# `scripts/es/02d_extract_terroir_facts.py` (ES). -EXTRACTED_BY_COUNTRY = { - "fr": ROOT / "raw" / "inao" / "cahier-extracted", - "es": ROOT / "raw" / "es" / "pliegos-extracted", - "gb": ROOT / "raw" / "gb" / "specs-extracted", -} -WIKI_BY_COUNTRY = { - "fr": ROOT / "raw" / "wikipedia" / "aocs" / "fr", - "es": ROOT / "raw" / "wikipedia" / "aocs" / "es", - # GB's source language is English (the DEFRA product specifications). - "gb": ROOT / "raw" / "wikipedia" / "aocs" / "en", -} -LIEN_FIELD_BY_COUNTRY = { - "fr": "lien_au_terroir", - "es": "link_to_terroir", - "gb": "link_to_terroir", -} - -FUZZY_THRESHOLD = 0.6 -BULLET_SOFT_CAP = 140 -WIKI_HINT_CHAR_CAP = 1500 -# Each country's stage 02d caps the Wikipedia hint at its own length; the -# audit must reproduce that cap or a perfectly-grounded `wiki` bullet whose -# quote sits past the audit's shorter cap reports as spuriously "eroded". -WIKI_HINT_CAP_BY_COUNTRY = { - "gb": 2800, # scripts/gb/02d_extract_terroir_facts.py -} - - -def wiki_hint_cap(country: str) -> int: - return WIKI_HINT_CAP_BY_COUNTRY.get(country, WIKI_HINT_CHAR_CAP) - -TOP_RE = re.compile(r"\b([1-9])°\s*[-–]\s*([A-ZÀ-Ý][^\n]{5,80})") -SUB_RE = re.compile(r"\b([a-c])\)\s*[-–]?\s*([A-ZÀ-Ý][^\n]{5,80})") - -WIKI_TO_SUBSECTION_FR: dict[str, list[str]] = { - "facteurs_naturels": [ - "Géologie et orographie", "Géologie", "Climat", "Climatologie", - "Aire d'appellation", "Vignoble", - ], - "facteurs_humains": [ - "Histoire", "Antiquité", "Moyen Âge", "Période moderne", - "Période contemporaine", "Étymologie", "Encépagement", - "Méthodes culturales et réglementaires", "Vinification et élevage", - ], - "produit": ["Vins", "Types de chablis", "Types de vins", "Gastronomie"], - "interactions": [], -} -WIKI_TO_SUBSECTION_FR["interactions"] = WIKI_TO_SUBSECTION_FR["facteurs_naturels"] - -# Spanish Wikipedia headings — mirror scripts/es/02d_extract_terroir_facts.py. -WIKI_TO_SUBSECTION_ES: dict[str, list[str]] = { - "facteurs_naturels": [ - "Geografía", "Geología", "Geología y orografía", "Suelos", - "Clima", "Climatología", "Zona de producción", "Zona geográfica", - "Subzonas", "Comarca", "Viñedo", "Localización", - ], - "facteurs_humains": [ - "Historia", "Antigüedad", "Edad Media", "Edad Moderna", - "Etimología", "Variedades autorizadas", "Variedades de uva", - "Elaboración", "Vinificación", "Cultivo de la vid", - "Crianza", "Tradición", - ], - "produit": [ - "Vinos", "Tipos de vinos", "Características de los vinos", - "Gastronomía", "Maridaje", - ], - "interactions": [], -} -WIKI_TO_SUBSECTION_ES["interactions"] = WIKI_TO_SUBSECTION_ES["facteurs_naturels"] - -# English Wikipedia headings — mirror scripts/gb/02d_extract_terroir_facts.py. -WIKI_TO_SUBSECTION_EN: dict[str, list[str]] = { - "facteurs_naturels": [ - "Geography", "Geology", "Climate", "Soil", "Soils", "Terroir", - "Wine regions", "Regions", "Viticulture", "Vineyards", - ], - "facteurs_humains": [ - "History", "Grapes", "Grape varieties", "Varieties", "Production", - "Winemaking", "Wineries", - ], - "produit": ["Wines", "Styles", "Wine styles", "Types of wine", "Production"], - "interactions": [], -} -WIKI_TO_SUBSECTION_EN["interactions"] = WIKI_TO_SUBSECTION_EN["facteurs_naturels"] - -WIKI_TO_SUBSECTION_BY_COUNTRY = { - "fr": WIKI_TO_SUBSECTION_FR, - "es": WIKI_TO_SUBSECTION_ES, - "gb": WIKI_TO_SUBSECTION_EN, -} - - -# Helpers duplicated from scripts/02d_extract_terroir_facts.py because the -# 02d module name starts with a digit and isn't directly importable. Keep in -# sync if either file changes. - -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - -def cahier_sha(text: str) -> str: - return hashlib.sha256(text.encode("utf-8")).hexdigest() - - -def fuzzy_coverage(quote: str, source: str) -> float: - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib import cache # noqa: E402 +from _lib.terroir_cache import LANGS, TERROIR, TRANSLATIONS # noqa: E402 +from _lib.terroir_chapters import is_shared, own_chapter # noqa: E402 +from _lib.terroir_coverage import ( # noqa: E402 + FUZZY_THRESHOLD, + SourceMatcher, + fuzzy_coverage, + normalize, +) +from _lib.terroir_dedupe import duplicate_reason, facts_sha # noqa: E402 +from _lib.terroir_feedback import load_feedback, recurrence_findings # noqa: E402 +from _lib.terroir_normalize import strip_colour_codes # noqa: E402 +from _lib.terroir_sources import COUNTRIES, Sources, resolve_sources # noqa: E402 + +FR_EXTRACTED = ROOT / "raw" / "inao" / "cahier-extracted" + +BULLET_SOFT_CAP = 240 + +NAME_GUARD_MIN_LIEN = 800 +NAME_GUARD_MIN_TOKEN = 4 +NAME_GUARD_LONG_TOKEN = 6 +# saone-et-loire: a correct lien that never names it. calvados-domfontais: SIQO +# misspells the name (the cahier and the register say "Domfrontais" — see +# scripts/_lib/fr/register_overrides.json), so the stem test cannot match. +NAME_GUARD_WHITELIST = {"saone-et-loire", "calvados-domfontais"} +FOREIGN_NAME_MIN_HITS = 5 +FOREIGN_NAME_MIN_CHARS = 6 +MASAF_SIDECARS = ROOT / "raw" / "it" / "masaf-disciplinari-extracted" +MASAF_CURRENT_TEMPLATE = "it-masaf-disciplinare-v3" +WIKI_AOCS = ROOT / "raw" / "wikipedia" / "aocs" +_WIKI_LANG = {"at": "de", "si": "sl", "gr": "el", "cz": "cs", "lu": "fr", "mt": "en", "gb": "en", "cy": "el"} +NAME_STOP_WORDS = frozenset( + "aoc aop igp vin vins de du des d l la le les et ou saint sainte grand grands cru crus " + "village villages cotes coteaux premier cote mont pays val vallee".split() +) + +SHARED_QUOTE_MIN_CHARS = 60 +SHARED_QUOTE_MIN_RECORDS = 3 +SHARED_QUOTE_TOP = 30 +NON_LATIN_RE = re.compile(r"[Ѐ-ӿͰ-Ͽ]") +ARROW_RE = re.compile("→") +# A one- or two-word label ("Colour:", "Rioja Oriental:"), no parenthesis — +# not an enumerating colon inside a sentence ("Three main soil types +# coexist:", "Zierfandler (Synonym: Spätrot)"). +LABEL_PREFIX_RE = re.compile(r"^\s*(?:[^\s:(\d]+\s){0,1}[^\s:(\d]+:\s") +META_RE = re.compile( + r"(?i)\b(wikipedia|according to the (production |product )?(document|specification|cahier|disciplinare|pliego)|confirmed by" + r"|the (cahier|disciplinare|specification|pliego) (states|says|notes)" + # source-language forms: "secondo il disciplinare", "selon le cahier des charges", + # "según el pliego", "laut (der) Produktspezifikation", "volgens het productdossier" + r"|secondo (il|quanto (previsto|indicato|riportato) (dal|nel)) disciplinare|il disciplinare (prevede|stabilisce|indica|riporta)" + r"|selon le cahier des charges|le cahier des charges (précise|indique|prévoit|stipule)" + r"|según (el|lo (establecido|indicado) en el) pliego|el pliego (establece|indica|recoge)" + r"|laut (der |dem )?(produktspezifikation|einzige[nm] dokument)|gemäß (der |dem )?(produktspezifikation|einzige[nm] dokument)" + r"|volgens het (productdossier|enig document)|conform het (productdossier|enig document)" + r"|segundo o caderno|de acordo com o caderno)\b" +) +TERMINAL_PUNCT = ".!?…" +MULTI_SENTENCE_RE = re.compile(r"[.!?]\s+[A-ZÀ-ÝΑ-ΩА-Я]") + +STRICT_CHECKS = frozenset({ + "non_latin", "colour_code", "no_terminal_punct", "intra_record_duplicates", + "name_guard", "no_own_chapter", "quote_outside_own_chapter", +}) +# Every check name, in report order. +CHECKS = ( + "non_latin", "colour_code", "no_terminal_punct", "arrow", "label_prefix", "meta_text", + "multi_sentence", "intra_record_duplicates", "cross_record_shared_quotes", + "cross_record_identical_en", "en_equals_src", "name_guard", "name_guard_other", "foreign_name", "wiki_binding", + "no_own_chapter", "quote_outside_own_chapter", "wiki_with_cahier_quote", + "feedback_recurrence", "masaf_sidecar_stale", "gate_pending", "rewrite_rejected", + "rewrite_missing", "translation_stale", +) +_LETTERS_RE = re.compile(r"[^\W\d_]+") + + +def log(msg: str) -> None: + print(f"[audit] {msg}", file=sys.stderr) + + +# ──────────────────────────────────────────────── pure checks (bullet) ── + + +_CHEM_PREFIX_RE = re.compile(r"\b[αβγδ]-(?=[A-Za-z])") + + +def has_non_latin(bullet: str) -> bool: + """Greek / Cyrillic script in a bullet — ignoring a Greek-letter + chemical prefix ("α-terpineol"), which is correct chemistry.""" + return bool(NON_LATIN_RE.search(_CHEM_PREFIX_RE.sub("", bullet or ""))) + + +def has_colour_code(bullet: str) -> bool: + """A grape colour code (` N` / ` B` / ` G` / ` Rs` / ` Rg`) after a grape + name — `strip_colour_codes` leaves any other capital alone.""" + return bool(bullet) and strip_colour_codes(bullet) != bullet + + +def has_arrow(bullet: str) -> bool: + return bool(ARROW_RE.search(bullet or "")) + + +def has_label_prefix(bullet: str) -> bool: + return bool(LABEL_PREFIX_RE.match(bullet or "")) + + +def missing_terminal_punct(bullet: str) -> bool: + b = (bullet or "").rstrip() + return not b or b[-1] not in TERMINAL_PUNCT -def _find_heading(full: str, heading: str) -> int: - idx = full.find(f"\n\n{heading}\n\n") - if idx == -1: - idx = full.find(f"\n{heading}\n") - return idx +def has_meta_text(bullet: str) -> bool: + return bool(META_RE.search(bullet or "")) + + +def has_multiple_sentences(bullet: str) -> bool: + """Two or more sentences: a terminator followed by a capital. A decimal + ("13.5 % vol") or an abbreviation before a lowercase word does not fire.""" + return bool(MULTI_SENTENCE_RE.search((bullet or "").strip()[:-1])) + + +STYLE_CHECKS = ( + ("colour_code", has_colour_code), + ("arrow", has_arrow), + ("label_prefix", has_label_prefix), + ("no_terminal_punct", missing_terminal_punct), + ("meta_text", has_meta_text), + ("multi_sentence", has_multiple_sentences), +) + + +def style_findings(bullet: str) -> list[str]: + """Names of the style checks that fire on `bullet`, in STYLE_CHECKS order.""" + return [name for name, check in STYLE_CHECKS if check(bullet)] -def _index_wiki_sections(full: str, headings: list[str]) -> dict[str, str]: - positions = sorted( - (idx, h) for h in headings if (idx := _find_heading(full, h)) != -1 - ) - section_text: dict[str, str] = {} - intro_end = positions[0][0] if positions else len(full) - section_text["__intro__"] = full[:intro_end].strip() - for i, (start, h) in enumerate(positions): - end = positions[i + 1][0] if i + 1 < len(positions) else len(full) - body_start = start + len(h) + 2 - section_text[h] = full[body_start:end].strip() - return section_text - - -def _build_subsection_hint_fr( - sub_key: str, wanted: list[str], section_text: dict[str, str], - char_cap: int = WIKI_HINT_CHAR_CAP, -) -> str: - """FR wiki-hint format used by `scripts/02d_extract_terroir_facts.py`.""" - chunks: list[str] = [] - if sub_key == "facteurs_naturels" and section_text.get("__intro__"): - chunks.append(section_text["__intro__"][:400]) - for h in wanted: - body = section_text.get(h, "").strip() - if body: - chunks.append(f"« {h} » : {body}") - joined = "\n\n".join(chunks) - if len(joined) > char_cap: - joined = joined[:char_cap].rsplit(" ", 1)[0] + " […]" - return joined - - -def _build_subsection_hint_es( - wiki_record: dict, headings: list[str], char_cap: int = WIKI_HINT_CHAR_CAP, -) -> str: - """ES wiki-hint format — mirrors `_wiki_hint_for_subsection` in - `scripts/es/02d_extract_terroir_facts.py`. Different from the FR - builder (lead intro always prepended; `# {h}` separator instead of - « {h} » :; raw char-cap, no word-boundary trim) — keeping them in - lockstep is what makes the audit's coverage check meaningful for ES - `wiki`-only bullets.""" - full = wiki_record.get("full_text") or "" - if not full: - return (wiki_record.get("lead_extract") or "")[:char_cap] - section_text = _index_wiki_sections(full, headings) - pieces = [section_text["__intro__"]] if section_text.get("__intro__") else [] - for h in headings: - if h in section_text: - pieces.append(f"# {h}\n{section_text[h]}") - blob = "\n\n".join(pieces).strip() - if blob: - return blob[:char_cap] - return (wiki_record.get("lead_extract") or "")[:char_cap] - - -def load_wiki_hints(slug: str, country: str) -> tuple[dict[str, str], dict | None]: - wiki_dir = WIKI_BY_COUNTRY[country] - headings_map = WIKI_TO_SUBSECTION_BY_COUNTRY[country] - cache = wiki_dir / f"{slug}.json" - empty = dict.fromkeys(headings_map, "") - if not cache.exists(): - return empty, None - data = json.loads(cache.read_text(encoding="utf-8")) - if data.get("missing") or data.get("error"): - return empty, data - # GB's stage 02d builds its hint the ES way — full intro, then - # "# {heading}" blocks, raw char cap — not the FR way. Routing it - # through the FR builder matches no heading at all and leaves only a - # 400-char intro, which reports well-grounded `wiki` bullets as eroded. - if country in ("es", "gb"): - cap = wiki_hint_cap(country) - out = { - sub_key: _build_subsection_hint_es(data, headings, cap) - for sub_key, headings in headings_map.items() - } - return out, data - section_text = _index_wiki_sections(data.get("full_text", ""), data.get("sections", [])) - cap = wiki_hint_cap(country) - out = { - sub_key: _build_subsection_hint_fr(sub_key, wanted, section_text, cap) - for sub_key, wanted in headings_map.items() - } - return out, data +# ───────────────────────────────────────────────── pure checks (record) ── -# ─────────────────────────────────────────────────────────────── audit ── +def fold(s: str) -> str: + s = unicodedata.normalize("NFKD", s or "") + s = "".join(c for c in s if not unicodedata.combining(c)) + return s.casefold() -def load_current_cahier(slug: str, country: str) -> tuple[str, str] | None: - """Return (lien_text, sha) for the current cahier/pliego extract, or None.""" - extracted_dir = EXTRACTED_BY_COUNTRY[country] - field = LIEN_FIELD_BY_COUNTRY[country] - p = extracted_dir / f"{slug}.json" - if not p.exists(): + +def name_tokens(name: str) -> list[str]: + """The appellation-name tokens a lien is expected to mention: letter + runs of the folded name, minus stop words and tokens under + NAME_GUARD_MIN_TOKEN letters.""" + return [ + t for t in _LETTERS_RE.findall(fold(name)) + if t not in NAME_STOP_WORDS and len(t) >= NAME_GUARD_MIN_TOKEN + ] + + +def lien_names_record(lien: str, name: str) -> bool | None: + """True / False: the folded lien contains the whole folded name minus + stop words, or — when the name has one — its longest token of at least + NAME_GUARD_LONG_TOKEN letters. Any 4-letter token was too weak: "tout" + and "grains" occur in every French cahier, so Bourgogne + Passe-tout-grains passed on the Beaujolais cahier (review 2026-09-12). + None: the name has no usable token, so nothing can be required.""" + tokens = name_tokens(name) + if not tokens: return None - try: - rec = json.loads(p.read_text(encoding="utf-8")) - except Exception: # noqa: BLE001 + folded = fold(lien) + whole = " ".join(tokens) + if whole in folded: + return True + longest = max(tokens, key=len) + if len(longest) >= NAME_GUARD_LONG_TOKEN and name_stem(longest) in folded: + return True + if len(longest) < NAME_GUARD_LONG_TOKEN: + return any(t in folded for t in tokens) + return False + + +def name_stem(token: str) -> str: + """A crude inflection-tolerant stem: the token minus its last two + letters (never shorter than 5). Greek, Slavic, Hungarian and Romanian + names decline — «Ρετσίνα Βοιωτίας» is named «Βοιωτία» in its own text, + «Šobes» appears as «Šobesu» — so the longest-token test matches the + stem, not the dictionary form.""" + return token[: max(5, len(token) - 2)] + + +def name_guard_finding(lien: str, name: str, slug: str) -> bool: + """True when the W2b guard fires: a lien of at least NAME_GUARD_MIN_LIEN + chars that never names the appellation, for a slug not whitelisted.""" + if slug in NAME_GUARD_WHITELIST or len(lien) < NAME_GUARD_MIN_LIEN: + return False + return lien_names_record(lien, name) is False + + +def foreign_names(lien: str, name: str, others: dict[str, str]) -> list[dict]: + """Other appellations of the same country whose whole folded name + (≥ FOREIGN_NAME_MIN_CHARS) occurs ≥ FOREIGN_NAME_MIN_HITS times in the + record's source text while the record's own name never does — the + signature of a pasted section, a mis-bound PDF or an annex. `others` + is {slug: folded name}. A name that is a substring of the record's own + name (Chianti inside Chianti Classico) is skipped.""" + if len(lien) < NAME_GUARD_MIN_LIEN or lien_names_record(lien, name) is not False: + return [] + folded = fold(lien) + own = " ".join(name_tokens(name)) + out: list[dict] = [] + for slug, other in others.items(): + key = name_stem(max(other.split(), key=len)) if other else "" + # A short stem ("terre", "colli", "pisa") is a generic word, not a name. + if len(key) < FOREIGN_NAME_MIN_CHARS or key in own or own in other: + continue + n = folded.count(key) + if n >= FOREIGN_NAME_MIN_HITS: + out.append({"other": slug, "hits": n}) + out.sort(key=lambda r: -r["hits"]) + return out[:5] + + +def wiki_binding_finding(country: str, source_lang: str, slug: str, name: str) -> dict | None: + """The bound Wikipedia article's title must share a name token (or a + 5-letter prefix) with the record name; curator pins are trusted.""" + lang = source_lang or _WIKI_LANG.get(country, country) + w = cache.read_json_or_none(WIKI_AOCS / lang / f"{slug}.json") + if not w or w.get("missing") or w.get("error") or w.get("override_source") == "curator": + return None + title = w.get("title") or w.get("wiki_title") or unquote((w.get("page_url") or "").rsplit("/", 1)[-1]).replace("_", " ") + if not title: return None - if country == "gb": - # GB's stage 02d grounds on a composite context, not the bare lien - # field — region + demarcated area + variety roster + the link - # section — and hashes that. Rebuild it identically here or every - # GB record reports a spurious `cahier-drift`. - # Mirrors `_cahier_context` in scripts/gb/02d_extract_terroir_facts.py. - lien = _cahier_context_gb(rec) - else: - lien = (rec.get(field) or "").strip() - return lien, cahier_sha(lien) - - -def _cahier_context_gb(record: dict) -> str: - pieces = [f"Region: {record.get('region') or ''} ({record.get('kind') or ''})"] - geo = record.get("geo_area_brief") or "" - if geo: - pieces.append(f"Demarcated area: {geo}") - grapes = (record.get("grapes") or {}).get("details") or [] - if grapes: - names = ", ".join(g.get("name") or g.get("slug") for g in grapes[:40]) - pieces.append(f"Authorised varieties: {names}") - link = record.get("link_to_terroir") or "" - if link: - pieces.append(f"Link with the geographical area:\n{link}") - return "\n".join(pieces) - - -class UnsupportedCountry(Exception): - """Raised for a terroir-facts cache whose country has no source - dispatch entry — see EXTRACTED_BY_COUNTRY.""" - - -def audit_one(cache_path: Path) -> dict: - """Audit one terroir-facts cache file. Returns a dict with the per-AOC - findings (drift flags + per-bullet coverage).""" - data = json.loads(cache_path.read_text(encoding="utf-8")) - slug = data.get("slug") or cache_path.stem + t_tokens = name_tokens(title) + n_tokens = name_tokens(name) + for a in n_tokens: + for b in t_tokens: + if a == b or a in b or b in a or (len(a) >= 5 and len(b) >= 5 and a[:5] == b[:5]): + return None + return {"title": title, "lang": lang} + + +def masaf_sidecar_stale(slug: str) -> bool: + d = cache.read_json_or_none(MASAF_SIDECARS / f"{slug}.json") + return bool(d) and d.get("parser_template") != MASAF_CURRENT_TEMPLATE + + +def gate_pending(data: dict) -> bool: + from _lib.terroir_gate import needs_gate # local import: keep the audit's imports flat + return needs_gate(data) + + +def own_chapter_findings(lien: str, name: str, facts: list[dict]) -> list[dict]: + """W2a rows for a record on a shared cahier: `no_own_chapter` when the + record's chapter is not found, else `quote_outside_own_chapter` for + every cahier-grounded fact (provenance `cahier` / `both`) whose + `cahier_quote` does not ground inside that chapter. A `wiki` fact is + skipped: its cahier quote never grounded at extraction either (02d keeps + a fact on either quote), so it says nothing about which chapter the + record was graded against. Empty for a lien that is not shared.""" + if not is_shared(lien): + return [] + window = own_chapter(lien, name) + if window is None: + return [{"check": "no_own_chapter"}] + matcher = SourceMatcher(lien[window[0]:window[1]]) + out: list[dict] = [] + for i, f in enumerate(facts): + quote = (f.get("cahier_quote") or "").strip() + if not quote or f.get("provenance") == "wiki": + continue + cov = matcher.coverage(quote) + if cov < FUZZY_THRESHOLD: + out.append({"check": "quote_outside_own_chapter", "index": i, "coverage": round(cov, 3)}) + return out + + +def intra_record_duplicates(facts: list[dict]) -> list[dict]: + """Every pair (i < j) of facts that `duplicate_reason` deems restatements.""" + out: list[dict] = [] + for i in range(len(facts)): + for j in range(i + 1, len(facts)): + reason = duplicate_reason(facts[i], facts[j]) + if reason: + out.append({"i": i, "j": j, "reason": reason}) + return out + + +def wiki_with_cahier_quote(facts: list[dict]) -> list[int]: + return [ + i for i, f in enumerate(facts) + if f.get("provenance") == "wiki" and (f.get("cahier_quote") or "").strip() + ] + + +def shared_quote_groups( + facts_by_slug: dict[str, list[dict]], + min_chars: int = SHARED_QUOTE_MIN_CHARS, + min_records: int = SHARED_QUOTE_MIN_RECORDS, +) -> list[dict]: + """Normalised `cahier_quote`s of at least `min_chars` shared by at least + `min_records` distinct records, largest group first.""" + slugs_by_quote: dict[str, set[str]] = defaultdict(set) + for slug, facts in facts_by_slug.items(): + for f in facts: + q = normalize(f.get("cahier_quote") or "") + if len(q) >= min_chars: + slugs_by_quote[q].add(slug) + groups = [ + {"quote": q, "count": len(slugs), "slugs": sorted(slugs)} + for q, slugs in slugs_by_quote.items() if len(slugs) >= min_records + ] + groups.sort(key=lambda g: (-g["count"], g["quote"])) + return groups + + +# ──────────────────────────────────────────────────────────── sources ── + + +class SourceResolver: + """Resolves each country's stage-02d sources once, on first request.""" + + def __init__(self) -> None: + self._by_country: dict[str, dict[str, Sources] | None] = {} + self.failed: dict[str, str] = {} + + def get(self, country: str) -> dict[str, Sources] | None: + if country not in self._by_country: + self._by_country[country] = self._resolve(country) + return self._by_country[country] + + def _resolve(self, country: str) -> dict[str, Sources] | None: + if country not in COUNTRIES: + self.failed[country] = "no stage-02d module for this country" + log(f"{country}: {self.failed[country]}") + return None + log(f"{country}: resolving stage-02d sources …") + t0 = time.monotonic() + try: + sources = resolve_sources(country) + except Exception as e: # noqa: BLE001 + self.failed[country] = repr(e) + log(f"{country}: FAILED to resolve sources: {e!r}") + return None + log(f"{country}: {len(sources)} source records in {time.monotonic() - t0:.1f}s") + return sources + + +def country_names() -> dict[str, dict[str, str]]: + """{country: {slug: folded appellation name minus stop words}} over every + terroir-facts cache (FR names from the extracted records) — the + candidate set for the foreign-name guard.""" + out: dict[str, dict[str, str]] = defaultdict(dict) + for p in TERROIR.glob("*.json"): + if p.name.startswith("manifest"): + continue + d = cache.read_json_or_none(p) + if not d: + continue + country = d.get("country") or "fr" + name = d.get("name") or (fr_record(p.stem)[0] if country == "fr" else p.stem) + tokens = name_tokens(name) + if tokens: + out[country][p.stem] = " ".join(tokens) + return out + + +def fr_record(slug: str) -> tuple[str, str]: + """(name, full lien) of the FR extracted record — FR caches carry no + `name`, and the W2a / W2b checks grade the unsliced lien.""" + rec = cache.read_json_or_none(FR_EXTRACTED / f"{slug}.json") or {} + return rec.get("name") or slug, (rec.get("lien_au_terroir") or "").strip() + + +# ──────────────────────────────────────────────────────────────── audit ── + + +def audit_one( + data: dict, src: Sources | None, source_status: str, name: str, fr_lien: str = "", + others: dict[str, str] | None = None, +) -> dict: + """Audit one source cache against its current sources. `source_status` + is `ok` (src given), `missing` (the slug is no longer a 02d target — + graded against empty sources) or `unresolved` (the country's 02d module + failed to load — source-dependent checks skipped). `fr_lien` is the FR + extracted record's full lien for the name-guard and own-chapter checks.""" + slug = data.get("slug") country = data.get("country") or "fr" facts = data.get("facts") or [] - if country not in EXTRACTED_BY_COUNTRY: - # This audit re-derives coverage from the country's own source - # documents, so it only covers countries wired into the dispatch - # tables above. Everything else is skipped explicitly rather than - # dying on a KeyError that the caller reports as a corrupt cache. - raise UnsupportedCountry(country) - - cur_cahier = load_current_cahier(slug, country) - cur_lien = cur_cahier[0] if cur_cahier else "" - cur_lien_sha = cur_cahier[1] if cur_cahier else "" - cahier_drift = bool(cur_lien_sha) and cur_lien_sha != data.get("cahier_source_sha") - - wiki_hints, wiki_data = load_wiki_hints(slug, country) - cur_wiki_rev = (wiki_data or {}).get("revision") + cahier_drift = src is not None and src.cahier_sha != data.get("cahier_source_sha") wiki_drift = ( - cur_wiki_rev is not None and cur_wiki_rev != data.get("wiki_source_revision") + src is not None and src.wiki_revision is not None + and src.wiki_revision != data.get("wiki_source_revision") ) + graded = source_status != "unresolved" - bullet_audits: list[dict] = [] + bullets: list[dict] = [] for f in facts: sub = f.get("subsection") or "facteurs_naturels" - cov_c = fuzzy_coverage(f.get("cahier_quote", ""), cur_lien) if f.get("cahier_quote") else 0.0 - cov_w = fuzzy_coverage(f.get("wiki_quote", ""), wiki_hints.get(sub, "")) if f.get("wiki_quote") else 0.0 - cur_keep_c = cov_c >= FUZZY_THRESHOLD - cur_keep_w = cov_w >= FUZZY_THRESHOLD - still_grounded = cur_keep_c or cur_keep_w - bullet_audits.append({ - "bullet": f.get("bullet", ""), + bullet = f.get("bullet") or "" + cq = f.get("cahier_quote") or "" + wq = f.get("wiki_quote") or "" + cov_c = src.matcher.coverage(cq) if (graded and src and cq) else 0.0 + cov_w = fuzzy_coverage(wq, src.hints.get(sub, "")) if (graded and src and wq) else 0.0 + still_grounded = (not graded) or cov_c >= FUZZY_THRESHOLD or cov_w >= FUZZY_THRESHOLD + bullets.append({ + "bullet": bullet, "subsection": sub, "provenance": f.get("provenance", ""), "cached_cahier_coverage": f.get("cahier_coverage", 0.0), @@ -337,46 +581,190 @@ def audit_one(cache_path: Path) -> dict: "current_cahier_coverage": round(cov_c, 3), "current_wiki_coverage": round(cov_w, 3), "still_grounded": still_grounded, - "over_length_cap": len(f.get("bullet", "")) > BULLET_SOFT_CAP, + "over_length_cap": len(bullet) > BULLET_SOFT_CAP, + "style": style_findings(bullet), }) + + name_guard = False + chapter: list[dict] = [] + foreign: list[dict] = [] + guard_text = fr_lien if country == "fr" else (src.cahier if (graded and src) else "") + if guard_text: + name_guard = name_guard_finding(guard_text, name, slug) + foreign = foreign_names(guard_text, name, others or {}) if name_guard else [] + if country == "fr" and fr_lien: + chapter = own_chapter_findings(fr_lien, name, facts) + wiki_binding = wiki_binding_finding(country, data.get("source_lang") or "", slug, name) + rejected = [ + i for i, f in enumerate(facts) if (f.get("support") or {}).get("verdict") == "rewrite-rejected" + ] + missing = [i for i, f in enumerate(facts) if (f.get("support") or {}).get("rewrite_missing")] + return { "slug": slug, + "country": country, + "name": name, "n_facts": len(facts), + "source_status": source_status, "cahier_drift": cahier_drift, "wiki_drift": wiki_drift, - "translator": data.get("translator") or data.get("model_id"), - "translator_kind": data.get("translator_kind") or data.get("translator_kind"), - "bullets": bullet_audits, + "translator": data.get("translator") or data.get("model_id") or data.get("model"), + "translator_kind": data.get("translator_kind") or data.get("model_kind"), + "bullets": bullets, + "duplicates": intra_record_duplicates(facts), + "wiki_with_cahier_quote": wiki_with_cahier_quote(facts), + "name_guard": name_guard, + "foreign_name": foreign, + "wiki_binding": wiki_binding, + "chapter": chapter, + "feedback_recurrence": recurrence_findings(load_feedback(slug), facts), + "masaf_sidecar_stale": country == "it" and masaf_sidecar_stale(slug), + "gate_pending": gate_pending(data), + "rewrite_rejected": rejected, + "rewrite_missing": missing, } -def summarize(audits: list[dict]) -> dict: - """Aggregate counters and medians over the per-AOC audits.""" - n_aocs = len(audits) +def audit_translations(slug: str, source_facts: list[dict] | None = None) -> tuple[dict[str, int], list[dict], list[str]]: + """(bullets per locale, finding rows, EN bullets) for the slug's + translation caches. `source_facts` enables `en_equals_src` (a + translated bullet identical to its source bullet, same index).""" + n_by_lang: dict[str, int] = {} + rows: list[dict] = [] + en_bullets: list[str] = [] + src_bullets = [(f.get("bullet") or "").strip() for f in (source_facts or [])] + src_sha = facts_sha(source_facts) if source_facts else None + for lang in LANGS: + t = cache.read_json_or_none(TRANSLATIONS / lang / f"{slug}.json") + if not t or t.get("mode") == "verbatim": + continue + facts = t.get("facts") or [] + n_by_lang[lang] = len(facts) + key = t.get("source_facts_sha") or "" + if src_sha and key != src_sha and not key.startswith("pending:"): + # keyed to a source that has since changed and not marked for 02e: + # invisible until 02e's next corpus-wide pass (two such caches sat + # in LU / AT until the 2026-09-14 smoke happened to pick them up) + rows.append({"check": "translation_stale", "slug": slug, "lang": lang}) + for i, f in enumerate(facts): + bullet = f.get("bullet") or "" + checks = [c for c in style_findings(bullet) if c != "multi_sentence"] + if has_non_latin(bullet): + checks.insert(0, "non_latin") + if i < len(src_bullets) and src_bullets[i] and bullet.strip() == src_bullets[i]: + checks.append("en_equals_src") + rows.extend({"check": c, "slug": slug, "lang": lang, "index": i} for c in checks) + if lang == "en": + en_bullets.append(bullet) + return n_by_lang, rows, en_bullets + + +def identical_en_groups(en_by_slug: dict[str, list[str]], min_records: int = 2) -> list[dict]: + """EN bullets (normalised) shared verbatim by ≥ `min_records` records.""" + slugs_by_bullet: dict[str, set[str]] = defaultdict(set) + for slug, bullets in en_by_slug.items(): + for b in bullets: + nb = normalize(b) + if len(nb) >= 20: + slugs_by_bullet[nb].add(slug) + groups = [{"bullet": b, "count": len(sl), "slugs": sorted(sl)} + for b, sl in slugs_by_bullet.items() if len(sl) >= min_records] + groups.sort(key=lambda g: (-g["count"], g["bullet"])) + return groups + + +def collect_findings( + audits: list[dict], translation_rows: list[dict], shared_groups: list[dict], + identical_en: list[dict] | None = None, +) -> dict[str, list[dict]]: + """The offending rows per check, source-cache rows first.""" + rows: dict[str, list[dict]] = {name: [] for name in CHECKS} + for a in audits: + slug = a["slug"] + for i, b in enumerate(a["bullets"]): + for check in b["style"]: + rows[check].append({"slug": slug, "index": i}) + for d in a["duplicates"]: + rows["intra_record_duplicates"].append({ + "slug": slug, "index": d["j"], "duplicate_of": d["i"], "reason": d["reason"], + }) + for i in a["wiki_with_cahier_quote"]: + rows["wiki_with_cahier_quote"].append({"slug": slug, "index": i}) + if a["name_guard"]: + key = "name_guard" if a["country"] == "fr" else "name_guard_other" + rows[key].append({"slug": slug, "name": a["name"], "country": a["country"]}) + if a.get("foreign_name"): + rows["foreign_name"].append({"slug": slug, "name": a["name"], "country": a["country"], + "names": a["foreign_name"]}) + if a.get("wiki_binding"): + rows["wiki_binding"].append({"slug": slug, "name": a["name"], **a["wiki_binding"]}) + for c in a["chapter"]: + rows[c["check"]].append({"slug": slug, **{k: v for k, v in c.items() if k != "check"}}) + for r in a.get("feedback_recurrence") or []: + rows["feedback_recurrence"].append({"slug": slug, **r}) + if a.get("masaf_sidecar_stale"): + rows["masaf_sidecar_stale"].append({"slug": slug}) + if a.get("gate_pending"): + rows["gate_pending"].append({"slug": slug}) + for i in a.get("rewrite_rejected") or []: + rows["rewrite_rejected"].append({"slug": slug, "index": i}) + for i in a.get("rewrite_missing") or []: + rows["rewrite_missing"].append({"slug": slug, "index": i}) + for r in translation_rows: + rows[r["check"]].append({k: v for k, v in r.items() if k != "check"}) + rows["cross_record_shared_quotes"] = shared_groups + rows["cross_record_identical_en"] = identical_en or [] + return rows + + +def summarize( + audits: list[dict], + findings: dict[str, list[dict]], + *, + translation_caches: int, + translated_bullets: int, + verbatim_skipped: int, + unresolved: dict[str, str], +) -> dict: + """Aggregate counters and medians over the per-AOC audits, plus one + `{count, strict}` entry per check.""" bullets_per_aoc = [a["n_facts"] for a in audits] by_provenance: Counter[str] = Counter() by_subsection: Counter[str] = Counter() by_translator_kind: Counter[str] = Counter() + by_country: Counter[str] = Counter() over_cap = 0 eroded_bullets = 0 eroded_aocs: list[str] = [] - cahier_drift_n = sum(1 for a in audits if a["cahier_drift"]) - wiki_drift_n = sum(1 for a in audits if a["wiki_drift"]) for a in audits: by_translator_kind[a["translator_kind"] or ""] += 1 + by_country[a["country"]] += 1 any_eroded = False for b in a["bullets"]: by_provenance[b["provenance"]] += 1 by_subsection[b["subsection"]] += 1 - if b["over_length_cap"]: - over_cap += 1 + over_cap += b["over_length_cap"] if not b["still_grounded"]: eroded_bullets += 1 any_eroded = True if any_eroded: eroded_aocs.append(a["slug"]) + + checks: dict[str, dict] = {} + for name in CHECKS: + rows = findings[name] + entry = {"strict": name in STRICT_CHECKS, "count": len(rows)} + if name in dict(STYLE_CHECKS) or name == "non_latin": + entry["source"] = sum(1 for r in rows if "lang" not in r) + entry["translated"] = len(rows) - entry["source"] + if name in ("intra_record_duplicates", "feedback_recurrence", "rewrite_rejected", "rewrite_missing", + "en_equals_src"): + entry["records"] = len({r["slug"] for r in rows}) + checks[name] = entry + strict_failures = sum(c["count"] for c in checks.values() if c["strict"]) + return { - "n_aocs": n_aocs, + "n_aocs": len(audits), "bullets_total": sum(bullets_per_aoc), "bullets_per_aoc_median": statistics.median(bullets_per_aoc) if bullets_per_aoc else 0, "bullets_per_aoc_min": min(bullets_per_aoc) if bullets_per_aoc else 0, @@ -384,16 +772,30 @@ def summarize(audits: list[dict]) -> dict: "by_provenance": dict(by_provenance), "by_subsection": dict(by_subsection), "by_translator_kind": dict(by_translator_kind), + "by_country": dict(sorted(by_country.items())), "over_length_cap": over_cap, "eroded_bullets": eroded_bullets, "eroded_aocs": eroded_aocs, - "cahier_drift_aocs": cahier_drift_n, - "wiki_drift_aocs": wiki_drift_n, + "cahier_drift_aocs": sum(1 for a in audits if a["cahier_drift"]), + "wiki_drift_aocs": sum(1 for a in audits if a["wiki_drift"]), + "no_source_today_aocs": [a["slug"] for a in audits if a["source_status"] == "missing"], + "source_unresolved_countries": unresolved, + "source_unresolved_aocs": sum(1 for a in audits if a["source_status"] == "unresolved"), + "verbatim_skipped": verbatim_skipped, + "translation_caches": translation_caches, + "translated_bullets_total": translated_bullets, + "checks": checks, + "strict_failures": strict_failures, } +# ───────────────────────────────────────────────────────────── output ── + + def print_per_aoc(audit: dict, verbose: bool) -> None: flags: list[str] = [] + if audit["source_status"] != "ok": + flags.append(f"source-{audit['source_status']}") if audit["cahier_drift"]: flags.append("cahier-drift") if audit["wiki_drift"]: @@ -404,23 +806,86 @@ def print_per_aoc(audit: dict, verbose: bool) -> None: over = sum(1 for b in audit["bullets"] if b["over_length_cap"]) if over: flags.append(f"over_cap={over}") + styled = sum(1 for b in audit["bullets"] if b["style"]) + if styled: + flags.append(f"style={styled}") + if audit["duplicates"]: + flags.append(f"dups={len(audit['duplicates'])}") + if audit["wiki_with_cahier_quote"]: + flags.append(f"wiki+cq={len(audit['wiki_with_cahier_quote'])}") + if audit["name_guard"]: + flags.append("name-guard") + if any(c["check"] == "no_own_chapter" for c in audit["chapter"]): + flags.append("no-own-chapter") + outside = sum(1 for c in audit["chapter"] if c["check"] == "quote_outside_own_chapter") + if outside: + flags.append(f"outside-chapter={outside}") + if audit.get("feedback_recurrence"): + flags.append(f"known-bad={len(audit['feedback_recurrence'])}") + if audit.get("foreign_name"): + flags.append("foreign-name=" + ",".join(f["other"] for f in audit["foreign_name"][:2])) + if audit.get("wiki_binding"): + flags.append(f"wiki-binding={audit['wiki_binding']['title'][:30]!r}") + if audit.get("gate_pending"): + flags.append("gate-pending") + if audit.get("rewrite_rejected"): + flags.append(f"rewrite-rejected={len(audit['rewrite_rejected'])}") + if audit.get("rewrite_missing"): + flags.append(f"rewrite-missing={len(audit['rewrite_missing'])}") flag_str = (" [" + ", ".join(flags) + "]") if flags else "" - print(f" {audit['slug']:40} n={audit['n_facts']:2} {audit['translator_kind'] or '-':12}{flag_str}") + print( + f" {audit['slug']:40} {audit['country']:2} n={audit['n_facts']:2} " + f"{audit['translator_kind'] or '-':12}{flag_str}" + ) if verbose: for b in audit["bullets"]: - note = "" + notes: list[str] = [] if not b["still_grounded"]: - note = ( - f" ⚠ eroded (now cahier={b['current_cahier_coverage']} " + notes.append( + f"eroded (now cahier={b['current_cahier_coverage']} " f"wiki={b['current_wiki_coverage']})" ) - elif b["over_length_cap"]: - note = " ⚠ over length cap" + if b["over_length_cap"]: + notes.append("over length cap") + notes.extend(b["style"]) + note = (" ⚠ " + "; ".join(notes)) if notes else "" print(f" {b['provenance']:6} [{b['subsection'][:8]:>8}] {b['bullet'][:80]}{note}") +# ─────────────────────────────────────────────────────────────── main ── + + +def select_caches( + slugs: set[str] | None, countries: set[str] | None, sample: int, +) -> tuple[list[tuple[Path, dict]], int]: + """(selected (path, cache) pairs in slug order, verbatim caches skipped).""" + selected: list[tuple[Path, dict]] = [] + verbatim = 0 + for p in sorted(TERROIR.glob("*.json")): + if p.name.startswith("manifest"): + continue + if slugs and p.stem not in slugs: + continue + d = cache.read_json_or_none(p) + if d is None: + log(f"err {p.stem}: unreadable cache") + continue + if d.get("mode") == "verbatim": + verbatim += 1 + continue + if countries and (d.get("country") or "fr") not in countries: + continue + selected.append((p, d)) + if sample and len(selected) > sample: + random.seed(0) + selected = sorted(random.sample(selected, sample), key=lambda t: t[0]) + return selected, verbatim + + def main() -> int: - ap = argparse.ArgumentParser(description=__doc__) + ap = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter, + ) ap.add_argument( "--sample", type=int, default=0, help="audit only N AOCs (random sample); 0 = all (default)", @@ -429,59 +894,111 @@ def main() -> int: "--slug", action="append", default=None, help="restrict to specific AOC slug(s); repeatable", ) + ap.add_argument( + "--country", action="append", default=None, + help="restrict to a country code (the cache's `country`, missing = fr); repeatable", + ) ap.add_argument("--verbose", action="store_true", help="print every bullet, not just AOC summary") ap.add_argument("--quiet", action="store_true", help="suppress per-AOC lines (totals only)") ap.add_argument("--report", metavar="PATH", default=None, help="write full audit JSON to PATH") + ap.add_argument( + "--strict", action="store_true", + help="exit 1 when any strict check (see module docstring) has a count > 0", + ) + ap.add_argument( + "--strict-labels", action="store_true", + help="also treat label_prefix as strict (once the pre-style-block corpus is re-extracted)", + ) args = ap.parse_args() + global STRICT_CHECKS + if args.strict_labels: + STRICT_CHECKS = STRICT_CHECKS | {"label_prefix"} - if not TERROIR_FACTS.exists(): + if not TERROIR.exists(): print("error: raw/terroir-facts is missing — run 02d first", file=sys.stderr) return 1 - files = sorted(p for p in TERROIR_FACTS.glob("*.json") if p.name != "manifest.json") - if args.slug: - wanted = set(args.slug) - files = [p for p in files if p.stem in wanted] - if args.sample and len(files) > args.sample: - import random - random.seed(0) - files = sorted(random.sample(files, args.sample)) - - if not files: - print("[audit] no terroir-facts caches to audit.", file=sys.stderr) + t_start = time.monotonic() + selected, verbatim_skipped = select_caches( + set(args.slug) if args.slug else None, + set(args.country) if args.country else None, + args.sample, + ) + if not selected: + log("no terroir-facts caches to audit.") return 0 + log(f"{len(selected)} AOC caches ({verbatim_skipped} verbatim skipped)") - print(f"[audit] {len(files)} AOC caches", file=sys.stderr) + resolver = SourceResolver() audits: list[dict] = [] - unsupported: dict[str, int] = {} - for p in files: + translation_rows: list[dict] = [] + translation_caches = 0 + translated_bullets = 0 + facts_by_country: dict[str, dict[str, list[dict]]] = defaultdict(dict) + en_by_slug: dict[str, list[str]] = {} + names_by_country = country_names() + for p, d in selected: + slug = d.get("slug") or p.stem + d["slug"] = slug + country = d.get("country") or "fr" + sources = resolver.get(country) + if sources is None: + src, status = None, "unresolved" + else: + src = sources.get(slug) + status = "ok" if src is not None else "missing" + name, fr_lien = fr_record(slug) if country == "fr" else (d.get("name") or slug, "") + others = {k: v for k, v in names_by_country.get(country, {}).items() if k != slug} try: - a = audit_one(p) - except UnsupportedCountry as e: - unsupported[str(e)] = unsupported.get(str(e), 0) + 1 - continue + a = audit_one(d, src, status, name, fr_lien, others) except Exception as e: # noqa: BLE001 - print(f" err {p.stem}: {e}", file=sys.stderr) + log(f"err {slug}: {e!r}") continue audits.append(a) + facts_by_country[country][slug] = d.get("facts") or [] + n_by_lang, rows, en_bullets = audit_translations(slug, d.get("facts") or []) + translation_caches += len(n_by_lang) + translated_bullets += sum(n_by_lang.values()) + translation_rows.extend(rows) + if en_bullets: + en_by_slug[slug] = en_bullets if not args.quiet: print_per_aoc(a, verbose=args.verbose) - if unsupported: - listed = ", ".join(f"{c}={n}" for c, n in sorted(unsupported.items())) - print(f" [skipped] countries with no source dispatch entry: {listed}", - file=sys.stderr) - summary = summarize(audits) + shared_groups: list[dict] = [] + for country, by_slug in sorted(facts_by_country.items()): + shared_groups.extend({"country": country, **g} for g in shared_quote_groups(by_slug)) + shared_groups.sort(key=lambda g: (-g["count"], g["country"], g["quote"])) + identical_en = identical_en_groups(en_by_slug) + + findings = collect_findings(audits, translation_rows, shared_groups, identical_en) + summary = summarize( + audits, findings, + translation_caches=translation_caches, + translated_bullets=translated_bullets, + verbatim_skipped=verbatim_skipped, + unresolved=resolver.failed, + ) + summary["elapsed_seconds"] = round(time.monotonic() - t_start, 1) print("\n[audit] summary:", file=sys.stderr) print(json.dumps(summary, ensure_ascii=False, indent=2, default=str), file=sys.stderr) if args.report: - Path(args.report).write_text(json.dumps({ + report_rows = dict(findings) + report_rows["cross_record_shared_quotes"] = shared_groups[:SHARED_QUOTE_TOP] + out = Path(args.report) + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text(json.dumps({ "generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), "summary": summary, + "findings": report_rows, "aocs": audits, }, ensure_ascii=False, indent=2, default=str) + "\n", encoding="utf-8") - print(f"[audit] full report → {args.report}", file=sys.stderr) + log(f"full report → {args.report}") + + if args.strict and summary["strict_failures"] > 0: + log(f"STRICT: {summary['strict_failures']} strict finding(s) — exit 1") + return 1 return 0 diff --git a/scripts/audit_terroir_facts_llm.py b/scripts/audit_terroir_facts_llm.py new file mode 100644 index 0000000..b650947 --- /dev/null +++ b/scripts/audit_terroir_facts_llm.py @@ -0,0 +1,408 @@ +"""LLM quality audit of the rendered terroir facts — the adversarial +verifier of the 2026-09-12 review as a repeatable script (R9). + +For a sample of records it grades every ENGLISH bullet (the locale +readers see; `--lang` for another) against the source-language bullet, +the grounding quotes, the full regulator text stage 02d graded against +(`_lib/terroir_sources`) and the Wikipedia hints, and asks: would a +reader of this bullet believe something the sources do not support? +One request per record; the default grader is `claude-opus-5` — a +different, stronger model than the stage-02d extractor and the +claim-support gate (claude-sonnet-4-6), so the measure is independent +of the machinery it measures. + +Paired before / after: `--from-backup RUN` grades the same records as +they were BEFORE a re-run (the snapshot under +`raw/terroir-facts-backup/<RUN>/`; a record the run did not touch is +read live), against the CURRENT sources. `--compare A.json B.json` +prints the paired comparison of two reports. + +Outputs a JSON report (`tmp/terroir-facts-review/llm-audit-<run>.json`): +per bullet verdict / reader_misled / where / tags / reason / +corrected_en, plus a summary (misleading share with a Wilson 95 % +interval, by country, by stage, by tag). Read-only: never writes a cache. + +Usage: + .venv/bin/python scripts/audit_terroir_facts_llm.py --sample 100 --batch + .venv/bin/python scripts/audit_terroir_facts_llm.py --sample 100 --batch --from-backup r1-2026-09-13 --report tmp/terroir-facts-review/llm-audit-before.json + .venv/bin/python scripts/audit_terroir_facts_llm.py --compare before.json after.json +""" + +from __future__ import annotations + +import argparse +import json +import math +import random +import sys +import time +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +from tqdm import tqdm + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib import batch, cache, llm_json, providers, terroir_backup # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import TERROIR, TRANSLATIONS # noqa: E402 +from _lib.terroir_dedupe import facts_sha # noqa: E402 +from _lib.terroir_feedback import _safe # noqa: E402 +from _lib.terroir_sources import COUNTRIES, Sources, resolve_sources # noqa: E402 + +DEFAULT_GRADER = providers.stage_default("audit")[0] +REPORT_DIR = ROOT / "tmp" / "terroir-facts-review" +BATCH_SIDECAR = ROOT / "raw" / ".batch" / "llm-audit.json" +MAX_TOKENS = 8000 +MAX_SOURCE_CHARS = 60_000 +TAGS = ("wrong-entity", "unsupported-causal-link", "invented-detail", "wrong-number-or-unit", + "wrong-direction", "hedge-dropped-or-changed", "sibling-or-subzone-confusion", + "narrowed-attribution", "added-qualifier", "mistranslation", "untranslated-term", + "label-fragment", "duplicate", "misfiled-subsection", "tautology", "other") + +SYSTEM = """You are the adversarial verifier for a terroir-fact quality audit (Open Wine Map: bullets an LLM extracted from wine-regulator appellation texts — cahier des charges, disciplinare, pliego, Einziges Dokument — then machine-translated for readers). For ONE record you receive every bullet as the reader sees it (the TARGET rendering), the source-language bullet it was translated from (SRC), the grounding quotes the extractor cited (CQ from the regulator text, WQ from Wikipedia), the record's full regulator SOURCE TEXT and the Wikipedia HINTS. Your job is to try to find, for each bullet, anything a reader would come to believe that the sources do not support. + +For EACH bullet decide: +- verdict: "faithful" (every assertion is supported; a simplification is not an error; a number formatted differently in the source — 13°5, 22, 5, anni '60, XVIIIe — is not missing; a Wikipedia-grounded bullet is legitimately grounded), "defective" (a real defect that does not mislead: a clumsy phrase, a source-language common noun left untranslated, a label-style fragment, a misfiled sub-section, a restatement of another bullet) or "misleading" (the reader would believe something false or unsupported: a wrong entity, an unsupported causal link, an invented detail or qualifier, a wrong number, unit or direction, a dropped or strengthened hedge, a sub-zone or neighbour presented as the whole, one factor credited with what the source credits to several, a mistranslation that changes the meaning). +- reader_misled: true only for "misleading". +- where: "extraction" (the defect is already in SRC), "translation" (SRC is right, the TARGET rendering is wrong), "both", or "none". +- tags: zero or more of {tags}. +- reason: two precise sentences quoting the decisive source words. +- corrected: the bullet as it should read in the TARGET language if not faithful, else "". + +Default to "faithful" when uncertain. Be strict about causation: "the soils give the wine minerality" is misleading unless a source sentence states that link. + +Answer ONLY with JSON, no text before or after: +{"bullets": [{"i": 0, "verdict": "faithful|defective|misleading", "reader_misled": false, "where": "none", "tags": [], "reason": "", "corrected": ""}, ...]} +One object per bullet, in order, with "i" equal to the bullet's index.""" + + +def log(msg: str) -> None: + print(f"[llm-audit] {msg}", file=sys.stderr) + + +# ───────────────────────────────────────────────────────────── loading ── + + +def _read(path: Path, run: str | None, kind: str, lang: str = "") -> dict | None: + """The live file, or — with `run` — the backup snapshot when the run + touched this slug (else live).""" + if run: + rd = terroir_backup.run_dir(run) + bp = rd / ("source" if kind == "source" else f"translations/{lang}") / path.name + entry = cache.read_json_or_none(rd / "entries" / path.name) + if entry is not None: + if kind == "source": + return cache.read_json_or_none(bp) if entry.get("source") else None + return cache.read_json_or_none(bp) if lang in (entry.get("translations") or []) else None + return cache.read_json_or_none(path) + + +def load_record(slug: str, lang: str, run: str | None) -> tuple[dict, dict] | str: + """(source cache, translation cache) or a skip reason.""" + src = _read(TERROIR / f"{slug}.json", run, "source") + if not src or src.get("mode") == "verbatim" or not src.get("facts"): + return "no-facts" + t = _read(TRANSLATIONS / lang / f"{slug}.json", run, "translation", lang) + if not t or t.get("mode") == "verbatim" or not t.get("facts"): + return "no-translation" + if t.get("source_facts_sha") != facts_sha(src["facts"]) or len(t["facts"]) != len(src["facts"]): + return "translation-misaligned" + return src, t + + +def build_user_message(*, name: str, country: str, source_lang: str, lang: str, src: dict, t: dict, sources: Sources) -> str: + text = sources.cahier or "" + if len(text) > MAX_SOURCE_CHARS: + text = text[:MAX_SOURCE_CHARS] + "\n[… truncated …]" + hints = "\n\n".join(f"[{k}]\n{v[:2000]}" for k, v in (sources.hints or {}).items() if v) or "(none)" + rows = [] + for i, (sf, tf) in enumerate(zip(src["facts"], t["facts"])): + rows.append( + f"#{i} [{sf.get('subsection') or ''} · {sf.get('provenance') or ''}]\n" + f"TARGET ({lang}): {tf.get('bullet') or ''}\n" + f"SRC ({source_lang}): {sf.get('bullet') or ''}\n" + f"CQ: {sf.get('cahier_quote') or ''}\nWQ: {sf.get('wiki_quote') or ''}" + ) + return _safe( + f"RECORD: {name} (country {country}); {len(rows)} bullets.\n\n" + f"SOURCE TEXT ({len(text)} chars)\n{text or '(no regulator text resolved — grade against the hints)'}\n\n" + f"WIKIPEDIA HINTS\n{hints}\n\nBULLETS\n" + "\n\n".join(rows) + ) + + +def parse_reply(raw: str, n: int) -> list[dict] | None: + s = llm_json.strip_fences(raw or "") + data = None + try: + data = json.loads(s) + except ValueError: + import re + m = re.search(r"\{.*\}", s, re.S) + if m: + try: + data = json.loads(m.group(0)) + except ValueError: + data = None + rows = data.get("bullets") if isinstance(data, dict) else None + if not isinstance(rows, list): + return None + out = [{"verdict": "faithful", "reader_misled": False, "where": "none", "tags": [], "reason": "", "corrected": ""} + for _ in range(n)] + seen = 0 + for r in rows: + if not isinstance(r, dict): + continue + try: + i = int(r.get("i")) + except (TypeError, ValueError): + continue + if not 0 <= i < n: + continue + verdict = str(r.get("verdict") or "faithful").lower() + verdict = verdict if verdict in ("faithful", "defective", "misleading") else "faithful" + out[i] = { + "verdict": verdict, + "reader_misled": bool(r.get("reader_misled")) and verdict == "misleading", + "where": str(r.get("where") or "none"), + "tags": [str(x) for x in (r.get("tags") or []) if isinstance(x, str)][:5], + "reason": " ".join(str(r.get("reason") or "").split())[:500], + "corrected": " ".join(str(r.get("corrected") or "").split())[:400], + } + seen += 1 + return out if seen or not n else None + + +# ─────────────────────────────────────────────────────────── grading ── + + +class SourceResolver: + def __init__(self) -> None: + self._by_country: dict[str, dict[str, Sources] | None] = {} + + def get(self, country: str) -> dict[str, Sources] | None: + if country not in self._by_country: + try: + self._by_country[country] = resolve_sources(country) if country in COUNTRIES else None + except Exception as e: # noqa: BLE001 + log(f"{country}: sources failed: {e!r}") + self._by_country[country] = None + return self._by_country[country] + + +def grade(provider, model_id: str, items: list[tuple[str, dict, dict]], resolver: SourceResolver, *, lang: str, quiet: bool) -> list[dict]: + rows: list[dict] = [] + for slug, src, t in tqdm(items, desc="llm-audit", leave=False, disable=quiet): + country = src.get("country") or "fr" + source_lang = src.get("source_lang") or ("fr" if country == "fr" else country) + sources = resolver.get(country) + s = sources.get(slug) if sources else None + if s is None: + rows.append({"slug": slug, "country": country, "status": "no_source"}) + continue + user = build_user_message(name=src.get("name") or slug, country=country, source_lang=source_lang, + lang=lang, src=src, t=t, sources=s) + try: + raw = provider.chat(system=mark_cached(SYSTEM.replace("{tags}", ", ".join(TAGS))), user=user, + max_tokens=MAX_TOKENS) + except Exception as e: # noqa: BLE001 + rows.append({"slug": slug, "country": country, "status": "error", "error": str(e)[:200]}) + continue + verdicts = parse_reply(raw, len(t["facts"])) + if verdicts is None: + rows.append({"slug": slug, "country": country, "status": "parse_error"}) + continue + bullets = [{"i": i, "bullet": t["facts"][i].get("bullet") or "", "src": src["facts"][i].get("bullet") or "", + "subsection": src["facts"][i].get("subsection"), **v} for i, v in enumerate(verdicts)] + rows.append({"slug": slug, "country": country, "status": "ok", "n": len(bullets), + "n_misleading": sum(1 for b in bullets if b["reader_misled"]), + "n_defective": sum(1 for b in bullets if b["verdict"] == "defective"), "bullets": bullets}) + if not quiet: + log(f"{slug:40} {country:2} n={len(bullets):2} misleading={rows[-1]['n_misleading']} defective={rows[-1]['n_defective']}") + return rows + + +def wilson(k: int, n: int, z: float = 1.96) -> tuple[float, float]: + if not n: + return 0.0, 0.0 + p = k / n + d = 1 + z * z / n + c = p + z * z / (2 * n) + h = z * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n)) + return (c - h) / d, (c + h) / d + + +def summarize(rows: list[dict]) -> dict: + ok = [r for r in rows if r.get("status") == "ok"] + n = sum(r["n"] for r in ok) + mis = sum(r["n_misleading"] for r in ok) + de = sum(r["n_defective"] for r in ok) + by_country: dict[str, dict] = {} + by_where: Counter = Counter() + by_tag: Counter = Counter() + for r in ok: + c = by_country.setdefault(r["country"], {"records": 0, "bullets": 0, "misleading": 0, "defective": 0}) + c["records"] += 1 + c["bullets"] += r["n"] + c["misleading"] += r["n_misleading"] + c["defective"] += r["n_defective"] + for b in r["bullets"]: + if b["reader_misled"]: + by_where[b["where"]] += 1 + for tag in b["tags"]: + by_tag[tag] += 1 + lo, hi = wilson(mis, n) + return { + "records": len(ok), "records_failed": len(rows) - len(ok), "bullets": n, + "misleading": mis, "misleading_share": round(mis / n, 4) if n else 0, + "misleading_ci95": [round(lo, 4), round(hi, 4)], + "defective": de, "defective_share": round(de / n, 4) if n else 0, + "records_with_misleading": sum(1 for r in ok if r["n_misleading"]), + "by_where": dict(by_where), "by_tag": dict(by_tag.most_common()), + "by_country": dict(sorted(by_country.items())), + } + + +def compare(a_path: Path, b_path: Path) -> dict: + a = json.loads(a_path.read_text(encoding="utf-8")) + b = json.loads(b_path.read_text(encoding="utf-8")) + ra = {r["slug"]: r for r in a["records"] if r.get("status") == "ok"} + rb = {r["slug"]: r for r in b["records"] if r.get("status") == "ok"} + common = sorted(set(ra) & set(rb)) + def _s(rs): + n = sum(rs[s]["n"] for s in common) + m = sum(rs[s]["n_misleading"] for s in common) + lo, hi = wilson(m, n) + return {"bullets": n, "misleading": m, "share": round(m / n, 4) if n else 0, "ci95": [round(lo, 4), round(hi, 4)], + "records_with_misleading": sum(1 for s in common if rs[s]["n_misleading"])} + better = sum(1 for s in common if rb[s]["n_misleading"] < ra[s]["n_misleading"]) + worse = sum(1 for s in common if rb[s]["n_misleading"] > ra[s]["n_misleading"]) + return {"paired_records": len(common), "before": _s(ra), "after": _s(rb), + "records_improved": better, "records_worse": worse, + "records_same": len(common) - better - worse, + "worse_slugs": [s for s in common if rb[s]["n_misleading"] > ra[s]["n_misleading"]]} + + +def emit_feedback(rows: list[dict], out_dir: Path, *, lang: str) -> int: + """Write the misleading verdicts in the review-evidence shape + `scripts/build_terroir_feedback.py` reads (`confirmed-misleading.json` + rows + a `merged.json` with empty record notes), so the next extraction + of each record sees them as do-not-claim constraints. Only verdicts + the grader marked reader-misleading; `mode` is the first tag.""" + out_dir.mkdir(parents=True, exist_ok=True) + misleading = [] + for r in rows: + if r.get("status") != "ok": + continue + for b in r["bullets"]: + if not b["reader_misled"]: + continue + misleading.append({ + "id": f"{r['slug']}#{b['i']}", "slug": r["slug"], "i": b["i"], "country": r["country"], + "subsection": b.get("subsection") or "", "mode": (b["tags"] or ["other"])[0], + "where": b["where"] if b["where"] in ("extraction", "translation", "both") else "extraction", + "severity": "high", "en" if lang == "en" else lang: b["bullet"], "src": b["src"], + "reason": b["reason"], "corrected_en": b["corrected"], + }) + (out_dir / "confirmed-misleading.json").write_text(json.dumps(misleading, ensure_ascii=False, indent=1) + "\n", encoding="utf-8") + (out_dir / "merged.json").write_text(json.dumps({"record_notes": {}}, indent=1) + "\n", encoding="utf-8") + return len(misleading) + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--provider", default="anthropic", choices=("anthropic", "mistral", "ollama")) + ap.add_argument("--model", default=DEFAULT_GRADER) + ap.add_argument("--thinking", default=None, choices=("disabled", "adaptive"), + help="anthropic thinking mode (default: the audit stage default, adaptive)") + ap.add_argument("--lang", default="en") + ap.add_argument("--sample", type=int, default=100) + ap.add_argument("--seed", type=int, default=0) + ap.add_argument("--only", action="append", default=None) + ap.add_argument("--slugs-file", default=None, help="JSON list of slugs to grade (overrides --sample)") + ap.add_argument("--country", action="append", default=None) + ap.add_argument("--from-backup", default=None, metavar="RUN", help="grade the pre-run state of the sample") + ap.add_argument("--batch", action="store_true") + ap.add_argument("--report", default=None) + ap.add_argument("--compare", nargs=2, metavar=("A", "B"), default=None) + ap.add_argument("--emit-feedback", default=None, metavar="DIR", + help="also write the misleading verdicts as a review-evidence dir " + "(confirmed-misleading.json + merged.json) for scripts/build_terroir_feedback.py") + ap.add_argument("--quiet", action="store_true") + args = ap.parse_args() + + if args.compare: + print(json.dumps(compare(Path(args.compare[0]), Path(args.compare[1])), ensure_ascii=False, indent=2)) + return 0 + + # Sample from the records that have facts today (so before / after share a frame). + candidates: list[str] = [] + for p in sorted(TERROIR.glob("*.json")): + if p.name.startswith("manifest"): + continue + d = cache.read_json_or_none(p) + if not d or d.get("mode") == "verbatim" or not d.get("facts"): + continue + if args.country and (d.get("country") or "fr") not in set(args.country): + continue + candidates.append(p.stem) + if args.slugs_file: + wanted = json.loads(Path(args.slugs_file).read_text(encoding="utf-8")) + wanted = wanted.get("slugs") if isinstance(wanted, dict) else wanted + slugs = [s for s in wanted if s in set(candidates)] + elif args.only: + slugs = [s for s in candidates if s in set(args.only)] + else: + random.seed(args.seed) + slugs = sorted(random.sample(candidates, min(args.sample, len(candidates)))) + items: list[tuple[str, dict, dict]] = [] + skipped: Counter = Counter() + for slug in slugs: + loaded = load_record(slug, args.lang, args.from_backup) + if isinstance(loaded, str): + skipped[loaded] += 1 + continue + items.append((slug, *loaded)) + log(f"{len(items)} records to grade (skipped {dict(skipped)}); state={'backup ' + args.from_backup if args.from_backup else 'live'}; lang={args.lang}") + if not items: + return 1 + resolver = SourceResolver() + t0 = time.monotonic() + rows: list[dict] = [] + if args.batch: + model_id = args.model + + def run_loop(prov): + nonlocal rows + rows = grade(prov, model_id, items, resolver, lang=args.lang, quiet=True) + + # One sidecar per graded state: a "before" batch still in flight must + # not be resumed by an "after" run. + sidecar = BATCH_SIDECAR.with_name(f"llm-audit-{args.from_backup or 'live'}-{args.lang}.json") + batch.run_two_pass(provider=args.provider, model=model_id, sidecar=sidecar, run_loop=run_loop, + thinking=args.thinking or batch.default_thinking(args.provider, stage="audit")) + else: + provider, model_id = providers.make_provider(args.provider, model=args.model, stage="audit", + thinking=args.thinking) + rows = grade(provider, model_id, items, resolver, lang=args.lang, quiet=args.quiet) + summary = summarize(rows) + summary.update({"model": model_id, "lang": args.lang, "state": args.from_backup or "live", + "seed": args.seed, "elapsed_seconds": round(time.monotonic() - t0, 1), + "generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), "skipped": dict(skipped)}) + stamp = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H%M%S") + report = Path(args.report) if args.report else REPORT_DIR / f"llm-audit-{stamp}.json" + report.parent.mkdir(parents=True, exist_ok=True) + report.write_text(json.dumps({"summary": summary, "records": rows}, ensure_ascii=False, indent=1) + "\n", encoding="utf-8") + print(json.dumps(summary, ensure_ascii=False, indent=2), file=sys.stderr) + log(f"report → {report}") + if args.emit_feedback: + n = emit_feedback(rows, Path(args.emit_feedback), lang=args.lang) + log(f"feedback evidence ({n} misleading bullets) → {args.emit_feedback}; merge with " + f"scripts/build_terroir_feedback.py --evidence {args.emit_feedback} --review-id llm-audit-{stamp}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/be/02d_extract_terroir_facts.py b/scripts/be/02d_extract_terroir_facts.py index 9040a05..e819994 100644 --- a/scripts/be/02d_extract_terroir_facts.py +++ b/scripts/be/02d_extract_terroir_facts.py @@ -32,7 +32,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -41,6 +40,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system, split_user_lead # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "be" / "dokumenten-extracted" WIKI_AOCS_ROOT = ROOT / "raw" / "wikipedia" / "aocs" @@ -172,7 +179,7 @@ - Citaten zijn WOORDELIJK (gekopieerd) uit de respectieve bron. Schrijf NOOIT tekst aan een bron toe die daar niet voorkomt. - Geen waardeoordelen ("uitzonderlijk", "prestigieus" …). - Geen externe conclusies. Geen cijfers die in geen van beide bronnen voorkomen. -- Maximaal {max_bullets} items, elk ≤ 140 tekens. +- Maximaal {max_bullets} items; elk item is een volledige zin van ongeveer 120–220 tekens — nooit een fragment in telegramstijl. - Als noch het enig document noch Wikipedia een concreet belangrijk feit voor deze subsectie bevat, retourneer dan een lege lijst. Antwoord ENKEL in JSON, zonder tekst ervoor of erna: @@ -199,13 +206,14 @@ - Les citations sont VERBATIM (copier-coller) depuis leur source. N'attribue JAMAIS à une source un texte qui n'y figure pas. - Pas de jugements de valeur ("exceptionnel", "prestigieux"…). - Pas de conclusions externes. Pas de chiffres absents des deux sources. -- Maximum {max_bullets} puces, ≤ 140 caractères chacune. +- Maximum {max_bullets} puces ; chaque puce est une phrase complète d'environ 120 à 220 caractères — jamais un fragment télégraphique. - Si ni le document unique ni Wikipédia ne contiennent de fait concret remarquable pour cette sous-section, renvoie une liste vide. Réponds UNIQUEMENT en JSON, sans préambule : {{"facts": [{{"bullet": "…", "cahier_quote": "…", "wiki_quote": "…"}}, ...]}} Utilise une chaîne vide "" pour la citation manquante.""", } +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) USER_LEAD: dict[str, str] = { @@ -217,25 +225,10 @@ # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -369,9 +362,13 @@ def _process_subsection( topics=topic, max_bullets=sub["max_bullets"], ) - user = USER_LEAD[lang].format(label=label, lien=lien) + system = with_feedback(system, record["slug"]) + # The regulator text is the cached leading block: the four sub-section calls share it. + user, doc = split_user_lead(USER_LEAD[lang], label=label, lien=lien) + system = cached_system(doc, system, phased=True) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -404,6 +401,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, lang) + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "be", "source_lang": lang, @@ -411,6 +414,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -421,7 +426,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -488,12 +493,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": label, "topics": topics[sub["key"]], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM[lang].format( + "system_prompt": with_feedback(EXTRACT_SYSTEM[lang].format( wiki_hint=wiki_hint or "(no Wikipedia extract)", label=label, topics=topics[sub["key"]], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -553,7 +558,7 @@ def _import_one_slug( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return "wrote" @@ -626,7 +631,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -659,7 +664,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-be.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -697,7 +702,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/be/02e_translate_terroir_facts.py b/scripts/be/02e_translate_terroir_facts.py index eb510e5..c62641d 100644 --- a/scripts/be/02e_translate_terroir_facts.py +++ b/scripts/be/02e_translate_terroir_facts.py @@ -29,6 +29,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -42,7 +45,6 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Dutch proper nouns verbatim: appellation names ("Hagelandse wijn", "Haspengouwse wijn", "Heuvellandse wijn", "Vlaamse landwijn", "Vlaamse mousserende kwaliteitswijn", "Maasvallei Limburg"), region names ("Vlaanderen", "Hageland", "Haspengouw", "Heuvelland", "Maasvallei"), commune and dorp names, grape variety names ("Acolon", "Dornfelder", "Pinotin", "Cabernet Cortis", "Regent", "Solaris", "Johanniter", "Auxerrois", "Riesling", "Gewürztraminer", "Müller-Thurgau", "Chardonnay", "Pinot blanc/gris/noir", "Siegerrebe"), named geological formations and soil types ("löss", "leem", "krijt", "mergel", "tuffeau", "zandleem", "alluviale grond"), named climatic features ("gematigd zeeklimaat", "gematigd maritiem klimaat"), and Belgian wine-law / EU GI terms ("BOB", "BGA", "enig document", "productdossier", "oorsprongsbenaming", "geografische aanduiding"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Dutch form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. @@ -53,7 +55,6 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve French proper nouns verbatim: appellation names ("Côtes de Sambre et Meuse", "Crémant de Wallonie", "Vin mousseux de qualité de Wallonie", "Vin de pays des jardins de Wallonie"), region names ("Wallonie", "Sambre", "Meuse", "Hesbaye", "Condroz", "Ardenne"), commune names, grape variety names ("Chardonnay", "Pinot noir/blanc/gris", "Auxerrois", "Müller-Thurgau", "Acolon", "Régent", "Solaris", "Johanniter", "Pinotin"), named geological formations and soil types ("craie", "limon", "loess", "marnes", "tuffeau", "alluvions", "sables", "schistes"), named climatic features ("climat tempéré océanique"), and EU GI / Belgian wine-law terms ("AOP", "IGP", "document unique", "cahier des charges", "appellation d'origine protégée", "indication géographique protégée"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the French form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. @@ -62,6 +63,11 @@ SYSTEM_PROMPTS = {"nl": SYSTEM_PROMPT_NL, "fr": SYSTEM_PROMPT_FR} +PROPER_NOUNS = { + "nl": """appellation names (Hagelandse wijn, Haspengouwse wijn, Heuvellandse wijn, Vlaamse landwijn, Vlaamse mousserende kwaliteitswijn, Maasvallei Limburg); region names (Vlaanderen, Hageland, Haspengouw, Heuvelland, Maasvallei); commune and village names; grape names (Acolon, Dornfelder, Pinotin, Cabernet Cortis, Regent, Solaris, Johanniter, Auxerrois, Riesling, Gewürztraminer, Müller-Thurgau, Chardonnay, Pinot blanc, Pinot gris, Pinot noir, Siegerrebe)""", + "fr": """appellation names (Côtes de Sambre et Meuse, Crémant de Wallonie, Vin mousseux de qualité de Wallonie, Vin de pays des jardins de Wallonie); region names (Wallonie, Sambre, Meuse, Hesbaye, Condroz, Ardenne); commune names; grape names (Chardonnay, Pinot noir, Pinot blanc, Pinot gris, Auxerrois, Müller-Thurgau, Acolon, Régent, Solaris, Johanniter, Pinotin)""", +} + # ─────────────────────────────────────────────────────────────── helpers ── @@ -125,7 +131,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -188,10 +194,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: src_lang = job["source_lang"] - system = SYSTEM_PROMPTS[src_lang].format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPTS[src_lang].format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=src_lang, target_lang=job["lang"], proper_nouns=PROPER_NOUNS[src_lang], + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -300,6 +311,7 @@ def _build_argparser() -> argparse.ArgumentParser: help="restrict to a specific locale (repeatable)") ap.add_argument("--limit", type=int, default=0) ap.add_argument("--workers", type=int, default=1) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true") ap.add_argument("--batch", action="store_true") roundtrip.add_arguments(ap) @@ -379,8 +391,10 @@ def _run_batch(args, languages: tuple[str, ...] | None) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -412,6 +426,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -420,7 +436,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: print(f"[02e/be] manual provider: {len(jobs)} entries need translation. " diff --git a/scripts/bg/02d_extract_terroir_facts.py b/scripts/bg/02d_extract_terroir_facts.py index 8fc5522..ee31343 100644 --- a/scripts/bg/02d_extract_terroir_facts.py +++ b/scripts/bg/02d_extract_terroir_facts.py @@ -31,7 +31,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -40,6 +39,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "bg" / "dokumenti-extracted" NATIONAL_SPECS = ROOT / "raw" / "bg" / "national-specs-extracted" @@ -133,36 +140,22 @@ - Цитатите са БУКВАЛНИ (копирани) от съответния източник. НИКОГА не приписвай на даден източник текст, който не се появява там. - Без оценъчни преценки ("изключителен", "престижен" …). - Без външни заключения. Без числа, които не се появяват в нито един от двата източника. -- Максимум {max_bullets} елемента, всеки ≤ 140 символа. +- Максимум {max_bullets} елемента; всеки елемент е пълно изречение от около 120–220 знака — никога телеграфен фрагмент. - Ако нито единният документ, нито Wikipedia съдържат конкретен факт, заслужаващ внимание за тази подсекция, върни празен списък. Отговори САМО в JSON, без текст преди или след: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} За липсващ цитат използвай празен низ "".""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -280,11 +273,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Подсекция за обработка: {label}\n\n" - f"Текст на единния документ (Описание на връзката):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Подсекция за обработка: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Текст на единния документ (Описание на връзката):\n\n{lien_text}" def _process_subsection( @@ -298,9 +292,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -335,6 +333,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "bg") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "bg", "source_lang": "bg", @@ -342,6 +346,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -352,7 +358,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -415,12 +421,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(извлечение от Wikipedia недостъпно)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -461,7 +467,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts(slug_items: list[dict], lien: str) -> list[dict]: @@ -576,7 +582,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -609,7 +615,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-bg.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -659,7 +665,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/bg/02e_translate_terroir_facts.py b/scripts/bg/02e_translate_terroir_facts.py index e95edec..cc5da38 100644 --- a/scripts/bg/02e_translate_terroir_facts.py +++ b/scripts/bg/02e_translate_terroir_facts.py @@ -27,6 +27,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -39,12 +42,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Bulgarian proper nouns verbatim: appellation names ("Мелник", "Поморие", "Сунгурларе", "Сандански", "Лясковец", "Карлово", "Хисаря", "Брестник", "Пловдив", "Асеновград", "Перущица", "Хасково", "Стамболово", "Любимец", "Ивайловград", "Сакар", "Стара Загора", "Нова Загора", "Шумен", "Велики Преслав", "Хан Крум", "Драгоево", "Видин", "Враца", "Ловеч", "Плевен", "Свищов", "Русе", "Търговище", "Сухиндол", "Павликени", "Лом", "Монтана", "Ново село", "Лозица", "Нови Пазар", "Оряховица", "Върбица", "Карнобат", "Поморие", "Варна", "Евксиноград", "Южно Черноморие", "Черноморски район", "Славянци", "Долината на Струма", "Хърсово", "Дунавска равнина", "Тракийска низина"), grape variety names ("Мавруд", "Широка мелнишка лоза", "Памид", "Димят", "Червен Мискет", "Тамянка", "Сандански Мискет", "Керацуда", "Ркацители", "Гъмза", "Рубин", "Руен", "Богдан", "Мелник 55", "Мелник 82", "Шевка", "Сторгозия", "Кайлъшки Мискет", "Варненски Мискет", "Букет"), named geographical features ("Стара планина", "Родопи", "Странджа", "Сакар", "Сърнена гора", "Средна гора", "Лудогорие", "Черно море", "Дунавска равнина", "Тракийска низина", "Розова долина", "Долината на Струма"), named soil types ("чернозем", "канелена горска почва", "смолница", "песъчливо-чакълеста", "льос", "мергел", "варовик", "пясъчник"), named climatic features ("умереноконтинентален климат", "преходноконтинентален", "средиземноморско влияние", "понтийско влияние", "фьон", "бора"), and Bulgarian wine-law terms ("лозарски район", "винарска област", "защитено наименование за произход", "защитено географско указание", "ЗНП", "ЗГУ", "ИАЛВ", "продуктова спецификация", "единен документ"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Bulgarian form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "bg" +PROPER_NOUNS = """appellation names, written in their Latin form (Melnik, Pomorie, Sungurlare, Sandanski, Lyaskovets, Karlovo, Hisarya, Brestnik, Plovdiv, Asenovgrad, Perushtitsa, Haskovo, Stambolovo, Lyubimets, Ivaylovgrad, Sakar, Stara Zagora, Nova Zagora, Shumen, Veliki Preslav, Han Krum, Dragoevo, Vidin, Vratsa, Lovech, Pleven, Svishtov, Ruse, Targovishte, Suhindol, Pavlikeni, Lom, Montana, Novo Selo, Lozitsa, Novi Pazar, Oryahovitsa, Varbitsa, Karnobat, Varna, Evksinograd, Yuzhno Chernomorie, Slavyantsi, Harsovo, Dolinata na Struma, Dunavska Ravnina, Trakiyska Nizina); grape names in their Latin form (Mavrud, Shiroka Melnishka Loza, Pamid, Dimyat, Cherven Misket, Tamyanka, Sandanski Misket, Kerasuda, Rkatsiteli, Gamza, Rubin, Ruen, Bogdan, Melnik 55, Melnik 82, Shevka, Storgozia, Kaylashki Misket, Varnenski Misket, Buket); named geographical features, in the established target-language form where one exists and transliterated otherwise (Stara Planina, Rodopi, Strandzha, Sakar, Sarnena Gora, Sredna Gora, Ludogorie, Black Sea, Danube Plain, Thracian Lowland, Rose Valley, Struma Valley)""" + def facts_sha(facts: list[dict]) -> str: blob = "\n".join((f.get("bullet") or "") for f in facts) @@ -95,7 +100,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -146,10 +151,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -259,6 +269,7 @@ def _build_argparser() -> argparse.ArgumentParser: ap.add_argument("--lang", action="append", choices=TARGET_LOCALES, default=None) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--workers", type=int, default=1) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true") ap.add_argument("--batch", action="store_true") roundtrip.add_arguments(ap) @@ -342,8 +353,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -375,6 +388,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] @@ -384,7 +399,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: for j in jobs: diff --git a/scripts/build_terroir_feedback.py b/scripts/build_terroir_feedback.py new file mode 100644 index 0000000..821ff18 --- /dev/null +++ b/scripts/build_terroir_feedback.py @@ -0,0 +1,224 @@ +"""Build / merge the per-record review feedback sidecars +(`raw/terroir-facts-feedback/<slug>.json`, see `_lib/terroir_feedback.py`) +from a quality review's evidence directory. + +Reads, from `--evidence DIR` (default: the 2026-09-12 full-corpus review): + + confirmed-misleading.json one row per verified misleading bullet: + id, slug, i, country, subsection, mode, + where, en, src, reason, corrected_en + merged.json `record_notes` — per-record MISSING / SIB / + WRONG_SOURCE / OTHER reviewer notes + +and, per slug, `raw/terroir-facts/<slug>.json` for the source shas the +review graded against. Existing sidecars are MERGED: a do-not-claim entry +whose source bullet is already present (token-set ratio ≥ 90) is kept +once, hints and cautions are deduplicated at ≥ 60 (two review lenses +paraphrase one observation), the review is +appended to `reviews`, and `history` is never touched — so a second +review adds to the trail instead of replacing it. + + .venv/bin/python scripts/build_terroir_feedback.py --dry-run + .venv/bin/python scripts/build_terroir_feedback.py + .venv/bin/python scripts/build_terroir_feedback.py --only taurasi --only volnay +""" + +from __future__ import annotations + +import argparse +import json +import sys +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +from rapidfuzz import fuzz + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib.terroir_feedback import FEEDBACK_DIR, clear_cache # noqa: E402 + +DEFAULT_EVIDENCE = ROOT / "tmp" / "terroir-facts-review" / "full-review-2026-09-12" +DEFAULT_REVIEW_ID = "review-2026-09-12" +DEFAULT_REVIEWER = "claude-opus-5 review agents, adversarially verified per record" +TERROIR_FACTS = ROOT / "raw" / "terroir-facts" + +MERGE_THRESHOLD = 90 # do-not-claim: same source bullet +NOTE_MERGE_THRESHOLD = 60 # hints / cautions: two lenses paraphrase one observation (pairs score 60–74) +NOTE_KIND = {"SIB": "sibling-text", "WRONG_SOURCE": "wrong-source", "OTHER": "other"} +_SKIP_NOTE_PREFIXES = ("Batch-wide", "batch-wide") +_SKIP_NOTE_MARKERS = ("Record header says", "record header says") + + +def log(msg: str) -> None: + print(f"[feedback] {msg}", file=sys.stderr) + + +def _present(text: str, pool: list[str], threshold: int = MERGE_THRESHOLD) -> bool: + return any(fuzz.token_set_ratio(text, p) >= threshold for p in pool if p) + + +def _caution_kind(note: str, tag: str) -> str: + low = note.lower() + if tag == "WRONG_SOURCE": + if "wiki" in low: + return "wrong-wikipedia" + return "wrong-source" + if tag == "SIB": + return "sibling-text" + if "wiki_url" in low or "wikipedia" in low and "points to" in low: + return "wrong-wikipedia" + if "typo" in low or "swapped" in low: + return "source-typo" + return "other" + + +def collect(evidence: Path, review_id: str) -> dict[str, dict]: + """slug → {do_not_claim, capture_if_present, record_cautions} from the + evidence directory.""" + misleading = json.loads((evidence / "confirmed-misleading.json").read_text(encoding="utf-8")) + merged = json.loads((evidence / "merged.json").read_text(encoding="utf-8")) + per: dict[str, dict] = {} + + def slot(slug: str) -> dict: + return per.setdefault(slug, {"do_not_claim": [], "capture_if_present": [], "record_cautions": []}) + + for row in misleading: + slot(row["slug"])["do_not_claim"].append({ + "claim_en": row.get("en") or "", + "claim_src": row.get("src") or "", + "subsection": row.get("subsection") or "", + "mode": row.get("mode") or "other", + "stage": row.get("where") or "extraction", + "why": row.get("reason") or "", + "fact_index": row.get("i"), + "review": review_id, + }) + for slug, notes in (merged.get("record_notes") or {}).items(): + for _lens, tag, note in notes: + note = " ".join((note or "").split()) + if not note: + continue + if tag == "MISSING": + s = slot(slug) + if not _present(note, [h["hint"] for h in s["capture_if_present"]], NOTE_MERGE_THRESHOLD): + s["capture_if_present"].append({"hint": note, "review": review_id}) + continue + if note.startswith(_SKIP_NOTE_PREFIXES) or any(m in note for m in _SKIP_NOTE_MARKERS): + continue + s = slot(slug) + if not _present(note, [c["note"] for c in s["record_cautions"]], NOTE_MERGE_THRESHOLD): + s["record_cautions"].append({ + "kind": _caution_kind(note, tag), "note": note, "review": review_id, + }) + return per + + +def _graded_against(slug: str) -> dict: + p = TERROIR_FACTS / f"{slug}.json" + if not p.exists(): + return {} + try: + d = json.loads(p.read_text(encoding="utf-8")) + except (ValueError, OSError): + return {} + return { + "country": d.get("country") or "fr", + "source_lang": d.get("source_lang"), + "cahier_source_sha": d.get("cahier_source_sha"), + "wiki_source_revision": d.get("wiki_source_revision"), + "facts_fetched_at": d.get("fetched_at"), + } + + +def merge_into(existing: dict | None, slug: str, new: dict, review: dict) -> tuple[dict, Counter]: + graded = _graded_against(slug) + fb = existing or { + "slug": slug, "country": graded.get("country"), "source_lang": graded.get("source_lang"), + "graded_against": {}, "reviews": [], "do_not_claim": [], "capture_if_present": [], + "record_cautions": [], "history": [], + } + fb.setdefault("history", []) + if graded: + fb["country"] = fb.get("country") or graded.get("country") + fb["source_lang"] = fb.get("source_lang") or graded.get("source_lang") + fb["graded_against"] = { + "cahier_source_sha": graded.get("cahier_source_sha"), + "wiki_source_revision": graded.get("wiki_source_revision"), + "facts_fetched_at": graded.get("facts_fetched_at"), + } + added = Counter() + if review["id"] not in {r.get("id") for r in fb["reviews"]}: + fb["reviews"].append(review) + pool = [c.get("claim_src") or c.get("claim_en") for c in fb["do_not_claim"]] + for c in new["do_not_claim"]: + if not _present(c["claim_src"] or c["claim_en"], pool): + fb["do_not_claim"].append(c) + pool.append(c["claim_src"] or c["claim_en"]) + added["do_not_claim"] += 1 + pool = [h["hint"] for h in fb["capture_if_present"]] + for h in new["capture_if_present"]: + if not _present(h["hint"], pool, NOTE_MERGE_THRESHOLD): + fb["capture_if_present"].append(h) + pool.append(h["hint"]) + added["capture_if_present"] += 1 + pool = [n["note"] for n in fb["record_cautions"]] + for n in new["record_cautions"]: + if not _present(n["note"], pool, NOTE_MERGE_THRESHOLD): + fb["record_cautions"].append(n) + pool.append(n["note"]) + added["record_cautions"] += 1 + return fb, added + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--evidence", type=Path, default=DEFAULT_EVIDENCE) + ap.add_argument("--review-id", default=DEFAULT_REVIEW_ID) + ap.add_argument("--reviewer", default=DEFAULT_REVIEWER) + ap.add_argument("--out", type=Path, default=FEEDBACK_DIR) + ap.add_argument("--only", action="append", default=None, metavar="SLUG") + ap.add_argument("--dry-run", action="store_true") + args = ap.parse_args() + + if not (args.evidence / "confirmed-misleading.json").exists(): + log(f"error: {args.evidence} has no confirmed-misleading.json") + return 1 + review = { + "id": args.review_id, "date": args.review_id.replace("review-", ""), + "reviewer": args.reviewer, "evidence": str(args.evidence.relative_to(ROOT)) if args.evidence.is_relative_to(ROOT) else str(args.evidence), + } + per = collect(args.evidence, args.review_id) + if args.only: + per = {s: v for s, v in per.items() if s in set(args.only)} + totals = Counter() + written = 0 + args.out.mkdir(parents=True, exist_ok=True) + for slug in sorted(per): + path = args.out / f"{slug}.json" + existing = json.loads(path.read_text(encoding="utf-8")) if path.exists() else None + fb, added = merge_into(existing, slug, per[slug], review) + totals.update(added) + if added and not args.dry_run: + path.write_text(json.dumps(fb, ensure_ascii=False, indent=1) + "\n", encoding="utf-8") + written += 1 + clear_cache() + manifest = { + "built_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "review": review, "records": len(per), "written": written, + "added": dict(totals), + "records_with_do_not_claim": sum(1 for v in per.values() if v["do_not_claim"]), + "records_with_capture_hints": sum(1 for v in per.values() if v["capture_if_present"]), + "records_with_cautions": sum(1 for v in per.values() if v["record_cautions"]), + } + if not args.dry_run: + (args.out / "manifest.json").write_text(json.dumps(manifest, ensure_ascii=False, indent=1) + "\n", encoding="utf-8") + log(("dry-run: " if args.dry_run else "") + json.dumps(manifest["added"]) + f" over {len(per)} records" + f" ({manifest['records_with_do_not_claim']} with do-not-claim, " + f"{manifest['records_with_capture_hints']} with hints, {manifest['records_with_cautions']} with cautions)") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/ch/02d_extract_terroir_facts.py b/scripts/ch/02d_extract_terroir_facts.py index 513b012..5f990b0 100644 --- a/scripts/ch/02d_extract_terroir_facts.py +++ b/scripts/ch/02d_extract_terroir_facts.py @@ -40,7 +40,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -49,6 +48,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system, split_user_lead # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "ch" / "dokumente-extracted" WIKI_AOCS_ROOT = ROOT / "raw" / "wikipedia" / "aocs" @@ -188,7 +195,7 @@ - Les citations sont VERBATIM (copier-coller) depuis leur source. N'attribue JAMAIS à une source un texte qui n'y figure pas. - Pas de jugements de valeur ("exceptionnel", "prestigieux"…). - Pas de chiffres absents des deux sources. -- Maximum {max_bullets} puces, ≤ 140 caractères chacune. +- Maximum {max_bullets} puces ; chaque puce est une phrase complète d'environ 120 à 220 caractères — jamais un fragment télégraphique. - Si ni Wikipédia ni le règlement ne contiennent de fait concret remarquable pour cette sous-section, renvoie une liste vide. Réponds UNIQUEMENT en JSON, sans préambule: @@ -213,7 +220,7 @@ - Zitate sind WÖRTLICH (kopieren und einfügen). Schreibe NIEMALS einer Quelle einen Text zu, der dort nicht vorkommt. - Keine Werturteile ("aussergewöhnlich", "prestigeträchtig"…). - Keine Zahlen, die in keiner der beiden Quellen stehen. -- Maximal {max_bullets} Einträge, je ≤ 140 Zeichen. +- Maximal {max_bullets} Einträge; jeder Eintrag ist ein vollständiger Satz von etwa 120–220 Zeichen — nie ein telegrafisches Fragment. - Wenn weder Wikipedia noch die Verordnung einen konkreten bemerkenswerten Fakt enthalten, gib eine leere Liste zurück. Antworte NUR in JSON, ohne Text davor oder danach: @@ -238,13 +245,14 @@ - Le citazioni sono TESTUALI (copia-incolla). NON attribuire MAI a una fonte un testo che non vi compare. - Niente giudizi di valore ("eccezionale", "prestigioso"…). - Niente cifre assenti da entrambe le fonti. -- Massimo {max_bullets} voci, ≤ 140 caratteri ciascuna. +- Massimo {max_bullets} voci; ogni voce è una frase completa di circa 120–220 caratteri — mai un frammento telegrafico. - Se né Wikipedia né il regolamento contengono un fatto concreto rilevante per questa sottosezione, restituisci una lista vuota. Rispondi SOLO in JSON, senza testo prima o dopo: {{"facts": [{{"bullet": "…", "cahier_quote": "…", "wiki_quote": "…"}}, ...]}} Usa una stringa vuota "" per la citazione mancante.""", } +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) USER_LEAD: dict[str, str] = { @@ -257,25 +265,10 @@ # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def wiki_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -448,9 +441,13 @@ def _process_subsection( topics=topic, max_bullets=sub["max_bullets"], ) - user = USER_LEAD[lang].format(label=label, ctx=cahier_ctx) + system = with_feedback(system, record["slug"]) + # The regulator text is the cached leading block: the four sub-section calls share it. + user, doc = split_user_lead(USER_LEAD[lang], label=label, ctx=cahier_ctx) + system = cached_system(doc, system, phased=True) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -485,6 +482,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: sources_list = record.get("sources") or [] reglement = next((s for s in sources_list if s.get("kind") == "cantonal-reglement"), {}) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, lang) + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "ch", "source_lang": lang, @@ -492,6 +495,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -502,7 +507,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki.get("page_url"), "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -561,12 +566,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection": sub["key"], "subsection_label": label, "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM[lang].format( + "system_prompt": with_feedback(EXTRACT_SYSTEM[lang].format( wiki_hint=wiki_hint or "(no Wikipedia extract)", label=label, topics=topics[sub["key"]], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_ctx": job["cahier_ctx"], "wiki_hint": wiki_hint, "cahier_source_sha": wiki_sha(job["cahier_ctx"]), @@ -603,7 +608,7 @@ def _write_imported_cache(*, slug: str, name: str, lang: str, facts: list[dict], "wiki_source_revision": wiki_record.get("revision"), "wiki_source_url": wiki_record.get("page_url"), } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts(slug_items: list[dict], cahier_ctx: str) -> list[dict]: @@ -701,7 +706,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -734,7 +739,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-ch.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -772,7 +777,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/ch/02e_translate_terroir_facts.py b/scripts/ch/02e_translate_terroir_facts.py index e0e61be..2c193eb 100644 --- a/scripts/ch/02e_translate_terroir_facts.py +++ b/scripts/ch/02e_translate_terroir_facts.py @@ -26,6 +26,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -43,11 +46,16 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. Array length MUST equal the input list length. -- Preserve Swiss proper nouns verbatim: appellation names (Valais, Vaud, Genève, Neuchâtel, Bündner Herrschaft, Lavaux, Dézaley, Calamin, Ticino, Bielersee, Thunersee, Mont-Vully, Chablais, La Côte, Côtes-de-l'Orbe, Bonvillars, Mandement, Rosso/Bianco/Rosato del Ticino), canton names in their native form (Valais/Wallis, Vaud, Genève, Neuchâtel, Ticino, Schwyz, Zürich, Aargau, Graubünden, Bern/Berne, Fribourg, Jura, Schaffhausen, etc.), Swiss grape varieties (Chasselas/Fendant, Petite Arvine, Humagne blanc, Humagne rouge, Amigne, Cornalin du Valais, Heida/Païen, Rèze, Räuschling, Completer, Bondola, Blauburgunder, Riesling-Sylvaner / Müller-Thurgau, Pinot noir, Gamay, Gamaret, Garanoir, Merlot, Diolinoir), commune names and lieu-dit names (Dézaley, Calamin, Saint-Saphorin, Lavaux), geological / soil terms (molasse, gneiss, calcaire, schiste, moraine, alluvial), and Swiss climatic features (foehn, brises lacustres, Lac Léman, Bielersee). - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +PROPER_NOUNS = { + "fr": """appellation names (Valais, Vaud, Genève, Neuchâtel, Lavaux, Dézaley, Calamin, Mont-Vully, Vully, Chablais, La Côte, Côtes-de-l'Orbe, Bonvillars, Mandement, Bielersee / Lac de Bienne, Thunersee / Lac de Thoune); canton names in their native form (Valais, Vaud, Genève, Neuchâtel, Fribourg, Bern, Jura); grape names (Chasselas, Fendant, Petite Arvine, Humagne blanche, Humagne rouge, Amigne, Cornalin, Heida, Païen, Rèze, Gamaret, Garanoir, Diolinoir, Pinot noir, Gamay, Merlot); commune and lieu-dit names (Saint-Saphorin, Dézaley, Calamin, Lavaux); lakes and named winds (Lac Léman, foehn)""", + "de": """appellation names (Bündner Herrschaft, Zürichsee, Bielersee, Thunersee, Schaffhausen, Aargau, Thurgau, Graubünden, Zürich, Schwyz, Basel, Luzern, St. Gallen); canton names in their native form (Zürich, Aargau, Graubünden, Bern, Schaffhausen, Thurgau, Schwyz, Luzern); grape names (Blauburgunder, Riesling-Sylvaner, Müller-Thurgau, Räuschling, Completer, Chasselas, Pinot noir, Merlot); commune and Lage names; lakes and named winds (Zürichsee, Bielersee, Bodensee, Föhn)""", + "it": """appellation names (Ticino, Rosso del Ticino, Bianco del Ticino, Rosato del Ticino, Mendrisiotto, Sopraceneri, Sottoceneri, Bellinzonese, Locarnese, Luganese); canton names in their native form (Ticino, Grigioni); grape names (Merlot, Bondola, Chardonnay, Sauvignon); commune names; lakes and named winds (Lago Maggiore, Lago di Lugano, favonio)""", +} + def facts_sha(facts: list[dict]) -> str: blob = "\n".join((f.get("bullet") or "") for f in facts) @@ -93,7 +101,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -151,13 +159,19 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT_TEMPLATE.format( - source_lang_name=SOURCE_LANG_NAME.get(job["source_lang"], job["source_lang"]), - lang_name=LOCALE_NAME[job["lang"]], + system = translation_system_prompt( + SYSTEM_PROMPT_TEMPLATE.format( + source_lang_name=SOURCE_LANG_NAME.get(job["source_lang"], job["source_lang"]), + lang_name=LOCALE_NAME[job["lang"]], + ), + source_lang=job["source_lang"], target_lang=job["lang"], + proper_nouns=PROPER_NOUNS.get(job["source_lang"], ""), ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -263,6 +277,7 @@ def _build_argparser() -> argparse.ArgumentParser: ap.add_argument("--lang", action="append", choices=TARGET_LOCALES, default=None) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--workers", type=int, default=1) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true") ap.add_argument("--batch", action="store_true") roundtrip.add_arguments(ap) @@ -318,8 +333,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -367,6 +384,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -375,7 +394,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: for j in jobs: diff --git a/scripts/cy/02d_extract_terroir_facts.py b/scripts/cy/02d_extract_terroir_facts.py index 914d05d..c58903f 100644 --- a/scripts/cy/02d_extract_terroir_facts.py +++ b/scripts/cy/02d_extract_terroir_facts.py @@ -31,7 +31,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -40,6 +39,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "cy" / "dokumenti-extracted" WIKI_AOCS = ROOT / "raw" / "wikipedia" / "aocs" / "el" @@ -134,36 +141,22 @@ - Τα αποσπάσματα είναι ΑΥΤΟΛΕΞΕΙ (αντιγραμμένα) από την αντίστοιχη πηγή. ΠΟΤΕ μην αποδώσεις σε μια πηγή κείμενο που δεν εμφανίζεται εκεί. - Χωρίς αξιολογικές κρίσεις ("εξαίσιος", "πρεστιζιακός" …). - Χωρίς εξωτερικά συμπεράσματα. Χωρίς αριθμούς που δεν εμφανίζονται σε καμία από τις δύο πηγές. -- Μέγιστο {max_bullets} στοιχεία, καθένα ≤ 140 χαρακτήρες. +- Μέγιστο {max_bullets} στοιχεία· κάθε στοιχείο είναι μία πλήρης πρόταση περίπου 120–220 χαρακτήρων — ποτέ τηλεγραφικό απόσπασμα. - Αν ούτε το ενιαίο έγγραφο ούτε η Wikipedia περιέχουν συγκεκριμένο γεγονός άξιο προσοχής για αυτή την υποενότητα, επέστρεψε κενή λίστα. Απάντησε ΜΟΝΟ σε JSON, χωρίς κείμενο πριν ή μετά: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Για ελλείπον απόσπασμα χρησιμοποίησε κενό αλφαριθμητικό "".""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -283,11 +276,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Υποενότητα προς επεξεργασία: {label}\n\n" - f"Κείμενο του ενιαίου εγγράφου (Περιγραφή του δεσμού):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Υποενότητα προς επεξεργασία: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Κείμενο του ενιαίου εγγράφου (Περιγραφή του δεσμού):\n\n{lien_text}" def _process_subsection( @@ -301,9 +295,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -338,6 +336,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "el") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "cy", "source_lang": "el", @@ -345,6 +349,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -355,7 +361,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -418,12 +424,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(απόσπασμα Wikipedia μη διαθέσιμο)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -464,7 +470,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts(slug_items: list[dict], lien: str) -> list[dict]: @@ -579,7 +585,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -612,7 +618,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-cy.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -662,7 +668,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/cy/02e_translate_terroir_facts.py b/scripts/cy/02e_translate_terroir_facts.py index 7703a7e..b72d25e 100644 --- a/scripts/cy/02e_translate_terroir_facts.py +++ b/scripts/cy/02e_translate_terroir_facts.py @@ -27,6 +27,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -39,13 +42,15 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Greek proper nouns verbatim: appellation names ("Σαντορίνη", "Νεμέα", "Νάουσα", "Μαντινεία", "Ραψάνη", "Πάτρα", "Μαυροδάφνη Πατρών", "Μοσχάτο Πατρών", "Μοσχάτος Ρίου Πάτρας", "Μονεμβασιά-Malvasia", "Ζίτσα", "Αμύνταιο", "Γουμένισσα", "Μεσενικόλα", "Πλαγιές Μελίτωνα", "Σάμος", "Ρόδος", "Λήμνος", "Πάρος", "Σητεία", "Δαφνές", "Πεζά", "Αρχάνες", "Χάνδακας - Candia", "Ρομπόλα Κεφαλληνίας", "Αγχίαλος", "Μοσχάτος Λήμνου", "Μοσχάτος Ρόδου", "Μαυροδάφνη Κεφαλληνίας", "Μοσχάτος Κεφαλληνίας", "Malvasia Πάρος", "Malvasia Σητείας", "Malvasia Χάνδακας-Candia"), grape variety names ("Ασύρτικο", "Ξινόμαυρο", "Αγιωργίτικο", "Μοσχοφίλερο", "Ροδίτης", "Ρομπόλα", "Λημνιό", "Λημνιώνα", "Μαυροδάφνη", "Μαλαγουζιά", "Σαββατιανό", "Βιδιανό", "Βιλάνα", "Λιάτικο", "Κοτσιφάλι", "Μανδηλαριά", "Αθήρι", "Αηδάνι", "Θραψαθήρι", "Ντεμπίνα", "Νεγκόσκα", "Σταυρωτό", "Κρασάτο", "Μπατίκι", "Vinsanto", "Νυχτέρι"), named geographical features ("Μακεδονία", "Θράκη", "Θεσσαλία", "Ήπειρος", "Πελοπόννησος", "Στερεά Ελλάδα", "Κρήτη", "Ιόνια Νησιά", "Νησιά Αιγαίου", "Όλυμπος", "Καλντέρα", "Παρνασσός", "Όρος Μέλιτων", "Πάικο", "Βέρμιο", "Πεντελικό", "Πάρνηθα"), named soil types ("ασπρα γη", "ηφαιστειακά εδάφη", "αλλουβιακά", "ασβεστόλιθος", "σχιστόλιθος", "μάργες", "πυριγενή πετρώματα", "μαυρόχωμα", "αμμώδες", "χαλικώδες", "αργιλώδες", "αμμοχαλικώδες"), named climatic features ("μεσογειακό κλίμα", "ηπειρωτικό κλίμα", "μελτέμια", "ετήσιοι άνεμοι", "αλίπνοες αύρες", "ορεινό μικροκλίμα"), and Greek wine-law terms ("προστατευόμενη ονομασία προέλευσης", "προστατευόμενη γεωγραφική ένδειξη", "ΠΟΠ", "ΠΓΕ", "οινολογικές πρακτικές", "αμπελουργική ζώνη", "ενιαίο έγγραφο", "προδιαγραφές προϊόντος", "αμπελοοινικός χάρτης"). -- Wine-style traditional terms (preserve verbatim): "Vinsanto", "Νυχτέρι", "Λιαστός οίνος", "Vin doux naturel" / "οίνος γλυκός φυσικός", "όψιμη συγκομιδή", "αφρώδης οίνος". +- Registered Greek wine terms stay, in their Latin form: "Vinsanto", "Nychteri" (for Νυχτέρι), "vin doux naturel" (for οίνος γλυκός φυσικός). Generic style categories (λιαστός οίνος, όψιμη συγκομιδή, αφρώδης οίνος) are common nouns and are translated. - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Greek form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "el" +PROPER_NOUNS = """appellation names, written in their EU-official Latin form (Koumandaria / Commandaria, Krasochoria Lemesou, Krasochoria Lemesou - Afamis, Krasochoria Lemesou - Laona, Laona Akama, Vouni Panagias - Ampelitis, Pitsilia, Lemesos, Pafos, Larnaka, Lefkosia); grape names in their Latin form (Xynisteri, Mavro, Maratheftiko, Ofthalmo, Giannoudi, Promara, Spourtiko, Morokanella, Kanella, Vlouriko, Vertzami, Lefkada, Assyrtiko, Agiorgitiko); named geographical features (Troodos, Akamas, Afamis, Madari); registered traditional terms (Commandaria)""" + def facts_sha(facts: list[dict]) -> str: blob = "\n".join((f.get("bullet") or "") for f in facts) @@ -96,7 +101,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -147,10 +152,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -260,6 +270,7 @@ def _build_argparser() -> argparse.ArgumentParser: ap.add_argument("--lang", action="append", choices=TARGET_LOCALES, default=None) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--workers", type=int, default=1) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true") ap.add_argument("--batch", action="store_true") roundtrip.add_arguments(ap) @@ -343,8 +354,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -376,6 +389,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] @@ -385,7 +400,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: for j in jobs: diff --git a/scripts/cz/02d_extract_terroir_facts.py b/scripts/cz/02d_extract_terroir_facts.py index fd2053c..9a02ec1 100644 --- a/scripts/cz/02d_extract_terroir_facts.py +++ b/scripts/cz/02d_extract_terroir_facts.py @@ -34,7 +34,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -51,6 +50,14 @@ roundtrip, terroir_verbatim, ) +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "cz" / "dokumenty-extracted" WIKI_AOCS = ROOT / "raw" / "wikipedia" / "aocs" / "cs" @@ -169,37 +176,22 @@ def _chzo_for_region(region: str) -> dict: - Citáty jsou DOSLOVNÉ (kopírované) z příslušného zdroje. NIKDY nepřipisuj zdroji text, který se tam nevyskytuje. - Bez hodnotových soudů ("vynikající", "prestižní" ...). - Bez vnějších závěrů. Bez čísel, která se nenacházejí ani v jednom ze zdrojů. -- Nejvýše {max_bullets} záznamů, každý ≤ 140 znaků. +- Nejvýše {max_bullets} záznamů; každý záznam je jedna úplná věta o zhruba 120–220 znacích — nikdy telegrafický fragment. - Pokud ani jednotný dokument ani Wikipedia neobsahují konkrétní důležitý fakt pro tuto podsekci, vrať prázdný seznam. Odpověz POUZE v JSON, bez textu před nebo za: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Pro chybějící citát použij prázdný řetězec "".""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - """Longest-contiguous-match coverage of `quote` in `source` (0–1).""" - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -338,11 +330,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Sledovaná podsekce: {label}\n\n" - f"Text jednotného dokumentu (Popis souvislostí):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Sledovaná podsekce: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Text jednotného dokumentu (Popis souvislostí):\n\n{lien_text}" def _process_subsection( @@ -358,9 +351,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -395,6 +392,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "cs") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "cz", "source_lang": "cs", @@ -402,6 +405,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -412,7 +417,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -481,12 +486,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(výňatek z Wikipedie není k dispozici)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -527,7 +532,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts(slug_items: list[dict], lien: str) -> list[dict]: @@ -665,7 +670,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -698,7 +703,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-cz.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -748,7 +753,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/cz/02e_translate_terroir_facts.py b/scripts/cz/02e_translate_terroir_facts.py index 059e830..9ffcc4f 100644 --- a/scripts/cz/02e_translate_terroir_facts.py +++ b/scripts/cz/02e_translate_terroir_facts.py @@ -27,6 +27,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -39,12 +42,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Czech proper nouns verbatim: appellation names ("Čechy", "Morava", "Litoměřická", "Mělnická", "Slovácká", "Znojemská", "Velkopavlovická", "Mikulovská", "Šobes", "Znojmo", "Novosedelské Slámové víno"), wine-region names ("Čechy", "Morava"), commune, vinařská podoblast and vineyard-site ("trať" / "viniční trať") names, grape variety names ("Veltlínské zelené", "Ryzlink rýnský", "Ryzlink vlašský", "Tramín červený", "Müller Thurgau", "Rulandské bílé", "Rulandské šedé", "Rulandské modré", "Frankovka", "Modrý Portugal", "Svatovavřinecké", "Zweigeltrebe", "André", "Pálava", "Aurelius", "Hibernal", "Cabernet Moravia", "Neronet", "Sauvignon", "Chardonnay"), named geological formations and soil types ("spraš", "černozem", "jíl", "vápenec", "opuka", "slínovec", "žula", "čedič", "hadec", "hnědozem", "písčitá hlína"), named climatic features ("panonské podnebí", "kontinentální podnebí", "vliv Karpat"), and Czech wine-law / predikát terms ("vinařská oblast", "vinařská podoblast", "viniční trať", "pozdní sběr", "výběr z hroznů", "výběr z bobulí", "výběr z cibéb", "ledové víno", "slámové víno", "CHOP", "CHZO"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Czech form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "cs" +PROPER_NOUNS = """appellation names (Čechy, Morava, Litoměřická, Mělnická, Slovácká, Znojemská, Velkopavlovická, Mikulovská, Šobes, Znojmo, Novosedelské Slámové víno); wine-region names (Čechy, Morava); commune, podoblast and vineyard-site names; grape names (Veltlínské zelené, Ryzlink rýnský, Ryzlink vlašský, Tramín červený, Müller Thurgau, Rulandské bílé, Rulandské šedé, Rulandské modré, Frankovka, Modrý Portugal, Svatovavřinecké, Zweigeltrebe, André, Pálava, Aurelius, Hibernal, Cabernet Moravia, Neronet, Sauvignon, Chardonnay)""" + # ─────────────────────────────────────────────────────────────── helpers ── @@ -102,7 +107,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -157,10 +162,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -291,6 +301,7 @@ def _build_argparser() -> argparse.ArgumentParser: "--workers", type=int, default=1, help="concurrent (lang, slug) pairs (default 1, keep 1 for Ollama)", ) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true", help="re-translate even if cached") ap.add_argument( "--batch", action="store_true", @@ -378,8 +389,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -411,6 +424,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] @@ -420,7 +435,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: for j in jobs: diff --git a/scripts/de/02d_extract_terroir_facts.py b/scripts/de/02d_extract_terroir_facts.py index 303da86..d49ded7 100644 --- a/scripts/de/02d_extract_terroir_facts.py +++ b/scripts/de/02d_extract_terroir_facts.py @@ -32,7 +32,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -41,6 +40,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "de" / "dokumente-extracted" PRODUKTSPEZIFIKATION = ROOT / "raw" / "de" / "produktspezifikationen-extracted" @@ -130,36 +137,22 @@ - Zitate sind WÖRTLICH (kopiert und eingefügt) aus der jeweiligen Quelle. Schreibe NIEMALS einer Quelle einen Text zu, der dort nicht vorkommt. - Keine Werturteile ("außergewöhnlich", "prestigeträchtig"...). - Keine externen Schlussfolgerungen. Keine Zahlen, die in keiner der beiden Quellen stehen. -- Maximal {max_bullets} Einträge, je ≤ 140 Zeichen. +- Maximal {max_bullets} Einträge; jeder Eintrag ist ein vollständiger Satz von etwa 120–220 Zeichen — nie ein telegrafisches Fragment. - Wenn weder das Einzige Dokument noch Wikipedia einen konkreten bemerkenswerten Fakt für diesen Unterabschnitt enthält, gib eine leere Liste zurück. Antworte NUR in JSON, ohne Text davor oder danach: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Verwende einen leeren String "" für das fehlende Zitat.""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -286,11 +279,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Zu behandelnder Unterabschnitt: {label}\n\n" - f"Text des Einzigen Dokuments (Beschreibung des Zusammenhangs):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Zu behandelnder Unterabschnitt: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Text des Einzigen Dokuments (Beschreibung des Zusammenhangs):\n\n{lien_text}" def _process_subsection( @@ -304,9 +298,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -341,6 +339,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "de") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "de", "source_lang": "de", @@ -348,6 +352,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -358,7 +364,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -421,12 +427,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(kein Wikipedia-Auszug verfügbar)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -467,7 +473,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts(slug_items: list[dict], lien: str) -> list[dict]: @@ -601,7 +607,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -634,7 +640,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-de.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -684,7 +690,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/de/02e_translate_terroir_facts.py b/scripts/de/02e_translate_terroir_facts.py index 480dd84..b4a8fb0 100644 --- a/scripts/de/02e_translate_terroir_facts.py +++ b/scripts/de/02e_translate_terroir_facts.py @@ -26,6 +26,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -38,12 +41,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve German proper nouns verbatim: Anbaugebiet names ("Ahr", "Baden", "Franken", "Hessische Bergstraße", "Mittelrhein", "Mosel", "Nahe", "Pfalz", "Rheingau", "Rheinhessen", "Saale-Unstrut", "Sachsen", "Württemberg"), Bereich / Großlage / Einzellage names ("Bürgstadter Berg", "Würzburger Stein-Berg", "Monzinger Niederberg", "Uhlen Blaufüsser Lay", "Uhlen Laubach", "Uhlen Roth Lay", "Bocksbeutel", "Niersteiner Gutes Domtal"), Bundesland names ("Rheinland-Pfalz", "Baden-Württemberg", "Bayern", "Hessen", "Sachsen", "Sachsen-Anhalt", "Thüringen", "Mecklenburg-Vorpommern", "Brandenburg", "Schleswig-Holstein", "Saarland"), commune and river names (Rhein, Mosel, Saar, Ruwer, Ahr, Nahe, Main, Neckar, Tauber, Saale, Unstrut, Elbe), grape variety names ("Riesling", "Spätburgunder", "Weißburgunder", "Grauburgunder", "Müller-Thurgau", "Silvaner", "Dornfelder", "Trollinger", "Lemberger", "Kerner", "Bacchus", "Scheurebe", "Portugieser", "Frühburgunder", "Schwarzriesling", "Müllerrebe", "Saint Laurent", "Auxerrois", "Elbling", "Gutedel"), named geological formations ("Buntsandstein", "Muschelkalk", "Keuper", "Schiefer", "Devon-Schiefer", "Blauschiefer", "Rotschiefer", "Löss", "Lehm", "Mergel", "Basalt", "Porphyr", "Granit", "Vulkangestein", "Quarzit"), named climatic features ("Föhn", "kontinentaler Einfluss", "atlantischer Einfluss"), and German wine-law / Prädikat terms ("Kabinett", "Spätlese", "Auslese", "Beerenauslese", "Trockenbeerenauslese", "Eiswein", "Sekt", "Qualitätswein", "Prädikatswein", "Großes Gewächs", "Erstes Gewächs", "VDP.Erste Lage", "VDP.Grosse Lage", "VDP.Ortswein", "Steillage", "Steillagenweinbau", "Großlage", "Einzellage", "Ortswein", "Lagenwein"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the German form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "de" +PROPER_NOUNS = """Anbaugebiet names (Ahr, Baden, Franken, Hessische Bergstraße, Mittelrhein, Mosel, Nahe, Pfalz, Rheingau, Rheinhessen, Saale-Unstrut, Sachsen, Württemberg); Bereich, Großlage and Einzellage names (Bürgstadter Berg, Würzburger Stein-Berg, Monzinger Niederberg, Uhlen Blaufüsser Lay, Uhlen Laubach, Uhlen Roth Lay, Bocksbeutel, Niersteiner Gutes Domtal); Bundesland names (Rheinland-Pfalz, Baden-Württemberg, Bayern, Hessen, Sachsen, Sachsen-Anhalt, Thüringen, Mecklenburg-Vorpommern, Brandenburg, Schleswig-Holstein, Saarland); commune names; river names, in the established target-language form where one exists (Rhine, Moselle, Elbe) and as in the source otherwise (Saar, Ruwer, Ahr, Nahe, Main, Neckar, Tauber, Saale, Unstrut); grape names (Riesling, Spätburgunder, Weißburgunder, Grauburgunder, Müller-Thurgau, Silvaner, Dornfelder, Trollinger, Lemberger, Kerner, Bacchus, Scheurebe, Portugieser, Frühburgunder, Schwarzriesling, Müllerrebe, Saint Laurent, Auxerrois, Elbling, Gutedel); named geological formations (Buntsandstein, Muschelkalk, Keuper, Rotliegend); named winds (Föhn); Prädikat tiers and registered classification terms (Kabinett, Spätlese, Auslese, Beerenauslese, Trockenbeerenauslese, Eiswein, Großes Gewächs, Erstes Gewächs, VDP.Erste Lage, VDP.Grosse Lage, VDP.Ortswein)""" + def facts_sha(facts: list[dict]) -> str: blob = "\n".join((f.get("bullet") or "") for f in facts) @@ -94,7 +99,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -145,10 +150,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -273,6 +283,7 @@ def _build_argparser() -> argparse.ArgumentParser: "--workers", type=int, default=1, help="concurrent (lang, slug) pairs (default 1, keep 1 for Ollama)", ) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true", help="re-translate even if cached") ap.add_argument( "--batch", action="store_true", @@ -360,8 +371,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -393,6 +406,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] @@ -402,7 +417,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: for j in jobs: diff --git a/scripts/dedupe_terroir_facts.py b/scripts/dedupe_terroir_facts.py new file mode 100644 index 0000000..9653693 --- /dev/null +++ b/scripts/dedupe_terroir_facts.py @@ -0,0 +1,157 @@ +"""Post-pass over the stage-02d terroir-fact caches: drop facts restated +inside one record. No LLM call. + +Why: 02d extracts each record in four sub-section calls over overlapping +source text, so one sentence of the cahier regularly comes back two or +three times ("facteurs naturels", again under "produit", again under +"interactions"). About a tenth of all bullets were such restatements. +The rule (`_lib.terroir_dedupe.dedupe_facts`) collapses near-identical +bullets, and same-quote bullets that overlap substantially, keeping the +more informative one; bullets with different numbers are never merged. + +Translations: the stage-02e caches are index-aligned with the source +facts and keyed on `source_facts_sha` (a hash of the source bullets). +Dropping a source fact would misalign them and, at the next 02e run, +re-translate the whole record. Instead this script prunes the same +indices from every aligned bullet-mode translation cache and updates its +`source_facts_sha`, so no translation is lost and no re-translation is +needed. A translation cache that is already out of step (hash or length +mismatch) is left alone and listed — 02e re-translates it. + +Sibling guard: a parent's bullets may lead with the name of one of its +sub-denominations ("Rioja Oriental: …"); stage 04's sibling filter keeps +such a bullet only on that sub-denomination's page. Two bullets leading +with different sub-denomination names (or one labelled, one not) are +therefore never merged, so no sub-zone loses the one bullet about it. +The roster comes from the last stage-04 startup blob (which includes the +sottozone stage 04 synthesises), falling back to `wiki/_index.json`. + +Usage: + .venv/bin/python scripts/dedupe_terroir_facts.py --dry-run + .venv/bin/python scripts/dedupe_terroir_facts.py [--only SLUG …] + [--report tmp/terroir-facts-review/dedupe.json] +""" + +from __future__ import annotations + +import argparse +import sys +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib import cache # noqa: E402 +from _lib.terroir_cache import prune_translations, write_source_cache # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts, facts_sha # noqa: E402 +from _lib.terroir_roster import load_children_names, roster_stats # noqa: E402 + +TERROIR = ROOT / "raw" / "terroir-facts" +DEFAULT_REPORT = ROOT / "tmp" / "terroir-facts-review" / "dedupe.json" + + +def log(msg: str) -> None: + print(f"[dedupe] {msg}", file=sys.stderr) + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--dry-run", action="store_true", help="report only; write nothing") + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") + ap.add_argument("--report", type=Path, default=DEFAULT_REPORT, help="JSON report path") + args = ap.parse_args() + + children_names = load_children_names() + st = roster_stats() + log(f"sibling roster: {st['index']} sub-denominations from wiki/_index.json + {st['blob']} synthesised " + f"ones from the startup blob, under {len(children_names)} parents") + n_records = n_touched = n_facts_before = n_facts_after = 0 + n_sibling_guarded = 0 + reasons: Counter = Counter() + per_country: dict[str, Counter] = {} + translations_pruned = 0 + stale: list[dict] = [] + records: list[dict] = [] + + for p in sorted(TERROIR.glob("*.json")): + if p.name.startswith("manifest"): + continue + d = cache.read_json_or_none(p) + if not d or d.get("mode") == "verbatim" or not d.get("facts"): + continue + slug = d.get("slug") or p.stem + if args.only and slug not in args.only: + continue + cc = d.get("country") or "fr" + facts = d["facts"] + n_records += 1 + n_facts_before += len(facts) + res = dedupe_facts(facts, protected_names=children_names.get(slug, ())) + if children_names.get(slug) and len(dedupe_facts(facts).drops) > len(res.drops): + n_sibling_guarded += 1 + n_facts_after += len(res.kept) + stats = per_country.setdefault(cc, Counter()) + stats["records"] += 1 + stats["facts_before"] += len(facts) + stats["facts_after"] += len(res.kept) + if not res.changed: + continue + n_touched += 1 + stats["records_touched"] += 1 + for drop in res.drops: + reasons[drop["reason"]] += 1 + old_sha = facts_sha(facts) + new_sha = facts_sha(res.kept) + pruned, rec_stale = prune_translations( + slug, old_sha, len(facts), res.kept_indices, new_sha, dry_run=args.dry_run, + ) + translations_pruned += len(pruned) + stale.extend(rec_stale) + records.append({ + "slug": slug, "country": cc, + "facts_before": len(facts), "facts_after": len(res.kept), + "translations_pruned": pruned, "drops": res.drops, + }) + if not args.dry_run: + d["facts"] = res.kept + d["n_deduped"] = int(d.get("n_deduped") or 0) + len(res.drops) + write_source_cache(p, d) + + verb = "would drop" if args.dry_run else "dropped" + n_dropped = n_facts_before - n_facts_after + log("") + log(f"{verb} {n_dropped} of {n_facts_before} facts ({100 * n_dropped / max(n_facts_before, 1):.1f} %) " + f"in {n_touched} of {n_records} records; " + f"{translations_pruned} translation caches pruned in step; " + f"{len(stale)} translation caches already misaligned (left for 02e)") + log("by reason: " + ", ".join(f"{r}: {n}" for r, n in reasons.most_common())) + log(f"records where the sibling-name guard kept a bullet apart: {n_sibling_guarded}") + log(f"{'cc':4} {'records':>7} {'touched':>7} {'before':>6} {'after':>6} {'dropped':>7}") + for cc, st in sorted(per_country.items()): + log(f"{cc:4} {st['records']:7} {st['records_touched']:7} {st['facts_before']:6} " + f"{st['facts_after']:6} {st['facts_before'] - st['facts_after']:7}") + + args.report.parent.mkdir(parents=True, exist_ok=True) + cache.write_json(args.report, { + "generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "dry_run": args.dry_run, + "summary": { + "records": n_records, "records_touched": n_touched, + "facts_before": n_facts_before, "facts_after": n_facts_after, + "dropped_by_reason": dict(reasons), + "translation_caches_pruned": translations_pruned, + "translation_caches_stale": len(stale), + "records_sibling_guarded": n_sibling_guarded, + "per_country": {cc: dict(st) for cc, st in sorted(per_country.items())}, + }, + "stale_translations": stale, + "records": records, + }) + log(f"report → {args.report}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/detect_untranslated_terroir_facts.py b/scripts/detect_untranslated_terroir_facts.py new file mode 100644 index 0000000..a5de50d --- /dev/null +++ b/scripts/detect_untranslated_terroir_facts.py @@ -0,0 +1,247 @@ +"""Detect untranslated source-language residue in the stage-02e caches (W1). + +Scans `raw/translations/terroir-facts/<lang>/*.json` for lang in en / fr / +es / nl (caches with `mode == "verbatim"` are skipped) and flags every +bullet that still carries non-Latin script or one of the common-noun leaks +the 2026-09-11 review catalogued in docs/plan-terroir-facts-quality.md — +soils, climates, harvest categories, site words and scheme abbreviations +that the old per-country "Preserve … verbatim" prompt lists told the model +to keep. Prints a per-country × per-locale count table to stderr and +writes a JSON report with the flagged rows plus, per country, the exact +`02e … --batch --provider anthropic --refresh --only …` command that +re-translates them. + +No LLM call, no cache write. `--strict` exits non-zero when anything is +flagged (for CI). + + .venv/bin/python scripts/detect_untranslated_terroir_facts.py + .venv/bin/python scripts/detect_untranslated_terroir_facts.py --strict +""" + +from __future__ import annotations + +import argparse +import json +import re +import sys +from collections import defaultdict +from datetime import datetime, timezone +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib.exonyms import exonym_hits, gi_forms_from_names # noqa: E402 + +TERROIR_FACTS = ROOT / "raw" / "terroir-facts" +CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" +DEFAULT_REPORT = ROOT / "tmp" / "terroir-facts-review" / "untranslated.json" + +TARGET_LOCALES = ("en", "fr", "es", "nl") + +# Cyrillic + Greek blocks — any hit in a Latin-target cache is a leak. +NON_LATIN = re.compile(r"[Ѐ-ӿͰ-Ͽ]") + +# Verbatim from docs/plan-terroir-facts-quality.md, section W1. +LEAK = re.compile(r"(?i)\b(lösz|mészkő|homokkő|argille|argilliti|arenari[ae]|calcar[ei]|marn[ae]|scisti|" + r"podgori\w*|mineralité|kasna berba|desertn\w+ vin\w*|predikatn\w+|pozna trgatev|" + r"ledeno vino|okoliš|vapnen\w+|crvenic\w+|fliš|apnen\w+|xisto\w*|leem|zandleem|mergel|" + r"viničn\w+|dűlő\w*|lege\b|Steillage\w*|Lagenwein\w*|Urgestein|typology|must varieties)\b") + +# The plan's regex is source-language-agnostic, so a few of its tokens are +# the *correct* word in one target locale: NL "leem" / "zandleem" / "mergel" +# are the Dutch for loam / sandy loam / marl and "lege" is Dutch for +# "empty"; FR "marne" is the French for marl. Those are not leaks in that +# locale. `--raw` disables the exemption. +_TARGET_NATIVE = { + "nl": {"leem", "zandleem", "mergel", "lege"}, + "fr": {"marne"}, +} +# "Marne" capitalised is the river / département (Haute-Marne, vallée de la +# Marne), a proper noun in every locale. +_PROPER_NOUN_TOKENS = {"Marne"} + + +def _load(path: Path) -> dict | None: + try: + return json.loads(path.read_text(encoding="utf-8")) + except Exception: # noqa: BLE001 + return None + + +def _country_of(slug: str, cache: dict, memo: dict[str, str]) -> str: + if slug in memo: + return memo[slug] + src = _load(TERROIR_FACTS / f"{slug}.json") + if src is not None: + cc = src.get("country") or "fr" + else: + cc = cache.get("country") or "fr" + memo[slug] = cc + return cc + + +def _script_for(country: str) -> str: + if country == "fr": + return "scripts/02e_translate_terroir_facts.py" + return f"scripts/{country}/02e_translate_terroir_facts.py" + + +_GI_FORMS: frozenset[str] | None = None + + +def _gi_forms() -> frozenset[str]: + """Exonym source forms that also occur in an appellation name (wiki/_index.json).""" + global _GI_FORMS + if _GI_FORMS is None: + index = ROOT / "wiki" / "_index.json" + names = [] + if index.exists(): + names = [v.get("name") or "" for v in json.loads(index.read_text(encoding="utf-8")).values()] + _GI_FORMS = gi_forms_from_names(names) + return _GI_FORMS + + +def _reasons(bullet: str, lang: str, *, raw: bool) -> list[str]: + reasons: list[str] = [] + m = NON_LATIN.search(bullet) + if m: + reasons.append(f"non-latin:{m.group(0)}") + for form in exonym_hits(bullet, lang, gi_forms=_gi_forms()): + reasons.append(f"exonym:{form}") + native = set() if raw else _TARGET_NATIVE.get(lang, set()) + for m in LEAK.finditer(bullet): + tok = m.group(1) + if not raw and (tok.lower() in native or tok in _PROPER_NOUN_TOKENS): + continue + reasons.append(f"leak:{tok}") + return reasons + + +def scan(languages: tuple[str, ...], *, raw: bool) -> list[dict]: + memo: dict[str, str] = {} + rows: list[dict] = [] + for lang in languages: + for path in sorted((CACHE_ROOT / lang).glob("*.json")): + d = _load(path) + if not d or d.get("mode") == "verbatim": + continue + slug = d.get("slug") or path.stem + country = _country_of(slug, d, memo) + for i, fact in enumerate(d.get("facts") or []): + bullet = fact.get("bullet") or "" + reasons = _reasons(bullet, lang, raw=raw) + if reasons: + rows.append({ + "slug": slug, "lang": lang, "country": country, "index": i, + "bullet": bullet, "reason": "; ".join(reasons), + }) + return rows + + +def _table(rows: list[dict], languages: tuple[str, ...], key) -> str: + cells: dict[str, dict[str, set | int]] = defaultdict(lambda: defaultdict(set)) + for r in rows: + cells[r["country"]][r["lang"]].add(key(r)) + countries = sorted(cells) + head = f"{'country':<8}" + "".join(f"{lang:>7}" for lang in languages) + f"{'total':>8}" + lines = [head, "-" * len(head)] + col_tot = defaultdict(int) + for cc in countries: + n = [len(cells[cc].get(lang, ())) for lang in languages] + for lang, v in zip(languages, n): + col_tot[lang] += v + lines.append(f"{cc:<8}" + "".join(f"{v:>7}" for v in n) + f"{sum(n):>8}") + lines.append("-" * len(head)) + tot = [col_tot[lang] for lang in languages] + lines.append(f"{'all':<8}" + "".join(f"{v:>7}" for v in tot) + f"{sum(tot):>8}") + return "\n".join(lines) + + +def _commands(rows: list[dict]) -> dict[str, dict]: + by_cc: dict[str, dict[str, set]] = defaultdict(lambda: {"slugs": set(), "langs": set()}) + for r in rows: + by_cc[r["country"]]["slugs"].add(r["slug"]) + by_cc[r["country"]]["langs"].add(r["lang"]) + out: dict[str, dict] = {} + for cc in sorted(by_cc): + slugs = sorted(by_cc[cc]["slugs"]) + langs = [lang for lang in TARGET_LOCALES if lang in by_cc[cc]["langs"]] + cmd = ( + f".venv/bin/python {_script_for(cc)} --batch --provider anthropic --refresh" + + "".join(f" --lang {lang}" for lang in langs) + + "".join(f" --only {s}" for s in slugs) + ) + out[cc] = { + "script": _script_for(cc), + "slugs": slugs, + "langs": langs, + "command": cmd, + "note": ( + "--refresh re-translates every listed slug in every listed locale, so a " + "slug flagged in only some of these locales is re-done in the others too." + ), + } + return out + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) + ap.add_argument("--report", default=str(DEFAULT_REPORT), help="JSON report path") + ap.add_argument( + "--lang", action="append", choices=TARGET_LOCALES, default=None, + help="restrict to a target locale (repeatable); default: all 4", + ) + ap.add_argument("--raw", action="store_true", + help="no target-native exemptions (flag NL leem/mergel, FR marne, Marne)") + ap.add_argument("--strict", action="store_true", help="exit non-zero when anything is flagged") + args = ap.parse_args() + + languages = tuple(args.lang) if args.lang else TARGET_LOCALES + if not CACHE_ROOT.exists(): + print(f"error: {CACHE_ROOT} is missing", file=sys.stderr) + return 1 + rows = scan(languages, raw=args.raw) + + pairs = {(r["slug"], r["lang"]) for r in rows} + slugs = {r["slug"] for r in rows} + print("[detect-untranslated] flagged bullets per country × locale:", file=sys.stderr) + print(_table(rows, languages, key=lambda r: (r["slug"], r["index"])), file=sys.stderr) + print("[detect-untranslated] flagged (slug, lang) pairs per country × locale:", file=sys.stderr) + print(_table(rows, languages, key=lambda r: r["slug"]), file=sys.stderr) + print( + f"[detect-untranslated] {len(rows)} bullets, {len(pairs)} (slug, lang) pairs, " + f"{len(slugs)} slugs flagged" + (" (raw)" if args.raw else ""), + file=sys.stderr, + ) + + by_country: dict[str, dict[str, dict[str, int]]] = defaultdict(dict) + for cc in sorted({r["country"] for r in rows}): + for lang in languages: + sub = [r for r in rows if r["country"] == cc and r["lang"] == lang] + if sub: + by_country[cc][lang] = { + "bullets": len(sub), "pairs": len({r["slug"] for r in sub}), + } + report = { + "generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "languages": list(languages), + "raw": args.raw, + "summary": { + "flagged_bullets": len(rows), + "flagged_pairs": len(pairs), + "flagged_slugs": len(slugs), + "by_country": by_country, + }, + "rows": rows, + "commands": _commands(rows), + } + out = Path(args.report) + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + print(f"[detect-untranslated] report → {out}", file=sys.stderr) + return 1 if (args.strict and rows) else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/es/00_fetch_data.py b/scripts/es/00_fetch_data.py index 53d14b6..9e6730f 100644 --- a/scripts/es/00_fetch_data.py +++ b/scripts/es/00_fetch_data.py @@ -27,6 +27,9 @@ - raw/es/eambrosia/manifest.json — fetch metadata for the eAmbrosia call - raw/es/figshare/EU_PDO.gpkg — Bétard 2022 wine-PDO polygons - raw/es/figshare/manifest.json — fetch metadata + license + sha for the gpkg +- raw/es/mapa/listado-dop-igp-vinos.pdf — MAPA's listado of EU-registered ES + wine DOPs/IGPs (carries the national traditional term per GI) +- raw/es/mapa/manifest.json — fetch metadata + license + sha for the listado Each ES wine GI record carries: - giIdentifier (e.g. EUGI00000003061) — internal EU id, unstable @@ -53,6 +56,11 @@ ROOT = Path(__file__).resolve().parents[2] sys.path.insert(0, str(ROOT / "scripts")) +from _lib.es.national_term import ( # noqa: E402 + LISTADO_DIR, + LISTADO_FILE, + LISTADO_URL, +) from _lib.es.zones import ( # noqa: E402 MAPA_LICENCE, MAPA_ZONES_FILE, @@ -342,12 +350,29 @@ def fetch_mapa_zones() -> None: ) +def fetch_mapa_listado() -> None: + """MAPA "Listado de DOPs e IGPs de vinos registradas en la UE" — the + per-GI national traditional term (DO / DOCa / VP / VC / VT) keyed by + EU file number; consumed via scripts/_lib/es/national_term.py.""" + _fetch_binary_with_manifest( + label="mapa-listado", + url=LISTADO_URL, + out_path=LISTADO_DIR / LISTADO_FILE, + manifest_path=LISTADO_DIR / "manifest.json", + extra_manifest={ + "license": MAPA_LICENCE, + "attribution": "Fuente: MAPA — reutilización permitida con atribución", + }, + ) + + def main() -> int: OUT_DIR.mkdir(parents=True, exist_ok=True) fetch_figshare_gpkg() fetch_gisco_lau() fetch_sigpac_comarques() fetch_mapa_zones() + fetch_mapa_listado() full, etag = fetch_list() es_wines_all = [ g for g in full diff --git a/scripts/es/02d_extract_terroir_facts.py b/scripts/es/02d_extract_terroir_facts.py index add2fb1..03edc8c 100644 --- a/scripts/es/02d_extract_terroir_facts.py +++ b/scripts/es/02d_extract_terroir_facts.py @@ -32,7 +32,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -40,7 +39,15 @@ ROOT = Path(__file__).resolve().parents[2] sys.path.insert(0, str(ROOT / "scripts")) -from _lib import batch, cache, llm_json, providers, terroir_verbatim # noqa: E402 +from _lib import batch, llm_json, providers, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "es" / "pliegos-extracted" WIKI_AOCS = ROOT / "raw" / "wikipedia" / "aocs" / "es" @@ -137,37 +144,22 @@ - Las citas son VERBATIM (copiadas y pegadas) de su fuente respectiva. NUNCA atribuyas a una fuente un texto que no figura en ella. - Sin juicios de valor ("excepcional", "extraordinario", "prestigioso"...). - Sin inferencia externa. Sin cifras ausentes de las dos fuentes. -- Máximo {max_bullets} viñetas, ≤ 140 caracteres cada una. +- Máximo {max_bullets} viñetas; cada viñeta es una frase completa de unos 120–220 caracteres — nunca un fragmento telegráfico. - Si ni el pliego ni Wikipedia contienen un hecho notable concreto para esta sub-sección, devuelve una lista vacía. Responde ÚNICAMENTE en JSON, sin texto antes o después: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Usa una cadena vacía "" para la cita ausente.""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - """Longest-contiguous-match coverage of `quote` in `source` (0–1).""" - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -263,11 +255,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Sub-sección a tratar: {label}\n\n" - f"Texto del pliego (Vínculo con la zona geográfica):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Sub-sección a tratar: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Texto del pliego (Vínculo con la zona geográfica):\n\n{lien_text}" def _process_subsection( @@ -283,9 +276,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) # Tolerant JSON extraction @@ -321,6 +318,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: all_facts.append(f) src = record.get("source") or {} + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "es") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "es", "source_lang": "es", @@ -328,6 +331,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -337,7 +342,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -374,7 +379,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -409,7 +414,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-es.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -467,7 +472,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/es/02e_translate_terroir_facts.py b/scripts/es/02e_translate_terroir_facts.py index cd50673..30be855 100644 --- a/scripts/es/02e_translate_terroir_facts.py +++ b/scripts/es/02e_translate_terroir_facts.py @@ -32,6 +32,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -44,12 +47,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Spanish proper nouns verbatim: appellation names ("Rioja", "Ribera del Duero", "Priorat"), region names ("La Rioja", "Castilla y León"), commune names, grape variety names ("Tempranillo", "Garnacha", "Albariño", "Mencía", "Bobal", "Monastrell", "Verdejo", "Viura"), named geological formations and soil types ("llicorella", "albariza", "calcáreo-arcilloso"), local landscape names ("ribera", "páramo", "comarca"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Spanish form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "es" +PROPER_NOUNS = """appellation names (Rioja, Ribera del Duero, Priorat); region names (La Rioja, Castilla y León); commune names; grape names (Tempranillo, Garnacha, Albariño, Mencía, Bobal, Monastrell, Verdejo, Viura); named geological formations (llicorella, albariza)""" + # ─────────────────────────────────────────────────────────────── helpers ── @@ -109,7 +114,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -167,10 +172,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -307,6 +317,7 @@ def _build_argparser() -> argparse.ArgumentParser: "--workers", type=int, default=1, help="concurrent (lang, slug) pairs to translate (default 1, keep 1 for Ollama on M1 32GB)", ) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true", help="re-translate even if cached") ap.add_argument( "--batch", action="store_true", @@ -345,7 +356,7 @@ def _dispatch_emit_or_import(args, languages: tuple[str, ...]) -> int | None: def _make_provider(args) -> tuple[object | None, str]: return providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) @@ -414,8 +425,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -447,6 +460,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] diff --git a/scripts/filter_terroir_boilerplate.py b/scripts/filter_terroir_boilerplate.py new file mode 100644 index 0000000..d35a271 --- /dev/null +++ b/scripts/filter_terroir_boilerplate.py @@ -0,0 +1,102 @@ +"""Post-pass over the terroir-fact caches: drop boilerplate facts — a +quote shared by ≥ 3 records of one country that matches a tautology +pattern for its language (`_lib.terroir_boilerplate`). No LLM call. + +The dropped indices are pruned from the index-aligned stage-02e +translation caches in step (their `source_facts_sha` is updated), the way +`dedupe_terroir_facts.py` does it, so nothing is re-translated. + +Usage: + .venv/bin/python scripts/filter_terroir_boilerplate.py --dry-run + .venv/bin/python scripts/filter_terroir_boilerplate.py [--only SLUG …] [--report PATH] +""" + +from __future__ import annotations + +import argparse +import sys +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib import cache # noqa: E402 +from _lib.terroir_boilerplate import find_boilerplate # noqa: E402 +from _lib.terroir_cache import TERROIR, prune_translations, write_source_cache # noqa: E402 +from _lib.terroir_dedupe import facts_sha # noqa: E402 + +DEFAULT_REPORT = ROOT / "tmp" / "terroir-facts-review" / "boilerplate.json" + + +def log(msg: str) -> None: + print(f"[boilerplate] {msg}", file=sys.stderr) + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--dry-run", action="store_true") + ap.add_argument("--only", action="append", default=[]) + ap.add_argument("--report", type=Path, default=DEFAULT_REPORT) + args = ap.parse_args() + + caches: dict[str, dict] = {} + paths: dict[str, Path] = {} + for p in sorted(TERROIR.glob("*.json")): + if p.name.startswith("manifest"): + continue + d = cache.read_json_or_none(p) + if not d or d.get("mode") == "verbatim" or not d.get("facts"): + continue + slug = d.get("slug") or p.stem + caches[slug] = d + paths[slug] = p + drops = find_boilerplate(caches) + if args.only: + drops = {s: v for s, v in drops.items() if s in args.only} + + per_country: Counter = Counter() + records: list[dict] = [] + stale: list[dict] = [] + n_pruned = 0 + for slug, idx in sorted(drops.items()): + d = caches[slug] + facts = d["facts"] + cc = d.get("country") or "fr" + per_country[cc] += len(idx) + kept_indices = [i for i in range(len(facts)) if i not in set(idx)] + old_sha = facts_sha(facts) + kept = [facts[i] for i in kept_indices] + new_sha = facts_sha(kept) + records.append({"slug": slug, "country": cc, "dropped": [ + {"index": i, "bullet": facts[i].get("bullet"), "cahier_quote": (facts[i].get("cahier_quote") or "")[:200]} for i in idx + ]}) + pruned, rec_stale = prune_translations(slug, old_sha, len(facts), kept_indices, new_sha, dry_run=args.dry_run) + n_pruned += len(pruned) + stale.extend(rec_stale) + if not args.dry_run: + d["facts"] = kept + d["n_boilerplate"] = int(d.get("n_boilerplate") or 0) + len(idx) + write_source_cache(paths[slug], d) + + verb = "would drop" if args.dry_run else "dropped" + total = sum(per_country.values()) + log(f"{verb} {total} boilerplate facts in {len(drops)} records; {n_pruned} translation caches pruned; " + f"{len(stale)} already misaligned (left for 02e)") + log("per country: " + ", ".join(f"{c}: {n}" for c, n in sorted(per_country.items()))) + args.report.parent.mkdir(parents=True, exist_ok=True) + cache.write_json(args.report, { + "generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "dry_run": args.dry_run, + "summary": {"records": len(drops), "facts_dropped": total, "per_country": dict(per_country), + "translation_caches_pruned": n_pruned, "translation_caches_stale": len(stale)}, + "stale_translations": stale, + "records": records, + }) + log(f"report → {args.report}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/gb/02d_extract_terroir_facts.py b/scripts/gb/02d_extract_terroir_facts.py index 45ae45e..768426b 100644 --- a/scripts/gb/02d_extract_terroir_facts.py +++ b/scripts/gb/02d_extract_terroir_facts.py @@ -43,7 +43,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -52,6 +51,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system, split_user_lead # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "gb" / "specs-extracted" WIKI_AOCS_ROOT = ROOT / "raw" / "wikipedia" / "aocs" @@ -123,12 +130,13 @@ - Quotes are VERBATIM (copy-paste) from their source. NEVER attribute to a source text that does not appear in it. - No value judgements ("exceptional", "prestigious"…). - No figures absent from both sources. -- At most {max_bullets} bullets, each ≤ 140 characters. +- At most {max_bullets} bullets; each bullet is one full sentence of roughly 120–220 characters — never a telegraphic fragment. - If neither the specification nor Wikipedia contains a concrete noteworthy fact for this sub-section, return an empty list. Reply ONLY in JSON, no preamble: {{"facts": [{{"bullet": "…", "cahier_quote": "…", "wiki_quote": "…"}}, ...]}} Use an empty string "" for the missing quote.""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) USER_LEAD = "Sub-section: {label}\n\nRegulator context:\n\n{ctx}" @@ -137,25 +145,10 @@ # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def wiki_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -304,9 +297,13 @@ def _process_subsection(provider, model_id: str, record: dict, sub: dict): wiki_hint=wiki_hint or "(no Wikipedia extract available)", label=label, topics=topic, max_bullets=sub["max_bullets"], ) - user = USER_LEAD.format(label=label, ctx=cahier_ctx) + system = with_feedback(system, record["slug"]) + # The regulator text is the cached leading block: the four sub-section calls share it. + user, doc = split_user_lead(USER_LEAD, label=label, ctx=cahier_ctx) + system = cached_system(doc, system, phased=True) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -334,6 +331,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, SOURCE_LANG) + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "gb", "source_lang": SOURCE_LANG, @@ -341,6 +344,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -351,7 +356,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki.get("page_url"), "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -395,11 +400,11 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection": sub["key"], "subsection_label": label, "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(no Wikipedia extract)", label=label, topics=SUBSECTION_TOPICS[sub["key"]], max_bullets=sub["max_bullets"], - ), + ), rec["slug"]), "cahier_ctx": rec.get("_cahier_ctx") or "", "wiki_hint": wiki_hint, "cahier_source_sha": wiki_sha(rec.get("_cahier_ctx") or ""), @@ -443,7 +448,7 @@ def import_todo(in_path: Path, *, translator_id: str, translator_kind: str) -> i f["subsection"] = it.get("subsection") or "facteurs_naturels" facts.append(f) wiki = rec.get("_wiki_record") or {} - cache.write_json(CACHE_DIR / f"{slug}.json", { + write_source_cache(CACHE_DIR / f"{slug}.json", { "country": "gb", "source_lang": SOURCE_LANG, "slug": slug, "name": rec.get("name") or slug, "facts": facts, "model": translator_id, "model_kind": translator_kind, @@ -498,7 +503,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -531,7 +536,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-gb.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -569,7 +574,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/gb/02e_translate_terroir_facts.py b/scripts/gb/02e_translate_terroir_facts.py index 5673a60..0a60ab4 100644 --- a/scripts/gb/02e_translate_terroir_facts.py +++ b/scripts/gb/02e_translate_terroir_facts.py @@ -27,6 +27,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -40,11 +43,12 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. Array length MUST equal the input list length. -- Preserve British proper nouns verbatim: appellation and place names (English, Welsh, English Regional, Welsh Regional, Sussex, East Sussex, West Sussex, Darnibole, Cornwall, the South Downs, the Weald, Camel Valley, Plumpton College), geological terms (Kimmeridgian, greensand, chalk, slate), grape names (Bacchus, Seyval Blanc, Madeleine Angevine, Chardonnay, Pinot Noir, Pinot Meunier, …), and the UK scheme terms PDO / PGI. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +PROPER_NOUNS = """appellation and place names (English, Welsh, English Regional, Welsh Regional, Sussex, East Sussex, West Sussex, Darnibole, Cornwall, the South Downs, the Weald, Camel Valley, Plumpton College); grape names (Bacchus, Seyval Blanc, Madeleine Angevine, Chardonnay, Pinot Noir, Pinot Meunier)""" + def facts_sha(facts: list[dict]) -> str: blob = "\n".join((f.get("bullet") or "") for f in facts) @@ -90,7 +94,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -141,13 +145,19 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT_TEMPLATE.format( - source_lang_name=SOURCE_LANG_NAME.get(job["source_lang"], job["source_lang"]), - lang_name=LOCALE_NAME[job["lang"]], + system = translation_system_prompt( + SYSTEM_PROMPT_TEMPLATE.format( + source_lang_name=SOURCE_LANG_NAME.get(job["source_lang"], job["source_lang"]), + lang_name=LOCALE_NAME[job["lang"]], + ), + source_lang=job["source_lang"], target_lang=job["lang"], + proper_nouns=PROPER_NOUNS, ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -237,6 +247,7 @@ def _build_argparser() -> argparse.ArgumentParser: ap.add_argument("--lang", action="append", choices=TARGET_LOCALES, default=None) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--workers", type=int, default=1) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true") ap.add_argument("--batch", action="store_true") roundtrip.add_arguments(ap) @@ -287,8 +298,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -332,6 +345,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -340,7 +355,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: print(f"[02e/gb] manual provider: {len(jobs)} entries need translation.", diff --git a/scripts/gr/02d_extract_terroir_facts.py b/scripts/gr/02d_extract_terroir_facts.py index 936cff6..194d848 100644 --- a/scripts/gr/02d_extract_terroir_facts.py +++ b/scripts/gr/02d_extract_terroir_facts.py @@ -31,7 +31,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -40,6 +39,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "gr" / "dokumenti-extracted" WIKI_AOCS = ROOT / "raw" / "wikipedia" / "aocs" / "el" @@ -134,36 +141,22 @@ - Τα αποσπάσματα είναι ΑΥΤΟΛΕΞΕΙ (αντιγραμμένα) από την αντίστοιχη πηγή. ΠΟΤΕ μην αποδώσεις σε μια πηγή κείμενο που δεν εμφανίζεται εκεί. - Χωρίς αξιολογικές κρίσεις ("εξαίσιος", "πρεστιζιακός" …). - Χωρίς εξωτερικά συμπεράσματα. Χωρίς αριθμούς που δεν εμφανίζονται σε καμία από τις δύο πηγές. -- Μέγιστο {max_bullets} στοιχεία, καθένα ≤ 140 χαρακτήρες. +- Μέγιστο {max_bullets} στοιχεία· κάθε στοιχείο είναι μία πλήρης πρόταση περίπου 120–220 χαρακτήρων — ποτέ τηλεγραφικό απόσπασμα. - Αν ούτε το ενιαίο έγγραφο ούτε η Wikipedia περιέχουν συγκεκριμένο γεγονός άξιο προσοχής για αυτή την υποενότητα, επέστρεψε κενή λίστα. Απάντησε ΜΟΝΟ σε JSON, χωρίς κείμενο πριν ή μετά: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Για ελλείπον απόσπασμα χρησιμοποίησε κενό αλφαριθμητικό "".""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -283,11 +276,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Υποενότητα προς επεξεργασία: {label}\n\n" - f"Κείμενο του ενιαίου εγγράφου (Περιγραφή του δεσμού):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Υποενότητα προς επεξεργασία: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Κείμενο του ενιαίου εγγράφου (Περιγραφή του δεσμού):\n\n{lien_text}" def _process_subsection( @@ -301,9 +295,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -338,6 +336,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "el") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "gr", "source_lang": "el", @@ -345,6 +349,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -355,7 +361,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -418,12 +424,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(απόσπασμα Wikipedia μη διαθέσιμο)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -464,7 +470,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts(slug_items: list[dict], lien: str) -> list[dict]: @@ -579,7 +585,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -612,7 +618,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-gr.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -662,7 +668,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/gr/02e_translate_terroir_facts.py b/scripts/gr/02e_translate_terroir_facts.py index ffb592e..61028f8 100644 --- a/scripts/gr/02e_translate_terroir_facts.py +++ b/scripts/gr/02e_translate_terroir_facts.py @@ -27,6 +27,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -39,13 +42,15 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Greek proper nouns verbatim: appellation names ("Σαντορίνη", "Νεμέα", "Νάουσα", "Μαντινεία", "Ραψάνη", "Πάτρα", "Μαυροδάφνη Πατρών", "Μοσχάτο Πατρών", "Μοσχάτος Ρίου Πάτρας", "Μονεμβασιά-Malvasia", "Ζίτσα", "Αμύνταιο", "Γουμένισσα", "Μεσενικόλα", "Πλαγιές Μελίτωνα", "Σάμος", "Ρόδος", "Λήμνος", "Πάρος", "Σητεία", "Δαφνές", "Πεζά", "Αρχάνες", "Χάνδακας - Candia", "Ρομπόλα Κεφαλληνίας", "Αγχίαλος", "Μοσχάτος Λήμνου", "Μοσχάτος Ρόδου", "Μαυροδάφνη Κεφαλληνίας", "Μοσχάτος Κεφαλληνίας", "Malvasia Πάρος", "Malvasia Σητείας", "Malvasia Χάνδακας-Candia"), grape variety names ("Ασύρτικο", "Ξινόμαυρο", "Αγιωργίτικο", "Μοσχοφίλερο", "Ροδίτης", "Ρομπόλα", "Λημνιό", "Λημνιώνα", "Μαυροδάφνη", "Μαλαγουζιά", "Σαββατιανό", "Βιδιανό", "Βιλάνα", "Λιάτικο", "Κοτσιφάλι", "Μανδηλαριά", "Αθήρι", "Αηδάνι", "Θραψαθήρι", "Ντεμπίνα", "Νεγκόσκα", "Σταυρωτό", "Κρασάτο", "Μπατίκι", "Vinsanto", "Νυχτέρι"), named geographical features ("Μακεδονία", "Θράκη", "Θεσσαλία", "Ήπειρος", "Πελοπόννησος", "Στερεά Ελλάδα", "Κρήτη", "Ιόνια Νησιά", "Νησιά Αιγαίου", "Όλυμπος", "Καλντέρα", "Παρνασσός", "Όρος Μέλιτων", "Πάικο", "Βέρμιο", "Πεντελικό", "Πάρνηθα"), named soil types ("ασπρα γη", "ηφαιστειακά εδάφη", "αλλουβιακά", "ασβεστόλιθος", "σχιστόλιθος", "μάργες", "πυριγενή πετρώματα", "μαυρόχωμα", "αμμώδες", "χαλικώδες", "αργιλώδες", "αμμοχαλικώδες"), named climatic features ("μεσογειακό κλίμα", "ηπειρωτικό κλίμα", "μελτέμια", "ετήσιοι άνεμοι", "αλίπνοες αύρες", "ορεινό μικροκλίμα"), and Greek wine-law terms ("προστατευόμενη ονομασία προέλευσης", "προστατευόμενη γεωγραφική ένδειξη", "ΠΟΠ", "ΠΓΕ", "οινολογικές πρακτικές", "αμπελουργική ζώνη", "ενιαίο έγγραφο", "προδιαγραφές προϊόντος", "αμπελοοινικός χάρτης"). -- Wine-style traditional terms (preserve verbatim): "Vinsanto", "Νυχτέρι", "Λιαστός οίνος", "Vin doux naturel" / "οίνος γλυκός φυσικός", "όψιμη συγκομιδή", "αφρώδης οίνος". +- Registered Greek wine terms stay, in their Latin form: "Vinsanto", "Nychteri" (for Νυχτέρι), "vin doux naturel" (for οίνος γλυκός φυσικός). Generic style categories (λιαστός οίνος, όψιμη συγκομιδή, αφρώδης οίνος) are common nouns and are translated. - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Greek form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "el" +PROPER_NOUNS = """appellation names, written in their EU-official Latin form (Santorini, Nemea, Naoussa, Mantinia, Rapsani, Patra, Mavrodafni Patron, Moschato Patron, Moschatos Riou Patras, Monemvasia-Malvasia, Zitsa, Amynteo, Goumenissa, Mesenikola, Playies Melitona, Samos, Rodos, Limnos, Paros, Sitia, Dafnes, Peza, Arhanes, Handakas-Candia, Robola Kefallinias, Anchialos, Moschatos Limnou, Moschato Rodou, Mavrodafni Kefallinias, Moschato Kefallinias, Malvasia Paros, Malvasia Sitia, Malvasia Handakas-Candia, Tyrnavos); grape names in their Latin form (Assyrtiko, Xinomavro, Agiorgitiko, Moschofilero, Roditis, Robola, Limnio, Limniona, Mavrodaphne, Malagousia, Savatiano, Vidiano, Vilana, Liatiko, Kotsifali, Mandilaria, Athiri, Aidani, Thrapsathiri, Debina, Negoska, Stavroto, Krasato, Batiki); named regions and mountains, in the established target-language form where one exists and transliterated otherwise (Macedonia, Thrace, Thessaly, Epirus, Peloponnese, Sterea Ellada, Crete, Ionian Islands, Aegean Islands, Olympus, Parnassus, Meliton, Paiko, Vermio, Penteli, Parnitha, the Santorini caldera); registered traditional terms (Vinsanto, Nychteri)""" + def facts_sha(facts: list[dict]) -> str: blob = "\n".join((f.get("bullet") or "") for f in facts) @@ -96,7 +101,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -147,10 +152,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -260,6 +270,7 @@ def _build_argparser() -> argparse.ArgumentParser: ap.add_argument("--lang", action="append", choices=TARGET_LOCALES, default=None) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--workers", type=int, default=1) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true") ap.add_argument("--batch", action="store_true") roundtrip.add_arguments(ap) @@ -343,8 +354,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -376,6 +389,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] @@ -385,7 +400,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: for j in jobs: diff --git a/scripts/hr/02d_extract_terroir_facts.py b/scripts/hr/02d_extract_terroir_facts.py index b7fec90..6a81636 100644 --- a/scripts/hr/02d_extract_terroir_facts.py +++ b/scripts/hr/02d_extract_terroir_facts.py @@ -33,7 +33,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -42,6 +41,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "hr" / "dokumenti-extracted" SPECIFIKACIJE = ROOT / "raw" / "hr" / "specifikacije-extracted" @@ -134,37 +141,22 @@ - Citati su DOSLOVNI (kopirani) iz odgovarajućeg izvora. NIKAD ne pripisuj izvoru tekst koji se tamo ne pojavljuje. - Bez vrijednosnih sudova ("izvanredan", "prestižan" ...). - Bez vanjskih zaključaka. Bez brojki kojih nema ni u jednom od dva izvora. -- Najviše {max_bullets} unosa, svaki ≤ 140 znakova. +- Najviše {max_bullets} unosa; svaki unos je jedna potpuna rečenica od otprilike 120–220 znakova — nikada telegrafski fragment. - Ako ni jedinstveni dokument ni Wikipedija ne sadrže konkretnu vrijednu činjenicu za ovaj pododjeljak, vrati prazan popis. Odgovori SAMO u JSON-u, bez teksta prije ili poslije: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Za citat koji nedostaje koristi prazan niz "".""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - """Longest-contiguous-match coverage of `quote` in `source` (0–1).""" - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -300,11 +292,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Pododjeljak za obradu: {label}\n\n" - f"Tekst jedinstvenog dokumenta (Opis povezanosti):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Pododjeljak za obradu: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Tekst jedinstvenog dokumenta (Opis povezanosti):\n\n{lien_text}" def _process_subsection( @@ -320,9 +313,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -357,6 +354,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "hr") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "hr", "source_lang": "hr", @@ -364,6 +367,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -374,7 +379,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -443,12 +448,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(izvadak Wikipedije nije dostupan)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -489,7 +494,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts(slug_items: list[dict], lien: str) -> list[dict]: @@ -627,7 +632,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -660,7 +665,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-hr.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -710,7 +715,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/hr/02e_translate_terroir_facts.py b/scripts/hr/02e_translate_terroir_facts.py index 0631895..74e270b 100644 --- a/scripts/hr/02e_translate_terroir_facts.py +++ b/scripts/hr/02e_translate_terroir_facts.py @@ -27,6 +27,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -39,12 +42,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Croatian proper nouns verbatim: appellation names ("Hrvatska Istra", "Hrvatsko primorje", "Dingač", "Ponikve", "Plešivica", "Pokuplje", "Moslavina", "Slavonija", "Sjeverna Dalmacija", "Srednja i Južna Dalmacija", "Dalmatinska zagora", "Hrvatsko Podunavlje", "Prigorje-Bilogora", "Zagorje – Međimurje", "Muškat momjanski", "Primorska Hrvatska", "Istočna kontinentalna Hrvatska", "Zapadna kontinentalna Hrvatska"), commune, vinogorje and vineyard-site names ("Pelješac", "Korčula", "Hvar", "Brač", "Vis", "Krk", "Momjan"), grape variety names ("Plavac mali", "Pošip", "Maraština", "Bogdanuša", "Vugava", "Grk", "Babić", "Tribidrag", "Crljenak Kaštelanski", "Graševina", "Malvazija istarska", "Žlahtina", "Debit", "Trbljan", "Muškat bijeli", "Refošk", "Teran"), named geological formations and soil types ("fliš", "lapor", "vapnenac", "pješčenjak", "ilovača", "crljenica", "crvenica", "terra rossa"), named climatic features ("mediteranska klima", "submediteranska klima", "kontinentalna klima", "panonski utjecaj", "bura", "jugo", "maestral"), and Croatian wine-law / Predikat terms ("vinogorje", "vinogradarska regija", "kasna berba", "izborna berba", "izborna berba bobica", "izborna berba prosušenih bobica", "ledeno vino", "prošek", "kvalitetno vino", "vrhunsko vino", "arhivsko vino", "desertno vino", "predikatno vino", "ZOI", "ZOZP", "KZP"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Croatian form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "hr" +PROPER_NOUNS = """appellation names (Hrvatska Istra, Hrvatsko primorje, Dingač, Ponikve, Plešivica, Pokuplje, Moslavina, Slavonija, Sjeverna Dalmacija, Srednja i Južna Dalmacija, Dalmatinska zagora, Hrvatsko Podunavlje, Prigorje-Bilogora, Zagorje – Međimurje, Muškat momjanski, Primorska Hrvatska, Istočna kontinentalna Hrvatska, Zapadna kontinentalna Hrvatska); commune, vinogorje and vineyard-site names (Pelješac, Korčula, Hvar, Brač, Vis, Krk, Momjan); grape names (Plavac mali, Pošip, Maraština, Bogdanuša, Vugava, Grk, Babić, Tribidrag, Crljenak Kaštelanski, Graševina, Malvazija istarska, Žlahtina, Debit, Trbljan, Muškat bijeli, Refošk, Teran); named winds (bura, jugo, maestral); registered traditional terms (Prošek)""" + # ─────────────────────────────────────────────────────────────── helpers ── @@ -102,7 +107,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -157,10 +162,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -291,6 +301,7 @@ def _build_argparser() -> argparse.ArgumentParser: "--workers", type=int, default=1, help="concurrent (lang, slug) pairs (default 1, keep 1 for Ollama)", ) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true", help="re-translate even if cached") ap.add_argument( "--batch", action="store_true", @@ -378,8 +389,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -411,6 +424,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] @@ -420,7 +435,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: for j in jobs: diff --git a/scripts/hu/02d_extract_terroir_facts.py b/scripts/hu/02d_extract_terroir_facts.py index b47c563..5dbe86e 100644 --- a/scripts/hu/02d_extract_terroir_facts.py +++ b/scripts/hu/02d_extract_terroir_facts.py @@ -32,7 +32,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -41,6 +40,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "hu" / "dokumentumok-extracted" NATIONAL_SPECS = ROOT / "raw" / "hu" / "national-specs-extracted" @@ -131,37 +138,22 @@ - Az idézetek SZÓ SZERINTIEK (másolt szöveg) a megfelelő forrásból. SOHA ne tulajdoníts egy forrásnak olyan szöveget, ami ott nem szerepel. - Ne használj értékítéletet ("kiváló", "rangos", …). - Ne vonj le külső következtetéseket. Ne adj meg olyan számokat, amelyek egyik forrásban sem szerepelnek. -- Legfeljebb {max_bullets} tétel, mindegyik ≤ 140 karakter. +- Legfeljebb {max_bullets} tétel; minden tétel egy teljes mondat, körülbelül 120–220 karakter — soha nem távirati stílusú töredék. - Ha sem az egységes dokumentum, sem a Wikipedia nem tartalmaz konkrét, értékes tényt ehhez az alfejezethez, üres listát adj vissza. CSAK JSON-t adj vissza, előtte és utána semmilyen szöveg: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Hiányzó idézethez használj üres karakterláncot "".""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - """Longest-contiguous-match coverage of `quote` in `source` (0–1).""" - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -284,11 +276,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Feldolgozandó alfejezet: {label}\n\n" - f"Az egységes dokumentum szövege (A kapcsolat(ok) leírása):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Feldolgozandó alfejezet: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Az egységes dokumentum szövege (A kapcsolat(ok) leírása):\n\n{lien_text}" def _process_subsection( @@ -302,9 +295,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -339,6 +336,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "hu") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "hu", "source_lang": "hu", @@ -346,6 +349,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -356,7 +361,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -422,12 +427,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(a Wikipedia-kivonat nem elérhető)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -468,7 +473,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts(slug_items: list[dict], lien: str) -> list[dict]: @@ -605,7 +610,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -638,7 +643,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-hu.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -688,7 +693,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/hu/02e_translate_terroir_facts.py b/scripts/hu/02e_translate_terroir_facts.py index b724f01..db86d72 100644 --- a/scripts/hu/02e_translate_terroir_facts.py +++ b/scripts/hu/02e_translate_terroir_facts.py @@ -27,6 +27,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -39,12 +42,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Hungarian proper nouns verbatim: appellation and wine-region names ("Tokaj", "Tokaji", "Eger", "Egri", "Villány", "Villányi", "Szekszárd", "Mátra", "Mátrai", "Bükk", "Bükki", "Sopron", "Soproni", "Pannonhalma", "Pannonhalmi", "Etyek-Buda", "Mór", "Móri", "Neszmély", "Nagy-Somló", "Somlói", "Badacsony", "Badacsonyi", "Balaton-felvidék", "Balatonfüred-Csopak", "Csopak", "Káli", "Tihany", "Füred", "Zala", "Pécs", "Pécsi", "Tolna", "Tolnai", "Kunság", "Hajós-Baja", "Csongrád", "Duna", "Debrői Hárslevelű", "Izsáki Arany Sárfehér", "Monor", "Soltvadkerti", "Etyeki Pezsgő", "Kőszeg", "Felső-Magyarország", "Felső-Pannon", "Duna-Tisza-közi", "Dunántúli", "Zemplén", "Balatonmelléki", "Pannon"), commune, dűlő and vineyard-site names ("Tokaj-Hegyalja", "Aszú", "Bikavér", "Egri Bikavér", "Egri Csillag", "Mád", "Tarcal", "Tállya", "Sárospatak", "Hercegkút", "Olaszliszka", "Eger", "Szentvér-dűlő", "Szépasszony-völgy"), grape variety names ("Furmint", "Hárslevelű", "Olaszrizling", "Kékfrankos", "Kadarka", "Kékoportó", "Cserszegi Fűszeres", "Irsai Olivér", "Királyleányka", "Leányka", "Juhfark", "Ezerjó", "Tramini", "Szürkebarát", "Csókaszőlő", "Kövérszőlő", "Olasz Rizling", "Sárga Muskotály", "Ottonel Muskotály", "Kékfrankos", "Pinot Noir", "Bíborkadarka"), named geological formations and soil types ("lösz", "nyirok", "riolittufa", "andezittufa", "andezit", "riolit", "bazalt", "mészkő", "dolomit", "agyagpala", "homokkő", "csernozjom", "barna erdőtalaj", "vulkáni talaj", "fekete talaj", "agyag", "sziklatalaj"), named climatic features ("pannon klíma", "kontinentális klíma", "mediterrán hatás", "atlanti hatás", "dunántúli klíma", "balatoni mikroklíma", "tokaji köd", "botrytis cinerea", "nemes rothadás", "északi szél"), and Hungarian wine-law / quality terms ("borvidék", "borrégió", "dűlő", "OEM", "OFJ", "Eredetvédett", "Tokaji Aszú", "Tokaji Szamorodni", "Tokaji Eszencia", "Tokaji Fordítás", "Tokaji Máslás", "Bikavér", "Csillag", "Cuvée", "Siller", "Pezsgő", "Gyöngyözőbor", "Klárét", "Késői Szüretelésű", "Jégbor", "Töppedt", "Likőrbor", "Válogatott Szüretelésű", "Edes", "Félédes", "Félszáraz", "Száraz"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Hungarian form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "hu" +PROPER_NOUNS = """appellation and wine-region names (Tokaj, Tokaji, Eger, Egri, Villány, Villányi, Szekszárd, Mátra, Mátrai, Bükk, Bükki, Sopron, Soproni, Pannonhalma, Pannonhalmi, Etyek-Buda, Mór, Móri, Neszmély, Nagy-Somló, Somlói, Badacsony, Badacsonyi, Balaton-felvidék, Balatonfüred-Csopak, Csopak, Káli, Tihany, Füred, Zala, Pécs, Pécsi, Tolna, Tolnai, Kunság, Hajós-Baja, Csongrád, Duna, Debrői Hárslevelű, Izsáki Arany Sárfehér, Monor, Soltvadkerti, Etyeki Pezsgő, Kőszeg, Felső-Magyarország, Felső-Pannon, Duna-Tisza-közi, Dunántúli, Zemplén, Balatonmelléki, Pannon); commune and vineyard-site names (Tokaj-Hegyalja, Mád, Tarcal, Tállya, Sárospatak, Hercegkút, Olaszliszka, Szentvér-dűlő, Szépasszony-völgy); grape names (Furmint, Hárslevelű, Olaszrizling, Kékfrankos, Kadarka, Kékoportó, Cserszegi Fűszeres, Irsai Olivér, Királyleányka, Leányka, Juhfark, Ezerjó, Tramini, Szürkebarát, Csókaszőlő, Kövérszőlő, Sárga Muskotály, Ottonel Muskotály, Pinot Noir, Bíborkadarka); registered traditional terms (Aszú, Tokaji Aszú, Szamorodni, Tokaji Szamorodni, Eszencia, Fordítás, Máslás, Bikavér, Egri Bikavér, Csillag, Egri Csillag, Siller, Klárét)""" + # ─────────────────────────────────────────────────────────────── helpers ── @@ -101,7 +106,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -155,10 +160,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -289,6 +299,7 @@ def _build_argparser() -> argparse.ArgumentParser: "--workers", type=int, default=1, help="concurrent (lang, slug) pairs (default 1, keep 1 for Ollama)", ) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true", help="re-translate even if cached") ap.add_argument( "--batch", action="store_true", @@ -376,8 +387,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -409,6 +422,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] @@ -418,7 +433,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: for j in jobs: diff --git a/scripts/it/00_fetch_data.py b/scripts/it/00_fetch_data.py index 25efe5d..61df774 100644 --- a/scripts/it/00_fetch_data.py +++ b/scripts/it/00_fetch_data.py @@ -18,6 +18,12 @@ the spine for stage 02f-MASAF: ~98 % of eAmbrosia IT wines (521 of 531) match a PDF inside one of these archives. +2b. **MASAF "Elenco alfabetico dei vini DOP / IGP italiani"** — two + small PDFs on the same IDPagina/4625 listing every Italian wine GI + with its traditional term (DOC / DOCG / IGT), eAmbrosia file number + and region. The DOP roster is the source of each DOP's DOC-vs-DOCG + term (scripts/_lib/it/national_term.py). + 3. **Figshare EU_PDO.gpkg** (Bétard et al. 2022, CC0, ~42 MB) — *already cached* by the ES pipeline at `raw/es/figshare/EU_PDO.gpkg`. Covers all EU PDOs including ~420 Italian DOPs (`PDO-IT-A*` file numbers). @@ -34,6 +40,7 @@ - raw/it/eambrosia/manifest.json — fetch metadata for the eAmbrosia call - raw/it/masaf-disciplinari/bundles/*.7z — 4 disciplinari archives - raw/it/masaf-disciplinari/bundles/manifest.json — per-bundle sha256 + URL +- raw/it/masaf-elenchi/elenco-{dop,igp}.pdf + manifest.json — DOC/DOCG/IGT rosters Each IT wine GI record carries: - giIdentifier (e.g. EUGI00000003500) — internal EU id, unstable @@ -182,6 +189,41 @@ "foreste (MASAF). Re-distribution permitted with attribution." ) +# "Elenco alfabetico dei vini DOP / IGP italiani" — the ministry's roster +# of every Italian wine GI with its traditional term (DOC / DOCG / IGT) +# and eAmbrosia file number. Linked from the same IDPagina/4625 as the +# bundles; the page is scraped for the current attachment URL (the BLOB +# hash rotates on every republication) with the last-known direct file +# URL as fallback. Consumed by scripts/_lib/it/national_term.py. +MASAF_ELENCHI_DIR = ROOT / "raw" / "it" / "masaf-elenchi" +MASAF_ELENCHI_MANIFEST = MASAF_ELENCHI_DIR / "manifest.json" +MASAF_ELENCHI_PAGE = "https://www.masaf.gov.it/flex/cm/pages/ServeBLOB.php/L/IT/IDPagina/4625" +MASAF_ELENCHI_LICENSE = "MASAF official act, public" +MASAF_ELENCHI = ( + { + "key": "dop", + "label": "Elenco alfabetico Vini DOP", + "filename": "elenco-dop.pdf", + "fallback_url": ( + "https://www.masaf.gov.it/flex/files/e/a/c/D.3b34b403c9bb8667ee4b/" + "Elenco_alfabetico_Vini_DOP_italiani.pdf" + ), + }, + { + "key": "igp", + "label": "Elenco alfabetico Vini IGP", + "filename": "elenco-igp.pdf", + "fallback_url": ( + "https://www.masaf.gov.it/flex/files/7/e/3/D.6effd5b02f25c6179834/" + "Elenco_alfabetico_Vini_IGP_italiani.pdf" + ), + }, +) +_ELENCO_LINK_RE = re.compile( + r"<a\s+title=['\"]Elenco alfabetico Vini (DOP|IGP)[^'\"]*['\"]\s+href=['\"]([^'\"]+)['\"]", + re.I, +) + # Shared upstream artifacts already fetched by the ES pipeline. Asserted # here so a fresh IT-only checkout fails loudly rather than silently # producing geometry-less records downstream. @@ -338,6 +380,99 @@ def fetch_masaf_bundles() -> dict: return manifest +def scrape_masaf_elenco_links() -> dict[str, str]: + """{key: attachment URL} for the elenco links on IDPagina/4625; {} when + the page is unreachable or has no such link (caller falls back).""" + try: + r = requests.get(MASAF_ELENCHI_PAGE, headers={"User-Agent": UA}, timeout=60) + r.raise_for_status() + except requests.RequestException as e: + print(f"[elenchi] WARNING page scrape failed: {e}", file=sys.stderr) + return {} + # FlexCMP serves ISO-8859-1; latin-1 decodes any byte, and the anchors are ASCII. + return { + kind.lower(): url.replace("&", "&") + for kind, url in _ELENCO_LINK_RE.findall(r.content.decode("latin-1")) + } + + +def fetch_masaf_elenchi() -> dict: + """Download the MASAF DOP + IGP elenchi (~750 KB total). Always + re-fetched — the URL is stable across republications only via the + page scrape — but a byte-identical PDF keeps its cached `fetched_at` + so a no-change re-run does not churn the manifest.""" + MASAF_ELENCHI_DIR.mkdir(parents=True, exist_ok=True) + existing: dict[str, dict] = {} + if MASAF_ELENCHI_MANIFEST.exists(): + try: + existing = json.loads( + MASAF_ELENCHI_MANIFEST.read_text(encoding="utf-8") + ).get("elenchi", {}) + except (ValueError, OSError): + existing = {} + + links = scrape_masaf_elenco_links() + by_key: dict[str, dict] = {} + for spec in MASAF_ELENCHI: + key = spec["key"] + url = links.get(key) + link_source = "page-scrape" + if not url: + url = spec["fallback_url"] + link_source = "known-url-fallback" + print(f"[elenchi] WARNING no {key.upper()} link on IDPagina/4625; " + f"falling back to {url}", file=sys.stderr) + print(f"[elenchi] fetch {url}", file=sys.stderr) + resp = requests.get(url, headers={"User-Agent": UA}, timeout=120, allow_redirects=True) + resp.raise_for_status() + body = resp.content + if not body.startswith(b"%PDF"): + raise RuntimeError( + f"MASAF returned non-PDF content for elenco {key} " + f"({len(body)} bytes, ct={resp.headers.get('content-type')}). " + "URL may have rotated; re-scrape IDPagina/4625." + ) + sha = hashlib.sha256(body).hexdigest() + served = re.search(r'filename="?([^";]+)', resp.headers.get("content-disposition") or "") + dest = MASAF_ELENCHI_DIR / spec["filename"] + cached = existing.get(key) or {} + from_cache = dest.exists() and cached.get("sha256") == sha + if not from_cache: + dest.write_bytes(body) + by_key[key] = { + "label": spec["label"], + "filename": spec["filename"], + "served_filename": served.group(1).strip() if served else "", + "url": url, + "link_source": link_source, + "sha256": sha, + "bytes": len(body), + "fetched_at": ( + cached["fetched_at"] if from_cache and cached.get("fetched_at") + else datetime.now(timezone.utc).isoformat(timespec="seconds") + ), + "from_cache": from_cache, + } + print( + f"[elenchi] {'cache hit ' if from_cache else 'saved '}" + f"{spec['filename']:16s} ({len(body):>8,} bytes, sha256={sha[:12]}…, " + f"{by_key[key]['served_filename'] or '-'})", + file=sys.stderr, + ) + + manifest = { + "generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "source_page": MASAF_ELENCHI_PAGE, + "licence": MASAF_ELENCHI_LICENSE, + "elenchi": by_key, + } + MASAF_ELENCHI_MANIFEST.write_text( + json.dumps(manifest, ensure_ascii=False, indent=2, sort_keys=True), + encoding="utf-8", + ) + return manifest + + def assert_shared_artifacts() -> None: missing = [] if not FIGSHARE_GPKG_PATH.exists(): @@ -581,6 +716,13 @@ def main() -> int: file=sys.stderr, ) + elenchi = fetch_masaf_elenchi() + print( + f"[done] MASAF elenchi: {len(elenchi['elenchi'])} rosters " + f"→ {MASAF_ELENCHI_DIR.relative_to(ROOT)}", + file=sys.stderr, + ) + fetch_istat_comuni() print( f"[done] ISTAT comuni registry → {ISTAT_COMUNI_DIR.relative_to(ROOT)}", diff --git a/scripts/it/02d_extract_terroir_facts.py b/scripts/it/02d_extract_terroir_facts.py index 42d2409..063da2c 100644 --- a/scripts/it/02d_extract_terroir_facts.py +++ b/scripts/it/02d_extract_terroir_facts.py @@ -36,7 +36,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -45,6 +44,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "it" / "disciplinari-extracted" MASAF_EXTRACTED = ROOT / "raw" / "it" / "masaf-disciplinari-extracted" @@ -143,37 +150,22 @@ - Le citazioni sono VERBATIM (copiate e incollate) dalla rispettiva fonte. MAI attribuire a una fonte un testo che non vi compare. - Niente giudizi di valore ("eccezionale", "straordinario", "prestigioso"...). - Niente inferenze esterne. Niente numeri assenti dalle due fonti. -- Massimo {max_bullets} voci, ≤ 140 caratteri ciascuna. +- Massimo {max_bullets} voci; ogni voce è una frase completa di circa 120–220 caratteri — mai un frammento telegrafico. - Se né il disciplinare né Wikipedia contengono un fatto notevole concreto per questa sotto-sezione, restituisci una lista vuota. Rispondi SOLO in JSON, senza testo prima o dopo: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Usa una stringa vuota "" per la citazione assente.""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - """Longest-contiguous-match coverage of `quote` in `source` (0–1).""" - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -274,7 +266,9 @@ def _resolve_lien_and_source(rec: dict) -> tuple[str, dict]: sidecar = json.loads(sidecar_path.read_text(encoding="utf-8")) except (ValueError, OSError): return on_disk_lien, {"pdf_url": eu_url, "kind": "none"} - masaf_lien = sidecar.get("link_to_terroir") or "" + # The whole Art. 9 (v3 sidecars); the panel-length `link_to_terroir` + # is a 4,000-char cut that hid 3.7 M characters of terroir text. + masaf_lien = sidecar.get("link_to_terroir_full") or sidecar.get("link_to_terroir") or "" msrc = sidecar.get("source") or {} # MASAF bundle matches have no direct URL; overrides do (`url` field). masaf_url = msrc.get("url") or eu_url @@ -305,11 +299,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Sotto-sezione da trattare: {label}\n\n" - f"Testo del disciplinare (Legame con la zona geografica):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Sotto-sezione da trattare: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Testo del disciplinare (Legame con la zona geografica):\n\n{lien_text}" def _process_subsection( @@ -325,9 +320,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -362,6 +361,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "it") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "it", "source_lang": "it", @@ -369,6 +374,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -379,7 +386,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -449,12 +456,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(nessun estratto Wikipedia disponibile)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -495,7 +502,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts( @@ -639,7 +646,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -674,7 +681,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-it.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -723,7 +730,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/it/02e_translate_terroir_facts.py b/scripts/it/02e_translate_terroir_facts.py index accddbb..239dab5 100644 --- a/scripts/it/02e_translate_terroir_facts.py +++ b/scripts/it/02e_translate_terroir_facts.py @@ -36,6 +36,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -48,12 +51,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Italian proper nouns verbatim: appellation names ("Barolo", "Brunello di Montalcino", "Chianti Classico", "Amarone della Valpolicella", "Soave", "Prosecco", "Franciacorta", "Bolgheri", "Etna", "Taurasi"), region names ("Piemonte", "Toscana", "Veneto", "Trentino-Alto Adige", "Friuli-Venezia Giulia", "Emilia-Romagna", "Marche", "Sicilia", "Sardegna", "Puglia"), commune and locality names, grape variety names ("Nebbiolo", "Sangiovese", "Barbera", "Dolcetto", "Corvina", "Glera", "Garganega", "Aglianico", "Nero d'Avola", "Vermentino", "Pignoletto", "Lagrein", "Lambrusco", "Greco di Tufo", "Fiano", "Trebbiano", "Verdicchio", "Cesanese", "Primitivo", "Negroamaro", "Cannonau", "Carricante"), named geological formations and soil types ("tufo", "galestro", "alberese", "calcare", "arenaria", "scisti", "marne", "argille", "morene", "porfido"), named winds and local climatic features ("bora", "scirocco", "ora del Garda", "tramontana", "garbino"), training systems and local vinification terms ("pergola", "tendone", "alberello", "guyot", "cordone speronato", "appassimento", "ripasso", "governo", "metodo classico", "metodo Martinotti", "vendemmia tardiva"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Italian form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "it" +PROPER_NOUNS = """appellation names (Barolo, Brunello di Montalcino, Chianti Classico, Amarone della Valpolicella, Soave, Prosecco, Franciacorta, Bolgheri, Etna, Taurasi); region names (Piemonte, Toscana, Veneto, Trentino-Alto Adige, Friuli-Venezia Giulia, Emilia-Romagna, Marche, Sicilia, Sardegna, Puglia); commune and locality names; grape names (Nebbiolo, Sangiovese, Barbera, Dolcetto, Corvina, Glera, Garganega, Aglianico, Nero d'Avola, Vermentino, Pignoletto, Lagrein, Lambrusco, Greco di Tufo, Fiano, Trebbiano, Verdicchio, Cesanese, Primitivo, Negroamaro, Cannonau, Carricante); named geological formations (galestro, alberese); named winds (bora, scirocco, ora del Garda, tramontana, garbino); named techniques and registered terms (appassimento, ripasso, governo all'uso toscano, metodo classico, metodo Martinotti)""" + # ─────────────────────────────────────────────────────────────── helpers ── @@ -113,7 +118,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -171,10 +176,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -311,6 +321,7 @@ def _build_argparser() -> argparse.ArgumentParser: "--workers", type=int, default=1, help="concurrent (lang, slug) pairs to translate (default 1, keep 1 for Ollama on M1 32GB)", ) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true", help="re-translate even if cached") ap.add_argument( "--batch", action="store_true", @@ -349,7 +360,7 @@ def _dispatch_emit_or_import(args, languages: tuple[str, ...]) -> int | None: def _make_provider(args) -> tuple[object | None, str]: return providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) @@ -418,8 +429,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -451,6 +464,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] diff --git a/scripts/it/02f_extract_masaf.py b/scripts/it/02f_extract_masaf.py index 96d3982..e290fab 100644 --- a/scripts/it/02f_extract_masaf.py +++ b/scripts/it/02f_extract_masaf.py @@ -66,9 +66,10 @@ from _lib.it.masaf import ( # noqa: E402 PdfRecord, build_pdf_index, + cap_at_sentence, derive_geo_area, derive_summary, - extract_articles, + extract_article_runs, match_wines_to_pdfs, parse_annex_grapes_with, parse_grapes_with, @@ -91,7 +92,9 @@ OVERRIDES_PATH = BUNDLES_DIR.parent / "manual_overrides.json" -PARSER_VERSION = "it-masaf-disciplinare-v1" +PARSER_VERSION = "it-masaf-disciplinare-v3" +# Panel length of the Art. 9 terroir text; the extractor reads the full body. +TERROIR_BRIEF_CHARS = 4000 def load_overrides() -> dict: @@ -235,7 +238,8 @@ def collapse_whitespace(s: str) -> str: def build_record(wine: dict, articles: dict[int, str], pdf_meta: dict, - match_info: dict, comune_map: dict, raw_text: str = "") -> dict: + match_info: dict, comune_map: dict, raw_text: str = "", + annexes: list[dict] | None = None) -> dict: """Build the sidecar JSON. The shape mirrors the doc-unico extracted record where it can (slug / name / kind / file_number / id_eambrosia / regione / grapes / styles / sections_present) so stage 04 can @@ -257,7 +261,12 @@ def build_record(wine: dict, articles: dict[int, str], pdf_meta: dict, summary = derive_summary(articles.get(1, "")) geo_area = derive_geo_area(articles.get(3, "")) - terroir_article_num, terroir = pick_terroir_article(articles, raw_text=raw_text) + # The panel reads the sentence-capped `link_to_terroir`; 02d, the gate + # and the audits read `link_to_terroir_full` — the whole article. + terroir_article_num, terroir_full = pick_terroir_article( + articles, raw_text=raw_text, max_chars=None + ) + terroir = cap_at_sentence(terroir_full, TERROIR_BRIEF_CHARS) # Wine-style tags: scan the denominazione/tipologie block (art 1) + the # organoleptic "Caratteristiche al consumo" (art 6). Those are colour- and @@ -296,6 +305,8 @@ def build_record(wine: dict, articles: dict[int, str], pdf_meta: dict, "menzioni": menzioni, "geo_area_brief": geo_area, "link_to_terroir": terroir, + "link_to_terroir_full": terroir_full, + "terroir_article": terroir_article_num, "articles_present": sorted(articles.keys()), # Subsection of articles useful to downstream consumers — we # keep articles 1 / 3 / 9 verbatim so 02d-style terroir @@ -305,6 +316,21 @@ def build_record(wine: dict, articles: dict[int, str], pdf_meta: dict, for n in sorted({1, 2, 3, 9, terroir_article_num}) if articles.get(n) }, + # Per-sottozona sub-disciplinari appended to the parent's PDF + # (ALLEGATO N — SOTTOZONA «…»), each with its own Art. 1 / 3 / 9: + # kept as chapters so a sottozona can be grounded on its own text + # rather than on the parent's (the Alsace `terroir_chapters` idea). + # Never merged into the parent's fields above. + "annexes": [ + { + "title": a.get("title") or "", + "article_bodies": { + str(n): body for n, body in sorted((a.get("articles") or {}).items()) + if n in (1, 2, 3, 8, 9) and body + }, + } + for a in (annexes or []) + ], "source": pdf_meta, "match": match_info, } @@ -427,7 +453,7 @@ def process_slug( return {"slug": slug, "status": "pdftotext-failed", "reason": e.stderr[:160] if e.stderr else str(e)[:160]} - articles = extract_articles(text) + articles, annexes = extract_article_runs(text) if not articles: if strict: raise SystemExit( @@ -436,7 +462,8 @@ def process_slug( ) return {"slug": slug, "status": "no-articles", "reason": "no-anchors"} - record = build_record(wine, articles, pdf_meta, match_info, comune_map, raw_text=text) + record = build_record(wine, articles, pdf_meta, match_info, comune_map, raw_text=text, + annexes=annexes) OUT_DIR.mkdir(parents=True, exist_ok=True) out_path = OUT_DIR / f"{slug}.json" out_path.write_text(json.dumps(record, ensure_ascii=False, indent=2), encoding="utf-8") diff --git a/scripts/lu/02d_extract_terroir_facts.py b/scripts/lu/02d_extract_terroir_facts.py index d296f50..7112a6a 100644 --- a/scripts/lu/02d_extract_terroir_facts.py +++ b/scripts/lu/02d_extract_terroir_facts.py @@ -32,7 +32,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -41,6 +40,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "lu" / "cahier-extracted" WIKI_AOCS = ROOT / "raw" / "wikipedia" / "aocs" / "fr" @@ -136,36 +143,22 @@ - Citations VERBATIM (copiées) de la source concernée. NE JAMAIS attribuer à une source un texte qui ne s'y trouve pas. - Pas de jugements de valeur ("excellent", "prestigieux"…). - Pas de conclusions externes. Pas de chiffres qui ne figurent dans aucune source. -- Maximum {max_bullets} entrées, chacune ≤ 140 caractères. +- Maximum {max_bullets} entrées ; chaque entrée est une phrase complète d'environ 120 à 220 caractères — jamais un fragment télégraphique. - Si ni le cahier ni Wikipédia ne contient de fait important pour cette sous-section, retourner une liste vide. Répondre UNIQUEMENT en JSON, sans texte avant ou après : {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Pour une citation manquante, utiliser la chaîne vide "".""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -272,11 +265,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Sous-section visée : {label}\n\n" - f"Texte du cahier des charges (Lien avec l'aire géographique) :\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Sous-section visée : {label}" + + +def _document_block(lien_text: str) -> str: + return f"Texte du cahier des charges (Lien avec l'aire géographique) :\n\n{lien_text}" def _process_subsection( @@ -290,9 +284,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -327,6 +325,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "fr") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "lu", "source_lang": "fr", @@ -334,6 +338,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -344,7 +350,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -410,12 +416,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(extrait Wikipédia indisponible)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -456,7 +462,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts(slug_items: list[dict], lien: str) -> list[dict]: @@ -575,7 +581,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -608,7 +614,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-lu.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -656,7 +662,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/lu/02e_translate_terroir_facts.py b/scripts/lu/02e_translate_terroir_facts.py index ad87496..4adc27a 100644 --- a/scripts/lu/02e_translate_terroir_facts.py +++ b/scripts/lu/02e_translate_terroir_facts.py @@ -25,6 +25,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -37,12 +40,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve French proper nouns verbatim: appellation name ("AOP Moselle Luxembourgeoise", "Moselle luxembourgeoise"), regulatory bodies and names ("Institut Viti-Vinicole", "IVV", "Marque Nationale", "Office national des appellations d'origine protégées", "ONAOP"), commune names ("Schengen", "Wormeldange", "Remich", "Grevenmacher", "Stadtbredimus", "Lenningen", "Mertert", "Bous-Waldbredimus", "Rosport - Mompach", "Flaxweiler", "Mondorf-les-Bains", and the historic communes "Burmerange", "Wellenstein", "Mompach", "Waldbredimus", "Ahn", "Ehnen", "Machtum", "Greiveldange", "Hettermillen", "Schwebsange", "Niederdonven", "Oberdonven"), the Moselle (river), canton names ("Canton de Remich", "Canton de Grevenmacher"), grape variety names ("Elbling", "Rivaner", "Sylvaner", "Auxerrois", "Pinot blanc", "Chardonnay", "Pinot gris", "Riesling", "Gewürztraminer", "Muscat-Ottonel", "Pinot noir", "Pinot noir précoce", "Saint Laurent", "Gamay"), named geological formations and soil types ("Trias", "marnes keupériennes", "calcaire conchylien", "gypse"), the Luxembourg wine mentions ("Crémant de Luxembourg", "Vendanges Tardives", "Vin de Glace", "Vin de Paille", "Premier cru", "Grand premier cru", "Vin classé"), private quality charters ("Domaine et Tradition", "Charta Privatwënzer", "Charta Schengen Prestige"), and Luxembourg wine-law terms ("AOP", "cahier des charges", "périmètre viticole", "Règlement grand-ducal"). - Geological era labels (Trias, Keuper): translate to the standard {lang_name} form when one exists. When unsure, keep the French form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "fr" +PROPER_NOUNS = """appellation name (AOP Moselle Luxembourgeoise, Moselle luxembourgeoise); institutions (Institut Viti-Vinicole, IVV, Marque Nationale, Office national des appellations d'origine protégées, ONAOP); commune names (Schengen, Wormeldange, Remich, Grevenmacher, Stadtbredimus, Lenningen, Mertert, Bous-Waldbredimus, Rosport - Mompach, Flaxweiler, Mondorf-les-Bains, and the historic communes Burmerange, Wellenstein, Mompach, Waldbredimus, Ahn, Ehnen, Machtum, Greiveldange, Hettermillen, Schwebsange, Niederdonven, Oberdonven); the Moselle river and canton names (Canton de Remich, Canton de Grevenmacher); grape names (Elbling, Rivaner, Sylvaner, Auxerrois, Pinot blanc, Chardonnay, Pinot gris, Riesling, Gewürztraminer, Muscat-Ottonel, Pinot noir, Pinot noir précoce, Saint Laurent, Gamay); registered Luxembourg mentions (Crémant de Luxembourg, Vendanges Tardives, Vin de Glace, Vin de Paille, Premier cru, Grand premier cru, Vin classé); private quality charters (Domaine et Tradition, Charta Privatwënzer, Charta Schengen Prestige)""" + # ─────────────────────────────────────────────────────────────── helpers ── @@ -99,7 +104,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -153,10 +158,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -276,6 +286,7 @@ def _build_argparser() -> argparse.ArgumentParser: ) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--workers", type=int, default=1) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true") ap.add_argument("--batch", action="store_true") roundtrip.add_arguments(ap) @@ -359,8 +370,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -392,6 +405,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -400,7 +415,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: for j in jobs: diff --git a/scripts/mt/02d_extract_terroir_facts.py b/scripts/mt/02d_extract_terroir_facts.py index 1182034..c316eb7 100644 --- a/scripts/mt/02d_extract_terroir_facts.py +++ b/scripts/mt/02d_extract_terroir_facts.py @@ -35,7 +35,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -44,6 +43,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system, split_user_lead # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "mt" / "dokumente-extracted" WIKI_AOCS_ROOT = ROOT / "raw" / "wikipedia" / "aocs" @@ -115,12 +122,13 @@ - Quotes are VERBATIM (copy-paste) from their source. NEVER attribute to a source text that does not appear in it. - No value judgements ("exceptional", "prestigious"…). - No figures absent from both sources. -- At most {max_bullets} bullets, each ≤ 140 characters. +- At most {max_bullets} bullets; each bullet is one full sentence of roughly 120–220 characters — never a telegraphic fragment. - If neither Wikipedia nor the regulator context contains a concrete noteworthy fact for this sub-section, return an empty list. Reply ONLY in JSON, no preamble: {{"facts": [{{"bullet": "…", "cahier_quote": "…", "wiki_quote": "…"}}, ...]}} Use an empty string "" for the missing quote.""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) USER_LEAD = "Sub-section: {label}\n\nRegulator context:\n\n{ctx}" @@ -129,25 +137,10 @@ # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def wiki_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -291,9 +284,13 @@ def _process_subsection(provider, model_id: str, record: dict, sub: dict): wiki_hint=wiki_hint or "(no Wikipedia extract available)", label=label, topics=topic, max_bullets=sub["max_bullets"], ) - user = USER_LEAD.format(label=label, ctx=cahier_ctx) + system = with_feedback(system, record["slug"]) + # The regulator text is the cached leading block: the four sub-section calls share it. + user, doc = split_user_lead(USER_LEAD, label=label, ctx=cahier_ctx) + system = cached_system(doc, system, phased=True) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -321,6 +318,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, SOURCE_LANG) + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "mt", "source_lang": SOURCE_LANG, @@ -328,6 +331,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -338,7 +343,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki.get("page_url"), "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -382,11 +387,11 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection": sub["key"], "subsection_label": label, "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(no Wikipedia extract)", label=label, topics=SUBSECTION_TOPICS[sub["key"]], max_bullets=sub["max_bullets"], - ), + ), rec["slug"]), "cahier_ctx": rec.get("_cahier_ctx") or "", "wiki_hint": wiki_hint, "cahier_source_sha": wiki_sha(rec.get("_cahier_ctx") or ""), @@ -430,7 +435,7 @@ def import_todo(in_path: Path, *, translator_id: str, translator_kind: str) -> i f["subsection"] = it.get("subsection") or "facteurs_naturels" facts.append(f) wiki = rec.get("_wiki_record") or {} - cache.write_json(CACHE_DIR / f"{slug}.json", { + write_source_cache(CACHE_DIR / f"{slug}.json", { "country": "mt", "source_lang": SOURCE_LANG, "slug": slug, "name": rec.get("name") or slug, "facts": facts, "model": translator_id, "model_kind": translator_kind, @@ -485,7 +490,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -518,7 +523,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-mt.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -556,7 +561,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/mt/02e_translate_terroir_facts.py b/scripts/mt/02e_translate_terroir_facts.py index c4906e1..a359066 100644 --- a/scripts/mt/02e_translate_terroir_facts.py +++ b/scripts/mt/02e_translate_terroir_facts.py @@ -26,6 +26,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -39,11 +42,12 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. Array length MUST equal the input list length. -- Preserve Maltese proper nouns verbatim: appellation / island names (Malta, Gozo, Għawdex, Maltese Islands), the indigenous grape varieties Ġellewża (red) and Girgentina (white), other grape names (Chardonnay, Cabernet Sauvignon, Syrah, Vermentino, Moscato, …), geological / soil terms (Globigerina limestone, Coralline limestone, blue clay, terra rossa), and Maltese wine terms (Passito, Imqadded, Riżerva). - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +PROPER_NOUNS = """appellation and island names (Malta, Gozo, Għawdex, Maltese Islands); the indigenous grapes Ġellewża and Girgentina and other grape names (Chardonnay, Cabernet Sauvignon, Syrah, Vermentino, Moscato); named geological formations (Globigerina Limestone, Coralline Limestone, Blue Clay); Maltese wine terms (Passito, Imqadded)""" + def facts_sha(facts: list[dict]) -> str: blob = "\n".join((f.get("bullet") or "") for f in facts) @@ -89,7 +93,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -140,13 +144,19 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT_TEMPLATE.format( - source_lang_name=SOURCE_LANG_NAME.get(job["source_lang"], job["source_lang"]), - lang_name=LOCALE_NAME[job["lang"]], + system = translation_system_prompt( + SYSTEM_PROMPT_TEMPLATE.format( + source_lang_name=SOURCE_LANG_NAME.get(job["source_lang"], job["source_lang"]), + lang_name=LOCALE_NAME[job["lang"]], + ), + source_lang=job["source_lang"], target_lang=job["lang"], + proper_nouns=PROPER_NOUNS, ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -236,6 +246,7 @@ def _build_argparser() -> argparse.ArgumentParser: ap.add_argument("--lang", action="append", choices=TARGET_LOCALES, default=None) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--workers", type=int, default=1) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true") ap.add_argument("--batch", action="store_true") roundtrip.add_arguments(ap) @@ -286,8 +297,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -331,6 +344,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -339,7 +354,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: print(f"[02e/mt] manual provider: {len(jobs)} entries need translation.", diff --git a/scripts/nl/02d_extract_terroir_facts.py b/scripts/nl/02d_extract_terroir_facts.py index 1b0b155..6799f0d 100644 --- a/scripts/nl/02d_extract_terroir_facts.py +++ b/scripts/nl/02d_extract_terroir_facts.py @@ -22,7 +22,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -31,6 +30,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "nl" / "dokumenten-extracted" WIKI_AOCS = ROOT / "raw" / "wikipedia" / "aocs" / "nl" @@ -122,36 +129,22 @@ - Citaten zijn WOORDELIJK (gekopieerd) uit de respectieve bron. Schrijf NOOIT tekst aan een bron toe die daar niet voorkomt. - Geen waardeoordelen ("uitzonderlijk", "prestigieus" …). - Geen externe conclusies. Geen cijfers die in geen van beide bronnen voorkomen. -- Maximaal {max_bullets} items, elk ≤ 140 tekens. +- Maximaal {max_bullets} items; elk item is een volledige zin van ongeveer 120–220 tekens — nooit een fragment in telegramstijl. - Als noch het enig document noch Wikipedia een concreet belangrijk feit voor deze subsectie bevat, retourneer dan een lege lijst. Antwoord ENKEL in JSON, zonder tekst ervoor of erna: {{"facts": [{{"bullet": "…", "cahier_quote": "…", "wiki_quote": "…"}}, ...]}} Gebruik een lege string "" voor een ontbrekend citaat.""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -247,11 +240,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Behandelde subsectie: {label}\n\n" - f"Tekst van het enig document (Beschrijving van het verband):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Behandelde subsectie: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Tekst van het enig document (Beschrijving van het verband):\n\n{lien_text}" def _process_subsection( @@ -265,9 +259,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -302,6 +300,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "nl") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "nl", "source_lang": "nl", @@ -309,6 +313,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -319,7 +325,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -382,12 +388,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(geen Wikipedia-uittreksel beschikbaar)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -443,7 +449,7 @@ def _import_one_slug( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return "wrote" @@ -506,7 +512,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -539,7 +545,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-nl.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -582,7 +588,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/nl/02e_translate_terroir_facts.py b/scripts/nl/02e_translate_terroir_facts.py index 46d51e4..109f644 100644 --- a/scripts/nl/02e_translate_terroir_facts.py +++ b/scripts/nl/02e_translate_terroir_facts.py @@ -25,6 +25,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -37,12 +40,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Dutch proper nouns verbatim: province names ("Limburg", "Gelderland", "Zeeland", "Noord-Brabant", "Zuid-Holland", "Noord-Holland", "Utrecht", "Overijssel", "Flevoland", "Drenthe", "Groningen", "Friesland"), appellation names ("Mergelland", "Vijlen", "Oolde", "Ambt Delden", "Achterhoek - Winterswijk", "Rivierenland", "Schouwen-Duiveland", "De Voerendaalse Bergen", "Twente"), commune and dorp names, grape variety names ("Solaris", "Johanniter", "Regent", "Acolon", "Pinotin", "Cabernet Cortis", "Auxerrois", "Riesling", "Gewürztraminer", "Müller-Thurgau", "Chardonnay", "Pinot blanc/gris/noir", "Souvignier Gris", "Cabernet Cantor", "Dornfelder"), named geological formations and soil types ("krijt", "kalksteen", "löss", "leem", "mergel", "tuffeau", "zandleem", "klei", "rivierklei", "alluviale grond"), named climatic features ("gematigd zeeklimaat", "gematigd maritiem klimaat", "invloed Noordzee"), and Dutch wine-law / EU GI terms ("BOB", "BGA", "enig document", "productdossier", "oorsprongsbenaming", "geografische aanduiding"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Dutch form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "nl" +PROPER_NOUNS = """province names (Limburg, Gelderland, Zeeland, Noord-Brabant, Zuid-Holland, Noord-Holland, Utrecht, Overijssel, Flevoland, Drenthe, Groningen, Friesland); appellation names (Mergelland, Vijlen, Oolde, Ambt Delden, Achterhoek - Winterswijk, Rivierenland, Schouwen-Duiveland, De Voerendaalse Bergen, Twente); commune and village names; grape names (Solaris, Johanniter, Regent, Acolon, Pinotin, Cabernet Cortis, Auxerrois, Riesling, Gewürztraminer, Müller-Thurgau, Chardonnay, Pinot blanc, Pinot gris, Pinot noir, Souvignier Gris, Cabernet Cantor, Dornfelder)""" + # ─────────────────────────────────────────────────────────────── helpers ── @@ -91,7 +96,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -142,10 +147,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -246,6 +256,7 @@ def _build_argparser() -> argparse.ArgumentParser: ap.add_argument("--lang", action="append", choices=TARGET_LOCALES, default=None) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--workers", type=int, default=1) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true") ap.add_argument("--batch", action="store_true") roundtrip.add_arguments(ap) @@ -321,8 +332,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -350,6 +363,8 @@ def main() -> int: if args.batch: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -357,7 +372,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: print(f"[02e/nl] manual provider: {len(jobs)} entries need translation.", diff --git a/scripts/normalize_terroir_facts.py b/scripts/normalize_terroir_facts.py new file mode 100644 index 0000000..32848a4 --- /dev/null +++ b/scripts/normalize_terroir_facts.py @@ -0,0 +1,128 @@ +"""Post-pass over the terroir-fact caches: apply the deterministic bullet +clean-up of `_lib.terroir_normalize` (regulatory colour codes after grape +names, VT / SGN expansion, terminal period) to the source caches AND the +translation caches, in step. No LLM call. + +Stage 04 applies the same normaliser at render time, so this pass is not +needed for the map; it exists so the caches themselves — what the audit +checks and what stage 02e re-translates from — are clean. Rewriting a +source bullet changes `source_facts_sha`; every aligned translation cache +(same hash, same length) is re-keyed to the new hash after its own bullets +are normalised, so nothing is re-translated. A cache already out of step +is left alone and reported. + +Usage: + .venv/bin/python scripts/normalize_terroir_facts.py --dry-run + .venv/bin/python scripts/normalize_terroir_facts.py [--only SLUG …] [--report PATH] +""" + +from __future__ import annotations + +import argparse +import sys +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib import cache # noqa: E402 +from _lib.terroir_cache import ( # noqa: E402 + LANGS, + TERROIR, + TRANSLATIONS, + rekey_translations, + write_source_cache, + write_translation_cache, +) +from _lib.terroir_dedupe import facts_sha # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 + +DEFAULT_REPORT = ROOT / "tmp" / "terroir-facts-review" / "normalize.json" + + +def log(msg: str) -> None: + print(f"[normalize] {msg}", file=sys.stderr) + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--dry-run", action="store_true") + ap.add_argument("--only", action="append", default=[]) + ap.add_argument("--report", type=Path, default=DEFAULT_REPORT) + args = ap.parse_args() + + per_country: dict[str, Counter] = {} + n_src_records = n_src_bullets = n_tr_records = n_tr_bullets = 0 + misaligned: list[dict] = [] + changes: list[dict] = [] + for p in sorted(TERROIR.glob("*.json")): + if p.name.startswith("manifest"): + continue + d = cache.read_json_or_none(p) + if not d or d.get("mode") == "verbatim" or not d.get("facts"): + continue + slug = d.get("slug") or p.stem + if args.only and slug not in args.only: + continue + cc = d.get("country") or "fr" + st = per_country.setdefault(cc, Counter()) + facts = d["facts"] + old_sha = facts_sha(facts) + before = [f.get("bullet") for f in facts] + n = normalize_facts(facts) + st["source_bullets"] += n + if n: + n_src_records += 1 + n_src_bullets += n + changes.extend({"slug": slug, "lang": d.get("source_lang") or cc, "index": i, "before": b, "after": f["bullet"]} + for i, (b, f) in enumerate(zip(before, facts)) if b != f["bullet"]) + new_sha = facts_sha(facts) + # translation bullets first (they hold their own text), then the re-key + for lang in LANGS: + tp = TRANSLATIONS / lang / f"{slug}.json" + if not tp.exists(): + continue + t = cache.read_json_or_none(tp) + if not t or t.get("mode") == "verbatim" or not t.get("facts"): + continue + tb = [f.get("bullet") for f in t["facts"]] + tn = normalize_facts(t["facts"], lang) + if tn: + n_tr_records += 1 + n_tr_bullets += tn + st[f"translated_bullets_{lang}"] += tn + changes.extend({"slug": slug, "lang": lang, "index": i, "before": b, "after": f["bullet"]} + for i, (b, f) in enumerate(zip(tb, t["facts"])) if b != f["bullet"]) + if not args.dry_run: + write_translation_cache(tp, t) + if n: + if not args.dry_run: + write_source_cache(p, d) + _, mis = rekey_translations(slug, old_sha, len(facts), new_sha, dry_run=args.dry_run) + misaligned.extend({"slug": slug, "lang": lang} for lang in mis) + + verb = "would normalise" if args.dry_run else "normalised" + log(f"{verb} {n_src_bullets} source bullets in {n_src_records} records and " + f"{n_tr_bullets} translated bullets in {n_tr_records} caches; " + f"{len(misaligned)} translation caches misaligned (left for 02e)") + log(f"{'cc':4} {'src':>5} " + " ".join(f"{lg:>5}" for lg in LANGS)) + for cc, st in sorted(per_country.items()): + log(f"{cc:4} {st['source_bullets']:5} " + " ".join(f"{st[f'translated_bullets_{lg}']:5}" for lg in LANGS)) + args.report.parent.mkdir(parents=True, exist_ok=True) + cache.write_json(args.report, { + "generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "dry_run": args.dry_run, + "summary": {"source_records": n_src_records, "source_bullets": n_src_bullets, + "translation_caches": n_tr_records, "translated_bullets": n_tr_bullets, + "misaligned": len(misaligned), "per_country": {c: dict(s) for c, s in sorted(per_country.items())}}, + "misaligned": misaligned, + "changes": changes, + }) + log(f"report → {args.report}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/pt/02d_extract_terroir_facts.py b/scripts/pt/02d_extract_terroir_facts.py index c9c6909..76863a1 100644 --- a/scripts/pt/02d_extract_terroir_facts.py +++ b/scripts/pt/02d_extract_terroir_facts.py @@ -35,7 +35,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -44,6 +43,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "pt" / "cadernos-extracted" WIKI_AOCS = ROOT / "raw" / "wikipedia" / "aocs" / "pt" @@ -139,37 +146,22 @@ - As citações são VERBATIM (copiadas e coladas) da sua fonte respetiva. NUNCA atribuas a uma fonte um texto que não consta dela. - Sem juízos de valor ("excecional", "extraordinário", "prestigiado"...). - Sem inferência externa. Sem números ausentes das duas fontes. -- Máximo {max_bullets} itens, ≤ 140 caracteres cada. +- Máximo {max_bullets} itens; cada item é uma frase completa de cerca de 120–220 caracteres — nunca um fragmento telegráfico. - Se nem o caderno nem a Wikipédia contêm um facto notável concreto para esta sub-secção, devolve uma lista vazia. Responde APENAS em JSON, sem texto antes ou depois: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Usa uma string vazia "" para a citação ausente.""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - """Longest-contiguous-match coverage of `quote` in `source` (0–1).""" - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -265,11 +257,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Sub-secção a tratar: {label}\n\n" - f"Texto do caderno (Relação com a área geográfica):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Sub-secção a tratar: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Texto do caderno (Relação com a área geográfica):\n\n{lien_text}" def _process_subsection( @@ -285,9 +278,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -322,6 +319,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: all_facts.append(f) src = record.get("source") or {} + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "pt") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "pt", "source_lang": "pt", @@ -329,6 +332,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -338,7 +343,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -408,12 +413,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(sem resumo Wikipédia disponível)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -453,7 +458,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts( @@ -597,7 +602,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -632,7 +637,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-pt.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -681,7 +686,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/pt/02e_translate_terroir_facts.py b/scripts/pt/02e_translate_terroir_facts.py index 9351c04..ade569d 100644 --- a/scripts/pt/02e_translate_terroir_facts.py +++ b/scripts/pt/02e_translate_terroir_facts.py @@ -33,6 +33,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -45,12 +48,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Portuguese proper nouns verbatim: appellation names ("Douro", "Vinho Verde", "Alentejo", "Dão", "Bairrada", "Madeira", "Porto"), region names ("Trás-os-Montes", "Beira Interior", "Península de Setúbal"), commune names, grape variety names ("Touriga Nacional", "Touriga Franca", "Tinta Roriz", "Trincadeira", "Alvarinho", "Loureiro", "Arinto", "Encruzado", "Baga", "Castelão", "Fernão Pires"), named geological formations and soil types ("xisto", "granito", "schistos", "complexo xisto-grauváquico"), named winds ("nortada", "suestada"), local landscape and historical terms ("socalcos", "quinta", "lagar", "talha", "estufagem", "canteiro"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Portuguese form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "pt" +PROPER_NOUNS = """appellation names (Douro, Vinho Verde, Alentejo, Dão, Bairrada, Madeira, Porto); region names (Trás-os-Montes, Beira Interior, Península de Setúbal); commune names; grape names (Touriga Nacional, Touriga Franca, Tinta Roriz, Trincadeira, Alvarinho, Loureiro, Arinto, Encruzado, Baga, Castelão, Fernão Pires); named winds (nortada, suestada); named Madeira processes (canteiro, estufagem)""" + # ─────────────────────────────────────────────────────────────── helpers ── @@ -110,7 +115,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -168,10 +173,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -308,6 +318,7 @@ def _build_argparser() -> argparse.ArgumentParser: "--workers", type=int, default=1, help="concurrent (lang, slug) pairs to translate (default 1, keep 1 for Ollama on M1 32GB)", ) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true", help="re-translate even if cached") ap.add_argument( "--batch", action="store_true", @@ -346,7 +357,7 @@ def _dispatch_emit_or_import(args, languages: tuple[str, ...]) -> int | None: def _make_provider(args) -> tuple[object | None, str]: return providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) @@ -415,8 +426,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -448,6 +461,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] diff --git a/scripts/recompute_terroir_provenance.py b/scripts/recompute_terroir_provenance.py new file mode 100644 index 0000000..bcbb05e --- /dev/null +++ b/scripts/recompute_terroir_provenance.py @@ -0,0 +1,209 @@ +"""Post-pass over the stage-02d terroir-fact caches: re-grade every fact's +quotes with the ellipsis-aware coverage rule and rewrite +`cahier_coverage` / `wiki_coverage` / `provenance` in place. No LLM call. + +Why: the model legitimately joins two source spans with "[…]" in a quote. +The original single-contiguous-match test then scored such a quote below +the 0.6 threshold, so hundreds of facts whose `cahier_quote` is verbatim +cahier text carried `provenance: wiki` (and the map panel attributed them +to Wikipedia). `_lib.terroir_coverage.fuzzy_coverage` now grades a +multi-span quote span by span; this script applies that rule to the +caches that were graded before it existed. + +How: each country's stage-02d module is loaded and asked for the exact +source text it graded against — the lien (for CH/MT/GB the règlement / +spec context) plus the per-sub-section Wikipedia hint — so the result is +precisely what 02d computes today. A cache whose `cahier_source_sha` or +`wiki_source_revision` no longer matches the current sources is skipped +and listed: it is stale for 02d anyway. Nothing is dropped: the new grade +of a previously-kept fact is never lower than its old one, so provenance +only ever moves towards `cahier` / `both`. + +The stage-02e translation caches copy each fact's `provenance`; the +aligned ones (same `source_facts_sha`, same length) are updated in the +same pass so the four locales agree with the source cache. Stage 04 reads +provenance from the source cache, so the panel picks the change up at the +next build. + +Usage: + .venv/bin/python scripts/recompute_terroir_provenance.py --dry-run + .venv/bin/python scripts/recompute_terroir_provenance.py [--country cc …] [--only SLUG …] + [--report tmp/terroir-facts-review/provenance-recompute.json] +""" + +from __future__ import annotations + +import argparse +import sys +from collections import Counter +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib import cache # noqa: E402 +from _lib.terroir_cache import sync_translation_provenance, write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage, provenance_for # noqa: E402 +from _lib.terroir_sources import Sources, resolve_sources # noqa: E402 + +TERROIR = ROOT / "raw" / "terroir-facts" +TRANSLATIONS = ROOT / "raw" / "translations" / "terroir-facts" +LANGS = ("en", "fr", "es", "nl") +DEFAULT_REPORT = ROOT / "tmp" / "terroir-facts-review" / "provenance-recompute.json" + + +def log(msg: str) -> None: + print(f"[provenance] {msg}", file=sys.stderr) + + +@dataclass +class Regrade: + changed: list[dict] + coverage_only: int = 0 + ungrounded_now: int = 0 + + +def regrade_facts(facts: list[dict], src: Sources) -> Regrade: + out = Regrade(changed=[]) + for i, f in enumerate(facts): + cq = (f.get("cahier_quote") or "").strip() + wq = (f.get("wiki_quote") or "").strip() + cc = src.matcher.coverage(cq) if cq else 0.0 + wc = fuzzy_coverage(wq, src.hints.get(f.get("subsection") or "", "")) if wq else 0.0 + prov = provenance_for(cc, wc) + if prov is None: + out.ungrounded_now += 1 + continue + old = (f.get("cahier_coverage"), f.get("wiki_coverage"), f.get("provenance")) + new = (round(cc, 3), round(wc, 3), prov) + if new == old: + continue + if new[2] == old[2]: + out.coverage_only += 1 + else: + out.changed.append({ + "index": i, + "old_provenance": old[2], "new_provenance": prov, + "old_cahier_coverage": old[0], "new_cahier_coverage": new[0], + "old_wiki_coverage": old[1], "new_wiki_coverage": new[1], + "bullet": f.get("bullet") or "", + "cahier_quote": cq[:200], + }) + f["cahier_coverage"], f["wiki_coverage"], f["provenance"] = new + return out + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--dry-run", action="store_true", help="report only; write nothing") + ap.add_argument("--country", action="append", default=[], help="restrict to a country code (repeatable)") + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") + ap.add_argument("--report", type=Path, default=DEFAULT_REPORT, help="JSON report path") + args = ap.parse_args() + + by_country: dict[str, list[Path]] = {} + for p in sorted(TERROIR.glob("*.json")): + if p.name.startswith("manifest"): + continue + d = cache.read_json_or_none(p) + if not d or d.get("mode") == "verbatim" or not d.get("facts"): + continue + cc = d.get("country") or "fr" + if args.country and cc not in args.country: + continue + if args.only and d.get("slug") not in args.only: + continue + by_country.setdefault(cc, []).append(p) + + transitions: Counter = Counter() + per_country: dict[str, Counter] = {} + skipped: list[dict] = [] + changes: list[dict] = [] + n_records_written = 0 + n_translations_written = 0 + misaligned_translations: list[dict] = [] + + for cc in sorted(by_country): + paths = by_country[cc] + log(f"{cc}: resolving sources for {len(paths)} cached records …") + try: + sources = resolve_sources(cc) + except Exception as e: # noqa: BLE001 + log(f"{cc}: FAILED to load stage 02d sources: {e!r} — skipping country") + skipped.extend({"slug": p.stem, "country": cc, "reason": f"stage-load-error: {e!r}"} for p in paths) + continue + stats = per_country.setdefault(cc, Counter()) + for p in paths: + d = cache.read_json_or_none(p) + slug = d.get("slug") or p.stem + src = sources.get(slug) + if src is None: + skipped.append({"slug": slug, "country": cc, "reason": "no-source-today"}) + stats["skipped"] += 1 + continue + if d.get("cahier_source_sha") != src.cahier_sha: + skipped.append({"slug": slug, "country": cc, "reason": "cahier-sha-drift"}) + stats["skipped"] += 1 + continue + if d.get("wiki_source_revision") != src.wiki_revision: + skipped.append({"slug": slug, "country": cc, "reason": "wiki-revision-drift"}) + stats["skipped"] += 1 + continue + facts = d["facts"] + res = regrade_facts(facts, src) + stats["facts"] += len(facts) + stats["coverage_only"] += res.coverage_only + stats["ungrounded_now"] += res.ungrounded_now + stats["provenance_changed"] += len(res.changed) + for ch in res.changed: + transitions[(ch["old_provenance"], ch["new_provenance"])] += 1 + changes.append({"slug": slug, "country": cc, **ch}) + if res.changed or res.coverage_only: + n_records_written += 1 + if not args.dry_run: + write_source_cache(p, d) + if res.changed: + n_t, mis = sync_translation_provenance(slug, facts, dry_run=args.dry_run) + n_translations_written += n_t + misaligned_translations.extend({"slug": slug, "lang": lang} for lang in mis) + + verb = "would rewrite" if args.dry_run else "rewrote" + log("") + log(f"{verb} {n_records_written} source caches; provenance changed on {len(changes)} facts; " + f"{n_translations_written} translation caches synced; " + f"{len(misaligned_translations)} translation caches misaligned (left for 02e); " + f"{len(skipped)} records skipped") + log("transitions: " + ", ".join(f"{a}→{b}: {n}" for (a, b), n in transitions.most_common())) + log(f"{'cc':4} {'facts':>6} {'prov-chg':>8} {'cov-only':>8} {'ungrnd':>6} {'skip':>5}") + for cc, st in sorted(per_country.items()): + log(f"{cc:4} {st['facts']:6} {st['provenance_changed']:8} {st['coverage_only']:8} " + f"{st['ungrounded_now']:6} {st['skipped']:5}") + if skipped: + reasons = Counter(s["reason"] for s in skipped) + log("skipped by reason: " + ", ".join(f"{r}: {n}" for r, n in reasons.most_common())) + + args.report.parent.mkdir(parents=True, exist_ok=True) + cache.write_json(args.report, { + "generated_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "dry_run": args.dry_run, + "summary": { + "records_rewritten": n_records_written, + "facts_provenance_changed": len(changes), + "translation_caches_synced": n_translations_written, + "translation_caches_misaligned": len(misaligned_translations), + "records_skipped": len(skipped), + "transitions": {f"{a}->{b}": n for (a, b), n in transitions.most_common()}, + "per_country": {cc: dict(st) for cc, st in sorted(per_country.items())}, + }, + "skipped": skipped, + "misaligned_translations": misaligned_translations, + "changes": changes, + }) + log(f"report → {args.report}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/rerun_terroir_facts.py b/scripts/rerun_terroir_facts.py new file mode 100644 index 0000000..21634a9 --- /dev/null +++ b/scripts/rerun_terroir_facts.py @@ -0,0 +1,264 @@ +"""Orchestrate a scoped terroir-facts re-run as ONE rollback unit: + + snapshot + mark stale → 02d --batch (per country, parallel) + → 02d_verify --batch → 02e --batch (per country, parallel) + → 02e_verify --batch → audit_terroir_facts + +Every step runs under one `OWM_TERROIR_RUN` id, so every cache the chain +touches is snapshotted into `raw/terroir-facts-backup/<run>/` before its +first write and `scripts/rollback_terroir_facts.py --run <run>` undoes +the whole chain (source caches and translations together). + +Scope. The 21 country scripts spell their record filter differently +(`--slug` exact for FR, `--only` substring elsewhere), so the +orchestrator does not pass slugs down. It marks each scoped record's +cache stale instead — `cahier_source_sha` prefixed `stale:` — which every +02d's cache check treats as "re-extract"; the per-country batch then +enumerates exactly those (plus records whose sources changed, which are +due anyway). A stale mark is written through `write_source_cache`, so the +pre-run copy is in the backup before the mark lands. + + --scope FILE JSON list of slugs (or {"slugs": [...]}) + --slug S add a slug (repeatable) + --country CC restrict the 02d step to these countries (default: + the countries of the scoped slugs). The gate, 02e + and the back-check always run corpus-wide on + whatever is ungated / stale / unchecked. + --parallel N per-country batch processes at once (default 6) + --scoped-02d pass the scope down to each 02d (--slug / --only) so a + country whose sources all changed (IT after the MASAF + full-text change) re-extracts only the scope + --scoped-gate gate only the scoped slugs (a smoke run right after a + GATE_VERSION bump would otherwise re-gate the corpus) + --scoped-backcheck back-check only the scoped slugs (same reason, after a + BACKCHECK_VERSION bump). A smoke run passes all three + --scoped-* flags; a corpus migration passes none. + --skip-02d / --skip-gate / --skip-02e / --skip-02e-verify / --skip-audit + --dry-run print the plan, touch nothing + +Logs: /tmp/owm-<run>/<step>-<cc>.log (tee'd, readable while running). + +Usage: + .venv/bin/python scripts/rerun_terroir_facts.py --scope tmp/r1-scope.json + .venv/bin/python scripts/rerun_terroir_facts.py --slug chablis --slug barolo +""" + +from __future__ import annotations + +import argparse +import json +import os +import subprocess +import sys +import time +from concurrent.futures import ThreadPoolExecutor, as_completed +from datetime import datetime, timezone +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib import batch, cache, terroir_backup # noqa: E402 +from _lib.terroir_cache import TERROIR, write_source_cache # noqa: E402 +from _lib.terroir_sources import COUNTRIES # noqa: E402 + +PY = ROOT / ".venv" / "bin" / "python" +STALE_PREFIX = "stale:" + + +def log(msg: str) -> None: + print(f"[rerun] {datetime.now().strftime('%H:%M:%S')} {msg}", file=sys.stderr) + + +def stage_script(stage: str, cc: str) -> Path: + name = {"02d": "02d_extract_terroir_facts.py", "02e": "02e_translate_terroir_facts.py"}[stage] + return ROOT / "scripts" / name if cc == "fr" else ROOT / "scripts" / cc / name + + +def load_scope(args) -> list[str]: + slugs: list[str] = list(args.slug or []) + if args.scope: + data = json.loads(Path(args.scope).read_text(encoding="utf-8")) + if isinstance(data, dict): + data = data.get("slugs") or data.get("union") or [] + slugs.extend(str(s) for s in data) + return sorted(set(slugs)) + + +def mark_stale(slugs: list[str], *, dry_run: bool) -> dict[str, list[str]]: + """Snapshot + mark each scoped cache stale. Returns {country: [slugs]}.""" + by_cc: dict[str, list[str]] = {} + for slug in slugs: + p = TERROIR / f"{slug}.json" + d = cache.read_json_or_none(p) + if not d: + log(f" {slug}: no cache — will be extracted if its country's 02d finds a source") + continue + cc = d.get("country") or "fr" + by_cc.setdefault(cc, []).append(slug) + sha = d.get("cahier_source_sha") or "" + if sha.startswith(STALE_PREFIX): + continue + if not dry_run: + d["cahier_source_sha"] = STALE_PREFIX + sha + write_source_cache(p, d) + return by_cc + + +def run_step(cmd: list[str], logfile: Path, env: dict) -> int: + logfile.parent.mkdir(parents=True, exist_ok=True) + with logfile.open("w", encoding="utf-8") as fh: + fh.write("$ " + " ".join(cmd) + "\n") + fh.flush() + proc = subprocess.run(cmd, stdout=fh, stderr=subprocess.STDOUT, env=env, cwd=ROOT) + return proc.returncode + + +def run_parallel(label: str, jobs: list[tuple[str, list[str]]], logdir: Path, env: dict, parallel: int) -> dict[str, int]: + rcs: dict[str, int] = {} + t0 = time.monotonic() + with ThreadPoolExecutor(max_workers=max(1, parallel)) as ex: + futs = {ex.submit(run_step, cmd, logdir / f"{label}-{cc}.log", env): cc for cc, cmd in jobs} + for fut in as_completed(futs): + cc = futs[fut] + rcs[cc] = fut.result() + log(f" {label} {cc}: exit {rcs[cc]} ({(time.monotonic() - t0) / 60:.1f} min) — {logdir / f'{label}-{cc}.log'}") + return rcs + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--scope", default=None) + ap.add_argument("--slug", action="append", default=[]) + ap.add_argument("--country", action="append", default=None) + ap.add_argument("--run", default=None, help="run id (default: timestamp)") + ap.add_argument("--parallel", type=int, default=6) + ap.add_argument("--provider", default="anthropic", choices=("anthropic", "mistral")) + ap.add_argument("--model", default=None, help="02d extractor model override (default: STAGE_DEFAULTS['02d']); " + "the gate, 02e and the back-check keep their own stage defaults") + for step in ("02d", "gate", "02e", "02e-verify", "audit"): + ap.add_argument(f"--skip-{step}", action="store_true") + ap.add_argument("--scoped-02d", action="store_true", + help="pass the scoped slugs down to each 02d (--slug for FR, --only elsewhere) so a " + "country whose sources all changed re-extracts only the scope") + ap.add_argument("--scoped-gate", action="store_true", + help="gate only the scoped slugs (default: corpus-wide on whatever is ungated — " + "after a GATE_VERSION bump that is the whole corpus)") + ap.add_argument("--scoped-backcheck", action="store_true", + help="back-check only the scoped slugs (default: corpus-wide on whatever is unchecked — " + "after a BACKCHECK_VERSION bump that is the whole corpus)") + ap.add_argument("--dry-run", action="store_true") + args = ap.parse_args() + + run = args.run or datetime.now(timezone.utc).strftime("%Y-%m-%dT%H%M%S") + os.environ[terroir_backup.RUN_ENV] = run + env = {**os.environ, terroir_backup.RUN_ENV: run} + logdir = Path("/tmp") / f"owm-{run}" + slugs = load_scope(args) + log(f"run {run}: {len(slugs)} scoped slugs; logs → {logdir}") + + by_cc = mark_stale(slugs, dry_run=args.dry_run) + countries = sorted(set(args.country) if args.country else set(by_cc)) + unknown = [c for c in countries if c not in COUNTRIES] + if unknown: + log(f"unknown countries {unknown}") + return 1 + for cc in countries: + log(f" {cc}: {len(by_cc.get(cc, []))} scoped records") + if args.dry_run: + log("dry run — nothing marked, nothing launched.") + return 0 + + model = ["--model", args.model] if args.model else [] + if not args.skip_02d and countries: + log(f"02d --batch for {len(countries)} countries (parallel {args.parallel}) …") + jobs = [] + for cc in countries: + scoped: list[str] = [] + if args.scoped_02d: + flag = "--slug" if cc == "fr" else "--only" + scoped = [x for slug in by_cc.get(cc, []) for x in (flag, slug)] + jobs.append((cc, [str(PY), str(stage_script("02d", cc)), "--batch", "--provider", args.provider, + *model, *scoped])) + rcs = run_parallel("02d", jobs, logdir, env, args.parallel) + if any(rcs.values()): + log(f"02d failed for {[c for c, rc in rcs.items() if rc]} — re-run the same command to resume; stopping.") + return 1 + + if not args.skip_gate: + gate_scope: list[str] = [] + if args.scoped_gate: + scope_file = logdir / "gate-scope.json" + scope_file.write_text(json.dumps({"slugs": slugs}), encoding="utf-8") + gate_scope = ["--only-file", str(scope_file)] + log(f"02d_verify --batch ({'scoped' if gate_scope else 'corpus-wide, ungated records'}) …") + rc = run_step([str(PY), str(ROOT / "scripts" / "02d_verify_terroir_facts.py"), "--batch", + "--provider", args.provider, "--quiet", *gate_scope], logdir / "gate.log", env) + log(f" gate: exit {rc} — {logdir / 'gate.log'}") + if rc: + return 1 + + if not args.skip_02e: + # Every country: the gate may have rewritten records outside the scope. + cc_02e = sorted(COUNTRIES) + log(f"02e --batch for {len(cc_02e)} countries …") + jobs = [(cc, [str(PY), str(stage_script("02e", cc)), "--batch", "--provider", args.provider]) + for cc in cc_02e] + rcs = run_parallel("02e", jobs, logdir, env, args.parallel) + if any(rcs.values()): + log(f"02e failed for {[c for c, rc in rcs.items() if rc]} — re-run to resume; continuing to the checks.") + + if not args.skip_02e_verify: + check_scope: list[str] = [] + if args.scoped_backcheck: + scope_file = logdir / "backcheck-scope.json" + scope_file.write_text(json.dumps({"slugs": slugs}), encoding="utf-8") + check_scope = ["--only-file", str(scope_file)] + log(f"02e_verify --batch ({'scoped' if check_scope else 'corpus-wide, unchecked translations'}) …") + rc = run_step([str(PY), str(ROOT / "scripts" / "02e_verify_terroir_facts.py"), "--batch", + "--provider", args.provider, "--quiet", *check_scope], logdir / "02e-verify.log", env) + log(f" 02e-verify: exit {rc} — {logdir / '02e-verify.log'}") + + if not args.skip_audit: + report = ROOT / "tmp" / "terroir-facts-review" / f"audit-{run}.json" + rc = run_step([str(PY), str(ROOT / "scripts" / "audit_terroir_facts.py"), "--quiet", "--report", str(report)], + logdir / "audit.log", env) + log(f" audit: exit {rc} — {report.relative_to(ROOT)}; strict summary in {logdir / 'audit.log'}") + + log(cost_report(run)) + log(f"done. Rollback with: scripts/rollback_terroir_facts.py --run {run}") + return 0 + + +_STAGE_LABEL = {"02d-verify": "gate", "02e-verify": "backcheck", "llm-audit": "audit"} + + +def cost_report(run: str) -> str: + """The run's spend per stage from the batch ledger (`raw/.batch/costs.jsonl`).""" + if not batch.COSTS_LEDGER.exists(): + return "cost: no batch ledger" + per_stage: dict[str, float] = {} + unpriced = 0 + for line in batch.COSTS_LEDGER.read_text(encoding="utf-8").splitlines(): + try: + row = json.loads(line) + except ValueError: + continue + if row.get("run") != run: + continue + stage = _STAGE_LABEL.get(row.get("stage") or "", (row.get("stage") or "?").split("-")[0]) + cost = row.get("cost_usd") + if cost is None: + unpriced += 1 + continue + per_stage[stage] = per_stage.get(stage, 0.0) + cost + if not per_stage and not unpriced: + return "cost: no batches recorded for this run" + parts = [f"{k} ${v:,.2f}" for k, v in sorted(per_stage.items())] + total = sum(per_stage.values()) + return (f"cost (Batch API): {' · '.join(parts)} — total ${total:,.2f}" + + (f" (+{unpriced} unpriced batches)" if unpriced else "")) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/ro/02d_extract_terroir_facts.py b/scripts/ro/02d_extract_terroir_facts.py index 6b52aa6..c383600 100644 --- a/scripts/ro/02d_extract_terroir_facts.py +++ b/scripts/ro/02d_extract_terroir_facts.py @@ -32,7 +32,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -41,6 +40,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "ro" / "dokumente-extracted" NATIONAL_SPECS = ROOT / "raw" / "ro" / "national-specs-extracted" @@ -134,37 +141,22 @@ - Citatele sunt LITERALE (copiate) din sursa corespunzătoare. NICIODATĂ nu atribui unei surse un text care nu apare acolo. - Fără judecăți de valoare ("excepțional", "prestigios" …). - Fără concluzii externe. Fără cifre care nu apar în niciuna dintre cele două surse. -- Maximum {max_bullets} elemente, fiecare ≤ 140 caractere. +- Maximum {max_bullets} elemente; fiecare element este o propoziție completă de aproximativ 120–220 de caractere — niciodată un fragment telegrafic. - Dacă nici documentul unic, nici Wikipedia nu conțin un fapt concret demn de menționat pentru această subsecțiune, returnează o listă goală. Răspunde DOAR în JSON, fără text înainte sau după: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Pentru un citat care lipsește, folosește un șir gol "".""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - """Longest-contiguous-match coverage of `quote` in `source` (0–1).""" - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -293,11 +285,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Subsecțiunea de procesat: {label}\n\n" - f"Textul documentului unic (Descrierea legăturii):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Subsecțiunea de procesat: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Textul documentului unic (Descrierea legăturii):\n\n{lien_text}" def _process_subsection( @@ -313,9 +306,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -350,6 +347,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "ro") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "ro", "source_lang": "ro", @@ -357,6 +360,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -367,7 +372,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -436,12 +441,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(extras Wikipedia indisponibil)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -482,7 +487,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts(slug_items: list[dict], lien: str) -> list[dict]: @@ -620,7 +625,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -653,7 +658,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-ro.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -703,7 +708,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/ro/02e_translate_terroir_facts.py b/scripts/ro/02e_translate_terroir_facts.py index 4269e22..163345d 100644 --- a/scripts/ro/02e_translate_terroir_facts.py +++ b/scripts/ro/02e_translate_terroir_facts.py @@ -27,6 +27,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -39,12 +42,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Romanian proper nouns verbatim: appellation names ("Cotnari", "Murfatlar", "Drăgășani", "Recaș", "Dealu Mare", "Târnave", "Iași", "Huși", "Odobești", "Panciu", "Bohotin", "Coteşti", "Banat", "Crișana", "Miniș", "Diosig", "Sâmburești", "Banu Mărăcine", "Mehedinți", "Severin", "Sarica Niculițel", "Babadag", "Aiud", "Lechința", "Sebeș-Apold", "Pietroasa", "Ștefănești", "Dealurile Munteniei", "Dealurile Olteniei", "Dealurile Crișanei", "Dealurile Sătmarului", "Viile Timișului", "Colinele Dobrogei", "Terasele Dunării"), commune and vineyard-site names ("Cernavodă", "Ostrov", "Mărăcineni", "Bujoreni", "Cotești"), grape variety names ("Fetească Albă", "Fetească Regală", "Fetească Neagră", "Tămâioasă Românească", "Grasă de Cotnari", "Băbească Neagră", "Negru de Drăgășani", "Novac", "Crâmpoșie Selecționată", "Frâncușă", "Galbenă de Odobești", "Plăvaie", "Zghihară de Huși", "Șarbă", "Busuioacă de Bohotin"), named geological formations and soil types ("cernoziom", "brun-roșcat", "loess", "marnă", "calcar", "gresie", "sol scheletic", "podzol"), named climatic features ("climă continentală", "climă temperat-continentală", "influență mediteraneană", "climă pontică", "crivăț", "austru", "băltăreț"), and Romanian wine-law terms ("podgorie", "regiune viticolă", "indicație geografică", "denumire de origine", "DOP", "IGP", "ONVPV", "Caiet de sarcini"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Romanian form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "ro" +PROPER_NOUNS = """appellation names (Cotnari, Murfatlar, Drăgășani, Recaș, Dealu Mare, Târnave, Iași, Huși, Odobești, Panciu, Bohotin, Coteşti, Banat, Crișana, Miniș, Diosig, Sâmburești, Banu Mărăcine, Mehedinți, Severin, Sarica Niculițel, Babadag, Aiud, Lechința, Sebeș-Apold, Pietroasa, Ștefănești, Dealurile Munteniei, Dealurile Olteniei, Dealurile Crișanei, Dealurile Sătmarului, Viile Timișului, Colinele Dobrogei, Terasele Dunării); commune and vineyard-site names (Cernavodă, Ostrov, Mărăcineni, Bujoreni, Cotești); grape names (Fetească Albă, Fetească Regală, Fetească Neagră, Tămâioasă Românească, Grasă de Cotnari, Băbească Neagră, Negru de Drăgășani, Novac, Crâmpoșie Selecționată, Frâncușă, Galbenă de Odobești, Plăvaie, Zghihară de Huși, Șarbă, Busuioacă de Bohotin); named winds (crivăț, austru, băltăreț); institutions (ONVPV)""" + # ─────────────────────────────────────────────────────────────── helpers ── @@ -102,7 +107,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -157,10 +162,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -291,6 +301,7 @@ def _build_argparser() -> argparse.ArgumentParser: "--workers", type=int, default=1, help="concurrent (lang, slug) pairs (default 1, keep 1 for Ollama)", ) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true", help="re-translate even if cached") ap.add_argument( "--batch", action="store_true", @@ -378,8 +389,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -411,6 +424,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] @@ -420,7 +435,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: for j in jobs: diff --git a/scripts/rollback_terroir_facts.py b/scripts/rollback_terroir_facts.py new file mode 100644 index 0000000..190d13c --- /dev/null +++ b/scripts/rollback_terroir_facts.py @@ -0,0 +1,97 @@ +"""Roll a terroir-facts run back from its backup. + +Every write to `raw/terroir-facts/` and `raw/translations/terroir-facts/` +(stage 02d, the claim-support gate, stage 02e, the post-passes) first +snapshots the slug's source + translation caches into +`raw/terroir-facts-backup/<run>/` (`_lib/terroir_backup.py`). This script +puts a run's slugs back to their pre-run state: files the run overwrote +are restored, files it created are deleted, so the source cache and its +four index-aligned translation caches stay in step. + + --list every run with its slug count and start time + --run ID the run to roll back (required unless --list) + --only SLUG restrict to these slugs (repeatable) + --dry-run print what would change, write nothing + +A rollback is itself a write and is snapshotted under a new run id +(`rollback-of-<ID>-<ts>`), so a rollback can be rolled back. After a +rollback, re-run stage 04 to rebuild the map from the restored caches. + +Usage: + .venv/bin/python scripts/rollback_terroir_facts.py --list + .venv/bin/python scripts/rollback_terroir_facts.py --run 2026-09-13T1200 --dry-run + .venv/bin/python scripts/rollback_terroir_facts.py --run 2026-09-13T1200 --only chablis +""" + +from __future__ import annotations + +import argparse +import os +import sys +from datetime import datetime, timezone +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(ROOT / "scripts")) + +from _lib import terroir_backup as tb # noqa: E402 + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--list", action="store_true", help="list the runs that can be rolled back") + ap.add_argument("--run", default=None, help="run id (a directory under raw/terroir-facts-backup/)") + ap.add_argument("--only", action="append", default=[], help="restrict to these slugs (repeatable)") + ap.add_argument("--dry-run", action="store_true") + args = ap.parse_args() + + if args.list: + runs = tb.list_runs() + if not runs: + print("[rollback] no backup runs under raw/terroir-facts-backup/", file=sys.stderr) + return 0 + for run, m in runs: + slugs = m.get("slugs") or {} + argv = " ".join(m.get("argv") or []) + print(f" {run:32} {len(slugs):5} slugs started {m.get('started_at') or '?'} {argv[:70]}") + return 0 + + if not args.run: + ap.error("--run ID is required (see --list)") + m = tb.load_manifest(args.run) + slugs = m.get("slugs") or {} + if not slugs: + print(f"error: run {args.run!r} has no manifest under {tb.BACKUP_ROOT}", file=sys.stderr) + return 1 + wanted = [s for s in sorted(slugs) if not args.only or s in set(args.only)] + unknown = sorted(set(args.only) - set(slugs)) + for s in unknown: + print(f" skip {s}: not in run {args.run}", file=sys.stderr) + if not wanted: + print("[rollback] nothing to do.", file=sys.stderr) + return 0 + + # The rollback's own snapshot goes under a fresh run id so it is undoable. + if not args.dry_run: + stamp = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H%M%S") + os.environ[tb.RUN_ENV] = f"rollback-of-{args.run}-{stamp}" + tb._run_id = None # noqa: SLF001 — re-read the env for this process + n_restored = n_deleted = 0 + for slug in wanted: + if not args.dry_run: + tb.snapshot_slug(slug, note=f"before rollback of {args.run}") + r = tb.restore_slug(slug, args.run, dry_run=args.dry_run) + n_restored += len(r["restored"]) + n_deleted += len(r["deleted"]) + what = ", ".join([f"restore {p}" for p in r["restored"]] + [f"delete {p}" for p in r["deleted"]]) + print(f" {slug}: {what or 'no change'}", file=sys.stderr) + verb = "would restore" if args.dry_run else "restored" + print(f"[rollback] run {args.run}: {len(wanted)} slugs — {verb} {n_restored} files, " + f"{'would delete' if args.dry_run else 'deleted'} {n_deleted} files created by the run." + f"{'' if args.dry_run else ' Re-run scripts/04_build_maps.py to rebuild the map.'}", + file=sys.stderr) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/serve.py b/scripts/serve.py index a231411..d9907f0 100644 --- a/scripts/serve.py +++ b/scripts/serve.py @@ -29,6 +29,16 @@ class RangeHandler(http.server.SimpleHTTPRequestHandler): + def end_headers(self): + # Local QA only: the per-slug panel JSON (data/d/<locale>/<slug>.json) + # and the entity pages are fetched at runtime on stable paths, so a + # rebuild does not change their URL and the browser keeps serving its + # cached copy across a #hash navigation (2026-09-15: a re-translated + # bullet stayed stale on screen). Never cached here; production + # caching is the CDN's business. + self.send_header("Cache-Control", "no-store") + super().end_headers() + def send_head(self): # Locale-aware SPA fallback: appellations deep-link as real paths # (/<lang>/<slug>, EN at /<slug>), which have no file behind them. diff --git a/scripts/si/02d_extract_terroir_facts.py b/scripts/si/02d_extract_terroir_facts.py index 88e0839..b30db6e 100644 --- a/scripts/si/02d_extract_terroir_facts.py +++ b/scripts/si/02d_extract_terroir_facts.py @@ -32,7 +32,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -49,6 +48,14 @@ roundtrip, terroir_verbatim, ) +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "si" / "dokumenti-extracted" SPECIFIKACIJE = ROOT / "raw" / "si" / "specifikacije-extracted" @@ -141,37 +148,22 @@ - Citati so DOBESEDNI (kopirani) iz ustreznega vira. NIKOLI ne pripiši viru besedila, ki se tam ne pojavi. - Brez vrednostnih sodb ("izjemno", "prestižno" ...). - Brez zunanjih sklepov. Brez številk, ki jih ni v nobenem od obeh virov. -- Največ {max_bullets} vnosov, vsak ≤ 140 znakov. +- Največ {max_bullets} vnosov; vsak vnos je en poln stavek z približno 120–220 znaki — nikoli telegrafski fragment. - Če niti enotni dokument niti Wikipedija ne vsebujeta konkretnega omembe vrednega dejstva za ta podrazdelek, vrni prazen seznam. Odgovori SAMO v JSON, brez besedila pred ali za njim: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Za manjkajoči citat uporabi prazen niz "".""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - """Longest-contiguous-match coverage of `quote` in `source` (0–1).""" - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -324,11 +316,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Podrazdelek za obravnavo: {label}\n\n" - f"Besedilo enotnega dokumenta (Povezava z geografskim območjem):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Podrazdelek za obravnavo: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Besedilo enotnega dokumenta (Povezava z geografskim območjem):\n\n{lien_text}" def _process_subsection( @@ -344,9 +337,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -381,6 +378,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "sl") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "si", "source_lang": "sl", @@ -388,6 +391,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -398,7 +403,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -467,12 +472,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(izvleček Wikipedije ni na voljo)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -513,7 +518,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts(slug_items: list[dict], lien: str) -> list[dict]: @@ -651,7 +656,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -684,7 +689,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-si.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -734,7 +739,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/si/02e_translate_terroir_facts.py b/scripts/si/02e_translate_terroir_facts.py index 79a1d0f..94f8c62 100644 --- a/scripts/si/02e_translate_terroir_facts.py +++ b/scripts/si/02e_translate_terroir_facts.py @@ -27,6 +27,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -39,12 +42,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Slovenian proper nouns verbatim: appellation names ("Cviček", "Teran", "Goriška Brda", "Vipavska dolina", "Kras", "Slovenska Istra", "Štajerska Slovenija", "Prekmurje", "Bizeljsko Sremič", "Dolenjska", "Bela krajina", "Belokranjec", "Bizeljčan", "Metliška črnina"), wine-region names ("Podravje", "Posavje", "Primorska"), commune, vinorodni-okoliš and vineyard-site ("lega") names, grape variety names ("Žametovka", "Modra frankinja", "Refošk", "Rebula", "Malvazija", "Laški rizling", "Renski rizling", "Zelen", "Pinela", "Šipon", "Kraljevina", "Ranfol", "Rumeni plavec", "Sauvignon", "Chardonnay"), named geological formations and soil types ("fliš", "opoka", "apnenec", "peščenjak", "ilovica", "terra rossa", "laporovec"), named climatic features ("submediteransko podnebje", "celinsko podnebje", "panonski vpliv"), and Slovenian wine-law / Predikat terms ("vinorodni okoliš", "vinorodna dežela", "vinorodni podokoliš", "pozna trgatev", "izbor", "jagodni izbor", "suhi jagodni izbor", "ledeno vino", "slamno vino", "PTP", "ZOP", "ZGP"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Slovenian form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "sl" +PROPER_NOUNS = """appellation names (Cviček, Teran, Goriška Brda, Vipavska dolina, Kras, Slovenska Istra, Štajerska Slovenija, Prekmurje, Bizeljsko Sremič, Dolenjska, Bela krajina, Belokranjec, Bizeljčan, Metliška črnina); wine-region names (Podravje, Posavje, Primorska); commune, okoliš and vineyard-site names; grape names (Žametovka, Modra frankinja, Refošk, Rebula, Malvazija, Laški rizling, Renski rizling, Zelen, Pinela, Šipon, Kraljevina, Ranfol, Rumeni plavec, Sauvignon, Chardonnay)""" + # ─────────────────────────────────────────────────────────────── helpers ── @@ -102,7 +107,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -157,10 +162,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -291,6 +301,7 @@ def _build_argparser() -> argparse.ArgumentParser: "--workers", type=int, default=1, help="concurrent (lang, slug) pairs (default 1, keep 1 for Ollama)", ) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true", help="re-translate even if cached") ap.add_argument( "--batch", action="store_true", @@ -378,8 +389,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -411,6 +424,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] @@ -420,7 +435,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: for j in jobs: diff --git a/scripts/sk/02d_extract_terroir_facts.py b/scripts/sk/02d_extract_terroir_facts.py index 16cde46..859221b 100644 --- a/scripts/sk/02d_extract_terroir_facts.py +++ b/scripts/sk/02d_extract_terroir_facts.py @@ -34,7 +34,6 @@ import time from concurrent.futures import ThreadPoolExecutor, as_completed from datetime import datetime, timezone -from difflib import SequenceMatcher from pathlib import Path from tqdm import tqdm @@ -43,6 +42,14 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip, terroir_verbatim # noqa: E402 +from _lib.prompt_cache import cached_system # noqa: E402 +from _lib.terroir_cache import write_source_cache # noqa: E402 +from _lib.terroir_coverage import fuzzy_coverage # noqa: E402 +from _lib.terroir_dedupe import dedupe_facts # noqa: E402 +from _lib.terroir_feedback import with_feedback # noqa: E402 +from _lib.terroir_interactions import earn_interactions # noqa: E402 +from _lib.terroir_normalize import normalize_facts # noqa: E402 +from _lib.terroir_prompts import with_style_rules # noqa: E402 EXTRACTED = ROOT / "raw" / "sk" / "dokumenty-extracted" NATIONAL_SPECS = ROOT / "raw" / "sk" / "national-specs-extracted" @@ -136,37 +143,22 @@ - Citáty sú DOSLOVNÉ (kopírované) z príslušného zdroja. NIKDY nepripíš zdroju text, ktorý sa tam nevyskytuje. - Bez hodnotových súdov ("vynikajúci", "prestížny" ...). - Bez vonkajších záverov. Bez čísel, ktoré sa nenachádzajú ani v jednom zo zdrojov. -- Najviac {max_bullets} záznamov, každý ≤ 140 znakov. +- Najviac {max_bullets} záznamov; každý záznam je jedna úplná veta s približne 120–220 znakmi — nikdy telegrafický fragment. - Ak ani jednotný dokument ani Wikipédia neobsahujú konkrétny dôležitý fakt pre tento podrazdelka, vráť prázdny zoznam. Odpovedz IBA v JSON, bez textu pred alebo za: {{"facts": [{{"bullet": "...", "cahier_quote": "...", "wiki_quote": "..."}}, ...]}} Pre chýbajúci citát použi prázdny reťazec "".""" +EXTRACT_SYSTEM = with_style_rules(EXTRACT_SYSTEM) # ─────────────────────────────────────────────────────────────── helpers ── -def normalize(s: str) -> str: - return " ".join((s or "").split()).lower() - - def cahier_sha(text: str) -> str: return hashlib.sha256(text.encode("utf-8")).hexdigest() -def fuzzy_coverage(quote: str, source: str) -> float: - """Longest-contiguous-match coverage of `quote` in `source` (0–1).""" - q = normalize(quote) - s = normalize(source) - if not q: - return 0.0 - match = SequenceMatcher(None, q, s, autojunk=False).find_longest_match( - 0, len(q), 0, len(s) - ) - return match.size / len(q) - - def _find_heading(full: str, heading: str) -> int: idx = full.find(f"\n\n{heading}\n\n") if idx == -1: @@ -296,11 +288,12 @@ def collect_targets() -> list[dict]: return out -def _build_user_message(label: str, lien_text: str) -> str: - return ( - f"Sledovaná podsekcia: {label}\n\n" - f"Text jednotného dokumentu (Opis súvislostí):\n\n{lien_text}" - ) +def _ask_line(label: str) -> str: + return f"Sledovaná podsekcia: {label}" + + +def _document_block(lien_text: str) -> str: + return f"Text jednotného dokumentu (Opis súvislostí):\n\n{lien_text}" def _process_subsection( @@ -316,9 +309,13 @@ def _process_subsection( topics=sub["topics"], max_bullets=sub["max_bullets"], ) - user = _build_user_message(sub["label"], lien) + system = with_feedback(system, record["slug"]) + # The lien is the cached leading block: the four sub-section calls share it. + system = cached_system(_document_block(lien), system, phased=True) + user = _ask_line(sub["label"]) try: - raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192) + raw = provider.chat(system=system, user=user, max_tokens=1500, num_ctx=8192, + cache_phase=sub["key"]) except Exception as e: # noqa: BLE001 return [], 0, str(e) payload, perr = llm_json.parse_facts(raw) @@ -353,6 +350,12 @@ def _process_record(provider, model_id: str, record: dict) -> dict: f["subsection"] = sub["key"] all_facts.append(f) + deduped = dedupe_facts(all_facts) + all_facts = deduped.kept + earned = earn_interactions(all_facts, "sk") + all_facts = earned.kept + normalize_facts(all_facts) + payload = { "country": "sk", "source_lang": "sk", @@ -360,6 +363,8 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "name": record.get("name") or slug, "facts": all_facts, "n_dropped": n_dropped_total, + "n_deduped": len(deduped.drops), + "n_unearned_interactions": len(earned.dropped), "model": model_id, "model_kind": provider.kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), @@ -370,7 +375,7 @@ def _process_record(provider, model_id: str, record: dict) -> dict: "wiki_source_url": wiki_url, "subsection_errors": sub_errors, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) return payload @@ -439,12 +444,12 @@ def emit_todo(out_path: Path, *, skip_cached: bool, limit: int = 0) -> int: "subsection_label": sub["label"], "topics": sub["topics"], "max_bullets": sub["max_bullets"], - "system_prompt": EXTRACT_SYSTEM.format( + "system_prompt": with_feedback(EXTRACT_SYSTEM.format( wiki_hint=wiki_hint or "(výňatok z Wikipédie nie je k dispozícii)", label=sub["label"], topics=sub["topics"], max_bullets=sub["max_bullets"], - ), + ), job["slug"]), "cahier_text": job["lien"], "wiki_hint": wiki_hint, "cahier_source_sha": job["lien_sha"], @@ -485,7 +490,7 @@ def _write_imported_cache( "wiki_source_revision": wiki_record.get("revision") if wiki_record else None, "wiki_source_url": wiki_record.get("page_url") if wiki_record else None, } - cache.write_json(CACHE_DIR / f"{slug}.json", payload) + write_source_cache(CACHE_DIR / f"{slug}.json", payload) def _classify_imported_facts(slug_items: list[dict], lien: str) -> list[dict]: @@ -623,7 +628,7 @@ def _run_batch(args) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02d") targets = collect_targets() if args.only: needles = [s.lower() for s in args.only] @@ -656,7 +661,7 @@ def run_loop(prov): batch.run_two_pass( provider=args.provider, model=model_id, sidecar=ROOT / "raw" / ".batch" / "02d-sk.json", - run_loop=run_loop, + run_loop=run_loop, thinking=batch.default_thinking(args.provider, stage="02d"), ) return 0 @@ -706,7 +711,7 @@ def main() -> int: return 0 provider, model_id = providers.make_provider( - args.provider, model=args.model, ollama_url=args.ollama_url, + args.provider, model=args.model, stage="02d", ollama_url=args.ollama_url, mistral_url=args.mistral_url, ) if provider is None: diff --git a/scripts/sk/02e_translate_terroir_facts.py b/scripts/sk/02e_translate_terroir_facts.py index 1d0ebbe..317e4fa 100644 --- a/scripts/sk/02e_translate_terroir_facts.py +++ b/scripts/sk/02e_translate_terroir_facts.py @@ -27,6 +27,9 @@ sys.path.insert(0, str(ROOT / "scripts")) from _lib import batch, cache, llm_json, providers, roundtrip # noqa: E402 +from _lib.prompt_cache import mark_cached # noqa: E402 +from _lib.terroir_cache import write_translation_cache # noqa: E402 +from _lib.terroir_prompts import translation_system_prompt, with_appellation_context # noqa: E402 TERROIR_FACTS = ROOT / "raw" / "terroir-facts" CACHE_ROOT = ROOT / "raw" / "translations" / "terroir-facts" @@ -39,12 +42,14 @@ Rules: - Output a JSON array of strings, one translated bullet per input bullet, in the SAME order. The array length must equal the input list length. -- Preserve Slovak proper nouns verbatim: appellation names ("Vinohradnícka oblasť Tokaj", "Tokajské víno", "Malokarpatská", "Južnoslovenská", "Stredoslovenská", "Východoslovenská", "Nitrianska", "Karpatská perla", "Skalický rubín", "Slovenská"), wine-region names ("Malokarpatská", "Južnoslovenská", "Nitrianska", "Stredoslovenská", "Východoslovenská", "Tokaj"), commune, vinohradnícky rajón and vineyard-site names, grape variety names ("Frankovka modrá", "Veltlínske zelené", "Devín", "Dunaj", "Hron", "Rimava", "Váh", "Nitria", "Furmint", "Lipovina", "Muškát žltý", "Svätovavrinecké", "Modrý Portugal", "Rizling vlašský", "Rizling rýnsky", "Tramín červený", "Pesecká leánka", "Müller Thurgau", "Cabernet Sauvignon"), named geological formations and soil types ("spraš", "černozem", "ílovica", "vápenec", "dolomit", "andezit", "ryolit", "vulkanická pôda", "hlina"), named climatic features ("panónske podnebie", "kontinentálne podnebie", "vplyv Karpát"), and Slovak wine-law / Tokaj-predikát terms ("vinohradnícka oblasť", "vinohradnícky rajón", "neskorý zber", "výber z hrozna", "bobuľový výber", "hrozienkový výber", "ľadové víno", "slamové víno", "tokajský výber", "tokajská esencia", "samorodné", "CHOP", "CHZO"). - Geological era labels: translate to the standard {lang_name} form when one exists. When unsure, keep the Slovak form. - Translate descriptive vocabulary naturally for a wine-literate reader. - Match each source bullet's length and register; do not add commentary, footnotes, or explanations. - Output ONLY the JSON array, no preface, no markdown fences.""" +SOURCE_LANG = "sk" +PROPER_NOUNS = """appellation names (Vinohradnícka oblasť Tokaj, Tokajské víno, Malokarpatská, Južnoslovenská, Stredoslovenská, Východoslovenská, Nitrianska, Karpatská perla, Skalický rubín, Slovenská); wine-region names (Malokarpatská, Južnoslovenská, Nitrianska, Stredoslovenská, Východoslovenská, Tokaj); commune, rajón and vineyard-site names; grape names (Frankovka modrá, Veltlínske zelené, Devín, Dunaj, Hron, Rimava, Váh, Nitria, Furmint, Lipovina, Muškát žltý, Svätovavrinecké, Modrý Portugal, Rizling vlašský, Rizling rýnsky, Tramín červený, Pesecká leánka, Müller Thurgau, Cabernet Sauvignon); registered Tokaj terms (tokajský výber, tokajská esencia, samorodné)""" + # ─────────────────────────────────────────────────────────────── helpers ── @@ -102,7 +107,7 @@ def write_cache( "translator_kind": translator_kind, "fetched_at": datetime.now(timezone.utc).isoformat(timespec="seconds"), } - cache.write_json(cache_path(lang, slug), payload) + write_translation_cache(cache_path(lang, slug), payload) def _is_fresh_cache(existing: dict | None, sha: str, expected_len: int) -> bool: @@ -157,10 +162,15 @@ def build_user_prompt(src_facts: list[dict]) -> str: def translate_one(provider, job: dict) -> tuple[list[str] | None, str | None]: - system = SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]) + system = translation_system_prompt( + SYSTEM_PROMPT.format(lang_name=LOCALE_NAME[job["lang"]]), + source_lang=SOURCE_LANG, target_lang=job["lang"], proper_nouns=PROPER_NOUNS, + ) user = build_user_prompt(job["src_facts"]) + user = with_appellation_context(user, job["slug"]) # names the appellation on sub-denomination pages try: - raw = provider.chat(system=system, user=user, max_tokens=2000, num_ctx=8192) + # One system prompt per locale, shared by every record of the batch: cached. + raw = provider.chat(system=mark_cached(system), user=user, max_tokens=2000, num_ctx=8192) except Exception as e: # noqa: BLE001 return None, f"call: {e}" parsed = parse_array(raw, len(job["src_facts"])) @@ -291,6 +301,7 @@ def _build_argparser() -> argparse.ArgumentParser: "--workers", type=int, default=1, help="concurrent (lang, slug) pairs (default 1, keep 1 for Ollama)", ) + ap.add_argument("--only", action="append", default=[], help="restrict to a slug (repeatable)") ap.add_argument("--refresh", action="store_true", help="re-translate even if cached") ap.add_argument( "--batch", action="store_true", @@ -378,8 +389,10 @@ def _run_batch(args, languages: tuple[str, ...]) -> int: if not batch.supports(args.provider): print("error: --batch requires --provider anthropic|mistral", file=sys.stderr) return 1 - model_id = args.model or batch.default_model(args.provider) + model_id = args.model or batch.default_model(args.provider, stage="02e") jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] if not jobs: @@ -411,6 +424,8 @@ def main() -> int: return _run_batch(args, languages) jobs = enumerate_jobs(languages, skip_cached=not args.refresh) + if args.only: + jobs = [j for j in jobs if j["slug"] in set(args.only)] if args.limit: jobs = jobs[: args.limit] @@ -420,7 +435,7 @@ def main() -> int: provider, model_id = providers.make_provider( args.provider, model=args.model, ollama_url=args.ollama_url, - mistral_url=args.mistral_url, + mistral_url=args.mistral_url, stage="02e", ) if provider is None: for j in jobs: diff --git a/tests/fixtures/es_mapa_listado_excerpt.txt b/tests/fixtures/es_mapa_listado_excerpt.txt new file mode 100644 index 0000000..261c47d --- /dev/null +++ b/tests/fixtures/es_mapa_listado_excerpt.txt @@ -0,0 +1,25 @@ + LISTADO DE DENOMINACIONES DE ORIGEN PROTEGIDAS E INDICACIONES + Total: 149 + Actualizado a: jueves, 2 de julio de 2026 + SUPRAAUTONÓMICO Total: 4 + Nombre oficial Término tradicional (*) Nº expediente UE + DOP Cava DO PDO-ES-A0735 + DOP Rioja DOCa PDO-ES-A0117 + Total DOPs: 3 + IGP Ribera del Queiles VT PGI-ES-A0083 + Total IGPs: 1 + ARAGÓN Total: 11 + DOP Aylés VP PDO-ES-A1522 + DOP Urbezo DO PDO-ES-02585 + IGP Bajo Aragón VT PGI-ES-A1362 +....JL...//.../INFORMES/IIGGs_NEW4 Página 1 de 6 + DOP Lebrija VC PDO-ES-A1478 + DOP Jerez-Xérès-Sherry / Jerez / Xérès / Sherry DO PDO-ES-A1483 + DOP Priorat / Priorato DOCa PDO-ES-A1560 + DOP Abadía Retuerta VP PDO-ES-02481 + DOP El Vicario VP PDO-ES-N1634 + DOP Tharsys VP PDO-ES-02086 + IGP Castelló VT PGI-ES-A1173 + (*) En vinos, la expresión “Denominación de Origen Protegida” (*) Por su parte, en vinos la expresión "Indicación Geográfica + DOCa: “Denominación de Origen Calificada” VT: “Vino de la Tierra” + VP: “Vino de Pago” Click en el nombre conduce a la URL de eAmbrosia de la COM diff --git a/tests/fixtures/it_masaf_elenco_excerpt.txt b/tests/fixtures/it_masaf_elenco_excerpt.txt new file mode 100644 index 0000000..a5692ef --- /dev/null +++ b/tests/fixtures/it_masaf_elenco_excerpt.txt @@ -0,0 +1,21 @@ + ELENCO ALFABETICO DEI VINI DOP ITALIANI + Menzione Regione + DENOMINAZIONE Espressione tradizionale Numero fascicolo +N° o + VINO comunitaria art. 112, lett. a) del Reg. eAmbrosia + (UE) 1308/2013 Provincia Autonoma +1 Abruzzo DOP DOC PDO-IT-A0880 ABRUZZO +2 Aglianico del Taburno DOP DOCG PDO-IT-A0277 CAMPANIA + Bagnoli Friularo +19 DOP DOCG PDO-IT-A0467 VENETO + Friularo di Bagnoli +73 Casauria DOP DOCG PDO-IT-02972 ABRUZZO +96 Cirò DOP DOC PDO-IT-A0610 CALABRIA +97 Cirò Classico DOP DOCG PDO-IT-03209 CALABRIA + FRIULI VENEZIA GIULIA +209 Lison DOP DOCG PDO-IT-A0457 + VENETO + Sforzato di Valtellina +344 DOP DOCG PDO-IT-A1035 LOMBARDIA + Sfursat di Valtellina +388 Valtellina Superiore DOP DOCG PDO-IT-A1036 LOMBARDIA diff --git a/tests/test_batch_usage.py b/tests/test_batch_usage.py new file mode 100644 index 0000000..12d640f --- /dev/null +++ b/tests/test_batch_usage.py @@ -0,0 +1,40 @@ +"""Batch usage accounting and pricing (scripts/_lib/batch.py).""" +from __future__ import annotations + +import json + +from _lib import batch + + +def test_batch_cost_is_half_list_price_with_cache_rates(): + usage = {"input_tokens": 1_000_000, "output_tokens": 100_000, + "cache_creation_input_tokens": 0, "cache_read_input_tokens": 0} + assert batch.batch_cost_usd(usage, "claude-sonnet-5") == round((2.0 + 1.0) * 0.5, 4) + assert batch.batch_cost_usd(usage, "claude-opus-5") == round((5.0 + 2.5) * 0.5, 4) + cached = {**usage, "cache_read_input_tokens": 1_000_000, "cache_creation_input_tokens": 1_000_000} + assert batch.batch_cost_usd(cached, "claude-sonnet-4-6") == round((3.0 + 1.5 + 0.3 + 3.75) * 0.5, 4) + assert batch.batch_cost_usd(usage, "some-unknown-model") is None + + +def test_usage_summary_sums_only_results_with_usage(): + results = { + "a": {"text": "x", "usage": {"input_tokens": 10, "output_tokens": 5}}, + "b": {"text": "y", "usage": {"input_tokens": 20, "output_tokens": 1, "cache_read_input_tokens": 7}}, + "c": {"error": "errored"}, + } + s = batch.usage_summary(results, "claude-sonnet-5") + assert (s["n_results"], s["n_with_usage"]) == (3, 2) + assert (s["input_tokens"], s["output_tokens"], s["cache_read_input_tokens"]) == (30, 6, 7) + assert s["cost_usd"] == batch.batch_cost_usd(s, "claude-sonnet-5") + + +def test_cost_ledger_row(tmp_path, monkeypatch): + ledger = tmp_path / "costs.jsonl" + monkeypatch.setattr(batch, "COSTS_LEDGER", ledger) + monkeypatch.setenv("OWM_TERROIR_RUN", "r-test") + summary = batch.usage_summary({"a": {"text": "", "usage": {"input_tokens": 1, "output_tokens": 1}}}, "claude-opus-5") + batch._record_cost(provider="anthropic", model="claude-opus-5", batch_id="msgbatch_1", + sidecar=tmp_path / "02d-it.json", thinking="adaptive", summary=summary) + row = json.loads(ledger.read_text().splitlines()[0]) + assert row["stage"] == "02d-it" and row["run"] == "r-test" and row["batch_id"] == "msgbatch_1" + assert row["cost_usd"] == summary["cost_usd"] and row["thinking"] == "adaptive" diff --git a/tests/test_content_block.py b/tests/test_content_block.py index 2f6f44b..540b6fb 100644 --- a/tests/test_content_block.py +++ b/tests/test_content_block.py @@ -215,10 +215,10 @@ def test_name_with_latin() -> None: def test_subappellations_section_rendered_when_children_passed() -> None: - rec = {"name": "Muscadet Sèvre et Maine", "kind": "AOC", "country": "fr"} + rec = {"name": "Muscadet Sèvre et Maine", "classification": "AOC", "country": "fr"} kids = [ - {"name": "Clisson", "path": "/en/clisson", "kind": "AOC"}, - {"name": "Le Pallet", "path": "/en/le-pallet", "kind": "AOC"}, + {"name": "Clisson", "path": "/en/clisson", "classification": "AOC"}, + {"name": "Le Pallet", "path": "/en/le-pallet", "classification": "AOC"}, ] out = render_content_block(rec, "muscadet", _ctx("en"), children=kids) # FR heading is the regulator's own term, not the generic UI label. @@ -246,7 +246,7 @@ def test_subappellations_heading_falls_back_to_generic_label() -> None: # A country with no clean regulator term (e.g. CH) uses the translated label. out = render_content_block({"name": "Vaud", "kind": "AOC", "country": "ch"}, "vaud", _ctx("en"), - children=[{"name": "La Côte", "path": "/en/la-cote", "kind": "AOC"}]) + children=[{"name": "La Côte", "path": "/en/la-cote", "classification": "AOC"}]) assert "<h2>Sub-appellations</h2>" in out # _LABELS['entity_nav_children'] diff --git a/tests/test_es_national_term.py b/tests/test_es_national_term.py new file mode 100644 index 0000000..21e9887 --- /dev/null +++ b/tests/test_es_national_term.py @@ -0,0 +1,150 @@ +"""Tests for the ES national traditional-term roster +(`scripts/_lib/es/national_term.py`). + +The fixture is a redacted `pdftotext -layout` excerpt of MAPA's listado — +header lines, region banners, "Total DOPs/IGPs" subtotals, a page-footer +line and the footnote block are all present so the row regex is proven to +skip everything that is not a GI row. +""" +from __future__ import annotations + +import json +import sys +from pathlib import Path + +import pytest + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.es import national_term as nt # noqa: E402 + +PLIEGOS_DIR = nt.ROOT / "raw" / "es" / "pliegos-extracted" + + +@pytest.fixture +def listado_rows(fixture_text): + return nt.parse_listado_text(fixture_text("es_mapa_listado_excerpt.txt")) + + +@pytest.fixture +def roster(monkeypatch, listado_rows): + display = {fn: nt.TERM_DISPLAY[t] for fn, t in listado_rows.items()} + monkeypatch.setattr(nt, "_TERMS_CACHE", display) + return display + + +def test_parse_rows_only(listado_rows): + assert listado_rows == { + "PDO-ES-A0735": "DO", + "PDO-ES-A0117": "DOCa", + "PGI-ES-A0083": "VT", + "PDO-ES-A1522": "VP", + "PDO-ES-02585": "DO", + "PGI-ES-A1362": "VT", + "PDO-ES-A1478": "VC", + "PDO-ES-A1483": "DO", + "PDO-ES-A1560": "DOCa", + "PDO-ES-02481": "VP", + "PDO-ES-N1634": "VP", + "PDO-ES-02086": "VP", + "PGI-ES-A1173": "VT", + } + + +def test_multi_alias_name_does_not_shift_columns(listado_rows): + assert listado_rows["PDO-ES-A1483"] == "DO" + + +def test_conflicting_term_for_one_file_number_raises(): + text = ( + " DOP Foo DO PDO-ES-A9999\n" + " DOP Foo VP PDO-ES-A9999\n" + ) + with pytest.raises(ValueError): + nt.parse_listado_text(text) + + +def test_term_display_is_bottle_wording(): + assert nt.TERM_DISPLAY["VP"] == "Vino de Pago" + assert nt.TERM_DISPLAY["VT"] == "Vino de la Tierra" + assert nt.TERM_DISPLAY["DOCa"] == "DOCa" + + +def test_exact_file_number(roster): + assert nt.es_term_for({"slug": "rioja", "file_number": "PDO-ES-A0117"}, "DOP") == "DOCa" + assert ( + nt.es_term_for({"slug": "ribera-del-queiles", "file_number": "PGI-ES-A0083"}, "IGP") + == "Vino de la Tierra" + ) + + +def test_numeric_tail_bridge(roster): + assert nt.es_term_for({"slug": "x", "file_number": "PDO-ES-1522"}, "DOP") == "Vino de Pago" + assert nt.es_term_for({"slug": "x", "file_number": "PGI-ES-1522"}, "IGP") == "" + + +def test_unknown_file_number_is_empty_not_kind_default(roster): + assert nt.es_term_for({"slug": "x", "file_number": "PDO-ES-A0000"}, "DOP") == "" + assert nt.es_term_for({"slug": "x", "file_number": "PGI-ES-A0000"}, "IGP") == "" + assert nt.es_term_for({"slug": "x"}, "IGP") == "" + + +def test_pgi_never_gets_pdo_term(roster): + with pytest.raises(AssertionError): + nt.es_term_for({"slug": "x", "file_number": "PDO-ES-A0117"}, "IGP") + with pytest.raises(AssertionError): + nt.es_term_for({"slug": "x", "file_number": "PGI-ES-A0083"}, "DOP") + + +def test_overrides_are_cited_and_win(roster): + overrides = nt.load_es_term_overrides() + for slug, entry in overrides.items(): + assert entry["sources"], slug + assert all(s["url"].startswith("https://") for s in entry["sources"]), slug + assert nt.es_term_for({"slug": "priorat", "file_number": "PDO-ES-A1560"}, "DOP") == "DOQ" + assert overrides["priorat"]["castilian_form"] == "DOCa" + assert ( + nt.es_term_for({"slug": "tharsys", "file_number": "PDO-ES-02980"}, "DOP") + == "Vino de Pago" + ) + assert ( + nt.es_term_for({"slug": "urbezo", "file_number": "PDO-ES-02585"}, "DOP") + == "Vino de Pago" + ) + + +def test_missing_pdf_warns_and_returns_empty(monkeypatch, tmp_path, capsys): + monkeypatch.setattr(nt, "_TERMS_CACHE", None) + monkeypatch.setattr(nt, "LISTADO_DIR", tmp_path) + assert nt.load_es_terms() == {} + assert "missing" in capsys.readouterr().err + + +def test_sub_denominations_follow_parent(roster): + records = [ + {"slug": "rioja", "file_number": "PDO-ES-A0117", "kind": "DOP"}, + {"slug": "rioja-rioja-alta", "file_number": "PDO-ES-A0117", "kind": "DOP", + "is_sub_denomination": True, "parent_slug": "rioja"}, + ] + assert nt.check_sub_denomination_terms(records) == [] + records[1]["file_number"] = "PDO-ES-A0735" + assert nt.check_sub_denomination_terms(records) == [ + "rioja-rioja-alta: 'DO' != parent rioja 'DOCa'" + ] + + +@pytest.mark.skipif( + not (PLIEGOS_DIR.exists() and (nt.LISTADO_DIR / nt.LISTADO_FILE).exists()), + reason="raw/ corpus + MAPA listado not present", +) +def test_live_corpus_joins_and_subzonas_agree(monkeypatch): + monkeypatch.setattr(nt, "_TERMS_CACHE", None) + records = [ + json.loads(p.read_text(encoding="utf-8")) + for p in sorted(PLIEGOS_DIR.glob("*.json")) if p.name != "_index.json" + ] + parents = [r for r in records if not r.get("is_sub_denomination")] + unresolved = [r["slug"] for r in parents if not nt.es_term_for(r, r["kind"])] + assert unresolved == [] + assert nt.check_sub_denomination_terms(records) == [] + assert nt.es_term_for(next(r for r in parents if r["slug"] == "rioja"), "DOP") == "DOCa" diff --git a/tests/test_exonyms.py b/tests/test_exonyms.py new file mode 100644 index 0000000..94e4a98 --- /dev/null +++ b/tests/test_exonyms.py @@ -0,0 +1,27 @@ +"""Geographic exonym detection for translated bullets (scripts/_lib/exonyms.py).""" +from __future__ import annotations + +from _lib.exonyms import EXONYMS, exonym_hits, gi_forms_from_names + +GI = gi_forms_from_names(["Toscana", "Bourgogne", "Alsace grand cru Rangen", "Wien", "Côtes du Rhône"]) + + +def test_forms_used_in_gi_names_are_detected(): + assert {"Toscana", "Bourgogne", "Alsace", "Wien", "Rhône"} <= set(GI) + assert "Vosges" not in GI + + +def test_plain_geographic_name_is_flagged_in_its_target_locale(): + assert exonym_hits("De Vosges vormen een scherm tegen oceanische invloeden.", "nl") == ["Vosges"] + assert exonym_hits("The Vosges shelter the vineyard.", "en") == [] # Vosges is the English form + + +def test_gi_homonym_is_flagged_only_as_a_place(): + assert exonym_hits("Hilly terrain in central Toscana, close to the Apennines.", "en", gi_forms=GI) == ["Toscana"] + assert exonym_hits("Reconnu en Toscana IGT en 1995.", "en", gi_forms=GI) == [] + assert exonym_hits("The Melon de Bourgogne variety spread in the sixteenth century.", "en", gi_forms=GI) == [] + assert exonym_hits("Recognised in 2003 by the Junta de Andalucía.", "en") == [] + + +def test_every_table_entry_has_a_valid_locale(): + assert all(set(t) <= {"en", "fr", "es", "nl"} for t in EXONYMS.values()) diff --git a/tests/test_fr_cahier_parser.py b/tests/test_fr_cahier_parser.py index 0e35782..ab38e65 100644 --- a/tests/test_fr_cahier_parser.py +++ b/tests/test_fr_cahier_parser.py @@ -148,3 +148,104 @@ def test_extract_sections_lowercase_chapitre_in_body_does_not_truncate(): bodies, _titles = extract.extract_sections(seg) assert "X" in bodies, "lowercase body 'chapitre' truncated the segment early" assert "coteaux exposés au sud" in bodies["X"] + + +# ----- extract_aire: section IV sub-block headers + sentence-form aires ----- +# +# Pouilly-Loché's 2024 PNOCDC writes "1 - Aire géographique" (no degree +# sign) and defines the aire as a sentence ("… sur le territoire de la +# commune de Mâcon du département de Saône-et-Loire") instead of a +# "Département de X :" list. Before the fix the block split missed, the +# whole section was scanned, and the aire de proximité immédiate list +# (366 Burgundy communes) was recorded as the aire géographique — which +# drew the AOC across all of Burgundy in the simple-mode map. + + +def test_extract_aire_degree_less_header_keeps_proximity_out_of_aire(): + iv = ( + "1 - Aire géographique\n\n" + "La récolte des raisins, la vinification, l’élaboration et l’élevage des vins " + "d’appellation d’origine\ncontrôlée « Pouilly-Loché » sont assurés sur le territoire " + "de la commune de Mâcon du département de\nSaône-et-Loire.\n\n" + "2 - Aire parcellaire délimitée\n\n" + "Les vins sont issus exclusivement de vignes situées dans l’aire parcellaire.\n\n" + "3 - Aire de proximité immédiate\n\n" + "L’aire de proximité immédiate est constituée par le territoire des communes suivantes :\n" + "- Département de la Côte-d’Or : Agencourt, Aloxe-Corton\n" + "- Département du Rhône : Anse, Belleville\n" + ) + aire = extract.extract_aire(iv) + assert aire["aire_geographique"] == {"Saône-et-Loire": ["Mâcon"]} + assert aire["aire_proximite_immediate"] == { + "Côte-d’Or": ["Agencourt", "Aloxe-Corton"], + "Rhône": ["Anse", "Belleville"], + } + + +def test_extract_aire_sentence_form_multi_commune_two_departements(): + iv = ( + "1° - Aire géographique\n\n" + "La récolte est assurée sur le territoire des communes de Chablis,\nPoinchy et Fyé " + "du département de l’Yonne et des communes de Dijon du département de la Côte-d’Or.\n\n" + "2° - Aire parcellaire délimitée\n\nx\n" + ) + aire = extract.extract_aire(iv) + assert aire["aire_geographique"] == { + "Yonne": ["Chablis", "Poinchy", "Fyé"], + "Côte-d’Or": ["Dijon"], + } + assert aire["aire_proximite_immediate"] == {} + + +def test_extract_aire_classic_degree_header_unchanged(): + iv = ( + "1° - Aire géographique\n\n- Département de la Marne : Reims, Épernay\n\n" + "2° - Aire parcellaire délimitée\n\nx\n\n" + "3° - Aire de proximité immédiate\n\n- Département de l’Aube : Troyes\n" + ) + aire = extract.extract_aire(iv) + assert aire["aire_geographique"] == {"Marne": ["Reims", "Épernay"]} + assert aire["aire_proximite_immediate"] == {"Aube": ["Troyes"]} + + +def test_extract_aire_list_form_wins_over_sentence_form_in_same_block(): + # A block that carries a "Département de X :" list must not ALSO pick up + # a stray sentence mention — the sentence fallback only runs when the + # list form found nothing. + iv = ( + "1° - Aire géographique\n\n" + "Les vins proviennent des communes de Nuits du département de la Côte-d’Or.\n" + "- Département de la Côte-d’Or : Nuits-Saint-Georges, Premeaux-Prissey\n\n" + "2° - Aire parcellaire délimitée\n\nx\n" + ) + aire = extract.extract_aire(iv) + assert aire["aire_geographique"] == {"Côte-d’Or": ["Nuits-Saint-Georges", "Premeaux-Prissey"]} + + +def test_aire_block_header_requires_a_marker(): + # "3 communes …" starts with a digit but carries no ° / dash / paren, so + # it must not open a sub-block and split a commune list in two. + import re + + pat = re.compile(extract._AIRE_BLOCK_HEADER_PATTERN, re.MULTILINE) + assert pat.search("1° - Aire géographique") + assert pat.search("1°- Aire parcellaire délimitée") + assert pat.search("1 - Aire géographique") + assert pat.search("2 – Aire parcellaire délimitée") + assert pat.search("1) Aire géographique") + assert not pat.search("3 communes du département") + assert not pat.search("12 - Aire géographique") + + +def test_extract_aire_sentence_form_drops_lowercase_asides(): + iv = ( + "1° Aire géographique :\n\nLa récolte des raisins est assurée sur le territoire de\n" + "la commune de Barsac, sur la base du code officiel géographique en date du 1er janvier 2025, " + "dans le\ndépartement de la Gironde.\n\n2° Aire parcellaire délimitée :\n\nx\n" + ) + assert extract.extract_aire(iv)["aire_geographique"] == {"Gironde": ["Barsac"]} + iv2 = ( + "1°- Aire géographique\n\nLes vins sont assurés sur le territoire de\nla commune de Loupiac, " + "située dans le département de la Gironde.\n\n2°- Aire parcellaire délimitée\n\nx\n" + ) + assert extract.extract_aire(iv2)["aire_geographique"] == {"Gironde": ["Loupiac"]} diff --git a/tests/test_fr_terroir_slicer.py b/tests/test_fr_terroir_slicer.py new file mode 100644 index 0000000..26c90e9 --- /dev/null +++ b/tests/test_fr_terroir_slicer.py @@ -0,0 +1,75 @@ +"""FR stage 02d: a record in a shared cahier is graded against its own chapter.""" +from __future__ import annotations + +import importlib.util +from pathlib import Path + +SCRIPT = Path(__file__).resolve().parents[1] / "scripts" / "02d_extract_terroir_facts.py" +spec = importlib.util.spec_from_file_location("fr02d", SCRIPT) +fr02d = importlib.util.module_from_spec(spec) +spec.loader.exec_module(fr02d) + +PAD = " Texte de remplissage sur les sols et le climat de ce cru, répété pour dépasser le seuil." * 8 + + +def chapter(name: str, body: str) -> str: + return ( + f" « Alsace grand cru {name} »\n\n" + f"1°– Informations sur la zone géographique\n\na) - Description des facteurs naturels\n\n{body}{PAD}\n\n" + f"b) - Description des facteurs humains\n\nVendanges manuelles.{PAD}\n\n" + f"2°– Informations sur la qualité et les caractéristiques des produits\n\nVins blancs secs.{PAD}\n\n" + f"3°– Interactions causales\n\nLe lien au terroir.{PAD}\n\n" + ) + + +LIEN = chapter("Kastelberg", "Schistes de Steige.") + chapter("Rangen", "Sols volcaniques.") + chapter("Zotzenberg", "Marnes denses.") + + +def record(name: str, slug: str, lien: str = LIEN) -> dict: + return {"name": name, "slug": slug, "lien_au_terroir": lien, "source": {}} + + +def test_shared_cahier_job_is_windowed_to_the_own_chapter(): + job = fr02d._job_from_record(record("Alsace grand cru Rangen", "alsace-grand-cru-rangen")) + assert job is not None + assert "Sols volcaniques" in job["lien"] + assert "Marnes denses" not in job["lien"] and "Schistes" not in job["lien"] + assert set(job["slices"]) >= {"facteurs_naturels", "facteurs_humains", "produit", "interactions"} + assert job["lien_sha"] == fr02d.cahier_sha(job["lien"]) + + +def test_shared_cahier_without_own_chapter_is_skipped(capsys): + assert fr02d._job_from_record(record("Alsace grand cru Sporen", "alsace-grand-cru-sporen")) is None + assert "no own chapter" in capsys.readouterr().err + + +def test_single_chapter_cahier_keeps_the_whole_lien(): + single = chapter("Rangen", "Sols volcaniques.") + job = fr02d._job_from_record(record("Alsace grand cru Rangen", "alsace-grand-cru-rangen", single)) + assert job is not None and job["lien"] == single.strip() + + +def _section_x(head: str, a_line: str) -> str: + body = " ".join(["Les sols sont argilo-calcaires sur le versant est du Mâconnais."] * 6) + return ( + f"{head}\n\n{a_line}\n\n{body}\n\n" + f"b) - Description des facteurs humains contribuant au lien\n\n{body}\n\n" + f"2°- Informations sur la qualité et les caractéristiques du produit\n\n{body}\n\n" + f"3°- Interactions causales\n\n{body}\n" + ) + + +def test_slicer_recovers_the_natural_factors_from_the_three_cahier_defects(): + ok = fr02d.slice_section_x(_section_x("1°- Informations sur la zone géographique", "a) - Description des facteurs naturels contribuant au lien")) + assert set(ok) == {"facteurs_naturels", "facteurs_humains", "produit", "interactions"} + # Pouilly-Vinzelles: pdftotext reads "1°" as "l°" + ocr = fr02d.slice_section_x(_section_x("l°- Informations sur la zone géographique", "a) - Description des facteurs naturels contribuant au lien")) + assert ocr["facteurs_naturels"] == ok["facteurs_naturels"].replace("1°", "l°") + # Menetou-Salon: no "1°" heading — the lien opens at a) + no_top = fr02d.slice_section_x(_section_x("", "a) - Description des facteurs naturels contribuant au lien").lstrip()) + assert "facteurs_naturels" in no_top and no_top["facteurs_naturels"].startswith("a) - Description des facteurs naturels") + assert set(no_top) == {"facteurs_naturels", "facteurs_humains", "produit", "interactions"} + # Floc de Gascogne: the a) heading lost its letter + no_a = fr02d.slice_section_x(_section_x("1°- Informations sur la zone géographique", "- Description des facteurs naturels contribuant au lien")) + assert "facteurs_naturels" in no_a and "Les sols sont" in no_a["facteurs_naturels"] + assert no_a["facteurs_humains"].startswith("b)") diff --git a/tests/test_gi_terms.py b/tests/test_gi_terms.py new file mode 100644 index 0000000..99e7e66 --- /dev/null +++ b/tests/test_gi_terms.py @@ -0,0 +1,125 @@ +"""Two naming axes (scripts/_lib/gi_terms.py): scheme derivation, the single +label composer, the facet key, the facet tree, and the SSR meta-line tokens. + +The ruling table (traditional_terms.json) is exercised only through the pure +helpers here; roster joins (MASAF / MAPA) have their own fixture tests.""" +from __future__ import annotations + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib import gi_terms as g # noqa: E402 +from _lib.content_block import RenderCtx, classification_html # noqa: E402 + +LABELS = {"scheme_pdo": "PDO", "scheme_pgi": "PGI", "scheme_spirit_gi": "spirit-drink GI"} +LABELS_FR = {"scheme_pdo": "AOP", "scheme_pgi": "IGP", "scheme_spirit_gi": "IG spiritueux"} + + +def test_scheme_from_siqo_signe_ue_for_france() -> None: + assert g.derive_eu_scheme({"country": "fr", "signe_ue": "AOP"}, "AOC") == "pdo" + assert g.derive_eu_scheme({"country": "fr", "signe_ue": "IGP"}, "IGP") == "pgi" + assert g.derive_eu_scheme({"country": "fr", "signe_ue": "IG"}, "EDV") == "spirit-gi" + + +def test_scheme_falls_back_on_the_derived_kind_when_signe_is_empty() -> None: + assert g.derive_eu_scheme({"country": "fr", "slug": "cote-roannaise"}, "AOC") == "pdo" + assert g.derive_eu_scheme({"country": "fr", "slug": "x"}, "EDV") == "spirit-gi" + + +def test_switzerland_has_no_scheme_and_uk_is_its_own_register() -> None: + assert g.derive_eu_scheme({"country": "ch"}, "AOC") == "none" + assert g.derive_eu_scheme({"country": "gb"}, "DOP") == "uk-pdo" + assert g.derive_eu_scheme({"country": "gb"}, "IGP") == "uk-pgi" + assert g.derive_eu_scheme({"country": "it"}, "DOP") == "pdo" + assert g.derive_eu_scheme({"country": "it"}, "IGP") == "pgi" + + +def test_label_is_term_then_localised_scheme_in_brackets() -> None: + assert g.classification_label("DOQ", "pdo", LABELS) == "DOQ (PDO)" + assert g.classification_label("DOCG", "pdo", LABELS_FR) == "DOCG (AOP)" + assert g.classification_label("AOC", "spirit-gi", LABELS) == "AOC (spirit-drink GI)" + + +def test_label_degrades_to_one_token_never_two_equal_ones() -> None: + assert g.classification_label("AOC", "none", LABELS) == "AOC" + assert g.classification_label("", "pgi", LABELS) == "PGI" + assert g.classification_label("", "uk-pdo", LABELS) == "PDO" + assert g.classification_label("IGP", "pgi", LABELS_FR) == "IGP" + assert g.classification_label("", "none", LABELS) == "" + + +def test_class_key_is_padded_scheme_then_slugified_term() -> None: + assert g.class_key("pdo", "it", "DOCG") == ";pdo;it:docg;" + assert g.class_key("pgi", "es", "Vino de la Tierra") == ";pgi;es:vino-de-la-tierra;" + assert g.class_key("none", "ch", "AOC") == ";none;ch:aoc;" + assert g.class_key("pdo", "de", "") == ";pdo;" + assert g.term_key("mt", "IĠT") == "mt:igt" + + +def test_term_tree_orders_schemes_then_terms_by_count() -> None: + tree, desc = g.build_term_tree({ + ";pdo;it:docg;": 79, ";pdo;it:doc;": 332, ";pdo;": 12, ";pgi;it:igt;": 112, + ";none;ch:aoc;": 30, ";spirit-gi;fr:aoc;": 28, + }) + assert [n["slug"] for n in tree if n["depth"] == 0] == ["pdo", "pgi", "spirit-gi", "none"] + pdo = next(n for n in tree if n["slug"] == "pdo") + assert pdo["count"] == 79 + 332 + 12 + kids = [n["slug"] for n in tree if n["parent"] == "pdo"] + assert kids == ["it:doc", "it:docg"] + assert desc["pdo"] == ["pdo", "it:doc", "it:docg"] + assert desc["it:docg"] == ["it:docg"] + + +def _ctx(terms_info: dict) -> RenderCtx: + return RenderCtx( + locale="en", labels={}, region_labels={}, country_labels={}, country_flag_emoji={}, + grapes_info={}, styles_info={}, style_labels={}, github_new_issue_url="", + terms_info=terms_info, + ) + + +def test_ssr_meta_line_splits_term_and_scheme_into_two_spans() -> None: + info = {"it:docg": {"full": "Denominazione di origine controllata e garantita", "note": "top tier"}, + "pdo": {"full": "Protected Designation of Origin", "note": "EU scheme"}} + rec = {"class_label": "DOCG (PDO)", "national_term": "DOCG", "eu_scheme": "pdo", + "class_key": ";pdo;it:docg;", "country": "it"} + html = classification_html(rec, _ctx(info)) + assert '<span class="gi-term has-info" tabindex="0" data-key="it:docg">' in html + assert '<span class="gi-scheme has-info" tabindex="0" data-key="pdo">' in html + assert "<abbr title=\"Denominazione di origine controllata e garantita — top tier\">DOCG</abbr>" in html + assert html.endswith("(PDO)</abbr></span>") + + +def test_ssr_meta_line_without_definitions_is_plain_text() -> None: + rec = {"class_label": "AOC", "national_term": "AOC", "eu_scheme": "none", + "class_key": ";none;ch:aoc;", "country": "ch"} + assert classification_html(rec, _ctx({})) == '<span class="gi-term" data-key="ch:aoc">AOC</span>' + rec = {"class_label": "PDO", "national_term": "", "eu_scheme": "uk-pdo", "country": "gb"} + assert classification_html(rec, _ctx({})) == '<span class="gi-scheme" data-key="uk-pdo">PDO</span>' + assert classification_html({"class_label": ""}, _ctx({})) == "" + + +def test_table_refuses_a_scheme_abbreviation_in_a_term_slot(tmp_path: Path) -> None: + import json + + import pytest + + bad = tmp_path / "t.json" + bad.write_text(json.dumps({"constants": {"hu": {"DOP": "OEM"}}}), encoding="utf-8") + with pytest.raises(ValueError, match="OEM"): + g.load_terms_table_from(bad) + bad.write_text(json.dumps({"terms": {"si:ZOP": {"scheme": "pdo"}}}), encoding="utf-8") + with pytest.raises(ValueError, match="ZOP"): + g.load_terms_table_from(bad) + ok = tmp_path / "ok.json" + ok.write_text(json.dumps({"constants": {"pt": {"DOP": "DOC", "IGP": "Vinho Regional"}}, + "terms": {"pt:DOC": {"scheme": "pdo"}}}), encoding="utf-8") + assert g.load_terms_table_from(ok)["constants"]["pt"]["DOP"] == "DOC" + + +def test_all_four_axis_fields_ship_in_the_startup_blob() -> None: + from _lib.map_template import STARTUP_AOCS_FIELDS + + assert {"eu_scheme", "national_term", "class_key", "class_label"} <= STARTUP_AOCS_FIELDS diff --git a/tests/test_it_national_term.py b/tests/test_it_national_term.py new file mode 100644 index 0000000..7731606 --- /dev/null +++ b/tests/test_it_national_term.py @@ -0,0 +1,92 @@ +"""Regression tests for the Italian DOC / DOCG / IGT term resolver +(scripts/_lib/it/national_term.py). + +The roster is the MASAF "Elenco alfabetico dei vini DOP italiani" +(pdftotext -layout). The fixture is a redacted excerpt of the 18.03.2026 +build covering the row shapes the parser must survive: a plain row, a +name wrapped over three lines (Bagnoli Friularo), a region wrapped around +the row (Lison), and both file-number spellings (`PDO-IT-A0277` vs the +post-2023 `PDO-IT-02972`). + +Pure-function tests — no pdftotext, no raw/. +""" + +from __future__ import annotations + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +from _lib.it import national_term # noqa: E402 +from _lib.it.national_term import ( # noqa: E402 + file_number_tail, + it_term_for, + load_it_term_overrides, + parse_elenco_text, +) + + +def test_file_number_tail_strips_scheme_country_letter_and_zeros(): + assert file_number_tail("PDO-IT-A1896") == "1896" + assert file_number_tail("PDO-IT-01896") == "1896" + assert file_number_tail("PGI-IT-A0852") == "852" + assert file_number_tail("PDO-IT-A0880") == "880" + assert file_number_tail("") == "" + + +def test_parse_elenco_fixture_rows(fixture_text): + roster = parse_elenco_text(fixture_text("it_masaf_elenco_excerpt.txt")) + assert len(roster) == 9 + assert sum(1 for t in roster.values() if t == "DOCG") == 7 + assert roster["880"] == "DOC" + assert roster["277"] == "DOCG" + # Name wrapped over three lines — term + file number still on one line. + assert roster["467"] == "DOCG" + # Region wrapped around the row. + assert roster["457"] == "DOCG" + # Post-2023 numeric-only file numbers join on the bare tail. + assert roster["2972"] == "DOCG" + assert roster["3209"] == "DOCG" + assert roster["610"] == "DOC" + + +def test_it_term_for_resolves_kind_and_roster(monkeypatch, fixture_text): + roster = parse_elenco_text(fixture_text("it_masaf_elenco_excerpt.txt")) + monkeypatch.setattr(national_term, "load_it_terms", lambda: roster) + monkeypatch.setattr(national_term, "load_it_term_overrides", lambda: {}) + + assert it_term_for({"slug": "toscana", "file_number": "PGI-IT-A1517"}, "IGP") == "IGT" + assert it_term_for({"slug": "abruzzo", "file_number": "PDO-IT-A0880"}, "DOP") == "DOC" + assert it_term_for({"slug": "casauria", "file_number": "PDO-IT-02972"}, "DOP") == "DOCG" + assert it_term_for({"slug": "nowhere", "file_number": "PDO-IT-A9999"}, "DOP") == "" + assert it_term_for({"slug": "abruzzo", "file_number": "PDO-IT-A0880"}, "AOC") == "" + + +def test_sottozona_resolves_through_parent_file_number(monkeypatch, fixture_text): + roster = parse_elenco_text(fixture_text("it_masaf_elenco_excerpt.txt")) + monkeypatch.setattr(national_term, "load_it_terms", lambda: roster) + monkeypatch.setattr(national_term, "load_it_term_overrides", lambda: {}) + sottozona = { + "slug": "valtellina-superiore-sassella", + "file_number": "PDO-IT-A1036", + "is_sub_denomination": True, + "parent_slug": "valtellina-superiore", + } + assert it_term_for(sottozona, "DOP") == "DOCG" + + +def test_override_takes_precedence_over_roster(monkeypatch): + monkeypatch.setattr(national_term, "load_it_terms", lambda: {"1188": "DOCG"}) + assert it_term_for({"slug": "valtenesi", "file_number": "PDO-IT-A1188"}, "DOP") == "DOC" + + +def test_overrides_file_is_well_formed(): + overrides = load_it_term_overrides() + assert {"valtenesi", "ciro-classico"} <= set(overrides) + for slug, entry in overrides.items(): + assert entry["term"] in {"DOC", "DOCG"}, slug + assert file_number_tail(entry["file_number"]), slug + assert entry["sources"], slug + for src in entry["sources"]: + assert src["label"] and src["url"].startswith("http"), slug diff --git a/tests/test_it_parser.py b/tests/test_it_parser.py index 402ecf4..8ec9607 100644 --- a/tests/test_it_parser.py +++ b/tests/test_it_parser.py @@ -46,9 +46,11 @@ from _lib.grape_entity import match_variety # noqa: E402 from _lib.it.masaf import ( # noqa: E402 article2_candidate_phrases, + cap_at_sentence, extract_articles, find_article_offsets, parse_grapes_with, + pick_terroir_article, ) from _lib.it.menzione import extract_menzioni # noqa: E402 from _lib.it.sottozona import extract_sottozone # noqa: E402 @@ -346,3 +348,55 @@ def test_masaf_article2_candidate_strips_percent_and_index(): for p in phrases: assert "%" not in p assert not p[:2].strip().rstrip(".").isdigit() + + +def test_masaf_extract_article_runs_keeps_the_parent_and_lists_annexes(): + # A consolidated disciplinare: TOC, the parent's own articles, then one + # sub-disciplinare per sottozona restarting at Art. 1. The parent's + # Art. 1 / 3 / 9 must come from ITS run, never from the last annex + # (review 2026-09-12: Montepulciano d'Abruzzo took San Martino's). + from _lib.it.masaf import extract_article_runs + body = "x" * 300 + text = ( + "Articolo 1 Denominazione\nArticolo 2 Base\nArticolo 3 Zona\nArticolo 9 Legame\n\n" + f"Articolo 1\nDenominazione e vini\nLa DOC «Parent» {body}\n" + f"Articolo 2\nBase ampelografica\nMontepulciano {body}\n" + f"Articolo 3\nZona di produzione\nParent communes {body}\n" + f"Articolo 9\nLegame con l'ambiente\nParent terroir {body}\n" + "17\nALLEGATO 1\n“PARENT” SOTTOZONA “ALTO TIRINO”\n" + f"Articolo 1\nDenominazione e vini\nLa sottozona Alto Tirino {body}\n" + f"Articolo 3\nZona di produzione\nAlto Tirino communes {body}\n" + f"Articolo 9\nLegame con l'ambiente\nAlto Tirino terroir {body}\n" + "ALLEGATO 2\n“PARENT” SOTTOZONA “TEATE”\n" + f"Articolo 1\nDenominazione e vini\nLa sottozona Teate {body}\n" + f"Articolo 9\nLegame con l'ambiente\nTeate terroir {body}\n" + ) + main, annexes = extract_article_runs(text) + assert sorted(main) == [1, 2, 3, 9] + assert "La DOC «Parent»" in main[1] and "Parent terroir" in main[9] + assert "Alto Tirino" not in main[9] and "Teate" not in main[1] + assert [a["title"] for a in annexes] == [ + "ALLEGATO 1 “PARENT” SOTTOZONA “ALTO TIRINO”", "ALLEGATO 2 “PARENT” SOTTOZONA “TEATE”", + ] + assert "Alto Tirino terroir" in annexes[0]["articles"][9] + assert sorted(annexes[1]["articles"]) == [1, 9] + assert extract_articles(text) == main + + +def test_masaf_terroir_uncapped_for_the_extractor_capped_for_the_panel(): + sentences = ["Il legame con l'ambiente geografico è antico e documentato."] + sentences += [f"La frase numero {i} descrive i suoli e il clima della zona." for i in range(200)] + body = "Legame con l'ambiente geografico\n" + " ".join(sentences) + articles = {1: "Denominazione\nLa denominazione…", 9: body} + + n, full = pick_terroir_article(articles, max_chars=None) + assert n == 9 + assert len(full) > 4000 + assert full.endswith("della zona.") + + n, brief = pick_terroir_article(articles) + assert n == 9 + assert brief == cap_at_sentence(full, 4000) + assert len(brief) <= 4000 and brief.endswith(".") + assert full.startswith(brief[:-1]) + assert cap_at_sentence(full, None) == full diff --git a/tests/test_prompt_cache.py b/tests/test_prompt_cache.py new file mode 100644 index 0000000..213412e --- /dev/null +++ b/tests/test_prompt_cache.py @@ -0,0 +1,77 @@ +"""Prompt-cache block shapes (scripts/_lib/prompt_cache.py) and the batch hash.""" +from __future__ import annotations + +from _lib import batch +from _lib import prompt_cache as pc + + +def test_cached_system_puts_the_shared_block_first(monkeypatch): + monkeypatch.delenv(pc.TTL_ENV, raising=False) + out = pc.cached_system("THE LIEN", "the instructions") + assert out == [ + {"type": "text", "text": "THE LIEN", "cache_control": {"type": "ephemeral"}}, + {"type": "text", "text": "the instructions"}, + ] + assert pc.system_text(out) == "THE LIEN\n\nthe instructions" + assert pc.cached_system("", "the instructions") == "the instructions" + + +def test_ttl_switches(monkeypatch): + monkeypatch.setenv(pc.TTL_ENV, "1h") + assert pc.cache_control() == {"type": "ephemeral", "ttl": "1h"} + assert pc.mark_cached("S")[0]["cache_control"]["ttl"] == "1h" + monkeypatch.setenv(pc.TTL_ENV, "off") + assert pc.cache_control() is None + assert pc.cached_system("A", "B") == "A\n\nB" + assert pc.mark_cached("S") == "S" + + +def test_split_user_lead_formats_both_halves(): + ask, doc = pc.split_user_lead("Sub-section: {label}\n\nText of the spec:\n\n{lien}", label="Soils", lien="Clay.") + assert (ask, doc) == ("Sub-section: Soils", "Text of the spec:\n\nClay.") + + +def test_batch_request_id_is_stable_for_block_systems_and_distinct_from_flat_text(): + blocks = pc.cached_system("A", "B") + assert batch._request_id(blocks, "u") == batch._request_id(list(blocks), "u") + assert batch._request_id(blocks, "u") != batch._request_id("A\n\nB", "u") + p = batch._anthropic_params("claude-sonnet-5", {"system": blocks, "user": "u", "max_tokens": 10}, "disabled") + assert p["system"] is blocks and p["messages"] == [{"role": "user", "content": "u"}] + + +def test_phase_groups_keep_first_appearance_order_and_untagged_requests(): + reqs = [{"custom_id": "a", "phase": "n"}, {"custom_id": "b", "phase": "h"}, {"custom_id": "c", "phase": "n"}, + {"custom_id": "d", "phase": None}] + groups = batch._phase_groups(reqs) + assert [(ph, [r["custom_id"] for r in rs]) for ph, rs in groups] == [("n", ["a", "c"]), ("h", ["b"]), (None, ["d"])] + + +def test_run_phased_submits_one_batch_per_phase_in_order(monkeypatch, tmp_path): + calls = [] + + def fake_run_batch(provider, model, reqs, *, sidecar, poll_interval=0, thinking=None): + calls.append((sidecar.name, [r["custom_id"] for r in reqs])) + return {r["custom_id"]: {"text": "ok"} for r in reqs} + + monkeypatch.setattr(batch, "run_batch", fake_run_batch) + reqs = [{"custom_id": "a", "phase": "n"}, {"custom_id": "b", "phase": "h"}, {"custom_id": "c", "phase": "n"}] + out = batch.run_phased("anthropic", "m", reqs, sidecar=tmp_path / "02d-it.json") + assert calls == [("02d-it.p0.json", ["a", "c"]), ("02d-it.p1.json", ["b"])] + assert set(out) == {"a", "b", "c"} + # a single phase falls through to one batch with the plain sidecar + calls.clear() + batch.run_phased("anthropic", "m", [{"custom_id": "x", "phase": None}], sidecar=tmp_path / "02d-it.json") + assert calls == [("02d-it.json", ["x"])] + + +def test_collecting_provider_records_the_phase_and_phased_lien_ttl(monkeypatch): + c = batch.CollectingProvider() + c.chat(system="s", user="u", cache_phase="facteurs_naturels") + assert c.requests[0]["phase"] == "facteurs_naturels" + monkeypatch.delenv(pc.TTL_ENV, raising=False) + monkeypatch.setenv(batch.PHASED_ENV, "1") + assert pc.cached_system("L", "R", phased=True)[0]["cache_control"] == {"type": "ephemeral", "ttl": "1h"} + assert pc.cached_system("L", "R")[0]["cache_control"] == {"type": "ephemeral"} + monkeypatch.setenv(batch.PHASED_ENV, "0") + assert not batch.phased() + assert pc.cached_system("L", "R", phased=True)[0]["cache_control"] == {"type": "ephemeral"} diff --git a/tests/test_stage_defaults.py b/tests/test_stage_defaults.py new file mode 100644 index 0000000..27f9b67 --- /dev/null +++ b/tests/test_stage_defaults.py @@ -0,0 +1,73 @@ +"""The decided model configuration (2026-09-14): Sonnet 5 extractor with +thinking off, Opus 5 gate / audit with adaptive thinking, Sonnet 4.6 for +translation and the back-check — and that every stage script asks for its +own stage default rather than the generic one.""" +from __future__ import annotations + +import re +from pathlib import Path + +from _lib import batch, providers + +ROOT = Path(__file__).resolve().parents[1] + + +def test_stage_defaults_are_the_decided_configuration(): + assert providers.stage_default("02d") == ("claude-sonnet-5", "disabled") + assert providers.stage_default("gate") == ("claude-opus-5", "adaptive") + assert providers.stage_default("audit") == ("claude-opus-5", "adaptive") + assert providers.stage_default("02e") == ("claude-sonnet-4-6", None) + assert providers.stage_default("backcheck") == ("claude-sonnet-4-6", None) + assert providers.stage_default(None) == (providers.DEFAULT_ANTHROPIC_MODEL, None) + assert batch.default_model("anthropic", "02d") == "claude-sonnet-5" + assert batch.default_thinking("anthropic", "gate") == "adaptive" + assert batch.default_model("mistral", "02d") == "mistral-medium-latest" + assert batch.default_thinking("mistral", "gate") is None + + +def test_every_02d_script_asks_for_the_02d_stage_default(): + paths = [ROOT / "scripts" / "02d_extract_terroir_facts.py"] + sorted(ROOT.glob("scripts/*/02d_extract_terroir_facts.py")) + assert len(paths) == 21 + for p in paths: + src = p.read_text(encoding="utf-8") + assert 'batch.default_model(args.provider, stage="02d")' in src, p + assert 'thinking=batch.default_thinking(args.provider, stage="02d")' in src, p + assert re.search(r'make_provider\(\s*args\.provider, model=args\.model, stage="02d"', src), p + + +def test_gate_backcheck_and_audit_use_their_stage(): + for name, stage in (("02d_verify_terroir_facts.py", "gate"), ("02e_verify_terroir_facts.py", "backcheck"), + ("audit_terroir_facts_llm.py", "audit")): + src = (ROOT / "scripts" / name).read_text(encoding="utf-8") + assert f'stage="{stage}"' in src, name + + +def test_every_02e_script_resolves_its_model_through_the_02e_stage(): + # Without stage="02e" a script silently takes the generic default and a + # change to STAGE_DEFAULTS["02e"] changes nothing (2026-09-15). + scripts = [ROOT / "scripts" / "02e_translate_terroir_facts.py"] + sorted( + (ROOT / "scripts").glob("*/02e_translate_terroir_facts.py")) + assert len(scripts) == 21 + for p in scripts: + src = p.read_text(encoding="utf-8") + assert src.count('stage="02e"') == 2, p + + +def test_anthropic_params_carry_the_thinking_mode(monkeypatch): + monkeypatch.delenv("OWM_BATCH_THINKING", raising=False) + r = {"custom_id": "x", "system": "s", "user": "u", "max_tokens": 10} + assert "thinking" not in batch._anthropic_params("m", r, None) + assert batch._anthropic_params("m", r, "adaptive")["thinking"] == {"type": "adaptive"} + monkeypatch.setenv("OWM_BATCH_THINKING", "disabled") + assert batch._anthropic_params("m", r, "adaptive")["thinking"] == {"type": "disabled"} + + +def test_claude5_models_get_thinking_disabled_when_the_stage_sets_no_mode(monkeypatch): + monkeypatch.delenv("OWM_BATCH_THINKING", raising=False) + assert providers.effective_thinking("claude-sonnet-5", None) == "disabled" + assert providers.effective_thinking("claude-opus-5", "adaptive") == "adaptive" + assert providers.effective_thinking("claude-sonnet-4-6", None) is None + r = {"custom_id": "x", "system": "s", "user": "u", "max_tokens": 10} + assert batch._anthropic_params("claude-sonnet-5", r, None)["thinking"] == {"type": "disabled"} + assert "thinking" not in batch._anthropic_params("claude-sonnet-4-6", r, None) + assert batch._anthropic_params("claude-opus-5", r, "adaptive")["thinking"] == {"type": "adaptive"} diff --git a/tests/test_terroir_audit_checks.py b/tests/test_terroir_audit_checks.py new file mode 100644 index 0000000..0f3857a --- /dev/null +++ b/tests/test_terroir_audit_checks.py @@ -0,0 +1,321 @@ +"""Pure checks of scripts/audit_terroir_facts.py on synthetic input — no raw/ access.""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +import pytest + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "scripts")) + +import audit_terroir_facts as audit # noqa: E402 + +# ───────────────────────────────────────────────── name guard (W2b) ── + + +def test_name_tokens_drop_stop_words_and_short_tokens(): + assert audit.name_tokens("Côtes du Rhône Villages") == ["rhone"] + assert audit.name_tokens("Saint-Pourçain") == ["pourcain"] + assert audit.name_tokens("L'Étoile") == ["etoile"] + assert audit.name_tokens("Grands-Echezeaux") == ["echezeaux"] + assert audit.name_tokens("Alsace grand cru Zotzenberg") == ["alsace", "zotzenberg"] + assert audit.name_tokens("Grand Cru") == [] + + +def _lien(sentence: str) -> str: + lien = (sentence + " ") * 30 + assert len(lien) >= audit.NAME_GUARD_MIN_LIEN + return lien + + +def test_name_guard_passes_when_the_lien_names_the_appellation(): + lien = _lien("Le vignoble de Pierrevert s'étend sur les collines de Haute-Provence.") + assert audit.lien_names_record(lien, "Pierrevert") is True + assert audit.name_guard_finding(lien, "Pierrevert", "pierrevert") is False + + +def test_name_guard_fires_when_the_lien_never_names_it(): + lien = _lien("Le vignoble de Saint-Pourçain s'étend sur les coteaux de l'Allier.") + assert audit.lien_names_record(lien, "Pierrevert") is False + assert audit.name_guard_finding(lien, "Pierrevert", "pierrevert") is True + + +def test_name_guard_folds_accents_and_case(): + lien = _lien("LE VIGNOBLE DE L'ETOILE DOMINE LA PLAINE DE LONS-LE-SAUNIER.") + assert audit.name_guard_finding(lien, "L'Étoile", "l-etoile") is False + + +def test_name_guard_respects_whitelist_short_lien_and_untestable_names(): + lien = _lien("Le vignoble bourguignon s'étend sur des coteaux calcaires exposés à l'est.") + assert audit.name_guard_finding(lien, "Saône-et-Loire", "saone-et-loire") is False + assert audit.name_guard_finding(lien, "Saône-et-Loire", "other-slug") is True + assert audit.name_guard_finding(lien[:400], "Pierrevert", "pierrevert") is False + assert audit.lien_names_record(lien, "Grand Cru") is None + assert audit.name_guard_finding(lien, "Grand Cru", "grand-cru") is False + + +# ────────────────────────────────────────────────── style checks (W5) ── + + +@pytest.mark.parametrize( + "bullet,expected", + [ + ("Cépages : pinot noir N et chardonnay B.", True), + ("Riesling B, gewurztraminer Rs et pinot gris G dominent.", True), + ("Pinot Noir N is the only red variety.", True), + ("Vignoble de Colmar N exposé au sud.", False), + ("Classé en 1936 en catégorie B du référentiel.", False), + ("Sols argilo-calcaires sur marnes du Kimméridgien.", False), + ], +) +def test_colour_code_needs_a_grape_name(bullet, expected): + assert audit.has_colour_code(bullet) is expected + + +def test_label_prefix_is_a_short_non_numeric_lead(): + assert audit.has_label_prefix("Rioja Oriental: climat méditerranéen plus chaud.") + assert audit.has_label_prefix(" Sols: marnes et calcaires.") + assert not audit.has_label_prefix("En 1936: premier classement.") + assert not audit.has_label_prefix("Sols argilo-calcaires sur marnes.") + assert not audit.has_label_prefix( + "Un très long préambule de plus de trente caractères ici: puis le fait." + ) + + +@pytest.mark.parametrize( + "bullet,expected", + [ + ("Sols argilo-calcaires", True), + ("Sols argilo-calcaires.", False), + ("Sols argilo-calcaires… ", False), + ("Vraiment ?", False), + ("Oui !", False), + ("Cité « Marnes »", True), + ("", True), + ], +) +def test_missing_terminal_punctuation(bullet, expected): + assert audit.missing_terminal_punct(bullet) is expected + + +def test_arrow_and_meta_text(): + assert audit.has_arrow("Marnes → vins puissants.") + assert not audit.has_arrow("Marnes, donc vins puissants.") + assert audit.has_meta_text("Slate soils, confirmed by Wikipedia.") + assert audit.has_meta_text("The disciplinare states that yields are capped.") + assert audit.has_meta_text("According to the specification, the area is 5 ha.") + assert not audit.has_meta_text("Slate soils on south-facing slopes.") + # source-language forms + assert audit.has_meta_text("Un'epoca iniziata 70 milioni di anni fa secondo il disciplinare.") + assert audit.has_meta_text("Les sols sont calcaires, selon le cahier des charges.") + assert audit.has_meta_text("Laut Produktspezifikation dominieren Schieferböden.") + assert not audit.has_meta_text("Il disciplinare del 1966 fu il primo in Italia.") # a dated fact about the rules, not a citation + + +def test_style_findings_lists_every_defect_in_order(): + assert audit.style_findings( + "Rioja Oriental: pinot noir N → vins, confirmed by Wikipedia" + ) == ["colour_code", "arrow", "label_prefix", "no_terminal_punct", "meta_text"] + assert audit.style_findings("Sols argilo-calcaires sur marnes.") == [] + + +# ─────────────────────────────────────────────────── non-Latin (W1) ── + + +@pytest.mark.parametrize( + "bullet,expected", + [ + ("Soils of чернозем dominate the plain.", True), + ("Vines on ασβεστόλιθος bedrock.", True), + ("Sols argilo-calcaires à Chablis, Kimméridgien.", False), + ("Ġellewża and Girgentina; Furmint, Hárslevelű, Žametovka.", False), + ("", False), + ], +) +def test_non_latin_detection(bullet, expected): + assert audit.has_non_latin(bullet) is expected + + +# ─────────────────────────────────────────────────── duplicates (W3) ── + + +def fact(bullet, *, cq="", wq="", prov="cahier", sub="facteurs_naturels"): + return {"bullet": bullet, "cahier_quote": cq, "wiki_quote": wq, "provenance": prov, "subsection": sub} + + +def test_intra_record_duplicates_reports_each_pair(): + facts = [ + fact("Marne calcaree argillose favoriscono potenza e longevità; macigno toscano apporta serbevolezza."), + fact("Marne calcaree danno potenza e longevità; macigno toscano conferisce serbevolezza.", sub="produit"), + fact("Clima mediterraneo con estati calde e inverni miti."), + ] + assert audit.intra_record_duplicates(facts) == [{"i": 0, "j": 1, "reason": "similar-bullet"}] + assert audit.intra_record_duplicates(facts[1:]) == [] + + +# ──────────────────────────────────────────── own chapter (W2a) & W4 ── + + +def chapter(name: str, body: str) -> str: + return ( + f" « Alsace grand cru {name} »\n\n" + f"1°– Informations sur la zone géographique\n\na) - Description des facteurs naturels\n\n{body}\n\n" + f"2°– Informations sur la qualité et les caractéristiques des produits\n\nVins blancs.\n\n" + f"3°– Interactions causales\n\nLe lien.\n\n" + ) + + +SHARED_LIEN = ( + chapter("Altenberg de Bergheim", "Marnes et calcaires du Keuper sur une pente forte exposée au sud.") + + chapter("Kastelberg", "Schistes de Steige, sombres et friables, sur un coteau escarpé.") + + chapter("Zotzenberg", "Marnes très denses sur socle calcaire, vignoble en amphithéâtre.") +) + + +def test_own_chapter_findings_flag_quotes_taken_from_another_cru(): + facts = [ + fact("Schistes de Steige.", cq="Schistes de Steige, sombres et friables"), + fact("Marnes denses.", cq="Marnes très denses sur socle calcaire, vignoble en amphithéâtre"), + fact("Vins blancs.", cq=""), + ] + rows = audit.own_chapter_findings(SHARED_LIEN, "Alsace grand cru Kastelberg", facts) + assert [(r["check"], r["index"]) for r in rows] == [("quote_outside_own_chapter", 1)] + assert rows[0]["coverage"] < audit.FUZZY_THRESHOLD + assert audit.own_chapter_findings(SHARED_LIEN, "Kastelberg", facts) == rows + + +def test_own_chapter_findings_when_the_record_has_no_chapter_or_the_lien_is_not_shared(): + facts = [fact("x", cq="Schistes de Steige")] + assert audit.own_chapter_findings(SHARED_LIEN, "Alsace grand cru Rangen", facts) == [ + {"check": "no_own_chapter"} + ] + assert audit.own_chapter_findings(chapter("Rangen", "Roches volcaniques."), "Chablis", facts) == [] + + +def test_wiki_provenance_with_a_cahier_quote(): + facts = [ + fact("a", cq="quote", prov="wiki"), + fact("b", cq="", prov="wiki"), + fact("c", cq="quote", prov="both"), + fact("d", cq=" ", prov="wiki"), + ] + assert audit.wiki_with_cahier_quote(facts) == [0] + + +def test_shared_quote_groups_need_three_records_and_sixty_chars(): + long_quote = "Les sols argilo-calcaires du Kimméridgien confèrent minéralité et tension aux vins" + short_quote = "Sols argilo-calcaires du Kimméridgien" + by_slug = { + "a": [fact("x", cq=long_quote), fact("y", cq=short_quote)], + "b": [fact("x", cq=" " + long_quote.upper() + " ")], + "c": [fact("x", cq=long_quote), fact("z", cq=long_quote)], + "d": [fact("x", cq=short_quote)], + "e": [fact("x", cq=short_quote)], + } + groups = audit.shared_quote_groups(by_slug) + assert groups == [{"quote": long_quote.lower(), "count": 3, "slugs": ["a", "b", "c"]}] + assert audit.shared_quote_groups({"a": by_slug["a"], "b": by_slug["b"]}) == [] + + +def test_own_chapter_check_skips_wiki_grounded_facts(): + lien = ( + " « Alsace grand cru Rangen »\n\n1°– Zone\n\nSols volcaniques du Rangen." + " x" * 40 + "\n\n" + " « Alsace grand cru Zotzenberg »\n\n1°– Zone\n\nMarnes denses du Zotzenberg." + " y" * 40 + "\n\n" + ) + facts = [ + {"cahier_quote": "Marnes denses du Zotzenberg", "provenance": "cahier"}, + {"cahier_quote": "Marnes denses du Zotzenberg", "provenance": "wiki"}, + {"cahier_quote": "Sols volcaniques du Rangen", "provenance": "both"}, + ] + rows = audit.own_chapter_findings(lien, "Alsace grand cru Rangen", facts) + assert [r["index"] for r in rows] == [0] + + +def test_greek_chemical_prefix_is_not_non_latin(): + assert not audit.has_non_latin("A bitter, resinous note of α-terpineol.") + assert audit.has_non_latin("Terraces (πεζούλες) up to 900 m.") + + +# ───────────────────────────────────────── review 2026-09-12 additions (R9) ── + + +def test_name_guard_requires_the_whole_name_or_a_long_token(): + from audit_terroir_facts import lien_names_record + beaujolais = "Le vignoble du Beaujolais s'étend sur tout le territoire; les grains sont petits. " * 20 + assert lien_names_record(beaujolais, "Bourgogne Passe-tout-grains") is False # "tout" / "grains" no longer enough + assert lien_names_record("… l'appellation Bourgogne Passe-tout-grains … " * 5, "Bourgogne Passe-tout-grains") is True + assert lien_names_record("… les vins de Bourgogne … " * 5, "Bourgogne Passe-tout-grains") is True # longest token ≥ 6 + assert lien_names_record("… le cru de Volnay … " * 5, "Volnay") is True + assert lien_names_record("… " * 50, "Volnay") is False + assert lien_names_record("x", "AOC de la") is None + + +def test_foreign_name_guard_flags_a_pasted_or_mis_bound_text(): + from audit_terroir_facts import foreign_names + text = ("Η ζώνη ΠΓΕ Παρνασσός καλύπτει τους δήμους … Παρνασσός … Παρνασσός … Παρνασσός … Παρνασσός … " + "λοιπά " * 300) + from audit_terroir_facts import name_tokens + others = {s: " ".join(name_tokens(n)) for s, n in + (("parnassos", "Παρνασσός"), ("fthiotida", "Φθιώτιδα"), ("attiki", "Αττική"))} + hits = foreign_names(text, "Φθιώτιδα", others) + assert [h["other"] for h in hits] == ["parnassos"] and hits[0]["hits"] >= 5 + assert foreign_names("Η ζώνη Φθιώτιδα … " + "Παρνασσός " * 6 + "x " * 500, "Φθιώτιδα", others) == [] + # a name contained in the record's own name is not foreign (Chianti in Chianti Classico) + assert foreign_names("Chianti Chianti Chianti " + "x " * 500, "Chianti Classico", {"chianti": "chianti"}) == [] + + +def test_multi_sentence_check(): + from audit_terroir_facts import has_multiple_sentences + assert has_multiple_sentences("The soils are clay. The climate is dry.") + assert not has_multiple_sentences("The wine reaches 13.5 % vol. in warm years.") + assert not has_multiple_sentences("Yields are capped at 60 hl/ha.") + + +def test_identical_en_groups(): + from audit_terroir_facts import identical_en_groups + groups = identical_en_groups({ + "a": ["Grand cru wines are aged for eighteen months in oak.", "unique"], + "b": ["Grand cru wines are aged for eighteen months in oak."], + "c": ["Short."], + }) + assert len(groups) == 1 and groups[0]["slugs"] == ["a", "b"] + + +def test_wiki_binding_check_reads_the_cached_article(tmp_path, monkeypatch): + import audit_terroir_facts as a + monkeypatch.setattr(a, "WIKI_AOCS", tmp_path) + (tmp_path / "de").mkdir() + (tmp_path / "de" / "tirol.json").write_text(json.dumps({"page_url": "https://de.wikipedia.org/wiki/Toro_(Weinbaugebiet)", "revision": "1"}), encoding="utf-8") + (tmp_path / "de" / "wachau.json").write_text(json.dumps({"page_url": "https://de.wikipedia.org/wiki/Wachau_(Weinbaugebiet)", "revision": "1"}), encoding="utf-8") + (tmp_path / "de" / "pinned.json").write_text(json.dumps({"page_url": "https://de.wikipedia.org/wiki/Anything", "override_source": "curator"}), encoding="utf-8") + (tmp_path / "el").mkdir() + (tmp_path / "el" / "samos.json").write_text(json.dumps({"page_url": "https://el.wikipedia.org/wiki/%CE%A3%CE%AC%CE%BC%CE%BF%CF%82", "revision": "1"}), encoding="utf-8") + assert a.wiki_binding_finding("at", "de", "tirol", "Tirol") == {"title": "Toro (Weinbaugebiet)", "lang": "de"} + assert a.wiki_binding_finding("at", "de", "wachau", "Wachau") is None + assert a.wiki_binding_finding("at", "de", "pinned", "Whatever") is None # curator pins are trusted + assert a.wiki_binding_finding("gr", "el", "samos", "Σάμος") is None # percent-encoded Greek title + assert a.wiki_binding_finding("gr", "el", "missing-file", "X") is None + + +def test_coverage_is_block_aware_across_a_pdftotext_artefact(): + from _lib.terroir_coverage import fuzzy_coverage + source = ("L'indice termico di Winkler, ossia la temperatura media attiva nel periodo aprile-ottobre, è compreso\n" + "tra 1.800 e 2.200 gradi- giorno, condizioni che garantiscono la maturazione ottimale del vitigno\nMontepulciano.") + quote = ("L'indice termico di Winkler, ossia la temperatura media attiva nel periodo aprile-ottobre, è compreso " + "tra 1.800 e 2.200 gradi-giorno, condizioni che garantiscono la maturazione ottimale del vitigno Montepulciano") + assert fuzzy_coverage(quote, source) >= 0.95 # one artefact ("gradi- giorno") no longer halves it + assert fuzzy_coverage("temperatura media attiva nel periodo aprile-ottobre", source) == 1.0 + assert fuzzy_coverage("clima temperato, suoli argillosi profondi, vigneti a 300 m", source) < 0.3 # scattered + assert fuzzy_coverage("Bordeaux gravel terraces beside the Gironde estuary", source) < 0.3 # foreign + + +def test_translation_stale_flags_an_outdated_key_but_not_pending(tmp_path, monkeypatch): + from _lib.terroir_dedupe import facts_sha + monkeypatch.setattr(audit, "TRANSLATIONS", tmp_path) + src = [{"bullet": "Les sols sont calcaires."}] + for lang, key in (("en", facts_sha(src)), ("es", "0" * 64), ("nl", "pending:" + "0" * 64)): + (tmp_path / lang).mkdir() + (tmp_path / lang / "x.json").write_text(json.dumps({"source_facts_sha": key, "facts": [{"bullet": "Soils are limestone."}]})) + _, rows, _ = audit.audit_translations("x", src) + assert [(r["lang"]) for r in rows if r["check"] == "translation_stale"] == ["es"] diff --git a/tests/test_terroir_backup.py b/tests/test_terroir_backup.py new file mode 100644 index 0000000..aff472d --- /dev/null +++ b/tests/test_terroir_backup.py @@ -0,0 +1,121 @@ +"""Per-run backups of the terroir-fact caches (scripts/_lib/terroir_backup.py), +the rollback (scripts/rollback_terroir_facts.py) and the wiring lint: every +02d / 02e script and post-pass writes the caches through the snapshotting +helpers, never through `cache.write_json` directly.""" +from __future__ import annotations + +import json +import re +from pathlib import Path + +import pytest +from _lib import terroir_backup as tb +from _lib import terroir_cache as tc + +ROOT = Path(__file__).resolve().parents[1] + + +@pytest.fixture +def sandbox(tmp_path, monkeypatch): + terroir = tmp_path / "terroir-facts" + trans = tmp_path / "translations" + backup = tmp_path / "backup" + for mod in (tb, tc): + monkeypatch.setattr(mod, "TERROIR", terroir) + monkeypatch.setattr(mod, "TRANSLATIONS", trans) + monkeypatch.setattr(tb, "BACKUP_ROOT", backup) + monkeypatch.setattr(tb, "ROOT", tmp_path) + monkeypatch.setenv(tb.RUN_ENV, "run-A") + monkeypatch.setattr(tb, "_run_id", None) + monkeypatch.setattr(tb, "_done", set()) + terroir.mkdir() + (terroir / "chablis.json").write_text(json.dumps({"slug": "chablis", "facts": [{"bullet": "old"}]}), encoding="utf-8") + (trans / "en").mkdir(parents=True) + (trans / "en" / "chablis.json").write_text(json.dumps({"facts": [{"bullet": "old-en"}]}), encoding="utf-8") + return tmp_path, terroir, trans, backup + + +def test_first_write_snapshots_source_and_translations_once(sandbox): + tmp, terroir, trans, backup = sandbox + tc.write_source_cache(terroir / "chablis.json", {"slug": "chablis", "facts": [{"bullet": "new"}]}) + tc.write_source_cache(terroir / "chablis.json", {"slug": "chablis", "facts": [{"bullet": "newer"}]}) + tc.write_translation_cache(trans / "en" / "chablis.json", {"facts": [{"bullet": "new-en"}]}) + tc.write_translation_cache(trans / "es" / "chablis.json", {"facts": [{"bullet": "new-es"}]}) + snap = json.loads((backup / "run-A" / "source" / "chablis.json").read_text(encoding="utf-8")) + assert snap["facts"][0]["bullet"] == "old" # the pre-run state, not the intermediate + assert json.loads((backup / "run-A" / "translations" / "en" / "chablis.json").read_text())["facts"][0]["bullet"] == "old-en" + assert not (backup / "run-A" / "translations" / "es").exists() # es did not exist before the run + m = tb.load_manifest("run-A") + assert m["slugs"]["chablis"] == {**m["slugs"]["chablis"], "source": True, "translations": ["en"]} + assert json.loads((terroir / "chablis.json").read_text())["facts"][0]["bullet"] == "newer" + + +def test_new_slug_is_recorded_as_created(sandbox): + tmp, terroir, trans, backup = sandbox + tc.write_source_cache(terroir / "new-aoc.json", {"slug": "new-aoc", "facts": []}) + m = tb.load_manifest("run-A") + assert m["slugs"]["new-aoc"]["source"] is False and m["slugs"]["new-aoc"]["translations"] == [] + assert not (backup / "run-A" / "source" / "new-aoc.json").exists() + + +def test_restore_puts_back_overwritten_and_deletes_created(sandbox): + tmp, terroir, trans, backup = sandbox + tc.write_source_cache(terroir / "chablis.json", {"slug": "chablis", "facts": [{"bullet": "new"}]}) + tc.write_translation_cache(trans / "es" / "chablis.json", {"facts": [{"bullet": "new-es"}]}) + tc.write_source_cache(terroir / "new-aoc.json", {"slug": "new-aoc", "facts": []}) + dry = tb.restore_slug("chablis", "run-A", dry_run=True) + assert json.loads((terroir / "chablis.json").read_text())["facts"][0]["bullet"] == "new" + assert any(p.endswith("terroir-facts/chablis.json") for p in dry["restored"]) + assert any(p.endswith("es/chablis.json") for p in dry["deleted"]) + r = tb.restore_slug("chablis", "run-A") + assert json.loads((terroir / "chablis.json").read_text())["facts"][0]["bullet"] == "old" + assert json.loads((trans / "en" / "chablis.json").read_text())["facts"][0]["bullet"] == "old-en" + assert not (trans / "es" / "chablis.json").exists() + assert r["missing"] is False + r2 = tb.restore_slug("new-aoc", "run-A") + assert not (terroir / "new-aoc.json").exists() and r2["deleted"] + assert tb.restore_slug("never-touched", "run-A")["missing"] is True + + +def test_run_id_comes_from_the_environment(sandbox, monkeypatch): + assert tb.run_id() == "run-A" + monkeypatch.setattr(tb, "_run_id", None) + monkeypatch.delenv(tb.RUN_ENV) + assert re.fullmatch(r"\d{4}-\d{2}-\d{2}T\d{6}", tb.run_id()) + + +# ───────────────────────────────────────────────────────── wiring lint ── + +STAGE_02D = [ROOT / "scripts" / "02d_extract_terroir_facts.py"] + sorted(ROOT.glob("scripts/*/02d_extract_terroir_facts.py")) +STAGE_02E = [ROOT / "scripts" / "02e_translate_terroir_facts.py"] + sorted(ROOT.glob("scripts/*/02e_translate_terroir_facts.py")) +POST_PASSES = [ROOT / "scripts" / n for n in ( + "dedupe_terroir_facts.py", "normalize_terroir_facts.py", "filter_terroir_boilerplate.py", + "recompute_terroir_provenance.py", "02d_verify_terroir_facts.py", "02e_verify_terroir_facts.py", +)] +_DIRECT_SOURCE_WRITE = re.compile(r"cache\.write_json\(\s*(CACHE_DIR\s*/|cache_path\(slug\)|p,|paths\[)") +_DIRECT_TRANSLATION_WRITE = re.compile(r"cache\.write_json\(\s*(cache_path\(lang, slug\)|tp,)") + + +def test_every_02d_writes_through_the_backup_helper(): + assert len(STAGE_02D) == 21 + for path in STAGE_02D: + src = path.read_text(encoding="utf-8") + assert "write_source_cache" in src, path + assert not _DIRECT_SOURCE_WRITE.search(src), f"{path}: writes the source cache directly" + + +def test_every_02e_writes_through_the_backup_helper(): + assert len(STAGE_02E) == 21 + for path in STAGE_02E: + src = path.read_text(encoding="utf-8") + assert "write_translation_cache" in src, path + assert not _DIRECT_TRANSLATION_WRITE.search(src), f"{path}: writes the translation cache directly" + + +def test_post_passes_write_through_the_backup_helpers(): + for path in POST_PASSES: + if not path.exists(): + continue + src = path.read_text(encoding="utf-8") + assert not _DIRECT_SOURCE_WRITE.search(src), f"{path}: writes the source cache directly" + assert not _DIRECT_TRANSLATION_WRITE.search(src), f"{path}: writes a translation cache directly" diff --git a/tests/test_terroir_boilerplate.py b/tests/test_terroir_boilerplate.py new file mode 100644 index 0000000..5efa094 --- /dev/null +++ b/tests/test_terroir_boilerplate.py @@ -0,0 +1,57 @@ +"""Boilerplate-fact detector (scripts/_lib/terroir_boilerplate.py).""" +from __future__ import annotations + +from _lib.terroir_boilerplate import find_boilerplate, is_tautology + +TAUT = ("Η μοναδικότητα των οίνων ΠΓΕ Χαλκιδική οφείλεται στα ιδιαίτερα χαρακτηριστικά της περιοχής " + "(έδαφος, κλίμα, επίδραση ανέμων)") +REAL = ("Το αμπέλι και το κρασί είναι άρρηκτα δεμένα με την πολιτιστική, κοινωνική και οικονομική ζωή " + "των ανθρώπων της περιοχής από την αρχαιότητα") + + +def cache(slug, facts, country="gr", lang="el"): + return {"slug": slug, "country": country, "source_lang": lang, + "facts": [{"bullet": b, "cahier_quote": q} for b, q in facts]} + + +def test_tautology_patterns_are_per_language(): + assert is_tautology(TAUT.casefold(), "el") + assert not is_tautology(TAUT.casefold(), "it") + assert not is_tautology(REAL.casefold(), "el") + + +def test_shared_tautology_is_dropped_but_shared_real_fact_is_kept(): + caches = { + f"pgi-{i}": cache(f"pgi-{i}", [("Ανάγλυφο και εδάφη.", REAL), ("Μοναδικότητα.", TAUT)]) + for i in range(3) + } + assert find_boilerplate(caches) == {"pgi-0": [1], "pgi-1": [1], "pgi-2": [1]} + + +def test_tautology_in_fewer_than_three_records_is_kept(): + caches = {f"pgi-{i}": cache(f"pgi-{i}", [("A.", REAL), ("B.", TAUT)]) for i in range(2)} + assert find_boilerplate(caches) == {} + + +def test_only_fact_is_never_dropped(): + caches = {f"pgi-{i}": cache(f"pgi-{i}", [("B.", TAUT)]) for i in range(3)} + assert find_boilerplate(caches) == {} + + +def test_country_partition_and_language_match(): + caches = {f"x-{i}": cache(f"x-{i}", [("A.", REAL), ("B.", TAUT)], country="cy", lang="el") for i in range(2)} + caches.update({f"y-{i}": cache(f"y-{i}", [("A.", REAL), ("B.", TAUT)], country="gr", lang="el") for i in range(2)}) + assert find_boilerplate(caches) == {} # 2 + 2, never 3 in one country + + +def test_embedded_appellation_name_and_span_length_do_not_break_the_grouping(): + caches = {} + for i, name in enumerate(("Χαλκιδική", "Κοζάνη", "Σιθωνία")): + q = TAUT.replace("Χαλκιδική", name) + " σε συνδυασμό" * i + caches[name] = cache(name, [("A.", REAL), ("B.", q)]) + assert find_boilerplate(caches) == {n: [1] for n in caches} + + +def test_bullet_with_a_number_is_never_boilerplate(): + caches = {f"pgi-{i}": cache(f"pgi-{i}", [("A.", REAL), ("Άνεμοι 25,5 °C.", TAUT)]) for i in range(3)} + assert find_boilerplate(caches) == {} diff --git a/tests/test_terroir_chapters.py b/tests/test_terroir_chapters.py new file mode 100644 index 0000000..bbdbeba --- /dev/null +++ b/tests/test_terroir_chapters.py @@ -0,0 +1,50 @@ +"""Own-chapter windows in a shared cahier (scripts/_lib/terroir_chapters.py).""" +from __future__ import annotations + +from _lib.terroir_chapters import chapter_windows, is_shared, own_chapter + + +def chapter(name: str, body: str) -> str: + return ( + f" « Alsace grand cru {name} »\n\n" + f"1°– Informations sur la zone géographique\n\na) - Description des facteurs naturels\n\n{body}\n\n" + f"2°– Informations sur la qualité et les caractéristiques des produits\n\nVins blancs.\n\n" + f"3°– Interactions causales\n\nLe lien.\n\n" + ) + + +LIEN = ( + "Préambule commun mentionnant « Alsace grand cru Zotzenberg » en passant.\n\n" + + chapter("Altenberg de Bergheim", "Marnes et calcaires.") + + chapter("Kastelberg", "Schistes de Steige.") + + chapter("Zotzenberg", "Marnes très denses sur socle calcaire.") +) + + +def test_windows_are_found_in_order_and_ignore_passing_mentions(): + names = [n for n, _, _ in chapter_windows(LIEN)] + assert names == ["Altenberg de Bergheim", "Kastelberg", "Zotzenberg"] + assert is_shared(LIEN) + + +def test_own_chapter_matches_the_record_name_accent_and_case_folded(): + s, e = own_chapter(LIEN, "Alsace grand cru KASTELBERG") + assert "Schistes de Steige" in LIEN[s:e] + assert "Zotzenberg" not in LIEN[s:e] + s2, e2 = own_chapter(LIEN, "Altenberg de Bergheim") + assert "Marnes et calcaires" in LIEN[s2:e2] and "Kastelberg" not in LIEN[s2:e2] + + +def test_last_chapter_runs_to_the_end(): + s, e = own_chapter(LIEN, "Alsace grand cru Zotzenberg") + assert e == len(LIEN) and "socle calcaire" in LIEN[s:e] + + +def test_missing_own_chapter_is_none_not_a_fallback(): + assert own_chapter(LIEN, "Alsace grand cru Rangen") is None + + +def test_single_chapter_lien_is_not_shared(): + single = chapter("Rangen", "Volcanique.") + assert not is_shared(single) + assert own_chapter(single, "Alsace grand cru Rangen") is None diff --git a/tests/test_terroir_coverage.py b/tests/test_terroir_coverage.py new file mode 100644 index 0000000..d91b93d --- /dev/null +++ b/tests/test_terroir_coverage.py @@ -0,0 +1,108 @@ +"""Ellipsis-aware grounding coverage (scripts/_lib/terroir_coverage.py).""" +from __future__ import annotations + +import pytest +from _lib.terroir_coverage import ( + FUZZY_THRESHOLD, + SourceMatcher, + fuzzy_coverage, + normalize, + provenance_for, + split_spans, +) + +SOURCE = ( + "Le climat est semi-continental à influence océanique. Les précipitations " + "annuelles sont d'environ 600 millimètres. Les vignes sont plantées sur des " + "sols argilo-calcaires du Kimméridgien, en coteaux exposés au sud." +) + + +def test_verbatim_quote_is_fully_covered(): + assert fuzzy_coverage("Le climat est semi-continental à influence océanique.", SOURCE) == 1.0 + + +def test_normalisation_ignores_case_and_whitespace(): + assert fuzzy_coverage("LE CLIMAT est semi-continental", SOURCE) == 1.0 + + +def test_absent_quote_scores_low(): + assert fuzzy_coverage("Les vendanges ont lieu en octobre sous la neige.", SOURCE) < FUZZY_THRESHOLD + + +def test_empty_quote_is_zero(): + assert fuzzy_coverage("", SOURCE) == 0.0 + assert fuzzy_coverage(" ", SOURCE) == 0.0 + + +@pytest.mark.parametrize("join", ["[…]", "[...]", "…", "...", "(…)", "(...)"]) +def test_ellipsis_joined_spans_ground_when_every_span_grounds(join): + quote = f"Le climat est semi-continental à influence océanique {join} sols argilo-calcaires du Kimméridgien" + # A single contiguous match covers only the longer span (< 0.6); span-wise both are verbatim. + assert SourceMatcher(SOURCE).contiguous(normalize(quote)) < FUZZY_THRESHOLD + assert fuzzy_coverage(quote, SOURCE) == 1.0 + + +def test_ellipsis_quote_with_one_ungrounded_span_keeps_whole_quote_grade(): + quote = "Le climat est semi-continental à influence océanique […] vendanges en octobre sous la neige" + whole = SourceMatcher(SOURCE).contiguous(normalize(quote)) + assert fuzzy_coverage(quote, SOURCE) == whole + assert fuzzy_coverage(quote, SOURCE) < FUZZY_THRESHOLD + + +def test_short_spans_are_not_graded_on_their_own(): + quote = "Le climat est semi-continental à influence océanique […] neige" + assert fuzzy_coverage(quote, SOURCE) == 1.0 + + +def test_split_spans_drops_empty_pieces(): + assert split_spans("a long enough first span […] […] second span here …") == [ + "a long enough first span", + "second span here", + ] + + +def test_source_matcher_reuses_index_across_quotes(): + m = SourceMatcher(SOURCE) + assert m.coverage("Les précipitations annuelles sont d'environ 600 millimètres.") == 1.0 + assert m.coverage("coteaux exposés au sud") == 1.0 + assert m.coverage("") == 0.0 + + +@pytest.mark.parametrize( + "cahier,wiki,expected", + [(1.0, 1.0, "both"), (0.6, 0.2, "cahier"), (0.59, 0.6, "wiki"), (0.5, 0.5, None)], +) +def test_provenance_for(cahier, wiki, expected): + assert provenance_for(cahier, wiki) == expected + + +@pytest.mark.parametrize( + "quote", + [ + "Les précipitations annuelles sont d’environ 600 millimètres.", # curly apostrophe + "Les précipitations annuelles sont d'environ 600 millimètres.", + "sols argilo‑calcaires du Kimméridgien", # non-breaking hyphen + "sols argilo–calcaires du Kimméridgien", # en dash + "sols argilo—calcaires du Kimméridgien", # em dash + ], +) +def test_typography_variants_of_a_verbatim_quote_match_fully(quote): + assert fuzzy_coverage(quote, SOURCE) == 1.0 + + +def test_soft_hyphen_bullet_glyph_and_zero_width_characters_are_dropped(): + source = "\u00ad De bodem bestaat uit l\u200bössleem op een kalkrijke ondergrond." + assert fuzzy_coverage("De bodem bestaat uit lössleem op een kalkrijke ondergrond.", source) == 1.0 + + +def test_hyphenated_line_break_in_the_source_is_closed_up(): + source = "La temperatura media è di 1.900 gradi- giorno nel periodo aprile- ottobre sul versante sud." + assert fuzzy_coverage("1.900 gradi-giorno nel periodo aprile-ottobre", source) == 1.0 + assert fuzzy_coverage("fra 300 - 400 m", "vigneti fra 300 - 400 m") == 1.0 # spaced dash untouched + + +def test_low_nine_and_guillemet_quotes_fold_to_straight(): + source = 'Die g.U. „Rosalia" liegt am Osthang; le « terroir » y est calcaire.' + assert fuzzy_coverage('Die g.U. "Rosalia" liegt am Osthang', source) == 1.0 + assert fuzzy_coverage('le "terroir" y est calcaire', source) == 1.0 diff --git a/tests/test_terroir_dedupe.py b/tests/test_terroir_dedupe.py new file mode 100644 index 0000000..c96a653 --- /dev/null +++ b/tests/test_terroir_dedupe.py @@ -0,0 +1,116 @@ +"""Intra-record fact de-duplication (scripts/_lib/terroir_dedupe.py).""" +from __future__ import annotations + +from _lib.terroir_dedupe import dedupe_facts, duplicate_reason, facts_sha + +QUOTE = "I terreni argillosi delle marne calcaree producono vini di potenza e longevità, mentre il macigno" + + +def fact(bullet, *, cq="", wq="", prov="cahier", sub="facteurs_naturels"): + return { + "bullet": bullet, "cahier_quote": cq, "wiki_quote": wq, + "provenance": prov, "subsection": sub, + } + + +def test_near_identical_bullets_collapse_to_the_earlier_one(): + facts = [ + fact("Marne calcaree argillose favoriscono potenza e longevità; macigno toscano apporta serbevolezza."), + fact("Marne calcaree danno potenza e longevità; macigno toscano conferisce serbevolezza.", sub="produit"), + ] + res = dedupe_facts(facts) + assert res.kept_indices == [0] + assert res.drops[0]["reason"] == "similar-bullet" + assert res.drops[0]["dropped_index"] == 1 and res.drops[0]["kept_index"] == 0 + + +def test_same_quote_with_substantial_bullet_overlap_is_a_duplicate(): + facts = [ + fact("Hlinitá až ílovito-hlinitá pôda dodáva vínam vyššiu mineralitu; priemerný bezcukorný extrakt dosahuje až 19,0 g/l.", cq=QUOTE), + fact("Černozemné ílovito-hlinité pôdy zvyšujú mineralitu vína; bezcukorný extrakt dosahuje až 19,0 g/l.", cq=QUOTE, sub="interactions"), + ] + res = dedupe_facts(facts) + assert res.kept_indices == [0] + assert res.drops[0]["reason"] == "same-quote" + + +def test_same_quote_but_different_facts_are_both_kept(): + facts = [ + fact("Origine geologica alluvionale: suoli di medio impasto tendenti all'argilloso.", cq=QUOTE), + fact("Disponibilità idrica garantita dal fiume Panaro e da acqua di falda nel sottosuolo.", cq=QUOTE), + ] + assert dedupe_facts(facts).kept_indices == [0, 1] + + +def test_bullets_with_different_numbers_are_never_merged(): + a = fact("Rendement maximal fixé à 50 hl/ha pour les vins rouges de l'appellation.", cq=QUOTE) + b = fact("Rendement maximal fixé à 60 hl/ha pour les vins rouges de l'appellation.", cq=QUOTE) + assert duplicate_reason(a, b) is None + assert dedupe_facts([a, b]).kept_indices == [0, 1] + + +def test_more_informative_superset_wins_and_takes_the_earlier_slot(): + a = fact("As condições climáticas favorecem a síntese de açúcares e a concentração de matérias corantes.") + b = fact("As 3000 horas de sol anuais favorecem a síntese de açúcares e a concentração de matérias corantes.", sub="interactions") + res = dedupe_facts([a, b]) + assert res.kept == [b] + assert res.kept_indices == [1] + assert res.drops[0]["dropped_index"] == 0 and res.drops[0]["kept_index"] == 1 + + +def test_both_provenance_beats_cahier_only(): + a = fact("Vulkanische Böden bedingen mineralischen Geruch der Weine, beschrieben als Geruch nach nassem Stein.", cq=QUOTE) + b = fact("Vulkanböden prägen mineralischen Geruch der Weine, beschrieben als Geruch nach nassem Stein.", cq=QUOTE, wq="x" * 40, prov="both") + res = dedupe_facts([a, b]) + assert res.kept == [b] + + +def test_short_quotes_do_not_count_as_shared(): + a = fact("Suoli vulcanici di origine piroclastica nell'area del Vulture con elevata fertilità.", cq="short quote") + b = fact("Suoli vulcanici piroclastici al Vulture conferiscono fertilità mentre le argille danno struttura.", cq="short quote") + assert dedupe_facts([a, b]).kept_indices == [0, 1] + + +def test_unchanged_record_reports_no_drops(): + facts = [fact("Un fait."), fact("Un autre fait, sans rapport avec le premier.")] + res = dedupe_facts(facts) + assert not res.changed and res.kept == facts and res.kept_indices == [0, 1] + + +def test_facts_sha_hashes_bullets_only(): + a = [fact("x", cq="q1"), fact("y", cq="q2")] + b = [fact("x", cq="other"), fact("y", wq="w")] + assert facts_sha(a) == facts_sha(b) + assert facts_sha(a) != facts_sha(a[:1]) + + +def test_one_quote_being_a_longer_cut_of_the_other_counts_as_shared(): + a = fact("El estrés hídrico estival genera uvas con alto contenido en polifenoles y graduación óptima, dando un aroma frutado.", cq=QUOTE) + b = fact("El estrés hídrico estival genera uva de bajo rendimiento con alto potencial vínico: polifenoles elevados y graduación óptima.", cq=QUOTE + " toscano, ricco di minerali, conferisce serbevolezza") + assert duplicate_reason(a, b) == "same-quote" + assert duplicate_reason(fact(a["bullet"], cq="unrelated quote long enough to count here"), b) is None + + +def test_restatement_of_a_dropped_fact_is_dropped_too(): + a = fact("Raues Klima mit hohen Tag-Nacht-Temperaturschwankungen sorgt für eine ausgeprägte Säurestruktur.") + b = fact("Raues Klima mit hohen Tag-Nacht-Temperaturschwankungen prägt eine ausgeprägte Säurestruktur der Tiroler Weine.") + c = fact("Raues Klima mit hohen Tag-Nacht-Temperaturunterschieden sorgt für eine gute Säurestruktur der Tiroler Weine.") + res = dedupe_facts([a, b, c]) + assert res.kept_indices == [0] + assert [d["dropped_index"] for d in res.drops] == [1, 2] + + +def test_bullets_leading_with_different_sub_denomination_names_are_kept_apart(): + names = ["Rioja Alavesa", "Rioja Alta", "Rioja Oriental"] + a = fact("Rioja Alavesa: suelos arcillo-calcáreos en terrazas y laderas, vinos de gran frescura y acidez.", cq=QUOTE) + b = fact("Rioja Oriental: suelos arcillo-calcáreos en terrazas y laderas, vinos de gran frescura y acidez.", cq=QUOTE) + c = fact("Suelos arcillo-calcáreos en terrazas y laderas dan vinos de gran frescura y acidez.", cq=QUOTE) + assert dedupe_facts([a, b, c], protected_names=names).kept_indices == [0, 1, 2] + assert dedupe_facts([a, b, c]).kept_indices == [0] + + +def test_protected_name_must_lead_the_bullet_as_a_whole_word(): + names = ["Rioja Alta"] + a = fact("Rioja Altamira: suelos arcillo-calcáreos, vinos frescos con buena acidez y fruta roja madura.") + b = fact("Rioja Altamira: suelos arcillo-calcáreos, vinos frescos con buena acidez y fruta roja.") + assert dedupe_facts([a, b], protected_names=names).kept_indices == [0] diff --git a/tests/test_terroir_feedback.py b/tests/test_terroir_feedback.py new file mode 100644 index 0000000..4a9f594 --- /dev/null +++ b/tests/test_terroir_feedback.py @@ -0,0 +1,177 @@ +"""Per-record review feedback (scripts/_lib/terroir_feedback.py, +scripts/build_terroir_feedback.py) and its wiring into the 21 stage-02d +scripts.""" +from __future__ import annotations + +import importlib.util +import json +import re +from pathlib import Path + +from _lib import terroir_feedback as tf + +ROOT = Path(__file__).resolve().parents[1] + + +def _fb(**over): + base = { + "slug": "taurasi", "country": "it", "source_lang": "it", + "graded_against": {"cahier_source_sha": "abc", "wiki_source_revision": "7"}, + "reviews": [{"id": "review-2026-09-12", "date": "2026-09-12"}], + "do_not_claim": [ + {"claim_en": "Volcanic material imparts minerality to Taurasi.", + "claim_src": "Il materiale piroclastico conferisce mineralità al Taurasi.", + "mode": "unsupported-causal-link", "stage": "extraction", + "why": "The source only records presence {of the material}."}, + {"claim_en": "Plots lie at 250–490 m.", "claim_src": "Parcelle tra 250 e 290 m.", + "mode": "wrong-number-or-unit", "stage": "translation", "why": "490 is a translation slip."}, + ], + "capture_if_present": [{"hint": "Aglianico's Greek origin is prominent in the source."}], + "record_cautions": [{"kind": "sibling-text", "note": "Section b) describes the neighbouring DOC."}], + "history": [], + } + base.update(over) + return base + + +def test_block_is_empty_without_feedback(): + assert tf.feedback_prompt_block(None) == "" + assert tf.feedback_prompt_block({"do_not_claim": [], "capture_if_present": [], "record_cautions": []}) == "" + + +def test_block_keeps_extraction_claims_and_drops_translation_ones(): + block = tf.feedback_prompt_block(_fb()) + assert "Do not assert «Volcanic material imparts minerality to Taurasi.»" in block + assert "250–490" not in block + assert "unless the source states it explicitly" in block + assert "capture: Aglianico's Greek origin" in block + assert "Caution (sibling-text)" in block + assert "(2026-09-12)" in block + + +def test_translation_stage_selects_the_other_bucket(): + block = tf.feedback_prompt_block(_fb(), stage="translation") + assert "250–490" in block and "Volcanic material" not in block + + +def test_block_never_carries_format_braces(): + block = tf.feedback_prompt_block(_fb()) + assert "{" not in block and "}" not in block + assert "(of the material)" in block + + +def test_block_caps_the_number_of_claims(): + claims = [{"claim_en": f"claim {i}", "claim_src": f"src {i}", "stage": "extraction", "why": "w"} for i in range(9)] + block = tf.feedback_prompt_block(_fb(do_not_claim=claims, capture_if_present=[], record_cautions=[])) + assert block.count("Do not assert") == tf.MAX_CLAIMS + assert "3 more claims" in block + + +def test_with_feedback_reads_the_sidecar(tmp_path, monkeypatch): + monkeypatch.setattr(tf, "FEEDBACK_DIR", tmp_path) + tf.clear_cache() + assert tf.with_feedback("SYSTEM", "taurasi") == "SYSTEM" + (tmp_path / "taurasi.json").write_text(json.dumps(_fb()), encoding="utf-8") + tf.clear_cache() + out = tf.with_feedback("SYSTEM\n", "taurasi") + assert out.startswith("SYSTEM\n\nLessons from the previous review") + assert tf.with_feedback("SYSTEM", "no-such-slug") == "SYSTEM" + tf.clear_cache() + + +def test_recurrence_matches_the_source_bullet_not_translation_entries(): + facts = [ + {"bullet": "Il materiale piroclastico conferisce mineralità e struttura al Taurasi."}, + {"bullet": "Parcelle tra 250 e 290 m."}, + ] + hits = tf.recurrence_findings(_fb(), facts) + assert [h["index"] for h in hits] == [0] + assert hits[0]["mode"] == "unsupported-causal-link" + assert tf.recurrence_findings(_fb(), [{"bullet": "Clima mediterraneo con estati calde."}]) == [] + assert tf.recurrence_findings(None, facts) == [] + + +def test_is_stale_tracks_the_graded_sources(): + fb = _fb() + assert not tf.is_stale(fb, "abc", "7") + assert tf.is_stale(fb, "def", "7") + assert tf.is_stale(fb, "abc", "8") + assert not tf.is_stale(fb, None, None) + + +def test_append_history_creates_and_extends(tmp_path, monkeypatch): + monkeypatch.setattr(tf, "FEEDBACK_DIR", tmp_path) + tf.clear_cache() + tf.append_history("x", {"run": "02d-2026-10", "kind": "gate", "dropped": [3]}) + tf.append_history("x", {"run": "02d-2026-11", "kind": "gate", "dropped": []}) + fb = json.loads((tmp_path / "x.json").read_text(encoding="utf-8")) + assert [h["run"] for h in fb["history"]] == ["02d-2026-10", "02d-2026-11"] + assert all("at" in h for h in fb["history"]) + tf.clear_cache() + + +# ───────────────────────────────────────────── builder (merge semantics) ── + + +def _load_builder(): + spec = importlib.util.spec_from_file_location("build_terroir_feedback", ROOT / "scripts" / "build_terroir_feedback.py") + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +def test_builder_collects_and_merges_without_duplicating(tmp_path): + b = _load_builder() + ev = tmp_path / "ev" + ev.mkdir() + (ev / "confirmed-misleading.json").write_text(json.dumps([ + {"slug": "x", "i": 2, "subsection": "interactions", "mode": "unsupported-causal-link", + "where": "extraction", "en": "Soils give minerality.", "src": "I suoli danno mineralità.", "reason": "no link in source"}, + ]), encoding="utf-8") + (ev / "merged.json").write_text(json.dumps({"record_notes": {"x": [ + ["fidelity", "MISSING", "The lake as climate regulator is absent."], + ["language", "MISSING", "The lake, a climate regulator, is not captured."], + ["fidelity", "OTHER", "Batch-wide: Rodopi should be Rhodopes."], + ["fidelity", "WRONG_SOURCE", "The source text is the neighbouring DOC's disciplinare."], + ]}}), encoding="utf-8") + per = b.collect(ev, "review-t") + assert set(per) == {"x"} + assert len(per["x"]["do_not_claim"]) == 1 + assert len(per["x"]["capture_if_present"]) == 1 # two lenses, one observation + assert [c["kind"] for c in per["x"]["record_cautions"]] == ["wrong-source"] # batch-wide note skipped + review = {"id": "review-t", "date": "t"} + fb, added = b.merge_into(None, "x", per["x"], review) + assert dict(added) == {"do_not_claim": 1, "capture_if_present": 1, "record_cautions": 1} + fb2, added2 = b.merge_into(fb, "x", per["x"], review) # re-running the same review adds nothing + assert dict(added2) == {} and len(fb2["reviews"]) == 1 and fb2["history"] == [] + + +# ─────────────────────────────────────────────── wiring lint (21 scripts) ── + + +STAGE_02D = [ROOT / "scripts" / "02d_extract_terroir_facts.py"] + sorted(ROOT.glob("scripts/*/02d_extract_terroir_facts.py")) + + +def test_all_02d_scripts_are_wired(): + assert len(STAGE_02D) == 21 + for path in STAGE_02D: + src = path.read_text(encoding="utf-8") + assert "from _lib.terroir_feedback import with_feedback" in src, path + calls = len(re.findall(r"with_feedback\(", src)) + # live call after the system prompt, plus the emit-todo prompt (ES has no manual round-trip yet) + expected = 1 if path.parent.name == "es" else 2 + assert calls >= expected, f"{path}: {calls} with_feedback calls, expected ≥ {expected}" + assert re.search(r"^\s*system = with_feedback\(system, (record|job)\[\"slug\"\]\)", src, re.M), path + + +def test_recurrence_counts_a_meaningful_gate_rewrite_as_resolved(): + probe = "Il clima mediterraneo conferisce ai vini una spiccata mineralità." + fb = {"do_not_claim": [{"claim_src": probe, "stage": "extraction", "mode": "causal"}]} + rewritten = {"bullet": "Il clima è mediterraneo e i vini mostrano una spiccata mineralità.", + "support": {"verdict": "rewrite", "original_bullet": probe}} + assert tf.recurrence_findings(fb, [rewritten]) == [] + # a cosmetic rewrite (punctuation only) is not a resolution + cosmetic = {"bullet": probe.rstrip(".") + ",", "support": {"verdict": "rewrite", "original_bullet": probe}} + assert len(tf.recurrence_findings(fb, [cosmetic])) == 1 + # the same claim re-extracted without a gate verdict recurs + assert len(tf.recurrence_findings(fb, [{"bullet": probe}])) == 1 diff --git a/tests/test_terroir_gate.py b/tests/test_terroir_gate.py new file mode 100644 index 0000000..d34f8a1 --- /dev/null +++ b/tests/test_terroir_gate.py @@ -0,0 +1,228 @@ +"""The claim-support gate (scripts/_lib/terroir_gate.py): verdict parsing, +rewrite guards and the deterministic application of verdicts.""" +from __future__ import annotations + +from _lib import terroir_gate as tg + +FACTS = [ + {"bullet": "I suoli vulcanici conferiscono mineralità e struttura al vino.", "subsection": "interactions", + "provenance": "cahier", "cahier_quote": "suoli di origine vulcanica"}, + {"bullet": "Il clima è mediterraneo con estati calde e secche.", "subsection": "facteurs_humains", + "provenance": "cahier", "cahier_quote": "clima mediterraneo con estati calde e secche"}, + {"bullet": "La vendemmia avviene esclusivamente a mano.", "subsection": "facteurs_humains", + "provenance": "cahier", "cahier_quote": "la vendemmia è di norma manuale"}, + {"bullet": "Le estati sono calde e secche, con clima mediterraneo.", "subsection": "produit", + "provenance": "cahier", "cahier_quote": "clima mediterraneo con estati calde e secche"}, +] +SOURCE = "I suoli sono di origine vulcanica. Il clima è mediterraneo con estati calde e secche. La vendemmia è di norma manuale." + + +def test_parse_verdicts_is_tolerant_and_aligned(): + raw = '```json\n{"facts": [{"i": 0, "verdict": "drop", "note": "n0"}, {"i": 2, "verdict": "REWRITE", "rewrite": "x", "restates": "", "subsection": "produit"}, {"i": 9, "verdict": "drop"}, "junk"]}\n```' + rows, err = tg.parse_verdicts(raw, 3) + assert err is None and len(rows) == 3 + assert rows[0]["verdict"] == "drop" and rows[1]["verdict"] == "supported" + assert rows[2]["verdict"] == "rewrite" and rows[2]["subsection"] == "produit" and rows[2]["restates"] is None + assert tg.parse_verdicts("no json here", 2) == (None, "no JSON object in reply") + assert tg.parse_verdicts('{"facts": []}', 2)[1] == "reply graded none of the bullets" + + +def test_rewrite_guards(): + assert tg.rewrite_ok("Piove 600 mm all'anno.", "Piove 600 mm all'anno sulla costa.", "600 mm sulla costa") is None + assert tg.rewrite_ok("Piove 600 mm.", "Piove 600 mm.", "") == "unchanged" + assert tg.rewrite_ok("Piove 600 mm.", "", "") == "empty" + assert tg.rewrite_ok("Piove molto.", "Piove 800 mm all'anno in collina.", "piove molto") == "new numbers ['800']" + assert tg.rewrite_ok("Piove molto.", "Suolo → vino.", "") == "arrow" + + +def test_apply_verdicts_drops_rewrites_moves_and_records_support(): + verdicts = [ + {"verdict": "rewrite", "note": "presence only", "rewrite": "I suoli sono di origine vulcanica", "restates": None, "subsection": "facteurs_naturels"}, + {"verdict": "supported", "note": "", "rewrite": "", "restates": None, "subsection": "facteurs_naturels"}, + {"verdict": "rewrite", "note": "hedge dropped", "rewrite": "La vendemmia è di norma manuale.", "restates": None, "subsection": None}, + {"verdict": "drop", "note": "restates #1", "rewrite": "", "restates": 1, "subsection": None}, + ] + res = tg.apply_verdicts(FACTS, verdicts, source=SOURCE, source_lang="it", run="r1", model="m") + assert [f["bullet"] for f in res["facts"]] == [ + "I suoli sono di origine vulcanica.", "Il clima è mediterraneo con estati calde e secche.", "La vendemmia è di norma manuale.", + ] + assert res["kept_indices"] == [0, 1, 2] + assert [f["subsection"] for f in res["facts"]] == ["facteurs_naturels", "facteurs_naturels", "facteurs_humains"] + assert res["facts"][0]["support"]["original_bullet"].startswith("I suoli vulcanici") + assert res["facts"][0]["support"]["moved_from"] == "interactions" + assert res["facts"][1]["support"] == {**res["facts"][1]["support"], "verdict": "supported", "moved_from": "facteurs_humains"} + assert [d["index"] for d in res["dropped"]] == [3] and res["dropped"][0]["restates"] == 1 + assert res["text_changed"] and len(res["rewritten"]) == 2 and len(res["moved"]) == 2 + + +def test_apply_verdicts_keeps_a_bullet_whose_twin_was_dropped_and_rejects_bad_rewrites(): + verdicts = [ + {"verdict": "drop", "note": "unsupported", "rewrite": "", "restates": None, "subsection": None}, + {"verdict": "drop", "note": "restates #0", "rewrite": "", "restates": 0, "subsection": None}, + {"verdict": "rewrite", "note": "x", "rewrite": "Vendemmia a mano su 1.200 ettari.", "restates": None, "subsection": None}, + {"verdict": "supported", "note": "", "rewrite": "", "restates": None, "subsection": None}, + ] + res = tg.apply_verdicts(FACTS, verdicts, source=SOURCE, source_lang="it", run="r1", model="m") + kept = {f["bullet"]: f["support"] for f in res["facts"]} + assert "Il clima è mediterraneo con estati calde e secche." in kept + assert kept["Il clima è mediterraneo con estati calde e secche."]["verdict"] == "supported" + assert kept["La vendemmia avviene esclusivamente a mano."]["verdict"] == "rewrite-rejected" + assert kept["La vendemmia avviene esclusivamente a mano."]["rejected_reason"].startswith("new numbers") + # #3 restates #1 (same quote, near-identical bullet): the lexical dedupe + # after the gate collapses it even though the model kept it. + assert [(d["index"], d["note"][:9]) for d in res["dropped"]] == [(0, "unsupport"), (3, "duplicate")] + assert res["kept_indices"] == [1, 2] + assert not res["text_changed"] + + +def test_user_message_carries_constraints_and_no_braces(): + fb = {"do_not_claim": [{"claim_src": "I suoli danno {mineralità}", "why": "presence only", "stage": "extraction"}, + {"claim_src": "x", "why": "y", "stage": "translation"}], + "record_cautions": [{"kind": "wrong-source", "note": "section b describes the neighbour"}]} + msg = tg.build_user_message(name="Taurasi", country="it", source_lang="it", cahier="testo", hints={"produit": "vino rosso"}, + facts=FACTS[:1], feedback=fb) + assert "Verified misleading before: «I suoli danno (mineralità)»" in msg + assert msg.count("Verified misleading before") == 1 + assert "Caution (wrong-source)" in msg and "[produit]\nvino rosso" in msg and "#0 [interactions · cahier]" in msg + assert "{" not in msg and "}" not in msg + + +def test_parse_verdicts_recovers_rows_with_unescaped_quotes_in_notes(): + raw = '''```json +{"facts": [ + {"i": 0, "verdict": "supported", "note": "Source states «la "montille" est calcaire», confirmed.", "rewrite": "", "restates": null, "subsection": null}, + {"i": 1, "verdict": "drop", "note": "Says "chiaretto" twice; restates #0.", "rewrite": "", "restates": 0, "subsection": null}, + {"i": 2, "verdict": "rewrite", "note": "ok", "rewrite": "Il suolo è "calcareo" e marnoso.", "restates": null, "subsection": "facteurs_naturels"} +]} +```''' + rows, err = tg.parse_verdicts(raw, 3) + assert err is None + assert [r["verdict"] for r in rows] == ["supported", "drop", "rewrite"] + assert rows[1]["restates"] == 0 and 'Says "chiaretto"' in rows[1]["note"] + assert rows[2]["rewrite"] == 'Il suolo è "calcareo" e marnoso.' and rows[2]["subsection"] == "facteurs_naturels" + + +# ───────────────────────────────────────────── 02e back-check (R6) ── + +from _lib import terroir_backcheck as tb # noqa: E402 + + +def test_backcheck_parse_and_apply_fixes_with_guards(): + raw = '{"facts": [{"i": 0, "verdict": "fix", "issue": "generoso is fortified", "fix": "Fortified wines (generoso) are aged under flor."}, {"i": 1, "verdict": "fix", "issue": "number", "fix": "Plots lie at 250–490 m."}, {"i": 2, "verdict": "ok", "issue": "", "fix": ""}]}' + checks, err = tb.parse_checks(raw, 3) + assert err is None and [c["verdict"] for c in checks] == ["fix", "fix", "ok"] + source = [{"bullet": "Los vinos generosos se crían bajo velo de flor."}, {"bullet": "Las parcelas están entre 250 y 290 m."}, {"bullet": "x"}] + translated = [{"bullet": "Generous wines are aged under flor.", "subsection": "produit"}, + {"bullet": "Plots lie at 250–290 m.", "subsection": "facteurs_naturels"}, {"bullet": "x", "subsection": "produit"}] + res = tb.apply_fixes(translated, source, checks, lang="en", run="r", model="m") + assert res["facts"][0]["bullet"] == "Fortified wines (generoso) are aged under flor." + assert res["facts"][0]["check"]["original"] == "Generous wines are aged under flor." + assert res["facts"][1]["bullet"] == "Plots lie at 250–290 m." # 490 is in neither source nor translation + assert res["facts"][1]["check"]["verdict"] == "fix-rejected" and res["facts"][1]["check"]["rejected_reason"].startswith("new numbers") + assert res["facts"][2]["check"]["verdict"] == "ok" and res["facts"][2]["subsection"] == "produit" + assert [f["index"] for f in res["fixed"]] == [0] and [r["index"] for r in res["rejected"]] == [1] + + +def test_backcheck_user_message_flags_exonyms_and_translation_feedback(): + fb = {"do_not_claim": [{"claim_en": "Plots lie at 250–490 m.", "why": "490 is a slip", "stage": "translation"}, + {"claim_en": "x", "why": "y", "stage": "extraction"}]} + msg = tb.build_user_message(name="Rueda", source_lang="es", target_lang="en", + source_facts=[{"bullet": "Al pie de los Pirineos.", "subsection": "facteurs_naturels"}], + translated=[{"bullet": "At the foot of the Pirineos."}], feedback=fb) + assert "DETECTOR: source-form place name(s) still present: Pirineos" in msg + assert "250–490" in msg and msg.count("previous review verified") == 1 + assert "Spanish → English" in msg and "{" not in msg + assert "generoso" in tb.system_prompt() and "{watch_list}" not in tb.system_prompt() + + +def test_cosmetic_rewrite_is_case_punctuation_or_short_words_only(): + orig = "Le vignoble s'étage entre 200 et 400 mètres, sur des marnes calcaires exposées au sud." + assert tg.is_cosmetic_rewrite(orig, orig) + assert tg.is_cosmetic_rewrite(orig, "Le vignoble s’étage entre 200 et 400 mètres, sur des marnes calcaires exposées au sud") + assert tg.is_cosmetic_rewrite(orig, "le vignoble s'étage entre 200 et 400 metres sur des marnes calcaires exposees au sud.") + assert tg.is_cosmetic_rewrite(orig, "Le vignoble s'étage entre 200 et 400 mètres, sur les marnes calcaires exposées au sud.") + # a hedge, a qualifier or an entity change is a real rewrite even at ratio ≥ 95 + assert not tg.is_cosmetic_rewrite(orig, "Le vignoble s'étage surtout entre 200 et 400 mètres, sur des marnes calcaires exposées au sud.") + assert not tg.is_cosmetic_rewrite(orig, "Le vignoble s'étage entre 200 et 400 mètres, sur des marnes calcaires exposées au nord.") + assert not tg.is_cosmetic_rewrite("La vendemmia avviene esclusivamente a mano.", "La vendemmia avviene a mano.") + + +def test_apply_verdicts_keeps_the_original_for_empty_and_cosmetic_rewrites(): + verdicts = [ + {"verdict": "rewrite", "note": "could not phrase", "rewrite": "", "restates": None, "subsection": None}, + {"verdict": "rewrite", "note": "punctuation", "rewrite": "Il clima è mediterraneo, con estati calde e secche", "restates": None, "subsection": None}, + {"verdict": "rewrite", "note": "hedge", "rewrite": "La vendemmia è di norma manuale.", "restates": None, "subsection": None}, + {"verdict": "supported", "note": "", "rewrite": "", "restates": None, "subsection": None}, + ] + res = tg.apply_verdicts(FACTS, verdicts, source=SOURCE, source_lang="it", run="r1", model="m") + s0, s1, s2 = (f["support"] for f in res["facts"][:3]) + assert s0["verdict"] == "supported" and s0["rewrite_missing"] and s0["note"] == "could not phrase" + assert s1["verdict"] == "supported" and s1["cosmetic_rewrite"].startswith("Il clima") + assert res["facts"][1]["bullet"] == FACTS[1]["bullet"] + assert s2["verdict"] == "rewrite" and res["facts"][2]["bullet"] == "La vendemmia è di norma manuale." + assert [m["index"] for m in res["missing_rewrites"]] == [0] + assert [c["index"] for c in res["cosmetic_rewrites"]] == [1] + assert [r["index"] for r in res["rewritten"]] == [2] and res["text_changed"] + assert not res["rejected_rewrites"] + + +def test_gate_demotes_an_interactions_bullet_whose_quote_states_no_link(): + facts = [ + {"bullet": "I suoli vulcanici conferiscono mineralità al vino.", "subsection": "interactions", + "provenance": "cahier", "cahier_quote": "i suoli vulcanici conferiscono mineralità al vino"}, + {"bullet": "I suoli sono di origine vulcanica.", "subsection": "interactions", + "provenance": "cahier", "cahier_quote": "suoli di origine vulcanica"}, + ] + verdicts = [{"verdict": "supported", "note": "", "rewrite": "", "restates": None, "subsection": None}] * 2 + res = tg.apply_verdicts(facts, verdicts, source="…", source_lang="it", run="r1", model="m") + assert [f["subsection"] for f in res["facts"]] == ["interactions", "facteurs_naturels"] + assert res["facts"][1]["support"]["moved_from"] == "interactions" + assert res["facts"][1]["support"]["unearned_interaction"] is True + assert res["moved"] == [{"index": 1, "from": "interactions", "to": "facteurs_naturels"}] + + +def test_needs_gate_keys_on_verdicts_and_source_not_on_an_exact_sha(): + gated = {"cahier_source_sha": "abc", "facts": [{"bullet": "A.", "support": {"verdict": "supported"}}], + "gate": {"version": tg.GATE_VERSION, "cahier_source_sha": "abc", "facts_sha_after": "stale"}} + assert not tg.needs_gate(gated) # a post-pass changed the bullets: still gated + assert tg.needs_gate(gated, refresh=True) + assert tg.needs_gate({**gated, "cahier_source_sha": "def"}) # source changed + assert tg.needs_gate({**gated, "gate": {**gated["gate"], "version": "gate-v1"}}) + assert tg.needs_gate({**gated, "facts": gated["facts"] + [{"bullet": "B."}]}) # a fresh, unverdicted fact + assert not tg.needs_gate({**gated, "facts": []}) + assert tg.needs_gate({"facts": [{"bullet": "A."}]}) + + +def test_backcheck_keeps_the_translation_when_a_fix_comes_back_empty(): + from _lib import terroir_backcheck as tb + translated = [{"bullet": "The soils are mostly clay."}] + source = [{"bullet": "Les sols sont surtout argileux."}] + checks = [{"verdict": "fix", "issue": "hedge", "fix": ""}] + res = tb.apply_fixes(translated, source, checks, lang="en", run="r", model="m") + c = res["facts"][0]["check"] + assert c["verdict"] == "ok" and c["fix_missing"] and c["issue"] == "hedge" + assert res["facts"][0]["bullet"] == "The soils are mostly clay." + assert res["missing_fixes"] == [{"index": 0, "issue": "hedge"}] and not res["rejected"] + + +def test_a_move_into_interactions_that_the_earned_rule_reverts_leaves_no_trace(): + facts = [{"bullet": "I suoli sono di origine vulcanica.", "subsection": "facteurs_naturels", + "provenance": "cahier", "cahier_quote": "suoli di origine vulcanica"}] + verdicts = [{"verdict": "supported", "note": "", "rewrite": "", "restates": None, "subsection": "interactions"}] + res = tg.apply_verdicts(facts, verdicts, source="…", source_lang="it", run="r1", model="m") + f = res["facts"][0] + assert f["subsection"] == "facteurs_naturels" and "moved_from" not in f["support"] + assert "unearned_interaction" not in f["support"] and res["moved"] == [] + + +def test_backcheck_user_message_carries_the_appellation_note_for_a_parent(monkeypatch): + from _lib import terroir_backcheck as tb + from _lib import terroir_roster as tr + monkeypatch.setattr(tr, "_load", lambda: ({"rioja": ["Rioja Alavesa", "Rioja Alta"]}, {"rioja": "Rioja"}, {})) + kw = dict(name="Rioja", source_lang="es", target_lang="en", + source_facts=[{"bullet": "Los suelos de la denominación son arcillosos."}], + translated=[{"bullet": "The soils of the Rioja appellation are clayey."}], feedback=None) + msg = tb.build_user_message(**kw, slug="rioja") + assert "CONTEXT FOR THE CHECK:" in msg and "do not flag it as an added or wrong entity" in msg + assert "CONTEXT" not in tb.build_user_message(**kw, slug="leaf") + assert "CONTEXT" not in tb.build_user_message(**kw) diff --git a/tests/test_terroir_interactions.py b/tests/test_terroir_interactions.py new file mode 100644 index 0000000..e2bb6e8 --- /dev/null +++ b/tests/test_terroir_interactions.py @@ -0,0 +1,129 @@ +"""The earned `interactions` rule (scripts/_lib/terroir_interactions.py).""" +from __future__ import annotations + +import pytest +from _lib.terroir_interactions import ( + CONNECTIVES, + earn_interactions, + has_connective, + quote_has_connective, + unearned_indices, +) + + +@pytest.mark.parametrize( + "lang,text", + [ + ("fr", "Les sols argilo-calcaires confèrent aux vins une belle structure tannique."), + ("fr", "Grâce à l'exposition sud, les raisins mûrissent tôt."), + ("it", "L'escursione termica determina un'elevata concentrazione aromatica."), + ("es", "Debido a la altitud, los vinos conservan la acidez."), + ("de", "Die Schieferböden speichern die Wärme, wodurch die Trauben voll ausreifen."), + ("nl", "Dankzij de zuidhelling rijpen de druiven vroeg."), + ("en", "The chalk soils give the wines their freshness."), + ("el", "Λόγω των μελτεμιών το κλίμα είναι ξηρό."), + ("bg", "Благодарение на почвите виното е плътно."), + ("hu", "A lösztalajnak köszönhetően a borok testesek."), + ("cs", "Díky spraši jsou vína plná."), + ("sk", "Vďaka sprašiam sú vína plné."), + ("sl", "Zaradi apnenca so vina mineralna."), + ("hr", "Zahvaljujući buri grožđe je zdravo."), + ("ro", "Datorită solurilor calcaroase vinurile sunt minerale."), + ("mt", "Thanks to the sea breeze the grapes ripen slowly."), # mt → en + ("gr", "Το ηφαιστειακό έδαφος προσδίδει ορυκτότητα."), # gr → el + ], +) +def test_connectives_per_language(lang, text): + assert has_connective(text, lang) + + +@pytest.mark.parametrize( + "lang,text", + [ + ("fr", "Les sols sont argilo-calcaires et les coteaux exposés au sud."), + ("it", "Il clima è mediterraneo con estati calde e secche."), + ("de", "Die Böden bestehen aus Schiefer und Quarzit."), + ("en", "The vineyards lie on chalk at 200 m."), + ("xx", "grâce à quelque chose"), # unknown language → never earned + ], +) +def test_no_connective(lang, text): + assert not has_connective(text, lang) + + +def test_word_initial_anchoring_and_stems(): + assert not has_connective("dankbar", "fr") # "dank" is German, not in the FR list + assert has_connective("undankbar dank", "de") # word-initial "dank" matches; "undankbar" alone would not + assert not has_connective("undankbar", "de") + assert has_connective("επιδρούν στο κλίμα", "el") # stem + + +def test_every_language_table_compiles_and_is_lowercase(): + for lang, terms in CONNECTIVES.items(): + assert terms and all(t == t.lower() for t in terms), lang + assert has_connective(f"x {terms[0]} y", lang) + + +def test_quote_has_connective_follows_provenance(): + f = {"provenance": "wiki", "cahier_quote": "grâce aux sols", "wiki_quote": "les sols sont calcaires"} + assert not quote_has_connective(f, "fr") + f["provenance"] = "cahier" + assert quote_has_connective(f, "fr") + f = {"provenance": "both", "cahier_quote": "sols calcaires", "wiki_quote": "ce qui explique la fraîcheur"} + assert quote_has_connective(f, "fr") + + +FACTS = [ + {"bullet": "Le sol est calcaire.", "subsection": "facteurs_naturels", "provenance": "cahier", + "cahier_quote": "le sol est calcaire"}, + {"bullet": "Le calcaire confère de la fraîcheur.", "subsection": "interactions", "provenance": "cahier", + "cahier_quote": "le calcaire confère de la fraîcheur"}, + {"bullet": "Le climat donne de la fraîcheur aux vins.", "subsection": "interactions", "provenance": "cahier", + "cahier_quote": "le climat est frais"}, # link only in the bullet + {"bullet": "L'exposition permet une maturité précoce.", "subsection": "interactions", "provenance": "cahier", + "cahier_quote": "l'exposition permet une maturité précoce"}, + {"bullet": "Le vent favorise la santé des raisins.", "subsection": "interactions", "provenance": "cahier", + "cahier_quote": "le vent favorise la santé des raisins"}, # third earned one: over the cap +] + + +def test_earn_interactions_drops_unearned_and_caps(): + res = earn_interactions(FACTS, "fr") + assert [f["bullet"][:12] for f in res.kept] == ["Le sol est c", "Le calcaire ", "L'exposition"] + assert [f["bullet"][:12] for f in res.dropped] == ["Le climat do", "Le vent favo"] + assert unearned_indices(FACTS, "fr") == [2, 4] + assert earn_interactions(FACTS, "fr", max_interactions=3).dropped == [FACTS[2]] + + +@pytest.mark.parametrize( + "lang,text", + [ + ("hr", "nezaobilazni i glavni čimbenik, uz navedene okolišne uvjete, vrhunske kakvoće grožđa"), + ("de", "Intensive Pflege wirkt sich stabilisierend aus. Sie fördert in hohem Maße die Qualität."), + ("el", "το ηφαιστειογενές έδαφος απορροφά την υγρασία και έτσι τρέφονται τα αμπέλια"), + ("es", "la especial influencia del clima atlántico, que hace que los vinos tengan cuerpo"), + ("fr", "La richesse des minéraux dans les sols déterminent la finesse des arômes des vins."), + ("fr", "Les sols maigres entrainent une faible production de la plante."), + ("nl", "Door zijn mengeling van gesteenten is het rijk aan mineralen, hetgeen zich vertaalt in wijnen"), + ("nl", "The climate helps to achieve the required ripeness."), # Ambt Delden: English source + ("ro", "Solul brun dă vinuri extractive; incluziunile au influenţe remarcabile"), # cedilla spelling + ("it", "rese naturalmente basse in quanto le radici affondano nel calcare"), + ("bg", "букет и вкус, резултат от съчетанието на тръпчивостта на Мавруд"), + ("hu", "A bazaltsapkák és a Balaton közelsége együttesen garantálják a magas mustfokot."), + ], +) +def test_connective_forms_the_smoke_and_the_corpus_samples_missed(lang, text): + assert has_connective(text, lang) + + +@pytest.mark.parametrize( + "lang,text", + [ + ("fr", "Les coteaux au caractère marqué sont exposés au sud."), # "car" must not match "caractère" + ("de", "Die Böden bestehen dennoch aus Schiefer."), # "denn" is not in the table + ("en", "Vines have been planted here since 1950."), # "since" is temporal here: not listed + ("hu", "A talaj lösz és agyag."), + ], +) +def test_word_initial_stems_do_not_overreach(lang, text): + assert not has_connective(text, lang) diff --git a/tests/test_terroir_normalize.py b/tests/test_terroir_normalize.py new file mode 100644 index 0000000..709f3c5 --- /dev/null +++ b/tests/test_terroir_normalize.py @@ -0,0 +1,104 @@ +"""Deterministic bullet clean-up (scripts/_lib/terroir_normalize.py).""" +from __future__ import annotations + +import pytest +from _lib.terroir_normalize import ( + ensure_terminal_period, + expand_mentions, + latinize_residual_script, + normalize_bullet, + normalize_facts, + strip_colour_codes, +) + + +@pytest.mark.parametrize( + "src,expected", + [ + ("Cépages : pinot noir N, chardonnay B et pinot gris G.", "Cépages : pinot noir, chardonnay et pinot gris."), + ("Riesling B, gewurztraminer Rs, pinot gris G et sylvaner B.", "Riesling, gewurztraminer, pinot gris et sylvaner."), + ("Seul grand cru à inclure le sylvaner B parmi ses cépages.", "Seul grand cru à inclure le sylvaner parmi ses cépages."), + ("Grenache G (Rg) et muscat à petits grains B dominent.", "Grenache et muscat à petits grains dominent."), + ("Vignoble de Colmar N exposé au sud.", "Vignoble de Colmar N exposé au sud."), + ("Classé en 1936 en catégorie B du référentiel.", "Classé en 1936 en catégorie B du référentiel."), + ], +) +def test_colour_codes_are_stripped_only_after_grape_names(src, expected): + assert strip_colour_codes(src) == expected + + +def test_vt_sgn_are_expanded(): + assert expand_mentions("VT : arômes exotiques ; SGN plus concentrés.") == ( + "Vendanges Tardives : arômes exotiques ; Sélection de Grains Nobles plus concentrés." + ) + assert expand_mentions("Mentions VT/SGN exigent 18 mois.") == ( + "Mentions Vendanges Tardives / Sélection de Grains Nobles exigent 18 mois." + ) + assert expand_mentions("La SGNV n'existe pas.") == "La SGNV n'existe pas." + + +@pytest.mark.parametrize( + "src,expected", + [ + ("Sols argilo-calcaires", "Sols argilo-calcaires."), + ("Sols argilo-calcaires.", "Sols argilo-calcaires."), + ("AOC reconnue en 1936 (JORF)", "AOC reconnue en 1936 (JORF)."), + ("Renommée « Montlouis-sur-Loire »", "Renommée « Montlouis-sur-Loire »."), + ("Trailing space ", "Trailing space."), + ("Déjà ponctué ?", "Déjà ponctué ?"), + ("", ""), + ], +) +def test_terminal_period(src, expected): + assert ensure_terminal_period(src) == expected + + +def test_normalize_facts_counts_changes_and_edits_in_place(): + facts = [{"bullet": "Pinot noir N dominant"}, {"bullet": "Déjà propre."}] + assert normalize_facts(facts) == 1 + assert facts[0]["bullet"] == "Pinot noir dominant." + assert normalize_bullet("") == "" + + +@pytest.mark.parametrize( + "src,expected", + [ + ("Thermoheliоhydric index 4,596–4,765.", "Thermoheliohydric index 4,596–4,765."), + ("Dr. Nik. Piniatorοs founded a company.", "Dr. Nik. Piniatoros founded a company."), + ("Terraces with dry-stone walls (ξερολιθιές) of 1–2 m.", "Terraces with dry-stone walls (xerolithies) of 1–2 m."), + ("Vertisols (смолници) and brown forest soils.", "Vertisols (smolnitsi) and brown forest soils."), + ("Bяло Мискет врачански: fine misket aroma.", "Bialo Misket vrachanski: fine misket aroma."), + ("A bitter, resinous note of α-terpineol.", "A bitter, resinous note of α-terpineol."), + ("Οι θερινοί άνεμοι αποτελούν παράγοντα μοναδικότητας των οίνων.", "Οι θερινοί άνεμοι αποτελούν παράγοντα μοναδικότητας των οίνων."), + ], +) +def test_residual_script_is_latinised_only_in_mostly_latin_bullets(src, expected): + assert latinize_residual_script(src) == expected + + +def test_latinisation_applies_to_target_locales_only(): + src = "Wijngaarden op terrassen met droogstenen muren (πεζούλες), tot 900 m hoogte." + assert normalize_bullet(src, "nl") == "Wijngaarden op terrassen met droogstenen muren (pezoules), tot 900 m hoogte." + assert normalize_bullet(src, "") == src + + +def test_trailing_meta_clause_is_dropped_and_the_fact_kept(): + from _lib.terroir_normalize import normalize_bullet, strip_trailing_meta + assert normalize_bullet("Un'epoca iniziata 70 milioni di anni fa secondo il disciplinare.") == "Un'epoca iniziata 70 milioni di anni fa." + assert normalize_bullet("Les sols sont argilo-calcaires, selon le cahier des charges.") == "Les sols sont argilo-calcaires." + assert normalize_bullet("Die Böden sind Schiefer laut Produktspezifikation") == "Die Böden sind Schiefer." + assert normalize_bullet("A period that began 70 million years ago according to the production specification.", "en") == "A period that began 70 million years ago." + # mid-sentence citations are left to the audit's meta_text check + assert strip_trailing_meta("Il disciplinare prevede una resa massima di 80 q/ha.") == "Il disciplinare prevede una resa massima di 80 q/ha." + + +def test_dutch_common_noun_appellatie_keeps_the_registered_french_term(): + from _lib.terroir_normalize import normalize_bullet + assert normalize_bullet("De appellation is gelegen in de streek Revermont.", "nl") == "De appellatie is gelegen in de streek Revermont." + assert normalize_bullet("In 2009 coëxisteerden de appellations Limoux en Crémant de Limoux.", "nl") == "In 2009 coëxisteerden de appellaties Limoux en Crémant de Limoux." + assert normalize_bullet("Appellation Alsace grand cru werd erkend in 1975.", "nl") == "Appellatie Alsace grand cru werd erkend in 1975." + kept = "De appellation d'origine contrôlée Alsace grand cru Muenchberg werd erkend in 1992." + assert normalize_bullet(kept, "nl") == kept + assert normalize_bullet("De appellation d’origine protégée omvat 12 gemeenten.", "nl") == "De appellation d’origine protégée omvat 12 gemeenten." + # other locales untouched + assert normalize_bullet("The appellation lies in the Revermont.", "en") == "The appellation lies in the Revermont." diff --git a/tests/test_terroir_prompts.py b/tests/test_terroir_prompts.py new file mode 100644 index 0000000..5a164bb --- /dev/null +++ b/tests/test_terroir_prompts.py @@ -0,0 +1,41 @@ +"""Shared extraction-prompt style block (scripts/_lib/terroir_prompts.py).""" +from __future__ import annotations + +from _lib.terroir_prompts import STYLE_RULES, with_style_rules + +PROMPT = "Intro {wiki_hint}\n\nRules:\n- a\n- b\n\nRéponds UNIQUEMENT en JSON :\n{{\"facts\": []}}" + + +def test_rules_land_before_the_json_paragraph_and_keep_format_fields(): + out = with_style_rules(PROMPT) + assert out.endswith("Réponds UNIQUEMENT en JSON :\n{\"facts\": []}".replace("{", "{{").replace("}", "}}")) + assert out.index(STYLE_RULES) < out.index("Réponds UNIQUEMENT") + assert "{" not in STYLE_RULES and "}" not in STYLE_RULES + assert out.format(wiki_hint="x").startswith("Intro x") + + +def test_dict_prompts_are_handled_per_language(): + out = with_style_rules({"fr": PROMPT, "de": "Kurz.\n\nAntworte NUR in JSON."}) + assert set(out) == {"fr", "de"} + assert out["de"] == f"Kurz.\n\n{STYLE_RULES}\n\nAntworte NUR in JSON." + + +def test_single_paragraph_prompt_gets_rules_appended(): + assert with_style_rules("Only one paragraph.") == f"Only one paragraph.\n\n{STYLE_RULES}" + + +def test_appellation_context_only_for_records_with_sub_denominations(monkeypatch): + from _lib import terroir_prompts as tp + from _lib import terroir_roster as tr + monkeypatch.setattr(tr, "_load", lambda: ({"rioja": ["Rioja Alavesa", "Rioja Alta", "Rioja Oriental"], + "big": [f"Sub {i}" for i in range(9)]}, + {"rioja": "Rioja", "big": "Big", "leaf": "Leaf"}, {})) + tr._load.cache_clear() if hasattr(tr._load, "cache_clear") else None + ctx = tp.appellation_context("rioja", for_translation=True) + assert "«Rioja»" in ctx and "3 sub-denominations (Rioja Alavesa, Rioja Alta, Rioja Oriental)" in ctx + assert "in the translation" in ctx and '"the Rioja appellation"' in ctx + assert "in the bullet" in tp.appellation_context("rioja", for_translation=False) + assert tp.appellation_context("leaf", for_translation=True) == "" + assert "Sub 5 and 3 more" in tp.appellation_context("big", for_translation=True) + assert tp.with_appellation_context("Translate:\n1. x", "leaf") == "Translate:\n1. x" + assert tp.with_appellation_context("Translate:\n1. x", "rioja").startswith("Translate:\n1. x\n\nCONTEXT:") diff --git a/tests/test_terroir_translation_rules.py b/tests/test_terroir_translation_rules.py new file mode 100644 index 0000000..7fb4e20 --- /dev/null +++ b/tests/test_terroir_translation_rules.py @@ -0,0 +1,98 @@ +"""Shared stage-02e translation rules (scripts/_lib/terroir_prompts.py, plan W1 + W5).""" +from __future__ import annotations + +import importlib.util +from pathlib import Path + +import pytest +from _lib.terroir_prompts import translation_rules, translation_system_prompt +from _lib.translation_glossary import glossary_for + +SCRIPTS = Path(__file__).resolve().parents[1] / "scripts" +COUNTRIES = ( + "at", "be", "bg", "ch", "cy", "cz", "de", "es", "gb", "gr", "hr", "hu", "it", "lu", "mt", + "nl", "pt", "ro", "si", "sk", +) +ROSTER = "appellation names (Barolo, Soave); grape names (Nebbiolo)" + + +def _load_script(path: Path, name: str): + spec = importlib.util.spec_from_file_location(name, path) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +@pytest.mark.parametrize("source_lang,target_lang", [("it", "en"), ("el", "nl"), ("bg", "fr"), ("de", "es")]) +def test_rules_block_has_no_braces(source_lang, target_lang): + rules = translation_rules(source_lang, target_lang, proper_nouns=ROSTER) + assert "{" not in rules and "}" not in rules + assert rules.startswith("Terminology rules (") + assert "In this corpus: " + ROSTER in rules + + +def test_roster_braces_are_neutralised_and_empty_roster_is_fine(): + rules = translation_rules("it", "en", proper_nouns="names {x}") + assert "{" not in rules and "}" not in rules + assert "In this corpus" not in translation_rules("it", "en", proper_nouns="") + + +def test_latin_script_rule_only_for_greek_and_cyrillic_sources(): + latin = "entirely in Latin script" + assert latin in translation_rules("el", "en", proper_nouns="") + assert latin in translation_rules("bg", "es", proper_nouns="") + for src in ("it", "fr", "hu", "de", "hr"): + assert latin not in translation_rules(src, "en", proper_nouns="") + + +def test_target_language_is_named_and_w5_rules_present(): + rules = translation_rules("it", "nl", proper_nouns="") + assert "Translate into Dutch" in rules + assert "PDO / PGI in their Dutch form" in rules + assert 'never "Pinot noir N"' in rules + assert "End every bullet with a period." in rules + assert "never strengthen them" in rules + + +def test_glossary_appended_for_en_and_nl_only(): + for target in ("en", "nl"): + out = translation_system_prompt("BASE", source_lang="it", target_lang=target, proper_nouns="") + assert out.endswith(glossary_for(target)) + out_es = translation_system_prompt("BASE", source_lang="it", target_lang="es", proper_nouns="") + assert glossary_for("en") not in out_es and glossary_for("nl") not in out_es + assert out_es.endswith(translation_rules("it", "es", proper_nouns="")) + + +def test_output_starts_with_base_prompt(): + base = "You translate short Italian bullets into English.\n\nRules:\n- a\n- b" + out = translation_system_prompt(base, source_lang="it", target_lang="en", proper_nouns=ROSTER) + assert out.startswith(base + "\n\n" + "Terminology rules (Italian → English):") + assert "{" not in out and "}" not in out + + +@pytest.mark.parametrize("cc", COUNTRIES) +def test_country_script_roster_and_only_flag(cc): + mod = _load_script(SCRIPTS / cc / "02e_translate_terroir_facts.py", f"owm_02e_{cc}") + roster = mod.PROPER_NOUNS + rosters = list(roster.values()) if isinstance(roster, dict) else [roster] + assert rosters and all(isinstance(r, str) and r for r in rosters) + for r in rosters: + assert "{" not in r and "}" not in r + for attr in ("SYSTEM_PROMPT", "SYSTEM_PROMPT_TEMPLATE", "SYSTEM_PROMPT_NL", "SYSTEM_PROMPT_FR"): + if hasattr(mod, attr): + assert "- Preserve " not in getattr(mod, attr) + assert mod.translation_system_prompt is translation_system_prompt + args = mod._build_argparser().parse_args(["--only", "a", "--only", "b"]) + assert args.only == ["a", "b"] and args.refresh is False + + +def test_fr_base_script_builds_through_the_shared_helper(): + mod = _load_script(SCRIPTS / "02e_translate_terroir_facts.py", "owm_02e_fr") + out = mod.build_system_prompt(source_lang="fr", target_lang="en") + assert out.startswith("You translate short French bullets") + assert "- Preserve " not in out + assert "Terminology rules (French → English):" in out + assert "Marnes à exogyra virgula" in out and "caillottes" in out + assert out.endswith(glossary_for("en")) + args = mod._build_argparser().parse_args(["--only", "chablis"]) + assert args.only == ["chablis"] diff --git a/tests/test_traditional_terms.py b/tests/test_traditional_terms.py new file mode 100644 index 0000000..fc7c7ba --- /dev/null +++ b/tests/test_traditional_terms.py @@ -0,0 +1,169 @@ +"""Shape guard for scripts/_lib/traditional_terms.json. + +The loader (scripts/_lib/gi_terms.py) is written against this exact schema; a +drift here silently drops a country's term from every panel, so the file is +validated structurally rather than trusted. +""" + +from __future__ import annotations + +import json +import re +from pathlib import Path + +import pytest + +PATH = Path(__file__).resolve().parents[1] / "scripts" / "_lib" / "traditional_terms.json" +LOCALES = {"en", "fr", "es", "nl"} +SCHEME_IDS = {"pdo", "pgi", "spirit-gi", "uk-pdo", "uk-pgi", "none"} +COUNTRIES = { + "fr", + "ch", + "it", + "es", + "pt", + "ro", + "at", + "de", + "mt", + "gb", + "lu", + "be", + "nl", + "si", + "hr", + "hu", + "bg", + "gr", + "cz", + "sk", + "cy", +} +ROSTER_KINDS = {("it", "DOP"), ("es", "DOP"), ("es", "IGP"), ("at", "DOP")} +SLUG_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$") +FILE_NUMBER_RE = re.compile(r"^(PDO|PGI)-[A-Z]{2}(\+[A-Z]{2})?-[A-Z0-9]+$") +VINTAGE_RE = re.compile(r"^\d{4}$") + + +@pytest.fixture(scope="module") +def raw() -> str: + return PATH.read_text(encoding="utf-8") + + +@pytest.fixture(scope="module") +def table(raw: str) -> dict: + return json.loads(raw) + + +def _assert_sources(sources: object, where: str) -> None: + assert isinstance(sources, list) and sources, f"{where}: sources must be a non-empty list" + for src in sources: + assert set(src) == {"label", "url"}, f"{where}: source keys must be label+url" + assert src["label"].strip(), f"{where}: empty source label" + assert re.match(r"^https?://", src["url"]), f"{where}: not a URL: {src['url']}" + + +def _assert_locales(block: object, where: str) -> None: + assert isinstance(block, dict) and set(block) == LOCALES, f"{where}: needs {sorted(LOCALES)}" + for lang, text in block.items(): + assert isinstance(text, str) and text.strip(), f"{where}.{lang}: empty" + + +def test_canonical_serialisation(raw: str, table: dict) -> None: + assert raw == json.dumps(table, ensure_ascii=False, indent=1, sort_keys=True) + "\n" + + +def test_top_level_keys(table: dict) -> None: + assert set(table) == {"__doc__", "schemes", "constants", "rulings", "pins", "terms"} + assert table["__doc__"].strip() + + +def test_schemes(table: dict) -> None: + assert set(table["schemes"]) == SCHEME_IDS + for sid, scheme in table["schemes"].items(): + assert set(scheme) == {"full", "note", "sources"}, sid + _assert_locales(scheme["full"], f"schemes.{sid}.full") + _assert_locales(scheme["note"], f"schemes.{sid}.note") + _assert_sources(scheme["sources"], f"schemes.{sid}") + + +def test_constants_cover_every_country(table: dict) -> None: + constants = table["constants"] + assert set(constants) == COUNTRIES + assert set(table["rulings"]) == COUNTRIES + for cc, kinds in constants.items(): + for kind, term in kinds.items(): + assert kind in {"AOC", "IGP", "EDV", "DOP"}, (cc, kind) + assert isinstance(term, str), (cc, kind) + assert (cc, kind) not in ROSTER_KINDS, f"{cc}:{kind} is roster-served, not constant" + for cc, kind in ROSTER_KINDS: + assert kind not in constants.get(cc, {}), f"{cc}:{kind} must be omitted (roster)" + + +def test_rulings(table: dict) -> None: + for cc, ruling in table["rulings"].items(): + assert set(ruling) == {"reason", "sources"}, cc + assert ruling["reason"].strip(), cc + _assert_sources(ruling["sources"], f"rulings.{cc}") + + +def test_pins(table: dict) -> None: + assert set(table["pins"]) <= COUNTRIES + for cc, by_file in table["pins"].items(): + assert by_file, cc + for key, entry in by_file.items(): + where = f"pins.{cc}.{key}" + # keyed by EU file number (rosters) or by slug (a record whose own + # fields cannot carry the fact, e.g. an empty SIQO categorie row) + assert FILE_NUMBER_RE.match(key) or SLUG_RE.match(key), where + assert {"term", "sources"} <= set(entry) <= { + "term", "since_vintage", "sources", "scheme", "note" + }, where + assert entry["term"].strip(), where + if "since_vintage" in entry: + assert VINTAGE_RE.match(entry["since_vintage"]), where + _assert_sources(entry["sources"], where) + + +def test_at_pins_match_eambrosia_dacs(table: dict) -> None: + index = Path(__file__).resolve().parents[1] / "raw" / "at" / "eambrosia" / "index.json" + if not index.exists(): + pytest.skip("raw/at/eambrosia/index.json not fetched") + wines = json.loads(index.read_text(encoding="utf-8"))["wines"] + by_file = {w["fileNumber"]: w for w in wines} + for file_number in table["pins"]["at"]: + assert file_number in by_file, file_number + assert by_file[file_number]["kind"] == "DOP", file_number + + +def test_terms(table: dict) -> None: + for key, term in table["terms"].items(): + cc, _, spelled = key.partition(":") + assert cc in COUNTRIES and spelled, key + assert {"full", "scheme", "note", "sources"} <= set(term), key + assert set(term) - {"full", "scheme", "note", "sources", "castilian_form"} == set(), key + assert term["scheme"] in table["schemes"], key + assert term["full"].strip(), key + _assert_locales(term["note"], f"terms.{key}.note") + _assert_sources(term["sources"], f"terms.{key}") + + +def test_every_constant_and_pin_term_is_defined(table: dict) -> None: + used = { + f"{cc}:{term}" + for cc, kinds in table["constants"].items() + for term in kinds.values() + if term + } + used |= { + f"{cc}:{entry['term']}" + for cc, by_file in table["pins"].items() + for entry in by_file.values() + } + missing = sorted(used - set(table["terms"])) + assert not missing, f"terms without a definition: {missing}" + + +def test_scheme_none_only_for_switzerland(table: dict) -> None: + for key, term in table["terms"].items(): + assert (term["scheme"] == "none") == key.startswith("ch:"), key