From 9553771f624bd7ef8dc89b660e93770509b0db33 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 04:39:46 +0000 Subject: [PATCH 1/5] Tooling API, stage 1: legaldown.grammar and legaldown.syntax Two new public modules that re-export, as the same objects, what downstream tools imported from private modules: the language's constants and rules that need no document (grammar), and the lexer, markers, fragments and Markdown helpers for reading source (syntax). Nothing moves and no diagnostic changes. New constants, defined where they are used: PARTY_TYPES and MAX_SECTION_LEVEL (validator.patterns, read by core and units), DRAFTING_MARKER (validator.templates, formerly private) and FINAL_CHECK_RULES (validator.core, read by the final check). slugify_identifier takes a keyword-only fallback= for text that yields no identifier. choose_problem(directive, questions) gives the first choose-invalid message for one directive; check_choose is built on the same logic. find_markers(document) defaults to the package lexer. Guard test tests/test_tooling_surface.py pins both __all__ lists, the identity of every re-export, and the public home of each deep import PactTrack and legaldown-render use today. README gains a Tooling API section. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_013WmBAc5T7UCKVdpUmg9qxz --- README.md | 84 +++++- src/legaldown/definitions.py | 2 +- src/legaldown/directives.py | 2 +- src/legaldown/grammar.py | 101 ++++++++ src/legaldown/syntax.py | 78 ++++++ src/legaldown/validator/core.py | 25 +- src/legaldown/validator/helpers.py | 20 +- src/legaldown/validator/patterns.py | 4 + src/legaldown/validator/templates.py | 51 ++-- src/legaldown/validator/units.py | 13 +- tests/test_tooling_surface.py | 374 +++++++++++++++++++++++++++ 11 files changed, 713 insertions(+), 41 deletions(-) create mode 100644 src/legaldown/grammar.py create mode 100644 src/legaldown/syntax.py create mode 100644 tests/test_tooling_surface.py diff --git a/README.md b/README.md index e179cb1..27494b0 100644 --- a/README.md +++ b/README.md @@ -279,8 +279,8 @@ The public API is what the `legaldown` and `legaldown.validator` packages export `__all__`). Changes to it are listed in the notes of each [GitHub release](https://github.com/ForLegalAI/legaldown-validator/releases); before 1.0 a minor release may change it, a patch release does not. Other modules are internal and may -change in any release; constants still only there are to be made public -([#34](https://github.com/ForLegalAI/legaldown-validator/issues/34)). +change in any release, except `legaldown.grammar` and `legaldown.syntax` (see +[Tooling API](#tooling-api)), which are public too. ### Working with the result @@ -429,6 +429,86 @@ or anchor inside it is recognized (§8.6, §11.4), so a clause commented out wit `` is not a section. As in CommonMark, a comment left unclosed runs to the end of the document. +## Tooling API + +Tools built on the validator (renderers, editors, importers) need the language itself: its +vocabulary, its value formats, and how the validator reads source text. Two modules hand that +over. They are part of the supported API and covered by the same versioning as `legaldown` +and `legaldown.validator`, and they are the only supported way to reach these names: every +module path below them (`legaldown.markdown`, `legaldown.validator.patterns`, …) is internal +and may be reorganised in any release. Their names are the validator's own objects, not copies, so +a tool and the validator cannot disagree. + +**`legaldown.grammar`** is what needs no document: constants and small rules. + +```python +from legaldown.grammar import ( + IDENTIFIER_RE, KNOWN_CURRENCIES, MAX_SECTION_LEVEL, PARTY_TYPES, VALID_PLACEHOLDER_TYPES, + is_valid_iso_date, slugify_identifier, +) + +is_valid_iso_date("2026-02-30") # False +slugify_identifier("1st Term") # "section-1st-term" (§5.3) +slugify_identifier("日本語", fallback="") # "": nothing usable, as against real text +``` + +| Names | What they are | +|---|---| +| `SPEC_VERSION`, `LEGALDOWN_EXTENSIONS` | The specification version implemented; the file extensions of LegalDown source (§2.1) | +| `KNOWN_DIRECTIVES`, `DIRECTIVE_PARAMS`, `PLACEHOLDER_TYPE_PARAMS` | The directive vocabulary and the parameters each defines (§11) | +| `IDENTIFIER_RE`, `RESERVED_VALUE_TYPES`, `VALID_DOC_TYPES`, `PARTY_TYPES`, `KNOWN_CURRENCIES`, `DELIMITER_PAIRS` | Identifier format (§5.3), reserved value-type names, document types, party types, ISO 4217 codes, accepted definition delimiters (§7.2) | +| `VALID_PLACEHOLDER_TYPES`, `DURATION_UNITS`, `VALID_DURATION_UNITS` | Placeholder types and duration units (§10); `DURATION_UNITS` is in the order the specification lists them | +| `LIST_KINDS`, `MAX_SECTION_LEVEL`, `MAX_QUOTE_DEPTH`, `MAX_LIST_DEPTH` | Block kinds that are lists; the deepest heading level (5); how deep the parser reads quotes and lists | +| `VALUE_QUESTION_TYPES`, `DECISION_QUESTION_TYPES`, `QUESTION_TYPES`, `DRAFTING_MARKER`, `FINAL_CHECK_RULES` | Template question types (§15.2); the drafting note's first line (§15.6); the rule ids of the final check (§15.9) | +| `slugify_identifier(text, *, fallback="section")`, `format_section_number` | The §5.3 identifier of a heading or term, and `fallback` where the text yields none (`""` tells that case apart); a dotted section number | +| `is_valid_iso_date`, `is_valid_numeric`, `is_valid_money_amount`, `is_positive_numeric` | Value checks (§3.10, §10) | +| `parse_condition` → `Condition`, `condition_problem`, `exclusive`, `Presence`, `ALWAYS` | Conditions (§15.3, §15.4): parse one, tell why one is invalid, tell whether two units can never appear together | +| `choose_problem(directive, questions)` | Why a `{{choose:}}` is invalid (choose-invalid, §15.5): the first message `validate` records for it, or `None` | + +**`legaldown.syntax`** is reading source text the way the validator does. + +```python +from legaldown import load +from legaldown.syntax import block_fragments, find_markers, lex + +document = load("contract.lgd") +for found in find_markers(document): # every marker and look-alike in the body + print(found.source, found.placed(template=False), found.misplaced) +for _section, _index, block in document.iter_blocks(): + for text, _anchor in block_fragments(block): + print([d.name for d in lex(text).directives]) +``` + +| Names | What they are | +|---|---| +| `lex`, `Lexed`, `Directive`, `iter_directives`, `is_escaped`, `format_value`, `collect_source_directives` | The directive lexer (§11.4) and its inverse; the `{{ref:}}` and `{{term:}}` targets of a document | +| `Marker`, `MARKER_RE`, `parse_marker`, `format_marker`, `is_look_alike` | Anchor and condition markers (§5.7, §15.3) | +| `find_markers(document)`, `FoundMarker`, `is_include_only` | Every marker and look-alike in a document's body, placed or not; `FoundMarker.placed(template)` says which apply (`validate` gives the placed ones as `result.index.placed_markers`) | +| `Fragment`, `ListFragment`, `block_fragments`, `list_fragments`, `text_fragments`, `list_items`, `item_text` | Where a block's text is, and a list's items | +| `Quote`, `block_quotes` | The block quotes in a block and whether each is a drafting note | +| `FRONTMATTER_RE`, `LINE_ENDING_RE`, `HTML_COMMENT_RE`, `FENCE_OPEN_RE`, `closes_fence`, `fence_end`, `dedent`, `indent_width`, `strip_text` | The Markdown rules the reading is built on: frontmatter, line endings, comments, fenced code, indentation | + +If you import one of these from a private module, use its public home: + +| From | Use | +|---|---| +| `legaldown.directives`: `PLACEHOLDER_TYPE_PARAMS` | `legaldown.grammar` | +| `legaldown.directives`: `format_value` | `legaldown.syntax` | +| `legaldown.definitions`: `text_fragments` | `legaldown.syntax` | +| `legaldown.definitions`: `DELIMITER_PAIRS` | `legaldown.grammar` (also top-level) | +| `legaldown.markdown`: `FENCE_OPEN_RE`, `HTML_COMMENT_RE`, `LINE_ENDING_RE`, `closes_fence`, `dedent`, `fence_end`, `indent_width`, `strip_text` | `legaldown.syntax` | +| `legaldown.markers`: `MARKER_RE`, `Marker`, `format_marker`, `is_look_alike`, `parse_marker` | `legaldown.syntax` | +| `legaldown.models`: `LIST_KINDS` | `legaldown.grammar` | +| `legaldown.parser`: `FRONTMATTER_RE` | `legaldown.syntax` | +| `legaldown.parser`: `MAX_QUOTE_DEPTH` | `legaldown.grammar` | +| `legaldown.validator.patterns`: `LEGALDOWN_EXTENSIONS`, `IDENTIFIER_RE`, `KNOWN_CURRENCIES`, `VALID_DOC_TYPES`, `VALID_DURATION_UNITS`, `VALID_PLACEHOLDER_TYPES` | `legaldown.grammar` | +| `legaldown.validator.templates`: `DECISION_QUESTION_TYPES`, `QUESTION_TYPES` | `legaldown.grammar` | +| `legaldown.validator.templates`: `block_quotes` | `legaldown.syntax` | +| `legaldown.validator.templates`: `check_choose` | `legaldown.grammar.choose_problem` | +| `legaldown.validator.units`: `find_markers`, `is_include_only` | `legaldown.syntax` | +| `legaldown.validator`: `parse_condition`, `condition_problem`, `exclusive`, `Presence`, `ALWAYS`, `is_valid_iso_date`, `is_valid_money_amount`, `is_positive_numeric`, `slugify_identifier` | `legaldown.grammar` | +| `legaldown.cli`: `_read_answers` | `legaldown.load_answers` | + ## What gets checked The full rule set with severities and examples lives in the specification (§16); this is the map: diff --git a/src/legaldown/definitions.py b/src/legaldown/definitions.py index caa8607..c2531ea 100644 --- a/src/legaldown/definitions.py +++ b/src/legaldown/definitions.py @@ -205,7 +205,7 @@ def block_fragments(block: Block) -> list[Fragment]: of a list item's first paragraph; a block quote or a table cell never is one. A list's fragments are its items' blocks', in order (``list_fragments``); a block quote's, those of the blocks it holds - (``parser.quote_content``), so that nothing written in one block — a + (each read as its own text), so that nothing written in one block — a code span or comment left open — runs into the next, and code and raw HTML in it hold none (§11.4). """ diff --git a/src/legaldown/directives.py b/src/legaldown/directives.py index 54780e6..126baaa 100644 --- a/src/legaldown/directives.py +++ b/src/legaldown/directives.py @@ -317,7 +317,7 @@ def lex(text: str) -> Lexed: *text* is inline text, such as a paragraph's: fenced code is not looked for. Text that can hold it (a code block's, a block quote's) has it blanked first, as block structure precedes inline structure - (``blank_fenced_code``; ``block_fragments`` gives such text so). It is + (``block_fragments`` gives such text so). It is read once, left to right, taking whichever of a directive, a comment, or a code span opens first. A directive is lexed from the source as written and consumes its own text, so a quoted diff --git a/src/legaldown/grammar.py b/src/legaldown/grammar.py new file mode 100644 index 0000000..ef1502f --- /dev/null +++ b/src/legaldown/grammar.py @@ -0,0 +1,101 @@ +"""legaldown.grammar — The language's constants and rules that need no document. + +The vocabulary the specification fixes: directive names and parameters +(§11), placeholder, duration and question types (§10, §15.2), party types +(§3.4), the identifier format (§5.3), currencies, file extensions (§2.1), +and the limits the parser and validator apply; and the small rules over them: +the §5.3 identifier of a text, the value formats of §10, and template +conditions (§15.3, §15.4). + +This module is part of the supported API and covered by semantic versioning. +Every name here is the validator's own object, not a copy: the validator and +a tool built on this module cannot disagree about the language. Import them +from here; the modules they are defined in are internal and may be +reorganised. + + from legaldown.grammar import VALID_PLACEHOLDER_TYPES, is_valid_iso_date + + VALID_PLACEHOLDER_TYPES # frozenset({"text", "date", "money", "duration"}) + is_valid_iso_date("2026-02-30") # False +""" +from __future__ import annotations + +from .definitions import DELIMITER_PAIRS +from .directives import DIRECTIVE_PARAMS, KNOWN_DIRECTIVES, PLACEHOLDER_TYPE_PARAMS +from .models import LIST_KINDS +from .parser import MAX_LIST_DEPTH, MAX_QUOTE_DEPTH +from .specification import SPEC_VERSION +from .validator.conditions import ALWAYS, Condition, Presence, condition_problem, exclusive, parse_condition +from .validator.core import FINAL_CHECK_RULES +from .validator.helpers import ( + format_section_number, + is_positive_numeric, + is_valid_iso_date, + is_valid_money_amount, + is_valid_numeric, + slugify_identifier, +) +from .validator.patterns import ( + DURATION_UNITS, + IDENTIFIER_RE, + KNOWN_CURRENCIES, + LEGALDOWN_EXTENSIONS, + MAX_SECTION_LEVEL, + PARTY_TYPES, + RESERVED_VALUE_TYPES, + VALID_DOC_TYPES, + VALID_DURATION_UNITS, + VALID_PLACEHOLDER_TYPES, +) +from .validator.templates import ( + DECISION_QUESTION_TYPES, + DRAFTING_MARKER, + QUESTION_TYPES, + VALUE_QUESTION_TYPES, + choose_problem, +) + +__all__ = [ + # The specification + "SPEC_VERSION", + "LEGALDOWN_EXTENSIONS", + # Directives (§11) + "KNOWN_DIRECTIVES", + "DIRECTIVE_PARAMS", + "PLACEHOLDER_TYPE_PARAMS", + # Identifiers and values (§3, §5.3, §10) + "IDENTIFIER_RE", + "RESERVED_VALUE_TYPES", + "VALID_DOC_TYPES", + "PARTY_TYPES", + "VALID_PLACEHOLDER_TYPES", + "DURATION_UNITS", + "VALID_DURATION_UNITS", + "KNOWN_CURRENCIES", + "DELIMITER_PAIRS", + # Structure and limits + "LIST_KINDS", + "MAX_SECTION_LEVEL", + "MAX_QUOTE_DEPTH", + "MAX_LIST_DEPTH", + # Templates (§15) + "VALUE_QUESTION_TYPES", + "DECISION_QUESTION_TYPES", + "QUESTION_TYPES", + "DRAFTING_MARKER", + "FINAL_CHECK_RULES", + # Rules over them + "slugify_identifier", + "format_section_number", + "is_valid_iso_date", + "is_valid_numeric", + "is_valid_money_amount", + "is_positive_numeric", + "parse_condition", + "Condition", + "condition_problem", + "exclusive", + "Presence", + "ALWAYS", + "choose_problem", +] diff --git a/src/legaldown/syntax.py b/src/legaldown/syntax.py new file mode 100644 index 0000000..9071033 --- /dev/null +++ b/src/legaldown/syntax.py @@ -0,0 +1,78 @@ +"""legaldown.syntax — Reading LegalDown source: lexer, markers, fragments, Markdown helpers. + +What a tool needs to read the text of a document the way the validator does: +the directive lexer (§11.4), anchor and condition markers (§5.7, §15.3), the +text fragments of a block that may hold either, block quotes and drafting +notes (§15.6), and the Markdown helpers (fences, indentation, comments) +those readings are built on. + +This module is part of the supported API and covered by semantic versioning. +Every name here is the validator's own object, not a copy, so a tool that +reads source with it finds what the validator finds. Import them from here; +the modules they are defined in are internal and may be reorganised. + + from legaldown import load + from legaldown.syntax import find_markers + + for found in find_markers(load("contract.lgd")): + print(found.source, found.misplaced or "placed") +""" +from __future__ import annotations + +from .definitions import Fragment, ListFragment, block_fragments, list_fragments, text_fragments +from .directives import Directive, Lexed, format_value, is_escaped, iter_directives, lex +from .markdown import ( + FENCE_OPEN_RE, + HTML_COMMENT_RE, + LINE_ENDING_RE, + closes_fence, + dedent, + fence_end, + indent_width, + strip_text, +) +from .markers import MARKER_RE, Marker, format_marker, is_look_alike, parse_marker +from .models import item_text, list_items +from .parser import FRONTMATTER_RE, collect_source_directives +from .validator.templates import Quote, block_quotes +from .validator.units import FoundMarker, find_markers, is_include_only + +__all__ = [ + # Directives (§11) + "lex", + "Lexed", + "Directive", + "iter_directives", + "is_escaped", + "format_value", + "collect_source_directives", + # Markers (§5.7, §15.3) + "Marker", + "MARKER_RE", + "parse_marker", + "format_marker", + "is_look_alike", + "FoundMarker", + "find_markers", + "is_include_only", + # Where text is: fragments, lists, quotes + "Fragment", + "ListFragment", + "block_fragments", + "list_fragments", + "text_fragments", + "Quote", + "block_quotes", + "list_items", + "item_text", + # Markdown + "FRONTMATTER_RE", + "LINE_ENDING_RE", + "HTML_COMMENT_RE", + "FENCE_OPEN_RE", + "closes_fence", + "fence_end", + "dedent", + "indent_width", + "strip_text", +] diff --git a/src/legaldown/validator/core.py b/src/legaldown/validator/core.py index 3e80d4a..be2983b 100644 --- a/src/legaldown/validator/core.py +++ b/src/legaldown/validator/core.py @@ -44,6 +44,8 @@ IDENTIFIER_RE, KNOWN_CURRENCIES, LEGALDOWN_EXTENSIONS, + MAX_SECTION_LEVEL, + PARTY_TYPES, RESERVED_VALUE_TYPES, VALID_DOC_TYPES, VALID_DURATION_UNITS, @@ -386,6 +388,12 @@ def _check_blank_codes(blanks: dict[str, Blank], result: _Recorder) -> None: ) +_UNFILLED = "placeholder-unfilled" +_CONSTRUCT_PRESENT = "template-construct-present" +#: The rules of the final check (§15.9): the only ones it reports. +FINAL_CHECK_RULES: frozenset[str] = frozenset({_UNFILLED, _CONSTRUCT_PRESENT}) + + def _check_final( placeholders: list[tuple[Directive, Line]], chooses: list[tuple[Directive, Line]], @@ -401,14 +409,14 @@ def _check_final( written, and their lines).""" for directive, line in placeholders: result.error( - "placeholder-unfilled", + _UNFILLED, f"'{directive.source}' is an unfilled blank in a document meant to be final (§15.9).", line=line, ) def construct(what: str, line: int | None) -> None: result.error( - "template-construct-present", + _CONSTRUCT_PRESENT, f"{what} remains in a document meant to be final (§15.9).", line=line, ) @@ -852,10 +860,11 @@ def named(text: str, name: str) -> list[Directive]: ) continue seen_party_names.add(party_name) - if party.type not in ("legal_entity", "natural_person"): + if party.type not in PARTY_TYPES: + allowed = " or ".join(map(repr, PARTY_TYPES)) result.error( "party-type-invalid", - f"Party '{party_name}' has invalid type '{party.type}'. Must be 'legal_entity' or 'natural_person'.", + f"Party '{party_name}' has invalid type '{party.type}'. Must be {allowed}.", line=where.key(*party_path, "type"), ) with result.at(where.key(*party_path, "date_of_birth")): @@ -1008,7 +1017,7 @@ def free_identifier(base: str, presence: Presence) -> str: # headings all start at ## (an attachment or include file, which has no # # heading) numbers them 1, 2, … A level that a heading skips counts # as 1 (see below), so no two sections get the same number (#38). - shallowest = min((min(max(s.level, 1), 5) for s in document.sections), default=1) + shallowest = min((min(max(s.level, 1), MAX_SECTION_LEVEL) for s in document.sections), default=1) path_stack: list[str] = [] # The level before the first heading: 0 in a main document, whose first # heading must be at level 1 (§4.1). A document without frontmatter may @@ -1027,13 +1036,13 @@ def free_identifier(base: str, presence: Presence) -> str: # one here would shift every later section's number and drop the last # one from rendered output. Numbering clamps into the valid range. level = section.level - if level < 1 or level > 5: + if level < 1 or level > MAX_SECTION_LEVEL: result.error( "heading-depth", f"Section '{section.title}' uses unsupported heading level " - f"{section.level}. LegalDown supports levels 1-5 (§4.1).", line=where.heading(section_index), + f"{section.level}. LegalDown supports levels 1-{MAX_SECTION_LEVEL} (§4.1).", line=where.heading(section_index), ) - level = min(max(level, 1), 5) + level = min(max(level, 1), MAX_SECTION_LEVEL) if last_level == 0 and level > 1: result.error( "heading-skip", diff --git a/src/legaldown/validator/helpers.py b/src/legaldown/validator/helpers.py index 5a2642b..a97264e 100644 --- a/src/legaldown/validator/helpers.py +++ b/src/legaldown/validator/helpers.py @@ -33,9 +33,10 @@ def _ascii_text(value: str) -> tuple[str, bool]: return "".join(c for c in text if c.isascii()), lossy -def generate_identifier(value: str) -> tuple[str, bool]: +def generate_identifier(value: str, fallback: str = "section") -> tuple[str, bool]: """The identifier §5.3 generates from heading or term text *value* - (without its trailing marker), and whether a letter or digit without an + (without its trailing marker), or *fallback* when the text yields none, + and whether a letter or digit without an ASCII form, such as Cyrillic or CJK text, was dropped: the identifier then lost information, and an explicit one is recommended. @@ -48,14 +49,21 @@ def generate_identifier(value: str) -> tuple[str, bool]: text = re.sub(r"-{2,}", "-", text).strip("-") text = text[:64].rstrip("-") if not text: - return "section", lossy + return fallback, lossy # The prefix is exempt from the 64-character maximum: no re-truncation. return (text if "a" <= text[0] <= "z" else f"section-{text}"), lossy -def slugify_identifier(value: str) -> str: - """The identifier §5.3 generates from *value* (see generate_identifier).""" - return generate_identifier(value)[0] +def slugify_identifier(value: str, *, fallback: str = "section") -> str: + """The identifier §5.3 generates from heading or term text *value*: + lower case, ASCII, hyphen-separated, at most 64 characters, and prefixed + ``section-`` when it would begin with a digit. + + *fallback* is returned when the text yields no identifier at all (it + holds no letter or digit with an ASCII form), which §5.3 names + ``section``; pass ``""`` to tell that case from real text. + """ + return generate_identifier(value, fallback)[0] def format_section_number(counters: list[int], level: int) -> str: diff --git a/src/legaldown/validator/patterns.py b/src/legaldown/validator/patterns.py index 58971fa..f4f4f2a 100644 --- a/src/legaldown/validator/patterns.py +++ b/src/legaldown/validator/patterns.py @@ -13,6 +13,10 @@ # the built-in field specs or placeholder types. RESERVED_VALUE_TYPES: frozenset[str] = frozenset({"date", "money", "duration", "party", "text"}) VALID_DOC_TYPES: frozenset[str] = frozenset({"contract", "unilateral_act", "collective_act"}) +# The `type` a party may have (§3.4). +PARTY_TYPES: tuple[str, ...] = ("legal_entity", "natural_person") +# The deepest heading level LegalDown supports (§4.1). +MAX_SECTION_LEVEL = 5 # §10.5: the bare unit "M" is deliberately undefined (ISO 8601 ambiguity); # validators reject it with a hint suggesting MIN (minutes) or MO (months). # In the order §10.5 lists them, for diagnostics. diff --git a/src/legaldown/validator/templates.py b/src/legaldown/validator/templates.py index 9f6ffee..947c64e 100644 --- a/src/legaldown/validator/templates.py +++ b/src/legaldown/validator/templates.py @@ -264,7 +264,8 @@ def invalid(message: str) -> None: #: The first line of a quote that looks like a GitHub-flavoured alert. _ALERT_RE = re.compile(r"\[![A-Za-z]+\]") -_DRAFTING_MARKER = "[!DRAFTING]" +#: The first line that makes a quote a drafting note, letters in any case (§15.6). +DRAFTING_MARKER = "[!DRAFTING]" @dataclass(frozen=True, slots=True) @@ -279,7 +280,7 @@ class Quote: def is_drafting_note(self) -> bool: """True if the quote is a drafting note: its first line is exactly ``[!DRAFTING]``, letters in any case (§15.6).""" - return self.first_line.upper() == _DRAFTING_MARKER + return self.first_line.upper() == DRAFTING_MARKER @property def is_unrecognized_alert(self) -> bool: @@ -385,17 +386,14 @@ def insertion_boundary_problem( BRACE_STRAY = "'{{' does not begin a directive and is literal text; write '\\{{' if that is intended (§11.4)." -def check_choose(directive: Directive, questions: Any, result: _Recorder) -> None: - """Report a ``{{choose:}}`` that does not list exactly the answers of a - declared decision question (choose-invalid, §15.5), and any ``{{`` in - its phrases, which is literal text (brace-stray).""" +def _choose_problems(directive: Directive, questions: Any) -> list[str]: + """Every way a ``{{choose:}}`` fails to list exactly the answers of a + declared decision question (choose-invalid, §15.5), as messages.""" + problems: list[str] = [] qid = directive.positional or "" qtype = question_type(questions, qid) if qtype not in DECISION_QUESTION_TYPES: - result.error( - "choose-invalid", - f"'{directive.source}' must name a declared boolean or choice question (§15.5).", - ) + problems.append(f"'{directive.source}' must name a declared boolean or choice question (§15.5).") elif qtype == "boolean" or _choices_problem(questions[qid].get("choices")) is None: # Malformed choices are question-invalid; there is nothing to match. answers = ( @@ -405,23 +403,38 @@ def check_choose(directive: Directive, questions: Any, result: _Recorder) -> Non extra = [param for param in directive.params if param not in answers] repeated = [param for param in dict.fromkeys(directive.duplicates) if param in answers] if missing: - result.error( - "choose-invalid", + problems.append( f"'{directive.source}' lists no phrase for {', '.join(missing)}: every answer " - f"to '{qid}' needs one (§15.5).", + f"to '{qid}' needs one (§15.5)." ) if extra: - result.error( - "choose-invalid", + problems.append( f"'{directive.source}' lists {', '.join(extra)}, which " - f"{'is not an answer' if len(extra) == 1 else 'are not answers'} to '{qid}' (§15.5).", + f"{'is not an answer' if len(extra) == 1 else 'are not answers'} to '{qid}' (§15.5)." ) if repeated: - result.error( - "choose-invalid", + problems.append( f"'{directive.source}' lists {', '.join(repeated)} more than once; each answer " - f"has one phrase (§15.5).", + f"has one phrase (§15.5)." ) + return problems + + +def choose_problem(directive: Directive, questions: Any) -> str | None: + """Why the ``{{choose:}}`` *directive* is not valid for *questions* (the + document's ``metadata.questions``): the first choose-invalid message + (§15.5) the validator reports for it, or None when it names a declared + boolean or choice question and lists exactly its answers.""" + problems = _choose_problems(directive, questions) + return problems[0] if problems else None + + +def check_choose(directive: Directive, questions: Any, result: _Recorder) -> None: + """Report a ``{{choose:}}`` that does not list exactly the answers of a + declared decision question (choose-invalid, §15.5), and any ``{{`` in + its phrases, which is literal text (brace-stray).""" + for problem in _choose_problems(directive, questions): + result.error("choose-invalid", problem) for offset in range(2, len(directive.source) - 1): if directive.source.startswith("{{", offset) and not is_escaped(directive.source, offset): result.warning("brace-stray", BRACE_STRAY) diff --git a/src/legaldown/validator/units.py b/src/legaldown/validator/units.py index 927157a..b280a5c 100644 --- a/src/legaldown/validator/units.py +++ b/src/legaldown/validator/units.py @@ -14,11 +14,12 @@ from dataclasses import dataclass from typing import Any -from ..directives import Directive, Lexed, is_escaped +from ..directives import Directive, Lexed, is_escaped, lex from ..markdown import HTML_COMMENT_RE from ..markers import MARKER_RE, Marker, is_look_alike, parse_marker from ..models import Document from .conditions import ALWAYS, Condition, Presence, condition_problem, parse_condition +from .patterns import MAX_SECTION_LEVEL _PARAGRAPHS = ("paragraph", "definition", "ref", "term") _LISTS = ("ordered_list", "unordered_list") @@ -102,9 +103,13 @@ def marker_matches(text: str, lexed: Lexed) -> Iterator[re.Match[str]]: yield match -def find_markers(document: Document, lex_fragment: Callable[[str], Lexed]) -> list[FoundMarker]: +def find_markers(document: Document, lex_fragment: Callable[[str], Lexed] = lex) -> list[FoundMarker]: """Every marker and look-alike in the document's body text, outside code, - comments, directives, and escapes (§11.4), with its place.""" + comments, directives, and escapes (§11.4), with its place: placed or not. + ``FoundMarker.placed(template)`` says which of them apply. + + *lex_fragment* reads a text's directives; by default, the package lexer. + """ # Imported here, as in core.py: definitions -> validator.helpers -> # validator/__init__ -> core -> units would otherwise be a cycle. from ..definitions import block_fragments @@ -190,7 +195,7 @@ def __init__( self._section_enclosing: list[Presence] = [] stack: list[tuple[int, Presence]] = [] # (level, presence) of open sections for section in document.sections: - level = min(max(section.level, 1), 5) # clamped, as numbering does + level = min(max(section.level, 1), MAX_SECTION_LEVEL) # clamped, as numbering does while stack and stack[-1][0] >= level: stack.pop() enclosing = stack[-1][1] if stack else ALWAYS diff --git a/tests/test_tooling_surface.py b/tests/test_tooling_surface.py new file mode 100644 index 0000000..fbd88e8 --- /dev/null +++ b/tests/test_tooling_surface.py @@ -0,0 +1,374 @@ +"""The tooling surface: ``legaldown.grammar`` and ``legaldown.syntax``, the +semver-covered home of what downstream tools used to import from private +modules, and the small helpers they were given with it.""" +from __future__ import annotations + +import importlib + +import pytest + +import legaldown +import legaldown.grammar as grammar +import legaldown.syntax as syntax +from legaldown import load_answers, parse, validate +from legaldown.directives import lex +from legaldown.syntax import find_markers +from legaldown.validator.core import FINAL_CHECK_RULES +from legaldown.validator.helpers import generate_identifier +from legaldown.validator.templates import check_choose, choose_problem + +GRAMMAR = [ + "SPEC_VERSION", "LEGALDOWN_EXTENSIONS", + "KNOWN_DIRECTIVES", "DIRECTIVE_PARAMS", "PLACEHOLDER_TYPE_PARAMS", + "IDENTIFIER_RE", "RESERVED_VALUE_TYPES", "VALID_DOC_TYPES", "PARTY_TYPES", + "VALID_PLACEHOLDER_TYPES", "DURATION_UNITS", "VALID_DURATION_UNITS", "KNOWN_CURRENCIES", + "DELIMITER_PAIRS", + "LIST_KINDS", "MAX_SECTION_LEVEL", "MAX_QUOTE_DEPTH", "MAX_LIST_DEPTH", + "VALUE_QUESTION_TYPES", "DECISION_QUESTION_TYPES", "QUESTION_TYPES", + "DRAFTING_MARKER", "FINAL_CHECK_RULES", + "slugify_identifier", "format_section_number", "is_valid_iso_date", "is_valid_numeric", + "is_valid_money_amount", "is_positive_numeric", + "parse_condition", "Condition", "condition_problem", "exclusive", "Presence", "ALWAYS", + "choose_problem", +] + +SYNTAX = [ + "lex", "Lexed", "Directive", "iter_directives", "is_escaped", "format_value", + "collect_source_directives", + "Marker", "MARKER_RE", "parse_marker", "format_marker", "is_look_alike", + "FoundMarker", "find_markers", "is_include_only", + "Fragment", "ListFragment", "block_fragments", "list_fragments", "text_fragments", + "Quote", "block_quotes", "list_items", "item_text", + "FRONTMATTER_RE", "LINE_ENDING_RE", "HTML_COMMENT_RE", "FENCE_OPEN_RE", + "closes_fence", "fence_end", "dedent", "indent_width", "strip_text", +] + +#: Where each public name is implemented. +IMPLEMENTATION = { + "DELIMITER_PAIRS": "legaldown.definitions", + "Fragment": "legaldown.definitions", + "ListFragment": "legaldown.definitions", + "block_fragments": "legaldown.definitions", + "list_fragments": "legaldown.definitions", + "text_fragments": "legaldown.definitions", + "DIRECTIVE_PARAMS": "legaldown.directives", + "KNOWN_DIRECTIVES": "legaldown.directives", + "PLACEHOLDER_TYPE_PARAMS": "legaldown.directives", + "Directive": "legaldown.directives", + "Lexed": "legaldown.directives", + "format_value": "legaldown.directives", + "is_escaped": "legaldown.directives", + "iter_directives": "legaldown.directives", + "lex": "legaldown.directives", + "FENCE_OPEN_RE": "legaldown.markdown", + "HTML_COMMENT_RE": "legaldown.markdown", + "LINE_ENDING_RE": "legaldown.markdown", + "closes_fence": "legaldown.markdown", + "dedent": "legaldown.markdown", + "fence_end": "legaldown.markdown", + "indent_width": "legaldown.markdown", + "strip_text": "legaldown.markdown", + "MARKER_RE": "legaldown.markers", + "Marker": "legaldown.markers", + "format_marker": "legaldown.markers", + "is_look_alike": "legaldown.markers", + "parse_marker": "legaldown.markers", + "LIST_KINDS": "legaldown.models", + "item_text": "legaldown.models", + "list_items": "legaldown.models", + "FRONTMATTER_RE": "legaldown.parser", + "MAX_LIST_DEPTH": "legaldown.parser", + "MAX_QUOTE_DEPTH": "legaldown.parser", + "collect_source_directives": "legaldown.parser", + "SPEC_VERSION": "legaldown.specification", + "ALWAYS": "legaldown.validator.conditions", + "Condition": "legaldown.validator.conditions", + "Presence": "legaldown.validator.conditions", + "condition_problem": "legaldown.validator.conditions", + "exclusive": "legaldown.validator.conditions", + "parse_condition": "legaldown.validator.conditions", + "FINAL_CHECK_RULES": "legaldown.validator.core", + "format_section_number": "legaldown.validator.helpers", + "is_positive_numeric": "legaldown.validator.helpers", + "is_valid_iso_date": "legaldown.validator.helpers", + "is_valid_money_amount": "legaldown.validator.helpers", + "is_valid_numeric": "legaldown.validator.helpers", + "slugify_identifier": "legaldown.validator.helpers", + "DURATION_UNITS": "legaldown.validator.patterns", + "IDENTIFIER_RE": "legaldown.validator.patterns", + "KNOWN_CURRENCIES": "legaldown.validator.patterns", + "LEGALDOWN_EXTENSIONS": "legaldown.validator.patterns", + "MAX_SECTION_LEVEL": "legaldown.validator.patterns", + "PARTY_TYPES": "legaldown.validator.patterns", + "RESERVED_VALUE_TYPES": "legaldown.validator.patterns", + "VALID_DOC_TYPES": "legaldown.validator.patterns", + "VALID_DURATION_UNITS": "legaldown.validator.patterns", + "VALID_PLACEHOLDER_TYPES": "legaldown.validator.patterns", + "DECISION_QUESTION_TYPES": "legaldown.validator.templates", + "DRAFTING_MARKER": "legaldown.validator.templates", + "QUESTION_TYPES": "legaldown.validator.templates", + "Quote": "legaldown.validator.templates", + "VALUE_QUESTION_TYPES": "legaldown.validator.templates", + "block_quotes": "legaldown.validator.templates", + "choose_problem": "legaldown.validator.templates", + "FoundMarker": "legaldown.validator.units", + "find_markers": "legaldown.validator.units", + "is_include_only": "legaldown.validator.units", +} + +#: What PactTrack and legaldown-render import today from private modules: +#: (module, name) -> (public module, public name). +DOWNSTREAM = { + # PactTrack + ("legaldown.definitions", "text_fragments"): ("legaldown.syntax", "text_fragments"), + ("legaldown.definitions", "DELIMITER_PAIRS"): ("legaldown.grammar", "DELIMITER_PAIRS"), + ("legaldown.directives", "PLACEHOLDER_TYPE_PARAMS"): ("legaldown.grammar", "PLACEHOLDER_TYPE_PARAMS"), + ("legaldown.directives", "format_value"): ("legaldown.syntax", "format_value"), + ("legaldown.markdown", "FENCE_OPEN_RE"): ("legaldown.syntax", "FENCE_OPEN_RE"), + ("legaldown.markdown", "HTML_COMMENT_RE"): ("legaldown.syntax", "HTML_COMMENT_RE"), + ("legaldown.markdown", "LINE_ENDING_RE"): ("legaldown.syntax", "LINE_ENDING_RE"), + ("legaldown.markdown", "dedent"): ("legaldown.syntax", "dedent"), + ("legaldown.markdown", "fence_end"): ("legaldown.syntax", "fence_end"), + ("legaldown.markdown", "strip_text"): ("legaldown.syntax", "strip_text"), + ("legaldown.markers", "MARKER_RE"): ("legaldown.syntax", "MARKER_RE"), + ("legaldown.markers", "Marker"): ("legaldown.syntax", "Marker"), + ("legaldown.markers", "format_marker"): ("legaldown.syntax", "format_marker"), + ("legaldown.markers", "is_look_alike"): ("legaldown.syntax", "is_look_alike"), + ("legaldown.markers", "parse_marker"): ("legaldown.syntax", "parse_marker"), + ("legaldown.models", "LIST_KINDS"): ("legaldown.grammar", "LIST_KINDS"), + ("legaldown.parser", "FRONTMATTER_RE"): ("legaldown.syntax", "FRONTMATTER_RE"), + ("legaldown.parser", "MAX_QUOTE_DEPTH"): ("legaldown.grammar", "MAX_QUOTE_DEPTH"), + ("legaldown.validator.patterns", "LEGALDOWN_EXTENSIONS"): ("legaldown.grammar", "LEGALDOWN_EXTENSIONS"), + ("legaldown.validator.templates", "DECISION_QUESTION_TYPES"): ("legaldown.grammar", "DECISION_QUESTION_TYPES"), + ("legaldown.validator.templates", "QUESTION_TYPES"): ("legaldown.grammar", "QUESTION_TYPES"), + ("legaldown.validator.templates", "block_quotes"): ("legaldown.syntax", "block_quotes"), + ("legaldown.validator.templates", "check_choose"): ("legaldown.grammar", "choose_problem"), + ("legaldown.validator.units", "find_markers"): ("legaldown.syntax", "find_markers"), + ("legaldown.validator.units", "is_include_only"): ("legaldown.syntax", "is_include_only"), + ("legaldown.validator", "IDENTIFIER_RE"): ("legaldown.grammar", "IDENTIFIER_RE"), + ("legaldown.validator", "KNOWN_CURRENCIES"): ("legaldown.grammar", "KNOWN_CURRENCIES"), + ("legaldown.validator", "VALID_DOC_TYPES"): ("legaldown.grammar", "VALID_DOC_TYPES"), + ("legaldown.validator", "VALID_DURATION_UNITS"): ("legaldown.grammar", "VALID_DURATION_UNITS"), + ("legaldown.validator", "VALID_PLACEHOLDER_TYPES"): ("legaldown.grammar", "VALID_PLACEHOLDER_TYPES"), + ("legaldown.validator", "is_valid_iso_date"): ("legaldown.grammar", "is_valid_iso_date"), + ("legaldown.validator", "parse_condition"): ("legaldown.grammar", "parse_condition"), + ("legaldown.validator", "slugify_identifier"): ("legaldown.grammar", "slugify_identifier"), + # legaldown-render + ("legaldown.markdown", "closes_fence"): ("legaldown.syntax", "closes_fence"), + ("legaldown.markdown", "indent_width"): ("legaldown.syntax", "indent_width"), + ("legaldown.cli", "_read_answers"): ("legaldown", "load_answers"), + ("legaldown.validator", "ALWAYS"): ("legaldown.grammar", "ALWAYS"), + ("legaldown.validator", "Presence"): ("legaldown.grammar", "Presence"), + ("legaldown.validator", "condition_problem"): ("legaldown.grammar", "condition_problem"), + ("legaldown.validator", "exclusive"): ("legaldown.grammar", "exclusive"), + ("legaldown.validator", "is_positive_numeric"): ("legaldown.grammar", "is_positive_numeric"), + ("legaldown.validator", "is_valid_money_amount"): ("legaldown.grammar", "is_valid_money_amount"), +} + +#: TODO: downstream imports whose public home arrives in a later stage. Not +#: asserted yet; the stage that adds the home moves the entry into DOWNSTREAM. +DOWNSTREAM_TODO = { + ("legaldown.parser", "quote_content"): "syntax.quote_blocks (stage 3)", + ("legaldown.parser", "_layout"): "Document.layout() (stage 3)", +} + + +def test_all_lists_are_the_expected_names(): + assert sorted(grammar.__all__) == sorted(GRAMMAR) + assert sorted(syntax.__all__) == sorted(SYNTAX) + assert len(set(grammar.__all__)) == len(grammar.__all__) + assert len(set(syntax.__all__)) == len(syntax.__all__) + + +def test_the_tooling_names_stay_out_of_the_top_level_package(): + top = set(legaldown.__all__) + assert top.isdisjoint({"PARTY_TYPES", "MAX_SECTION_LEVEL", "choose_problem", "find_markers"}) + + +@pytest.mark.parametrize("module", [grammar, syntax]) +def test_every_name_is_documented_and_defined_elsewhere(module): + for name in module.__all__: + implementation = importlib.import_module(IMPLEMENTATION[name]) + assert getattr(module, name) is getattr(implementation, name), name + # A constant has no docstring of its own; everything else must. + value = getattr(module, name) + if callable(value) or isinstance(value, type): + assert (value.__doc__ or "").strip(), name + + +def test_every_implementation_is_listed(): + assert set(IMPLEMENTATION) == set(GRAMMAR) | set(SYNTAX) + + +@pytest.mark.parametrize(("deep", "public"), DOWNSTREAM.items(), ids=lambda key: ".".join(key)) +def test_a_downstream_import_has_a_public_home(deep, public): + home = getattr(importlib.import_module(public[0]), public[1]) + old = getattr(importlib.import_module(deep[0]), deep[1]) # the deep path keeps working + if deep[1] == public[1]: + assert home is old + + +def test_the_downstream_todo_is_not_yet_a_public_home(): + # Remove an entry here when its stage lands, and add it to DOWNSTREAM. + for (module, name), home in DOWNSTREAM_TODO.items(): + assert hasattr(importlib.import_module(module), name), (module, name, home) + assert load_answers is not None + + +# ── The constants the validator now reads ───────────────────────── + + +_PARTY = "---\ntitle: T\nsides:\n - name: providers\n parties:\n - name: acme\n type: {type}\n" +_SECOND = " - name: clients\n parties:\n - name: beta\n type: legal_entity\n---\n\n# A\n\nText.\n" + + +def test_party_types_are_what_the_validator_accepts(): + assert grammar.PARTY_TYPES == ("legal_entity", "natural_person") + for party_type in grammar.PARTY_TYPES: + assert "party-type-invalid" not in validate(parse(_PARTY.format(type=party_type) + _SECOND)).rules() + result = validate(parse(_PARTY.format(type="company") + _SECOND)) + [message] = [d.message for d in result.diagnostics if d.rule == "party-type-invalid"] + assert message == ( + "Party 'acme' has invalid type 'company'. Must be 'legal_entity' or 'natural_person'." + ) + + +def test_the_section_level_limit_is_what_the_validator_enforces(): + assert grammar.MAX_SECTION_LEVEL == 5 + deepest = "---\ntitle: T\n---\n\n" + "".join(f"{'#' * n} H{n}\n\nText.\n\n" for n in range(1, 6)) + assert "heading-depth" not in validate(parse(deepest)).rules() + result = validate(parse(deepest + "###### H6\n\nText.\n")) + [message] = [d.message for d in result.diagnostics if d.rule == "heading-depth"] + assert message == ( + "Section 'H6' uses unsupported heading level 6. LegalDown supports levels 1-5 (§4.1)." + ) + + +def test_the_drafting_marker_makes_a_drafting_note(): + assert grammar.DRAFTING_MARKER == "[!DRAFTING]" + note = parse(f"---\ntitle: T\n---\n\n# A\n\n> {grammar.DRAFTING_MARKER}\n> Guidance.\n") + [quote] = syntax.block_quotes(note.sections[0].blocks[0]) + assert quote.is_drafting_note + + +_TEMPLATE = """--- +title: Fixture +sides: + - name: providers + parties: + - name: acme + type: legal_entity + - name: clients + parties: + - name: beta + type: legal_entity +questions: + vat: + type: boolean +--- + +# Terms {#terms when=vat} + +Pay {{placeholder: fee, currency=EUR}} {{choose: vat, true=a, false=b}}. + +> [!DRAFTING] +> Check the fee. +""" + + +def test_the_final_check_reports_only_its_rules_and_both_of_them(): + assert frozenset({"placeholder-unfilled", "template-construct-present"}) == grammar.FINAL_CHECK_RULES + assert FINAL_CHECK_RULES is grammar.FINAL_CHECK_RULES + document = parse(_TEMPLATE) + template_rules = validate(document).rules() + final_rules = validate(document, final=True).rules() + assert not FINAL_CHECK_RULES & template_rules + assert (final_rules - template_rules) == set(FINAL_CHECK_RULES) + + +# ── slugify_identifier(fallback=) ───────────────────────────────── + + +@pytest.mark.parametrize("text", ["", " ", "---", "日本語", "", "!?"]) +def test_a_text_without_an_identifier_yields_the_fallback(text): + assert grammar.slugify_identifier(text) == "section" + assert grammar.slugify_identifier(text, fallback="") == "" + assert grammar.slugify_identifier(text, fallback="x") == "x" + assert generate_identifier(text, "")[0] == "" + + +@pytest.mark.parametrize("text", ["Fees", "1st Term", "Zahlung & Frist", "Section"]) +def test_the_fallback_does_not_change_a_text_with_an_identifier(text): + assert grammar.slugify_identifier(text, fallback="") == grammar.slugify_identifier(text) + assert grammar.slugify_identifier(text, fallback="") != "" + + +def test_a_leading_digit_still_gets_the_section_prefix(): + assert grammar.slugify_identifier("1st Term", fallback="") == "section-1st-term" + + +def test_the_fallback_is_keyword_only(): + with pytest.raises(TypeError): + grammar.slugify_identifier("x", "y") # type: ignore[misc] + + +# ── choose_problem ──────────────────────────────────────────────── + +_QUESTIONS = { + "vat": {"type": "boolean"}, + "forum": {"type": "choice", "choices": {"courts": "Courts", "arbitration": "Arbitration"}}, + "broken": {"type": "choice", "choices": "nope"}, + "fee": {"type": "money"}, +} + + +@pytest.mark.parametrize( + "source", + [ + "{{choose: vat, true=a, false=b}}", + "{{choose: forum, courts=a, arbitration=b}}", + "{{choose: forum, courts=a}}", + "{{choose: vat, true=x, false=y, maybe=z}}", + "{{choose: forum, courts=a, arbitration=b, note=c, other=d}}", + "{{choose: vat, true=x, false=y, true=z}}", + "{{choose: forum, courts=a, courts=b}}", + "{{choose: fee, a=x, b=y}}", + "{{choose: missing, true=x, false=y}}", + "{{choose: broken, a=x}}", + "{{choose: vat}}", + "{{choose: }}", + "{{choose: vat, true=x{{y, false=z}}", + ], +) +def test_choose_problem_is_the_first_message_check_choose_records(source): + from legaldown.validator.result import _Recorder + + [directive] = lex(source).directives + recorded = _Recorder() + check_choose(directive, _QUESTIONS, recorded) + messages = [d.message for d in recorded.diagnostics if d.rule == "choose-invalid"] + assert choose_problem(directive, _QUESTIONS) == (messages[0] if messages else None) + + +def test_choose_problem_is_none_for_a_valid_choose(): + [directive] = lex("{{choose: vat, true=a, false=b}}").directives + assert grammar.choose_problem(directive, _QUESTIONS) is None + [directive] = lex("{{choose: forum, courts=a}}").directives + assert "arbitration" in (grammar.choose_problem(directive, _QUESTIONS) or "") + + +def test_choose_problem_without_questions(): + [directive] = lex("{{choose: vat, true=a, false=b}}").directives + assert "declared boolean or choice question" in (choose_problem(directive, None) or "") + + +# ── find_markers ────────────────────────────────────────────────── + + +def test_find_markers_reads_with_the_package_lexer_by_default(): + document = parse("---\ntitle: T\n---\n\n# A {#a}\n\nText {#b} more `{#c}` {#d when=}\n\nEnd. {#e}\n") + assert find_markers(document) == find_markers(document, lex) + found = find_markers(document) + # A heading's marker is not body text; a code span's is literal. + assert [f.source for f in found] == ["{#b}", "{#d when=}", "{#e}"] + assert [f.source for f in found if f.placed(False)] == ["{#e}"] From 25cfbc06eacd784dfede9bc885c8e96cd570fdd9 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 04:40:11 +0000 Subject: [PATCH 2/5] Answers in the validation result and directive locations (#31, #32, #33) DocumentIndex.blanks (public Blank) and SectionIndexEntry.alternative hand over what validate already decides; Directive.positional_span/param_spans and Lexed.literals say where values and literal regions are; iter_document_directives yields every directive with its location, and collect_source_directives is built on it. The validator's internal accumulator Blank is now _BlankState. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_013WmBAc5T7UCKVdpUmg9qxz --- src/legaldown/assembly.py | 22 +- src/legaldown/directives.py | 68 ++-- src/legaldown/parser.py | 77 ++++- src/legaldown/validator/__init__.py | 3 +- src/legaldown/validator/core.py | 23 +- src/legaldown/validator/result.py | 29 +- src/legaldown/validator/templates.py | 6 +- tests/test_result_answers.py | 477 +++++++++++++++++++++++++++ 8 files changed, 647 insertions(+), 58 deletions(-) create mode 100644 tests/test_result_answers.py diff --git a/src/legaldown/assembly.py b/src/legaldown/assembly.py index de342c8..ac20a2d 100644 --- a/src/legaldown/assembly.py +++ b/src/legaldown/assembly.py @@ -74,7 +74,7 @@ from .validator.result import Diagnostic from .validator.templates import ( DECISION_QUESTION_TYPES, - Blank, + _BlankState, answer_problem, is_drafting_note, question_type, @@ -143,14 +143,14 @@ class Question: currency: str | None = None unit: str | None = None declared: bool = True - #: What the placeholders of the question fix (``Blank``): every code, and + #: What the placeholders of the question fix (``_BlankState``): every code, and #: whether one is in frontmatter, which ``problem`` needs. Not a field: it is #: not part of the value (``asdict``, ``replace`` and equality leave it out). A #: question made without it behaves as if its placeholders fix the ``currency`` #: or ``unit`` it names, if any, and none is in frontmatter. - _blank: InitVar[Blank | None] = None + _blank: InitVar[_BlankState | None] = None - def __post_init__(self, _blank: Blank | None) -> None: + def __post_init__(self, _blank: _BlankState | None) -> None: object.__setattr__(self, "choices", dict(self.choices or {})) object.__setattr__(self, "default", copy.deepcopy(self.default)) object.__setattr__(self, "_info", _blank) @@ -166,12 +166,12 @@ def problem(self, answer: Any) -> str | None: self.type, answer, choices=self.choices if self.type == "choice" else None, blank=self._placeholders() ) - def _placeholders(self) -> Blank | None: + def _placeholders(self) -> _BlankState | None: """What the placeholders fix: as recorded, else as ``currency``/``unit`` name.""" if self._info is not None: return self._info fixed = self.currency if self.type == "money" else self.unit if self.type == "duration" else None - return Blank(type=self.type, codes={fixed}) if fixed else None + return _BlankState(type=self.type, codes={fixed}) if fixed else None @property def _fixed(self) -> str | None: @@ -859,7 +859,7 @@ class _Template: main: _Source subs: dict[str, _Source] declared: dict - blanks: dict[str, Blank] + blanks: dict[str, _BlankState] types: dict[str, str] # every question's effective type: declared, then implicit bom: str problems: list[Diagnostic] @@ -1006,16 +1006,16 @@ def _effective_type(directive: Directive, declared: dict) -> str: return qtype if qtype in VALID_PLACEHOLDER_TYPES else directive.params.get("type", "text") -def _blanks(occurrences: list[tuple[_Occurrence, bool]], declared: dict) -> dict[str, Blank]: +def _blanks(occurrences: list[tuple[_Occurrence, bool]], declared: dict) -> dict[str, _BlankState]: """Each placeholder id's blank, recorded as the validator records it: the type of its first occurrence, the currency or unit each occurrence of that type fixes, and whether one is in frontmatter.""" - blanks: dict[str, Blank] = {} + blanks: dict[str, _BlankState] = {} for occ, in_frontmatter in occurrences: if occ.directive.name != "placeholder" or not IDENTIFIER_RE.fullmatch(occ.qid): continue ptype = _effective_type(occ.directive, declared) - blank = blanks.setdefault(occ.qid, Blank()) + blank = blanks.setdefault(occ.qid, _BlankState()) blank.in_frontmatter |= in_frontmatter blank.type = blank.type or ptype if ptype == blank.type and ptype in PLACEHOLDER_TYPE_PARAMS: @@ -1023,7 +1023,7 @@ def _blanks(occurrences: list[tuple[_Occurrence, bool]], declared: dict) -> dict return blanks -def _fixed(blank: Blank | None) -> str | None: +def _fixed(blank: _BlankState | None) -> str | None: """The currency or unit every placeholder of *blank* fixes, if one.""" if blank is None or len(blank.codes) != 1 or "" in blank.codes: return None diff --git a/src/legaldown/directives.py b/src/legaldown/directives.py index 54780e6..45d38fc 100644 --- a/src/legaldown/directives.py +++ b/src/legaldown/directives.py @@ -15,7 +15,7 @@ import re from collections.abc import Iterator -from dataclasses import dataclass +from dataclasses import dataclass, field from .markdown import FENCE_OPEN_RE, HTML_COMMENT_RE, fence_end @@ -113,6 +113,13 @@ class Directive: is empty for a well-formed directive; a malformed directive carries no arguments, and runs through the first ``}}`` on its line or up to the next directive opener (§11.4 opener commitment), whichever comes first. + + ``positional_span`` and ``param_spans`` say where the values are: the + offsets ``(start, end)`` in the lexed text of each value as written, quotes + included, so ``text[start:end]`` is what to replace to change the value (a + parameter written without a value, ``label=``, has an empty span where its + value would go). ``param_spans`` has the first occurrence of a repeated + parameter, as ``params`` has its value. A malformed directive has none. """ name: str @@ -124,6 +131,8 @@ class Directive: end: int source: str unquoted: tuple[tuple[str, str], ...] = () + positional_span: tuple[int, int] | None = None + param_spans: dict[str, tuple[int, int]] = field(default_factory=dict) def curly_quoted(self) -> list[tuple[str, str]]: """The ``unquoted`` arguments whose value begins with a typographic @@ -177,12 +186,13 @@ def _skip_ws(text: str, pos: int) -> int: return pos -def _lex_value(text: str, pos: int) -> tuple[str, int]: - """Lex one value at *pos*; return it decoded, with the offset of the - ``,`` or ``}}`` that terminates it.""" +def _lex_value(text: str, pos: int) -> tuple[str, tuple[int, int], int]: + """Lex one value at *pos*; return it decoded, where it is written (quotes + included), and the offset of the ``,`` or ``}}`` that terminates it.""" pos = _skip_ws(text, pos) if text.startswith('"', pos): chars: list[str] = [] + opening = pos pos += 1 while True: if pos >= len(text) or text[pos] == "\n": @@ -195,12 +205,13 @@ def _lex_value(text: str, pos: int) -> tuple[str, int]: else: chars.append(text[pos]) pos += 1 - pos = _skip_ws(text, pos + 1) + closing = pos + 1 + pos = _skip_ws(text, closing) if not text.startswith((",", "}}"), pos): if pos >= len(text) or text[pos] == "\n": raise _Malformed(_UNCLOSED) raise _Malformed("text after a quoted value") - return "".join(chars), pos + return "".join(chars), (opening, closing), pos start = pos while not text.startswith((",", "}}"), pos): if pos >= len(text) or text[pos] == "\n": @@ -215,33 +226,38 @@ def _lex_value(text: str, pos: int) -> tuple[str, int]: # An argument must have content; ``label=`` names its parameter, so # only a bare empty argument (a stray comma) gets here. raise _Malformed("empty argument") - return value, pos + return value, (start, start + len(value)), pos -def _lex_named_value(text: str, pos: int) -> tuple[str, int]: +def _lex_named_value(text: str, pos: int) -> tuple[str, tuple[int, int], int]: """Lex a named parameter's value, which may be empty (``label=``).""" end = _skip_ws(text, pos) if text.startswith((",", "}}"), end): - return "", end + return "", (end, end), end return _lex_value(text, pos) def _lex_arguments( text: str, pos: int -) -> tuple[str | None, dict[str, str], list[str], list[tuple[str, str]], int]: +) -> tuple[ + str | None, tuple[int, int] | None, dict[str, str], dict[str, tuple[int, int]], + list[str], list[tuple[str, str]], int, +]: """Lex the arguments after a directive opener. - Returns ``(positional, params, duplicates, unquoted, end)``, where - *unquoted* is ``Directive.unquoted`` and *end* is the offset just past - the closing ``}}``. + Returns ``(positional, positional_span, params, param_spans, duplicates, + unquoted, end)``, where *unquoted* is ``Directive.unquoted`` and *end* is + the offset just past the closing ``}}``. """ positional: str | None = None + positional_span: tuple[int, int] | None = None params: dict[str, str] = {} + param_spans: dict[str, tuple[int, int]] = {} duplicates: list[str] = [] unquoted: list[tuple[str, str]] = [] pos = _skip_ws(text, pos) if text.startswith("}}", pos): - return positional, params, duplicates, unquoted, pos + 2 + return positional, positional_span, params, param_spans, duplicates, unquoted, pos + 2 while True: pos = _skip_ws(text, pos) named = _PARAM_NAME_RE.match(text, pos) @@ -249,24 +265,26 @@ def _lex_arguments( param = named.group(0)[:-1] # Quoting is syntax, lost on decoding (§11.3): note it first. quoted = text.startswith('"', _skip_ws(text, named.end())) - value, pos = _lex_named_value(text, named.end()) + value, span, pos = _lex_named_value(text, named.end()) if param in params: duplicates.append(param) else: params[param] = value + param_spans[param] = span else: param = "" quoted = text.startswith('"', pos) - value, pos = _lex_value(text, pos) + value, span, pos = _lex_value(text, pos) if positional is not None: raise _Malformed("more than one positional value") if params: raise _Malformed("positional value after a named parameter") positional = value + positional_span = span if value and not quoted: unquoted.append((param, value)) if text.startswith("}}", pos): - return positional, params, duplicates, unquoted, pos + 2 + return positional, positional_span, params, param_spans, duplicates, unquoted, pos + 2 pos += 1 # past the comma @@ -303,12 +321,15 @@ class Lexed: ``view`` has the same offsets as the text, with comments and code spans blanked. Callers that look around directives (the defined term before a ``{{def:}}``, anchor markers) use it so they agree with the - lexer about what is literal. + lexer about what is literal. ``literals`` lists those regions, in order, as + ``(kind, start, end)``: *kind* is ``"comment"`` or ``"code"`` (a code + span), and ``view[start:end]`` is what was blanked of ``text[start:end]``. """ directives: list[Directive] stray_braces: list[int] # offsets of each ``{{`` that opens no directive view: str + literals: list[tuple[str, int, int]] = field(default_factory=list) def lex(text: str) -> Lexed: @@ -327,6 +348,7 @@ def lex(text: str) -> Lexed: view = text or "" directives: list[Directive] = [] stray_braces: list[int] = [] + literals: list[tuple[str, int, int]] = [] pos = 0 while token := _INLINE_START_RE.search(view, pos): start = token.start() @@ -339,6 +361,7 @@ def lex(text: str) -> Lexed: pos = start + 4 # an unclosed comment is literal text else: view = _blank(view, start, comment.end()) + literals.append(("comment", start, comment.end())) pos = comment.end() continue if token.group(0).startswith("`"): @@ -351,6 +374,7 @@ def lex(text: str) -> Lexed: pos = token.end() # an unmatched backtick run is literal text else: view = _blank(view, start, close.end()) + literals.append(("code", start, close.end())) pos = close.end() continue opener = _OPENER_RE.match(view, start) @@ -361,13 +385,15 @@ def lex(text: str) -> Lexed: directive = _lex_directive(text, start, opener) directives.append(directive) pos = directive.end - return Lexed(directives, stray_braces, view) + return Lexed(directives, stray_braces, view, literals) def _lex_directive(text: str, start: int, opener: re.Match[str]) -> Directive: """Lex the directive whose opener (``{{name:``) is *opener*.""" try: - positional, params, duplicates, unquoted, end = _lex_arguments(text, opener.end()) + positional, positional_span, params, param_spans, duplicates, unquoted, end = _lex_arguments( + text, opener.end() + ) except _Malformed as exc: end = _malformed_end(text, start, opener.end()) return Directive( @@ -390,6 +416,8 @@ def _lex_directive(text: str, start: int, opener: re.Match[str]) -> Directive: end=end, source=text[start:end], unquoted=tuple(unquoted), + positional_span=positional_span, + param_spans=param_spans, ) diff --git a/src/legaldown/parser.py b/src/legaldown/parser.py index 4518e0c..1dc1aa9 100644 --- a/src/legaldown/parser.py +++ b/src/legaldown/parser.py @@ -16,12 +16,12 @@ from dataclasses import asdict, dataclass, field from functools import lru_cache from pathlib import Path -from typing import Any +from typing import Any, NamedTuple import yaml -from .definitions import DefinitionAnchor, find_definition_anchors, text_fragments -from .directives import Directive, iter_directives, lex +from .definitions import DefinitionAnchor, block_fragments, find_definition_anchors +from .directives import Directive, lex from .markdown import ( FENCE_OPEN_RE, HTML_BLOCK_START_RE, @@ -1457,6 +1457,57 @@ def parse_document(source: str, *, filename: str = "") -> Document: return parse(source, filename=filename) +class DirectiveLocation(NamedTuple): + """A directive and where it is (``iter_document_directives``): in block + *block* of the preamble (*section* None) or of section *section*, in + fragment *fragment* of ``block_fragments(block)``, which *directive*'s + offsets are into. *fragment* is None for the ``{{ref:}}`` or + ``{{term:}}`` the parser lifted into the block's own fields: its offsets + are into the directive as the serializer writes it.""" + + section: int | None + block: int + fragment: int | None + directive: Directive + + +def iter_document_directives(document: Document) -> Iterator[DirectiveLocation]: + """Every directive in the body of *document*, well-formed or malformed, in + document order, each with where it is (``DirectiveLocation``). + + It reads the body as ``validate`` does — the same fragments + (``block_fragments``), lexed the same way — so code spans, comments, code + blocks and raw HTML hold none (§11.4), and a block quote's or a list's are + those of the blocks it holds. A ``{{ref:}}`` or ``{{term:}}`` the parser + lifted into a block's fields is among them, between the directives of the + block's text before it and after it. The frontmatter and the headings are + not read. + """ + for section, index, block in document.iter_indexed_blocks(): + lifted = block.kind in ("ref", "term") and bool(block.target) + # The fragments before the lifted directive: its text, and its prefix. + before = bool(block.text) + bool(block.prefix) + for fragment, (text, _anchor) in enumerate(block_fragments(block)): + if lifted and fragment == before: + yield from _lifted_directive(section, index, block) + lifted = False + for directive in lex(text).directives: + yield DirectiveLocation(section, index, fragment, directive) + if lifted: + yield from _lifted_directive(section, index, block) + + +def _lifted_directive(section: int | None, index: int, block: Block) -> Iterator[DirectiveLocation]: + """The directive a ``ref`` or ``term`` block holds in its fields, as the + serializer writes it alone.""" + from .serializer import render_block # the serializer builds on this module + + source = render_block(Block(kind=block.kind, target=block.target, label=block.label)) + for directive in lex(source).directives: + yield DirectiveLocation(section, index, None, directive) + return + + def collect_source_directives(document: Document) -> tuple[set[str], set[str]]: """Collect all ref and term targets used in a document. @@ -1464,17 +1515,11 @@ def collect_source_directives(document: Document) -> tuple[set[str], set[str]]: """ refs: set[str] = set() terms: set[str] = set() - for _section, _index, block in document.iter_blocks(): - if block.kind == "ref" and block.target: - refs.add(block.target) - if block.kind == "term" and block.target: - terms.add(block.target) - for fragment in text_fragments(block): - for directive in iter_directives(fragment): - if directive.malformed or not directive.positional: - continue - if directive.name == "ref": - refs.add(directive.positional) - elif directive.name == "term": - terms.add(directive.positional) + for _section, _index, _fragment, directive in iter_document_directives(document): + if directive.malformed or not directive.positional: + continue + if directive.name == "ref": + refs.add(directive.positional) + elif directive.name == "term": + terms.add(directive.positional) return refs, terms diff --git a/src/legaldown/validator/__init__.py b/src/legaldown/validator/__init__.py index 660908c..b754d0b 100644 --- a/src/legaldown/validator/__init__.py +++ b/src/legaldown/validator/__init__.py @@ -19,7 +19,7 @@ VALID_DURATION_UNITS, VALID_PLACEHOLDER_TYPES, ) -from .result import Diagnostic, DocumentIndex, InlineValues, PlacedMarker, SectionIndexEntry, ValidationResult +from .result import Blank, Diagnostic, DocumentIndex, InlineValues, PlacedMarker, SectionIndexEntry, ValidationResult from .templates import is_drafting_note __all__ = [ @@ -35,6 +35,7 @@ "DocumentIndex", "InlineValues", "SectionIndexEntry", + "Blank", "Diagnostic", "PlacedMarker", # Conditions (§15.3, §15.4): a condition's presence, a set of them diff --git a/src/legaldown/validator/core.py b/src/legaldown/validator/core.py index 3e80d4a..b43a4ac 100644 --- a/src/legaldown/validator/core.py +++ b/src/legaldown/validator/core.py @@ -49,12 +49,13 @@ VALID_DURATION_UNITS, VALID_PLACEHOLDER_TYPES, ) -from .result import Line, PlacedMarker, SectionIndexEntry, ValidationResult, _Recorder +from .result import Blank, Line, PlacedMarker, SectionIndexEntry, ValidationResult, _Recorder from .templates import ( BRACE_STRAY, DECISION_QUESTION_TYPES, - Blank, Quote, + _BlankState, + _fixed_by_all, block_quotes, check_choose, check_questions, @@ -294,7 +295,7 @@ def _check_directive_arguments( def _check_placeholder( directive: Directive, result: _Recorder, - blanks: dict[str, Blank], + blanks: dict[str, _BlankState], questions: Any, *, in_frontmatter: bool = False, @@ -321,7 +322,7 @@ def _check_placeholder( f"Placeholder id '{pid}' is invalid — must match [a-z][a-z0-9-]*.", ) return - blank = blanks.setdefault(pid, Blank()) + blank = blanks.setdefault(pid, _BlankState()) blank.in_frontmatter |= in_frontmatter type_valid = written_type is None or written_type in VALID_PLACEHOLDER_TYPES if not type_valid: @@ -369,7 +370,7 @@ def _check_placeholder( _check_duration_unit(code, result) -def _check_blank_codes(blanks: dict[str, Blank], result: _Recorder) -> None: +def _check_blank_codes(blanks: dict[str, _BlankState], result: _Recorder) -> None: """Report each blank whose occurrences fix two currencies or units: one blank cannot hold two (§10.7).""" for pid, blank in blanks.items(): @@ -1122,6 +1123,7 @@ def free_identifier(base: str, presence: Presence) -> str: path=".".join(path_stack), level=level, number=number, + alternative=alternative, ) result.index.sections.append(entry) # Alternatives share an identifier: a reference resolves to @@ -1326,7 +1328,7 @@ def definition_line(ref: Any) -> int | None: result.index.definition_lookup.setdefault(def_id, term_text) # ── Inline directive validation ── - blanks: dict[str, Blank] = {} + blanks: dict[str, _BlankState] = {} referenced_attachments: set[str] = set() # ── Frontmatter placeholders (§3.10) ── @@ -1556,6 +1558,15 @@ def definition_line(ref: Any) -> int | None: ) _check_blank_codes(blanks, result) + result.index.blanks = { + pid: Blank( + id=pid, + type=state.type or "text", + fixed=_fixed_by_all(state.codes) or "", + in_frontmatter=state.in_frontmatter, + ) + for pid, state in blanks.items() + } # ── Templates (§15) ── # Every condition in a condition position (§15.3): on what, as written, diff --git a/src/legaldown/validator/result.py b/src/legaldown/validator/result.py index 2370169..2f2ba1a 100644 --- a/src/legaldown/validator/result.py +++ b/src/legaldown/validator/result.py @@ -15,12 +15,16 @@ class SectionIndexEntry: shallowest heading level. A level a heading skips (heading-skip) counts as 1, so ``# A``, ``### B``, ``## C`` are 1, 1.1.1, 1.2: no two sections share a number, except alternatives and what they contain (§15.8). - ``path`` joins the identifiers of the section and its ancestors.""" + ``path`` joins the identifiers of the section and its ancestors. + ``alternative`` is True when the section is an alternative to the one + before it — the same identifier, never present together (§15.4) — and so + shares that section's number (§15.8).""" title: str identifier: str path: str level: int number: str + alternative: bool = False @dataclass(slots=True, frozen=True) @@ -80,6 +84,24 @@ class PlacedMarker: line: int | None = None +@dataclass(slots=True, frozen=True) +class Blank: + """Every occurrence of one placeholder id — one logical blank (§10.7) — as + validating the document resolved it. A tool that asks for the answers + (a form, an interview) reads the blank's type and what it fixes here. + + ``type`` is the effective type of its first occurrence with a valid one + (§10.7, §15.2): ``text`` when none is written. ``fixed`` is the currency + of a money blank, or the unit of a duration blank, that every occurrence + fixes, and ``""`` when some occurrence fixes none or they disagree (the + disagreement is placeholder-type-inconsistent). ``in_frontmatter`` is + True when one of its occurrences is in the frontmatter (§3.10).""" + id: str + type: str + fixed: str = "" + in_frontmatter: bool = False + + #: A diagnostic's line (from 1), or a function giving it, called only when a #: diagnostic is recorded at it: finding a line costs more than knowing where. Line = int | None | Callable[[], "int | None"] @@ -127,6 +149,11 @@ class DocumentIndex: #: The markers in body text that apply (``PlacedMarker``), in document #: order. placed_markers: list[PlacedMarker] = field(default_factory=list) + #: The document's blanks (``Blank``), by placeholder id, in the order of + #: their first occurrence — frontmatter first. A document that is not a + #: template has them too. A placeholder whose arguments or id are + #: malformed is not among them (the diagnostics report it). + blanks: dict[str, Blank] = field(default_factory=dict) @dataclass(slots=True, kw_only=True) diff --git a/src/legaldown/validator/templates.py b/src/legaldown/validator/templates.py index 9f6ffee..fd1b9f6 100644 --- a/src/legaldown/validator/templates.py +++ b/src/legaldown/validator/templates.py @@ -45,7 +45,7 @@ @dataclass(slots=True) -class Blank: +class _BlankState: """Every occurrence of one placeholder id — one logical blank (§10.7).""" #: The effective type of its first occurrence with a valid one. type: str | None = None @@ -76,7 +76,7 @@ def _fixed_by_all(codes: set[str]) -> str | None: return next(iter(codes)) if len(codes) == 1 and "" not in codes else None -def answer_problem(qtype: str, answer: Any, *, choices: Any = None, blank: Blank | None = None) -> str | None: +def answer_problem(qtype: str, answer: Any, *, choices: Any = None, blank: _BlankState | None = None) -> str | None: """Why *answer* is not a valid answer to a *qtype* question (§15.7.1, §15.7.2 step 1), or None when it is. *blank* describes the question's placeholders — the currency or unit they fix, and whether one is in @@ -190,7 +190,7 @@ def _choices_problem(choices: Any) -> str | None: def check_questions( questions: Any, - blanks: dict[str, Blank], + blanks: dict[str, _BlankState], result: _Recorder, *, template: bool, diff --git a/tests/test_result_answers.py b/tests/test_result_answers.py new file mode 100644 index 0000000..a4fd2fb --- /dev/null +++ b/tests/test_result_answers.py @@ -0,0 +1,477 @@ +"""The answers in the validation result and the locations of directives +(#31, #32, #33): ``DocumentIndex.blanks``, ``SectionIndexEntry.alternative``, +directive argument spans, ``Lexed.literals``, and ``iter_document_directives``.""" +from __future__ import annotations + +import dataclasses +import os +from pathlib import Path + +import pytest + +import legaldown.validator +from legaldown import parse, validate +from legaldown.definitions import text_fragments +from legaldown.directives import Directive, iter_directives, lex +from legaldown.parser import ( + DirectiveLocation, + FrontmatterError, + collect_source_directives, + iter_document_directives, +) +from legaldown.serializer import render_block +from legaldown.validator import Blank +from legaldown.validator.result import SectionIndexEntry + +_FRONTMATTER = "---\ntitle: T\n---\n\n" + + +def _blanks(body: str, frontmatter: str = "") -> dict[str, Blank]: + return validate(parse(f"---\ntitle: T\n{frontmatter}---\n\n{body}\n")).index.blanks + + +# ── Blanks (#31) ────────────────────────────────────────────────── + + +def test_blank_is_public_and_frozen(): + assert "Blank" in legaldown.validator.__all__ + assert legaldown.validator.Blank is Blank + blank = Blank(id="a", type="text") + assert (blank.fixed, blank.in_frontmatter) == ("", False) + with pytest.raises(dataclasses.FrozenInstanceError): + blank.type = "date" # type: ignore[misc] + + +def test_a_document_without_blanks_has_none(): + assert _blanks("# A\n\nText.") == {} + + +def test_blanks_are_in_the_order_of_their_first_occurrence_in_a_document_that_is_no_template(): + result = validate( + parse( + _FRONTMATTER + + "# A\n\n{{placeholder: second}} {{placeholder: first, type=date}} {{placeholder: second}}\n" + ) + ) + assert not result.index.is_template + assert list(result.index.blanks) == ["second", "first"] + assert result.index.blanks["first"] == Blank(id="first", type="date") + assert result.index.blanks["second"] == Blank(id="second", type="text") + + +def test_a_blank_with_no_type_written_is_text(): + assert _blanks("# A\n\n{{placeholder: name}}")["name"].type == "text" + + +@pytest.mark.parametrize( + "occurrences, fixed", + [ + ("{{placeholder: p, type=money, currency=EUR}}", "EUR"), + ("{{placeholder: p, type=money, currency=EUR}} {{placeholder: p, type=money, currency=EUR}}", "EUR"), + ("{{placeholder: p, type=money}}", ""), + ("{{placeholder: p, type=money, currency=EUR}} {{placeholder: p, type=money}}", ""), + ("{{placeholder: p, type=money, currency=EUR}} {{placeholder: p, type=money, currency=USD}}", ""), + ], +) +def test_a_money_blank_fixes_the_currency_all_its_occurrences_fix(occurrences, fixed): + blank = _blanks(f"# A\n\n{occurrences}")["p"] + assert (blank.type, blank.fixed) == ("money", fixed) + + +@pytest.mark.parametrize( + "occurrences, fixed", + [ + ("{{placeholder: p, type=duration, unit=D}}", "D"), + ("{{placeholder: p, type=duration}} {{placeholder: p, type=duration, unit=D}}", ""), + ("{{placeholder: p, type=duration, unit=D}} {{placeholder: p, type=duration, unit=W}}", ""), + ], +) +def test_a_duration_blank_fixes_the_unit_all_its_occurrences_fix(occurrences, fixed): + blank = _blanks(f"# A\n\n{occurrences}")["p"] + assert (blank.type, blank.fixed) == ("duration", fixed) + + +def test_a_blank_has_the_type_of_its_first_occurrence_with_a_valid_one(): + blank = _blanks("# A\n\n{{placeholder: p, type=bogus}} {{placeholder: p, type=date}}")["p"] + assert blank.type == "date" + + +def test_a_blank_takes_its_type_from_its_question(): + blanks = _blanks("# A\n\n{{placeholder: fee}}", "questions:\n fee:\n type: money\n") + assert (blanks["fee"].type, blanks["fee"].fixed) == ("money", "") + + +def test_a_blank_is_in_frontmatter_when_one_occurrence_is(): + blanks = _blanks( + "# A\n\n{{placeholder: subject}} {{placeholder: other}}", + 'subtitle: "{{placeholder: subject}}"\n', + ) + assert blanks["subject"].in_frontmatter + assert not blanks["other"].in_frontmatter + assert list(blanks) == ["subject", "other"] # frontmatter is read first + + +def test_a_placeholder_with_a_malformed_id_or_arguments_is_no_blank(): + assert _blanks("# A\n\n{{placeholder: Bad_Id}}") == {} + assert _blanks('# A\n\n{{placeholder: "unterminated}}') == {} + + +# ── Alternatives (#31) ──────────────────────────────────────────── + +_QUESTIONS = "questions:\n forum:\n type: choice\n choices:\n courts: Courts\n arbitration: Arbitration\n" + + +def test_a_section_sharing_its_predecessors_number_is_an_alternative(): + body = ( + "# Disputes {#disputes when=forum:courts}\n\n## Venue\n\nText.\n\n" + "# Disputes {#disputes when=forum:arbitration}\n\n## Seat\n\nText.\n\n# Notices\n\nText." + ) + result = validate(parse(f"---\ntitle: T\n{_QUESTIONS}---\n\n{body}\n")) + sections = result.index.sections + assert [s.number for s in sections] == ["1", "1.1", "1", "1.1", "2"] + assert [s.alternative for s in sections] == [False, False, True, False, False] + + +def test_sections_with_one_identifier_that_can_appear_together_are_not_alternatives(): + result = validate(parse(_FRONTMATTER + "# A {#same}\n\nText.\n\n# A {#same}\n\nText.\n")) + assert not any(s.alternative for s in result.index.sections) + assert [s.number for s in result.index.sections] == ["1", "2"] + + +def test_alternative_is_the_last_field_and_defaults_to_false(): + entry = SectionIndexEntry("T", "t", "t", 1, "1") + assert entry.alternative is False + assert [f.name for f in dataclasses.fields(SectionIndexEntry)][-1] == "alternative" + + +# ── Directive argument spans (#32) ──────────────────────────────── + +_DIRECTIVES = [ + "{{ref: services}}", + "{{ref:services}}", + "{{ref: services }}", + '{{ref: "services"}}', + '{{ref: "a, b" }}', + r'{{ref: "say \"hi\" \\ there"}}', + "{{term: foo, label=Foo}}", + '{{term: foo, label="Foo, Inc."}}', + "{{term: foo, label=}}", + "{{term: foo, label= , note=x}}", + "{{money: 100, currency=EUR, note=n}}", + "{{money: 100, currency=EUR, currency=USD}}", + "{{placeholder: p, type=money, currency=EUR}}", + "{{def:}}", + "{{date:}}", + "{{unknown: a b, x=y z}}", + '{{choose: forum, courts="In court", arbitration=Arbitrate}}', +] + + +def _decode(span_text: str, param: str | None) -> str | None: + """What the lexer reads in *span_text* as the value of *param* (the + positional value when None).""" + argument = span_text if param is None else f"{param}={span_text}" + directive = lex("{{x: " + argument + "}}").directives[0] + return directive.positional if param is None else directive.params.get(param) + + +@pytest.mark.parametrize("source", _DIRECTIVES) +def test_spans_cover_the_value_as_written_and_decode_to_it(source): + directive = lex(source).directives[0] + assert not directive.malformed + if directive.positional is None: + assert directive.positional_span is None + else: + start, end = directive.positional_span + assert _decode(source[start:end], None) == directive.positional + assert list(directive.param_spans) == list(directive.params) + for param, (start, end) in directive.param_spans.items(): + assert _decode(source[start:end], param) == directive.params[param] + + +def test_spans_cover_a_quoted_value_with_its_quotes_and_an_unquoted_one_trimmed(): + text = 'a {{term: "x, y" , label= Foo }} b' + directive = lex(text).directives[0] + start, end = directive.positional_span + assert text[start:end] == '"x, y"' + start, end = directive.param_spans["label"] + assert text[start:end] == "Foo" + + +def test_a_repeated_parameter_has_the_span_of_its_first_occurrence(): + text = "{{money: 1, currency=EUR, currency=USD}}" + directive = lex(text).directives[0] + assert directive.duplicates == ("currency",) + start, end = directive.param_spans["currency"] + assert text[start:end] == "EUR" + + +def test_a_parameter_without_a_value_has_an_empty_span_where_it_goes(): + text = "{{term: foo, label=, note=x}}" + directive = lex(text).directives[0] + start, end = directive.param_spans["label"] + assert start == end + assert text[:start].endswith("label=") + assert directive.params["label"] == "" + + +def test_spans_are_offsets_into_the_lexed_text(): + text = "Before `code` {{ref: a}} and {{ref: b}}." + first, second = lex(text).directives + assert text[slice(*first.positional_span)] == "a" + assert text[slice(*second.positional_span)] == "b" + + +@pytest.mark.parametrize( + "text", + ['{{ref: "unterminated}}', "{{ref: a, b}}", "{{ref: a, x=1, b}}", "{{ref: a", "{{ref: a,, b}}", "{{ref: a} {{ref: b}}"], +) +def test_a_malformed_directive_has_no_spans(text): + directive = lex(text).directives[0] + assert directive.malformed + assert directive.positional_span is None + assert directive.param_spans == {} + + +def test_the_span_fields_come_last_and_default_so_a_directive_can_still_be_built_positionally(): + directive = Directive("ref", "a", {}, (), "", 0, 8, "{{ref: a}}") + assert directive.positional_span is None and directive.param_spans == {} + names = [f.name for f in dataclasses.fields(Directive)] + assert names[-2:] == ["positional_span", "param_spans"] + assert directive.param_spans is not Directive("ref", "a", {}, (), "", 0, 8, "{{ref: a}}").param_spans + + +# ── Literal regions (#32) ───────────────────────────────────────── + + +def test_literals_list_the_comments_and_code_spans_the_lexer_blanked(): + text = "a `{{x: 1}}` b c ``d ` e`` {{ref: z}}" + lexed = lex(text) + assert [(kind, text[start:end]) for kind, start, end in lexed.literals] == [ + ("code", "`{{x: 1}}`"), + ("comment", ""), + ("code", "``d ` e``"), + ] + for _kind, start, end in lexed.literals: + assert lexed.view[start:end].strip() == "" + assert len(lexed.view) == len(text) + assert [d.name for d in lexed.directives] == ["ref"] + + +def test_a_comment_or_backtick_inside_a_directive_is_no_literal_region(): + text = '{{note: "`. + +> Quoted {{term: foo, label=Foo}}. +> +> ``` +> {{ref: in-code}} +> ``` + +- Item one {{ref: third}} + - Nested {{term: bar}} + +``` +{{ref: fenced}} +``` + +{{ref: lifted}} after {{date: 2026-02-02}} + +Before {{term: tgt}} done + +# Second {#second} + +| H | +|---| +| {{ref: cell}} | + +{{ref: "a b"}} +""" + + +def test_directives_come_in_document_order_with_where_they_are(): + document = parse(_FRONTMATTER + _BODY) + found = list(iter_document_directives(document)) + assert all(isinstance(loc, DirectiveLocation) for loc in found) + assert [(loc.directive.name, loc.directive.positional) for loc in found] == [ + ("date", "2026-01-01"), + ("ref", "second"), + ("term", "foo"), + ("ref", "third"), + ("term", "bar"), + ("ref", "lifted"), + ("date", "2026-02-02"), + ("term", "tgt"), + ("ref", "cell"), + ("ref", "a b"), + ] + assert found[0][:3] == (None, 0, 0) # the preamble has section None + assert [loc.section for loc in found] == [None] + [0] * 7 + [1, 1] + + +def test_a_directive_is_at_its_offset_in_its_fragment(): + from legaldown import block_fragments + + document = parse(_FRONTMATTER + _BODY) + blocks = {(s, b): block for s, b, block in document.iter_indexed_blocks()} + for loc in iter_document_directives(document): + if loc.fragment is None: + continue + text = block_fragments(blocks[loc.section, loc.block])[loc.fragment].text + assert text[loc.directive.start:loc.directive.end] == loc.directive.source + + +def test_a_lifted_ref_or_term_has_no_fragment_and_comes_between_its_prefix_and_suffix(): + document = parse(_FRONTMATTER + "# A\n\nBefore {{date: 2026-01-01}} {{ref: x}} after {{date: 2026-02-02}}\n") + block = document.sections[0].blocks[0] + assert block.kind == "ref" + found = list(iter_document_directives(document)) + assert [(loc.directive.name, loc.fragment) for loc in found] == [("date", 0), ("ref", None), ("date", 1)] + lifted = found[1].directive + assert lifted.source == "{{ref: x}}" + assert render_block(dataclasses.replace(block, prefix="", suffix="")) == lifted.source + assert not lifted.malformed and lifted.positional == "x" + + +def test_a_lifted_term_carries_its_label_and_a_target_that_needs_quotes_is_decoded(): + document = parse(_FRONTMATTER + '# A\n\n{{term: "a, b", label=Foo}}\n') + (loc,) = iter_document_directives(document) + assert loc.fragment is None + assert (loc.directive.positional, loc.directive.params) == ("a, b", {"label": "Foo"}) + start, end = loc.directive.positional_span + assert loc.directive.source[start:end] == '"a, b"' + + +def test_a_lifted_directive_with_no_target_is_not_read(): + from legaldown import document_from_dict + + document = document_from_dict( + {"metadata": {"title": "T"}, "sections": [{"title": "A", "blocks": [{"kind": "ref", "target": ""}]}]} + ) + assert list(iter_document_directives(document)) == [] + + +def test_a_malformed_directive_is_yielded_too(): + document = parse(_FRONTMATTER + '# A\n\nText {{date: "oops}} more\n') + (loc,) = iter_document_directives(document) + assert loc.directive.malformed + + +def test_directives_agree_with_what_the_validator_reads(): + """The validator's reference diagnostics come from the same reading.""" + document = parse(_FRONTMATTER + "# A {#a}\n\nSee {{ref: a}} and {{ref: nowhere}} and {{term: nobody}}.\n") + result = validate(document) + targets = {loc.directive.positional for loc in iter_document_directives(document) if loc.directive.name == "ref"} + assert targets == {"a", "nowhere"} + assert result.rules("error") == {"ref-broken", "term-undefined"} + + +def test_collect_source_directives_is_built_on_it(): + document = parse(_FRONTMATTER + _BODY) + refs, terms = collect_source_directives(document) + assert refs == {"second", "third", "lifted", "cell", "a b"} + assert terms == {"foo", "bar", "tgt"} + + +def test_collect_source_directives_skips_malformed_and_empty_targets(): + document = parse(_FRONTMATTER + '# A\n\n{{ref: "oops}} {{ref:}} {{term:}} {{ref: ok}}\n') + assert collect_source_directives(document) == ({"ok"}, set()) + + +# ── Over the specification fixtures ─────────────────────────────── + +_FIXTURES = Path(os.environ.get("LEGALDOWN_FIXTURES_DIR", "")) +_FIXTURE_DOCUMENTS = sorted(_FIXTURES.rglob("*.lgd")) if _FIXTURES.is_dir() else [] + + +def _collected_before(document) -> tuple[set[str], set[str]]: + """The ref and term targets as ``collect_source_directives`` found them before it was built + on ``iter_document_directives``: from the lifted fields, and the lexed text fragments.""" + refs: set[str] = set() + terms: set[str] = set() + for _section, _index, block in document.iter_blocks(): + if block.kind == "ref" and block.target: + refs.add(block.target) + if block.kind == "term" and block.target: + terms.add(block.target) + for fragment in text_fragments(block): + for directive in iter_directives(fragment): + if directive.malformed or not directive.positional: + continue + if directive.name == "ref": + refs.add(directive.positional) + elif directive.name == "term": + terms.add(directive.positional) + return refs, terms + + +def _decoded(source: str, span: tuple[int, int], param: str | None) -> str | None: + """The value the lexer reads in ``source[span]``, as *param*'s (the positional when None).""" + return _decode(source[span[0]:span[1]], param) + + +@pytest.mark.skipif(not _FIXTURE_DOCUMENTS, reason="LEGALDOWN_FIXTURES_DIR not set (specification fixtures)") +@pytest.mark.parametrize( + "path", _FIXTURE_DOCUMENTS, ids=lambda p: "/".join(p.parts[-3:]), +) +def test_every_fixture_document_reads_consistently(path): + try: + document = parse(path.read_text(encoding="utf-8"), filename=path.name) + except FrontmatterError: + pytest.skip("unreadable frontmatter") + # The locations agree with the validator's reading of the same document. + for loc in iter_document_directives(document): + directive = loc.directive + if directive.malformed: + assert directive.positional_span is None and directive.param_spans == {} + continue + # Spans decode, through the lexer, to the values. + if directive.positional is None: + assert directive.positional_span is None + else: + assert _decoded(directive.source, _relative(directive, directive.positional_span), None) == directive.positional + for param, span in directive.param_spans.items(): + assert _decoded(directive.source, _relative(directive, span), param) == directive.params[param] + assert collect_source_directives(document) == _collected_before(document) + + # The blanks agree with the placeholders the validator met. + result = validate(document) + placeholders = result.index.values.placeholders + for pid, blank in result.index.blanks.items(): + assert blank.id == pid + met = [ptype for met_id, ptype in placeholders if met_id == pid] + assert met, pid + assert blank.type in met or blank.type == "text" + if blank.fixed: + assert blank.type in ("money", "duration") + assert set(result.index.blanks) <= {pid for pid, _ptype in placeholders} + for entry in result.index.sections: + assert isinstance(entry.alternative, bool) + + +def _relative(directive: Directive, span: tuple[int, int] | None) -> tuple[int, int]: + """*span*, which is an offset into the lexed text, as one into ``directive.source``.""" + assert span is not None + return span[0] - directive.start, span[1] - directive.start From c74344ae4a2180082d552339c7deed5fd9708e72 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 04:40:17 +0000 Subject: [PATCH 3/5] Model answers for tools: quote_blocks, drafting_note_blocks, code_content, SourceLayout, Document.layout()/line_of() (#88, #93, #27) Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_013WmBAc5T7UCKVdpUmg9qxz --- src/legaldown/markdown.py | 39 +++ src/legaldown/models.py | 30 +- src/legaldown/parser.py | 17 ++ src/legaldown/positions.py | 154 ++++++++++ src/legaldown/validator/templates.py | 26 ++ tests/test_model_answers.py | 441 +++++++++++++++++++++++++++ 6 files changed, 706 insertions(+), 1 deletion(-) create mode 100644 tests/test_model_answers.py diff --git a/src/legaldown/markdown.py b/src/legaldown/markdown.py index 35cac5a..5ea1aba 100644 --- a/src/legaldown/markdown.py +++ b/src/legaldown/markdown.py @@ -9,6 +9,10 @@ from __future__ import annotations import re +from typing import TYPE_CHECKING, NamedTuple + +if TYPE_CHECKING: + from .models import Block # A fenced code block opens with three or more backticks or tildes, indented # at most three columns; a backtick fence's info string cannot contain a @@ -294,3 +298,38 @@ def close_fences(text: str) -> str: return text + "\n" + indent + fence index = end return text + + +class CodeContent(NamedTuple): + """What a ``code`` block holds, read as CommonMark reads it + (``code_content``).""" + + #: A fenced block's info string, without the spaces around it; ``""`` + #: for an indented block, or a fence with none. + info: str + #: The code itself, each line ending with LF; ``""`` for no lines. + text: str + #: True for a fenced block, false for an indented one. + fenced: bool + + +def code_content(block: Block) -> CodeContent: + """The content of the ``code`` block *block*, as CommonMark reads it + (§11.4): of a fenced block, its info string and the lines between its + fences, the closing fence left out when it has one (``closes_fence``), + each line without as much indentation as the opening fence had + (``dedent``); of an indented block, its lines without their four columns + of indentation. Raises ``ValueError`` for a block that is not code.""" + if block.kind != "code": + raise ValueError(f"not a code block: {block.kind}") + lines = block.text.split("\n") if block.text else [] + opening = FENCE_OPEN_RE.match(lines[0]) if lines else None + if opening is None: + return CodeContent("", "".join(dedent(line, 4) + "\n" for line in lines), False) + body = lines[1:] + if body and closes_fence(body[-1], opening.group("fence")): + body = body[:-1] + indent = indent_width(lines[0]) + return CodeContent( + lines[0][opening.end():].strip(), "".join(dedent(line, indent) + "\n" for line in body), True + ) diff --git a/src/legaldown/models.py b/src/legaldown/models.py index 4ad60af..7c7a7b6 100644 --- a/src/legaldown/models.py +++ b/src/legaldown/models.py @@ -8,10 +8,13 @@ from collections.abc import Iterator from dataclasses import asdict, dataclass, field from pathlib import Path -from typing import Any +from typing import TYPE_CHECKING, Any from .markdown import LINE_ENDING_RE, is_blank, strip_text +if TYPE_CHECKING: + from .positions import SourceLayout + # --------------------------------------------------------------------------- # Dataclasses # --------------------------------------------------------------------------- @@ -211,6 +214,31 @@ def iter_indexed_blocks(self) -> Iterator[tuple[int | None, int, Block]]: for index, block in enumerate(section.blocks): yield section_index, index, block + def layout(self) -> SourceLayout | None: + """Where the document's frontmatter, headings, blocks and list items + lie in the file it was parsed from (§16.9): file lines counted from + 1, each span ending where the next begins. None for a document + built in code, and for one changed since it was parsed so that it + no longer fits its source: where its parts lie is not known then.""" + from .positions import source_layout # positions describes documents: it builds on this module + + return source_layout(self) + + def line_of(self, section: int | None, block: int | None = None, item: int | None = None) -> int | None: + """The first line, counted from 1, of a part of the document in its + file: section *section*'s heading (*block* None); a top-level block, + of a section or, with *section* None, of the preamble; or a list + item's marker, *item* counted in pre-order among all the list's + items, nested ones included (``list_fragments``). These are the + lines its diagnostics name (§16.9). None when the document has no + source to name a line of (``layout``). Raises ``IndexError`` for an + index out of range, and ``ValueError`` for a preamble without a + block, an item without a block, or an item of a block that is not a + list.""" + from .positions import line_of + + return line_of(self, section, block, item) + def iter_blocks(self) -> Iterator[tuple[Section | None, int, Block]]: """Every body block in document order, as ``(section, index, block)``: the preamble first, with ``section`` ``None``, then each section's.""" diff --git a/src/legaldown/parser.py b/src/legaldown/parser.py index 4518e0c..6d5d8a8 100644 --- a/src/legaldown/parser.py +++ b/src/legaldown/parser.py @@ -9,6 +9,7 @@ from __future__ import annotations import bisect +import copy import os import re import warnings @@ -1361,6 +1362,22 @@ def quote_content(text: str, depth: int = 0) -> tuple[tuple[Block, ...], tuple[_ return tuple(blocks), tuple(layout.preamble) +def quote_blocks(block: Block, *, depth: int = 0) -> list[Block]: + """The blocks the block quote *block* holds, as the validator reads them + (``quote_content``): each a fresh copy, free to change, since the + validator keeps its own reading of a quote's content. A heading in one + is a ``heading`` block, not a section, and no directive is lifted into + block fields (§4.1). *depth*: how many list items and quotes *block* is + in; a quote in ``MAX_QUOTE_DEPTH`` of them is read as the validator reads + it, as one text: a single paragraph, or no block when it holds none. + Raises ``ValueError`` for a block that is not a quote.""" + if block.kind != "quote": + raise ValueError(f"not a block quote: {block.kind}") + if depth >= MAX_QUOTE_DEPTH: + return [Block(kind="paragraph", text=block.text)] if block.text.strip() else [] + return copy.deepcopy(list(quote_content(block.text, depth + 1)[0])) + + def parse_item_content(text: str) -> list[Block]: """The blocks a list item holds whose content is *text*, as written in the item without its marker and indentation (``_parse_list``).""" diff --git a/src/legaldown/positions.py b/src/legaldown/positions.py index 6045973..9ec299f 100644 --- a/src/legaldown/positions.py +++ b/src/legaldown/positions.py @@ -85,6 +85,60 @@ def walk(node: yaml.Node, path: tuple[Any, ...], seen: frozenset[int]) -> None: return keys +@dataclass(frozen=True, slots=True) +class BlockSpan: + """Where a block lies in the file: lines ``[start, end)``, counted from 1. + *kind* is the block's. *items*: a list's items, each with where it lies + (the items of the list itself; a nested list's are in its item's blocks).""" + + kind: str + start: int + end: int + items: tuple[ItemSpan, ...] = () + + +@dataclass(frozen=True, slots=True) +class ItemSpan: + """Where a list item lies in the file: lines ``[start, end)``, its marker's + line first, and where each block of its content lies.""" + + start: int + end: int + blocks: tuple[BlockSpan, ...] + + +@dataclass(frozen=True, slots=True) +class HeadingSpan: + """Where a heading lies in the file: lines ``[start, end)``, and the line + its marker is on (an ATX heading's line, or a setext heading's last text + line, where a ``{#id}`` or ``{if:}`` marker is written, §5.2).""" + + start: int + end: int + marker_line: int + + +@dataclass(frozen=True, slots=True) +class SectionSpan: + """Where a section's heading lies, and its top-level blocks.""" + + heading: HeadingSpan + blocks: tuple[BlockSpan, ...] + + +@dataclass(frozen=True, slots=True) +class SourceLayout: + """Where a parsed document's parts lie in its file (``Document.layout``), + in lines counted from 1 and ending where the next begins: the frontmatter + as ``(start, end)`` with its ``---`` lines, or None without one; the + preamble's blocks (§4.4); each section's heading and blocks, in the + document's order, so that ``sections[n]`` is ``Document.sections[n]``'s.""" + + frontmatter: tuple[int, int] | None + preamble: tuple[BlockSpan, ...] + sections: tuple[SectionSpan, ...] + + @dataclass(slots=True) class SourceMap: """Where a parsed document's frontmatter keys, headings and blocks lie: @@ -137,6 +191,51 @@ def block(self, section: int | None, index: int) -> int: """The first line of a top-level block.""" return self.body_start + self.span(section, index).start + 1 + def to_layout(self) -> SourceLayout: + """``layout`` as the public ``SourceLayout``, in file lines.""" + base = self.body_start + 1 + + def block(span: Any) -> BlockSpan: + return BlockSpan(span.kind, base + span.start, base + span.end, tuple(item(each) for each in span.items)) + + def item(span: Any) -> ItemSpan: + return ItemSpan(base + span.start, base + span.end, tuple(block(each) for each in span.blocks)) + + frontmatter = None + if () in self.keys: + # The frontmatter's lines, its closing ``---`` too, which the + # source may end with no line ending after. + from .parser import FRONTMATTER_RE + + written = FRONTMATTER_RE.match("\n".join(self.lines)) + if written is not None: + text = written.group() + frontmatter = (1, 1 + text.count("\n") + (0 if text.endswith("\n") else 1)) + return SourceLayout( + frontmatter, + tuple(block(each) for each in self.layout.preamble), + tuple( + SectionSpan( + HeadingSpan(base + heading.start, base + heading.end, base + heading.marker_line), + tuple(block(each) for each in blocks), + ) + for heading, blocks in self.layout.sections + ), + ) + + def item(self, section: int | None, index: int, item: int) -> int: + """The line of list item *item* of top-level block *index* of a + section (or of the preamble): its marker's line. Items are numbered + in pre-order among all the list's items, nested ones included + (``definitions.list_fragments``).""" + span = self.span(section, index) + if not span.items: + raise ValueError("not a list") + found = [each for top in span.items for each in _preorder(top)] + if not 0 <= item < len(found): + raise IndexError(f"list item {item} out of range") + return self.body_start + found[item].start + 1 + def find( self, start: int, @@ -181,6 +280,15 @@ def find( return start +def _preorder(item: Any) -> Iterator[Any]: + """The item span *item*, then the item spans of the lists in its + blocks, each in the same order.""" + yield item + for block in item.blocks: + for inner in block.items: + yield from _preorder(inner) + + def _candidates(needle: str) -> list[str]: """What *needle* is looked for as: its start (24 characters, with a table cell's pipes escaped as the source writes them), then its opener.""" @@ -429,3 +537,49 @@ def _lifted(block: Block, own: list[str], first: int, lines: list[str], start: i places[number] = (0, len(joined) - len(suffix)) lifted = len(prefix) return _Leaf(start, end, [joined], places, lifted if lifted >= 0 else None) + + +def source_layout(document: Document) -> SourceLayout | None: + """``Document.layout``: where *document*'s parts lie in the source it was + parsed from; None without a source map, or when the document no longer + has the shape it was parsed with.""" + source_map = document.source_map + if source_map is None or not source_map.fits(document): + return None + return source_map.to_layout() + + +def line_of(document: Document, section: int | None, block: int | None, item: int | None) -> int | None: + """``Document.line_of``.""" + from .models import LIST_KINDS, list_items + + if section is not None and not 0 <= section < len(document.sections): + raise IndexError(f"section {section} out of range") + if block is None and item is not None: + raise ValueError("a list item is named by its block") + if block is None and section is None: + raise ValueError("the preamble has no heading") + if block is not None: + blocks = document.preamble if section is None else document.sections[section].blocks + if not 0 <= block < len(blocks): + raise IndexError(f"block {block} out of range") + if item is not None: + if blocks[block].kind not in LIST_KINDS: + raise ValueError("not a list") + + def count(items: list[Any]) -> int: + return sum( + 1 + sum(count(list_items(child)) for child in each.blocks if child.kind in LIST_KINDS) + for each in items + ) + + if not 0 <= item < count(list_items(blocks[block])): + raise IndexError(f"list item {item} out of range") + source_map = document.source_map + if source_map is None or not source_map.fits(document): + return None + if block is None: + return source_map.heading(section) # type: ignore[arg-type] + if item is None: + return source_map.block(section, block) + return source_map.item(section, block, item) diff --git a/src/legaldown/validator/templates.py b/src/legaldown/validator/templates.py index 9f6ffee..60ea046 100644 --- a/src/legaldown/validator/templates.py +++ b/src/legaldown/validator/templates.py @@ -295,6 +295,32 @@ def is_drafting_note(quote: Block) -> bool: return quote.kind == "quote" and Quote(quote.text.split("\n", 1)[0].strip(), range(0)).is_drafting_note +def drafting_note_blocks(quote: Block, *, depth: int = 0) -> list[Block]: + """The blocks of drafting note *quote* without its ``[!DRAFTING]`` marker + line (§15.6), as ``quote_blocks`` reads the note's content: fresh copies. + The marker starts the first paragraph or heading, and is cut from it, + the block going when nothing else is in it; where it does not (a marker + line indented as code), the note is what is written after its first + line. *depth*: as in ``quote_blocks``. Raises ``ValueError`` for a block + that is not a drafting note (``is_drafting_note``).""" + from ..parser import MAX_QUOTE_DEPTH, quote_blocks # see the import note in check_template_body + + if not is_drafting_note(quote): + raise ValueError("not a drafting note") + if depth >= MAX_QUOTE_DEPTH: + text = quote.text.partition("\n")[2] + return [Block(kind="paragraph", text=text)] if text.strip() else [] + blocks = quote_blocks(quote, depth=depth) + first = blocks[0] if blocks else None + if first is None or first.kind not in ("paragraph", "heading") or not first.text.upper().startswith(_DRAFTING_MARKER): + return quote_blocks(Block(kind="quote", text=quote.text.partition("\n")[2]), depth=depth) + rest = first.text[len(_DRAFTING_MARKER):].lstrip() + if not rest: + return blocks[1:] + first.text = rest + return blocks + + def block_quotes(block: Block) -> list[Quote]: """The block quotes in *block*, nested ones included: a quote block and the quotes in it, or those in a list's items.""" diff --git a/tests/test_model_answers.py b/tests/test_model_answers.py new file mode 100644 index 0000000..3671a18 --- /dev/null +++ b/tests/test_model_answers.py @@ -0,0 +1,441 @@ +"""Model answers for tools: a quote's content, a drafting note without its +marker, a code block's content, and where a document's parts lie in its file.""" +from __future__ import annotations + +import os +import re +from pathlib import Path + +import pytest + +from legaldown import document_from_dict, document_to_dict, parse +from legaldown.definitions import list_fragments +from legaldown.markdown import CodeContent, code_content +from legaldown.models import LIST_KINDS, Block, ListItem, item_text, list_items +from legaldown.parser import MAX_QUOTE_DEPTH, FrontmatterError, quote_blocks, quote_content +from legaldown.positions import BlockSpan, HeadingSpan, ItemSpan, SectionSpan, SourceLayout +from legaldown.validator import validate +from legaldown.validator.templates import drafting_note_blocks, is_drafting_note + +_HEAD = "---\ntitle: T\n---\n\n# A\n\n" + + +def _blocks(body: str) -> list[Block]: + return parse(_HEAD + body).sections[0].blocks + + +def _shown(blocks: list[Block]) -> list[tuple[str, str]]: + return [(block.kind, block.text) for block in blocks] + + +# ── Quote content ────────────────────────────────────────────────── + + +def test_quote_blocks_are_the_blocks_the_validator_reads(): + [quote] = _blocks("> One\n>\n> - a\n> - b\n>\n> > Inner\n") + blocks = quote_blocks(quote) + assert [block.kind for block in blocks] == ["paragraph", "unordered_list", "quote"] + assert blocks == list(quote_content(quote.text, 1)[0]) + assert _shown(quote_blocks(blocks[2], depth=1)) == [("paragraph", "Inner")] + + +def test_quote_blocks_are_fresh_copies(): + [quote] = _blocks("> One\n>\n> - a\n") + blocks = quote_blocks(quote) + blocks[0].text = "changed" + blocks[1].items[0].blocks[0].text = "changed" + blocks.append(Block()) + assert quote_blocks(quote) == list(quote_content(quote.text, 1)[0]) + assert quote_blocks(quote)[0].text == "One" + + +def test_a_heading_in_a_quote_is_a_heading_block(): + [quote] = _blocks("> # Title\n> text\n") + assert [(block.kind, block.level) for block in quote_blocks(quote)] == [("heading", 1), ("paragraph", 0)] + + +def test_quote_blocks_of_a_quote_in_a_list_item(): + [lst] = _blocks("- item\n\n > quoted\n") + quote = lst.items[0].blocks[1] + assert _shown(quote_blocks(quote, depth=1)) == [("paragraph", "quoted")] + + +def test_a_quote_past_the_quote_depth_is_one_text(): + [quote] = _blocks("> one\n>\n> two\n") + assert len(quote_blocks(quote, depth=MAX_QUOTE_DEPTH - 1)) == 2 + assert _shown(quote_blocks(quote, depth=MAX_QUOTE_DEPTH)) == [("paragraph", "one\n\ntwo")] + assert quote_blocks(Block(kind="quote", text=""), depth=MAX_QUOTE_DEPTH) == [] + + +@pytest.mark.parametrize("kind", ["paragraph", "code", "unordered_list", "rule"]) +def test_quote_blocks_of_a_block_that_is_no_quote(kind): + with pytest.raises(ValueError, match="quote"): + quote_blocks(Block(kind=kind, text="x")) + + +# ── Drafting notes (#88) ─────────────────────────────────────────── + + +@pytest.mark.parametrize( + ("body", "expected"), + [ + # The marker starts a paragraph. + ("> [!DRAFTING]\n> note\n", [("paragraph", "note")]), + # It is a setext heading's text: the heading goes with it. + ("> [!DRAFTING]\n> ---\n", []), + # Indented as code: the note is what follows the marker line. + ("> [!DRAFTING]\n> note\n", [("paragraph", "note")]), + # Letters in any case; later blocks stay. + ("> [!drafting]\n> first\n>\n> second\n", [("paragraph", "first"), ("paragraph", "second")]), + # Nothing but the marker. + ("> [!DRAFTING]\n", []), + ("> [!DRAFTING]\n>\n> - a\n> - b\n", [("unordered_list", "")]), + # Its first paragraph keeps its own lines. + ("> [!DRAFTING]\n> one\n> two\n", [("paragraph", "one\ntwo")]), + # A heading and a quote in the note. + ("> [!DRAFTING]\n>\n> # Head\n>\n> > inner\n", [("heading", "Head"), ("quote", "inner")]), + ], +) +def test_drafting_note_blocks(body, expected): + [note] = _blocks(body) + assert is_drafting_note(note) + assert _shown(drafting_note_blocks(note)) == expected + + +def test_drafting_note_blocks_are_fresh_copies(): + [note] = _blocks("> [!DRAFTING]\n> note\n>\n> second\n") + blocks = drafting_note_blocks(note) + blocks[0].text = "changed" + assert _shown(drafting_note_blocks(note)) == [("paragraph", "note"), ("paragraph", "second")] + assert _shown(quote_blocks(note))[0] == ("paragraph", "[!DRAFTING]\nnote") + + +def test_drafting_note_blocks_past_the_quote_depth(): + note = Block(kind="quote", text="[!DRAFTING]\nnote\n\nmore") + assert _shown(drafting_note_blocks(note, depth=MAX_QUOTE_DEPTH)) == [("paragraph", "note\n\nmore")] + assert drafting_note_blocks(Block(kind="quote", text="[!DRAFTING]"), depth=MAX_QUOTE_DEPTH) == [] + + +def test_drafting_note_blocks_of_what_is_no_note(): + for block in _blocks("> Plain\n\n> [!NOTE]\n> x\n\nText\n"): + with pytest.raises(ValueError, match="drafting note"): + drafting_note_blocks(block) + + +def test_a_drafting_note_in_a_quote(): + [outer] = _blocks("> text\n>\n> > [!DRAFTING]\n> > note\n") + [_text, inner] = quote_blocks(outer) + assert is_drafting_note(inner) + assert _shown(drafting_note_blocks(inner, depth=1)) == [("paragraph", "note")] + + +# ── Code blocks (#93) ────────────────────────────────────────────── + + +@pytest.mark.parametrize( + ("source", "expected"), + [ + ("```python\nx = 1\n\ny = 2\n```\n", CodeContent("python", "x = 1\n\ny = 2\n", True)), + ("``` python title=x \nx\n```\n", CodeContent("python title=x", "x\n", True)), + ("~~~\na\n~~~\n", CodeContent("", "a\n", True)), + ("~~~ js\n```\n~~~\n", CodeContent("js", "```\n", True)), + ("````\n```\nx\n```\n````\n", CodeContent("", "```\nx\n```\n", True)), + # Unclosed: runs to the end. + ("```\nx\ny\n", CodeContent("", "x\ny\n", True)), + # No lines at all. + ("```\n```\n", CodeContent("", "", True)), + ("```\n", CodeContent("", "", True)), + # An indented fence: each line loses as much as the fence had. + (" ```\n a\n b\n c\n ```\n", CodeContent("", "a\n b\nc\n", True)), + # Indented code: four columns. + (" a\n b\n\n c\n", CodeContent("", "a\n b\n\nc\n", False)), + ("\tx\n", CodeContent("", "x\n", False)), + ], +) +def test_code_content(source, expected): + [block] = _blocks(source) + assert code_content(block) == expected + + +def test_code_content_in_a_list_item_and_a_quote(): + [lst] = _blocks("- item\n\n ```sh\n ls\n ```\n") + assert code_content(lst.items[0].blocks[1]) == CodeContent("sh", "ls\n", True) + [quote] = _blocks("> ```sh\n> ls\n> ```\n") + assert code_content(quote_blocks(quote)[0]) == CodeContent("sh", "ls\n", True) + + +def test_code_content_of_a_model_built_in_code(): + assert code_content(Block(kind="code", text="")) == CodeContent("", "", False) + assert code_content(Block(kind="code", text="```py\nx\n```")) == CodeContent("py", "x\n", True) + + +@pytest.mark.parametrize("kind", ["paragraph", "quote", "html"]) +def test_code_content_of_a_block_that_is_no_code(kind): + with pytest.raises(ValueError, match="code"): + code_content(Block(kind=kind, text="x")) + + +# ── Layout ───────────────────────────────────────────────────────── + +_SOURCE = ( + "---\ntitle: T\n---\n" # 1-3 + "\n" # 4 + "Pre\n" # 5 + "\n" # 6 + "# A\n" # 7 + "\n" # 8 + "- a\n" # 9 + " - b\n" # 10 + " - c\n" # 11 + "- d\n" # 12 + "\n" # 13 + "1. x\n" # 14 + " 1. y\n" # 15 + " - z\n" # 16 + " 2. w\n" # 17 + "2. v\n" # 18 + "\n" # 19 + "> quote\n" # 20 + "\n" # 21 + "# B\n" # 22 + "Text\n" # 23 + "====\n" # 24 + "Last\n" # 25 +) + + +def test_the_layout_gives_the_file_lines_of_each_part(): + layout = parse(_SOURCE).layout() + assert isinstance(layout, SourceLayout) + assert layout.frontmatter == (1, 4) + assert layout.preamble == (BlockSpan("paragraph", 5, 6),) + assert [section.heading for section in layout.sections] == [ + HeadingSpan(7, 8, 7), + HeadingSpan(22, 23, 22), + HeadingSpan(23, 25, 23), + ] + first, _second, third = layout.sections + assert isinstance(first, SectionSpan) + assert [(block.kind, block.start, block.end) for block in first.blocks] == [ + ("unordered_list", 9, 13), + ("ordered_list", 14, 19), + ("quote", 20, 21), + ] + [a, d] = first.blocks[0].items + assert isinstance(a, ItemSpan) + assert (a.start, a.end, d.start, d.end) == (9, 12, 12, 13) + assert [(block.kind, block.start) for block in a.blocks] == [("paragraph", 9), ("unordered_list", 10)] + assert [(item.start, item.end) for item in a.blocks[1].items] == [(10, 11), (11, 12)] + assert third.blocks == (BlockSpan("paragraph", 25, 26),) + + +def test_the_layout_is_frozen(): + layout = parse(_SOURCE).layout() + assert layout is not None + with pytest.raises(AttributeError): + layout.frontmatter = None # type: ignore[misc] + + +def test_the_layout_without_frontmatter_or_body(): + layout = parse("# A\n\nText\n").layout() + assert layout is not None and layout.frontmatter is None and layout.preamble == () + assert layout.sections[0].blocks == (BlockSpan("paragraph", 3, 4),) + assert parse("").layout() == SourceLayout(None, (), ()) + + +@pytest.mark.parametrize( + ("source", "end"), + [ + ("---\ntitle: T\n---", 4), # no line ending after the closing line + ("---\ntitle: T\n---\n", 4), + ("---\n---\n\n# A\n", 3), + ("---\r\ntitle: T\r\n--- \r\n\r\n# A\r\n", 4), + ("---\ntitle: T\n---\n# A\n", 4), + ], +) +def test_the_layout_of_the_frontmatter(source, end): + layout = parse(source).layout() + assert layout is not None and layout.frontmatter == (1, end) + + +def test_a_thematic_break_is_not_frontmatter(): + layout = parse("---\n\n# A\n").layout() + assert layout is not None and layout.frontmatter is None + assert layout.preamble == (BlockSpan("rule", 1, 2),) + + +def test_the_layout_counts_lines_as_written(): + assert parse(_SOURCE.replace("\n", "\r\n")).layout() == parse(_SOURCE).layout() + + +def test_a_document_without_a_source_has_no_layout(): + rebuilt = document_from_dict(document_to_dict(parse(_SOURCE))) + assert rebuilt.layout() is None + assert rebuilt.line_of(0) is None and rebuilt.line_of(0, 0) is None and rebuilt.line_of(0, 0, 1) is None + + +def test_a_document_changed_since_parsing_has_no_layout(): + document = parse(_SOURCE) + document.sections[0].blocks.insert(0, Block(kind="paragraph", text="New.")) + assert document.layout() is None + assert document.line_of(0, 1) is None + with pytest.raises(IndexError): # misuse is still named + document.line_of(0, 99) + + +# ── Lines ────────────────────────────────────────────────────────── + + +def test_line_of_a_heading_a_block_and_an_item(): + document = parse(_SOURCE) + assert [document.line_of(index) for index in range(3)] == [7, 22, 23] + assert document.line_of(None, 0) == 5 + assert [document.line_of(0, index) for index in range(3)] == [9, 14, 20] + assert document.line_of(2, 0) == 25 + # Items in pre-order: an item, then those nested in it. + assert [document.line_of(0, 0, item) for item in range(4)] == [9, 10, 11, 12] + assert [document.line_of(0, 1, item) for item in range(5)] == [14, 15, 16, 17, 18] + + +def test_line_of_the_items_of_a_list_in_the_preamble(): + document = parse("- a\n- b\n - c\n\n# A\n") + assert [document.line_of(None, 0, item) for item in range(3)] == [1, 2, 3] + + +def test_line_of_items_are_numbered_as_list_fragments_number_them(): + document = parse(_SOURCE) + lines = _SOURCE.split("\n") + for index in (0, 1): + block = document.sections[0].blocks[index] + flat = _preorder(block) + for fragment in list_fragments(block): + if fragment.anchor: # the end of an item's first paragraph + assert fragment.text == item_text(flat[fragment.items[-1]]) + for number, item in enumerate(flat): + assert item_text(item) in lines[document.line_of(0, index, number) - 1] + + +def test_line_of_items_stop_at_a_quote(): + document = parse("- a\n\n > - q\n > - r\n- b\n") + # The quote's list items are no items of the list (§15.3). + assert [document.line_of(None, 0, item) for item in range(2)] == [1, 5] + with pytest.raises(IndexError): + document.line_of(None, 0, 2) + + +def test_line_of_in_a_document_without_frontmatter(): + document = parse("# A\n\nText\n") + assert (document.line_of(0), document.line_of(0, 0)) == (1, 3) + + +@pytest.mark.parametrize( + ("args", "error"), + [ + ((5,), IndexError), + ((-1,), IndexError), + ((0, 9), IndexError), + ((0, -1), IndexError), + ((None, 5), IndexError), + ((0, 0, 4), IndexError), + ((0, 0, -1), IndexError), + ((0, 2, 0), ValueError), # a quote is no list + ((0, None, 0), ValueError), + ((None,), ValueError), + ((None, None, 0), ValueError), + ], +) +def test_line_of_misuse(args, error): + with pytest.raises(error): + parse(_SOURCE).line_of(*args) + + +def test_line_of_agrees_with_the_lines_diagnostics_name(): + document = parse(_HEAD + "> [!DRAFT]\n> text\n\nAnd\n\n> [!DRAFT]\n> more\n\n### Skip\n\nText\n") + lines: dict[str, list[int | None]] = {} + for diagnostic in validate(document).diagnostics: + lines.setdefault(diagnostic.rule, []).append(diagnostic.line) + assert lines["drafting-note-unrecognized"] == [document.line_of(0, 0), document.line_of(0, 2)] + assert lines["heading-skip"] == [document.line_of(1)] + assert document.line_of(0, 0) == 7 + + +# ── Over the specification's documents ───────────────────────────── + +_ITEM_RE = re.compile(r"[ \t]*(?:[0-9]{1,9}[.)]|[-*+])(?:[ \t]|$)") + + +def _corpus(): + root = os.environ.get("LEGALDOWN_FIXTURES_DIR", "") + if not root or not Path(root).is_dir(): + return [] + return [pytest.param(path, id=path.relative_to(root).as_posix()) for path in sorted(Path(root).rglob("*.lgd"))] + + +def _preorder(block: Block) -> list[ListItem]: + found: list[ListItem] = [] + for item in list_items(block): + found.append(item) + for child in item.blocks: + if child.kind in LIST_KINDS: + found.extend(_preorder(child)) + return found + + +def _read(blocks: list[Block], depth: int) -> None: + """Ask the model for every quote and code block, nested ones too.""" + for block in blocks: + if block.kind in LIST_KINDS: + for item in list_items(block): + _read(item.blocks, depth + 1) + elif block.kind == "code": + assert code_content(block).text.endswith("\n") or not block.text.strip() + elif block.kind == "quote": + content = quote_blocks(block, depth=depth) + if is_drafting_note(block): + drafting_note_blocks(block, depth=depth) + _read(content, depth + 1) + + +@pytest.mark.parametrize("path", _corpus()) +def test_the_model_answers_hold_over_every_specification_document(path: Path): + text = path.read_text(encoding="utf-8") + try: + document = parse(text) + except FrontmatterError: + pytest.skip("unreadable frontmatter") + layout = document.layout() + assert layout is not None + lines = text.replace("\r\n", "\n").replace("\r", "\n").removeprefix("").split("\n") + count = len(lines) - (1 if lines[-1] == "" else 0) + if layout.frontmatter is not None: + assert layout.frontmatter[0] == 1 <= layout.frontmatter[1] <= count + 1 + assert len(layout.sections) == len(document.sections) + containers = [(None, document.preamble, layout.preamble)] + containers += [ + (index, section.blocks, span.blocks) + for index, (section, span) in enumerate(zip(document.sections, layout.sections, strict=True)) + ] + for index, blocks, spans in containers: + if index is not None: + heading = layout.sections[index].heading + assert document.line_of(index) == heading.start + assert 1 <= heading.start <= heading.marker_line < heading.end <= count + 1 + assert len(blocks) == len(spans) + for number, (block, span) in enumerate(zip(blocks, spans, strict=True)): + assert document.line_of(index, number) == span.start + assert 1 <= span.start < span.end <= count + 1 + assert span.kind == block.kind + if block.kind in LIST_KINDS: + flat = _preorder(block) + for item_number, item in enumerate(flat): + at = document.line_of(index, number, item_number) + assert at is not None and span.start <= at < span.end + assert _ITEM_RE.match(lines[at - 1]), (at, lines[at - 1]) + assert item_text(item).split("\n")[0].strip() in lines[at - 1] + with pytest.raises(IndexError): + document.line_of(index, number, len(flat)) + assert len(span.items) == len(block.items) + for item_span, item in zip(span.items, list_items(block), strict=True): + assert span.start <= item_span.start < item_span.end <= span.end + assert len(item_span.blocks) == len(item.blocks) + _read(blocks, 0) From b999d10260fb0d2655183569951caec858720ef7 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 04:43:07 +0000 Subject: [PATCH 4/5] Tooling API: wire stages 2 and 3 into legaldown.syntax and the README legaldown.syntax now also re-exports iter_document_directives and DirectiveLocation (#33), quote_blocks and drafting_note_blocks (#88, #93), code_content and CodeContent (#93), and the source-layout spans behind Document.layout() (#93, #27). The guard test covers them, and maps parser.quote_content and parser._layout to their public homes. README: result.index.blanks and SectionIndexEntry.alternative (#31) in the result table; the renderer helper table is replaced by a pointer to the Tooling API section, which gains the new names, a "where things are in the file" part, and migration rows. legaldown.validator names are no longer listed as private. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_013WmBAc5T7UCKVdpUmg9qxz --- README.md | 41 ++++++++++++++++++---------- src/legaldown/syntax.py | 25 +++++++++++++++-- src/legaldown/validator/templates.py | 4 +-- tests/test_tooling_surface.py | 38 ++++++++++++++++---------- 4 files changed, 76 insertions(+), 32 deletions(-) diff --git a/README.md b/README.md index 27494b0..062b64d 100644 --- a/README.md +++ b/README.md @@ -299,23 +299,16 @@ itself is what was found, kept nowhere but in `diagnostics`: | `result.index.…` | Contents | |---|---| -| `sections`, `section_lookup` | Numbered section index; resolves `{{ref:}}` targets. Numbers count from the shallowest heading level, and a level a heading skips counts as 1 (`#`, `###`, `##` → 1, 1.1.1, 1.2), so no two sections share a number except alternatives and what they contain (§15.8) | +| `sections`, `section_lookup` | Numbered section index; resolves `{{ref:}}` targets. Numbers count from the shallowest heading level, and a level a heading skips counts as 1 (`#`, `###`, `##` → 1, 1.1.1, 1.2), so no two sections share a number except alternatives and what they contain (§15.8); an entry's `alternative` is true for a section that is an alternative to the one before it and shares its number (§15.4) | | `definition_lookup`, `party_lookup`, `side_lookup`, `attachment_lookup` | Resolved display text | | `values` | The field-spec values the checks met, as written, as `InlineValues`: `dates`, `money`, `durations`, `fields`, `placeholders` (those of the frontmatter too; one with malformed arguments is not among them) | +| `blanks` | The document's blanks, a template's or not, by placeholder id in order of first occurrence: `Blank(id, type, fixed, in_frontmatter)` — `type` the effective type (§10.7, §15.2; `text` when none is written), `fixed` the currency of a money blank or the unit of a duration blank that every occurrence fixes (`""` when one fixes none or they disagree), `in_frontmatter` whether one occurrence is in the frontmatter (§3.10) | | `is_template` | Whether the document is a template (§15.1): it declares `questions`, carries a condition, or holds a `{{choose:}}` | | `placed_markers` | The markers in body text that apply (§5.7, §15.3), in document order: `PlacedMarker(section, block, fragment, offset, source, identifier, condition, field, item, include_only, line)` — in fragment `fragment` of `block_fragments(block)`, at `offset`, which is the block's `field` (`text`, or `suffix` after a lifted `{{ref:}}`/`{{term:}}`); `item` is the list item it marks, counted in pre-order over all the list's items, nested and empty ones included, as `list_fragments` counts them; `identifier` is `""` where it does not apply (an include-only paragraph, §12.2). Identifiers and conditions are as written: check `is_valid` before relying on them | -A renderer builds from these decisions rather than re-deriving them, with the helpers the -validator reads the document with. -[`legaldown-render`](https://github.com/ForLegalAI/legaldown-render) is built this way: - -| Helper | What it gives | -|---|---| -| `lex(text)` → `Lexed` | The directives in inline text (§11.4), and a `view` of it with comments and code spans blanked; `is_escaped(text, offset)` | -| `block_fragments(block)`, `list_fragments(block)`, `list_items(block)` | The texts of a block that hold directives and markers, in the order `PlacedMarker.fragment` counts them (`Fragment(text, anchor)`); the same for a list, with the items each is in (`ListFragment(text, anchor, items)`), numbered in pre-order: an item before the items nested in it; a list's items, as `ListItem`s | -| `is_template(document)`, `is_drafting_note(block)` | The template decision without validating (§15.1); whether a quote block is a drafting note (§15.6) | -| `legaldown.validator`: `parse_condition` → `Condition`, `condition_problem`, `exclusive`, `Presence`, `ALWAYS` | Conditions (§15.3, §15.4): parse one, tell why one is invalid, tell whether two units (each the set of conditions it appears under, `Presence`) can never appear together, given the document's `questions` | -| `legaldown.validator`: `is_valid_iso_date`, `is_valid_money_amount`, `is_positive_numeric`, `IDENTIFIER_RE`, `KNOWN_CURRENCIES` | Value checks (§3.10, §10) | +A renderer builds from these decisions rather than re-deriving them, reading the source with +the validator's own helpers ([Tooling API](#tooling-api)). +[`legaldown-render`](https://github.com/ForLegalAI/legaldown-render) is built this way. ### Reading and editing the document model @@ -481,13 +474,31 @@ for _section, _index, block in document.iter_blocks(): | Names | What they are | |---|---| -| `lex`, `Lexed`, `Directive`, `iter_directives`, `is_escaped`, `format_value`, `collect_source_directives` | The directive lexer (§11.4) and its inverse; the `{{ref:}}` and `{{term:}}` targets of a document | +| `lex`, `Lexed`, `Directive`, `iter_directives`, `is_escaped`, `format_value` | The directive lexer (§11.4) and its inverse. A `Directive`'s `positional_span` and `param_spans` are where each value is written, quotes included, so `text[start:end]` is what to replace to change it; `Lexed.literals` are the comments and code spans the lexer skipped, as `(kind, start, end)` | +| `iter_document_directives(document)` → `DirectiveLocation(section, block, fragment, directive)`, `collect_source_directives` | Every directive in a document's body, in order, read as `validate` reads it: `fragment` indexes `block_fragments(block)`, and is `None` for a `{{ref:}}`/`{{term:}}` the parser lifted into a block's fields; the `{{ref:}}` and `{{term:}}` targets of a document | | `Marker`, `MARKER_RE`, `parse_marker`, `format_marker`, `is_look_alike` | Anchor and condition markers (§5.7, §15.3) | | `find_markers(document)`, `FoundMarker`, `is_include_only` | Every marker and look-alike in a document's body, placed or not; `FoundMarker.placed(template)` says which apply (`validate` gives the placed ones as `result.index.placed_markers`) | | `Fragment`, `ListFragment`, `block_fragments`, `list_fragments`, `text_fragments`, `list_items`, `item_text` | Where a block's text is, and a list's items | | `Quote`, `block_quotes` | The block quotes in a block and whether each is a drafting note | +| `quote_blocks(block)`, `drafting_note_blocks(block)` | The blocks a block quote holds, as the validator reads them, as copies you may change; a drafting note's without its `[!DRAFTING]` marker (§15.6) | +| `code_content(block)` → `CodeContent(info, text, fenced)` | A code block read as CommonMark reads it: the info string, and the code without its fences or indentation | | `FRONTMATTER_RE`, `LINE_ENDING_RE`, `HTML_COMMENT_RE`, `FENCE_OPEN_RE`, `closes_fence`, `fence_end`, `dedent`, `indent_width`, `strip_text` | The Markdown rules the reading is built on: frontmatter, line endings, comments, fenced code, indentation | +**Where things are in the file.** `document.layout()` gives a `SourceLayout`: the frontmatter's +`(start, end)`, the preamble's `BlockSpan`s, and per section a `SectionSpan` (its `HeadingSpan` and +its blocks' spans), in `document.sections` order. Lines are file lines counted from 1, `end` +exclusive; a list's `BlockSpan` has its `ItemSpan`s, each with the spans of its blocks. +`document.line_of(section, block=None, item=None)` is one line: a section's heading, a top-level +block (`section=None` for the preamble), or a list item's marker, items counted in pre-order as +`PlacedMarker.item` counts them. These are the lines diagnostics name (§16.9). Both are `None` +for a document built in code, or changed since it was parsed. + +```python +document = load("contract.lgd") +for number, section in enumerate(document.layout().sections): + print(document.sections[number].title, section.heading.start, [b.start for b in section.blocks]) +``` + If you import one of these from a private module, use its public home: | From | Use | @@ -506,7 +517,9 @@ If you import one of these from a private module, use its public home: | `legaldown.validator.templates`: `block_quotes` | `legaldown.syntax` | | `legaldown.validator.templates`: `check_choose` | `legaldown.grammar.choose_problem` | | `legaldown.validator.units`: `find_markers`, `is_include_only` | `legaldown.syntax` | -| `legaldown.validator`: `parse_condition`, `condition_problem`, `exclusive`, `Presence`, `ALWAYS`, `is_valid_iso_date`, `is_valid_money_amount`, `is_positive_numeric`, `slugify_identifier` | `legaldown.grammar` | +| `legaldown.parser`: `quote_content` | `legaldown.syntax.quote_blocks` (and `drafting_note_blocks`) | +| `legaldown.parser`: `_layout` | `Document.layout()` (file lines, not body-relative) | +| your own fence stripping (`FENCE_OPEN_RE`, `closes_fence`, `dedent`) | `legaldown.syntax.code_content` | | `legaldown.cli`: `_read_answers` | `legaldown.load_answers` | ## What gets checked diff --git a/src/legaldown/syntax.py b/src/legaldown/syntax.py index 9071033..1bb894c 100644 --- a/src/legaldown/syntax.py +++ b/src/legaldown/syntax.py @@ -25,7 +25,9 @@ FENCE_OPEN_RE, HTML_COMMENT_RE, LINE_ENDING_RE, + CodeContent, closes_fence, + code_content, dedent, fence_end, indent_width, @@ -33,8 +35,15 @@ ) from .markers import MARKER_RE, Marker, format_marker, is_look_alike, parse_marker from .models import item_text, list_items -from .parser import FRONTMATTER_RE, collect_source_directives -from .validator.templates import Quote, block_quotes +from .parser import ( + FRONTMATTER_RE, + DirectiveLocation, + collect_source_directives, + iter_document_directives, + quote_blocks, +) +from .positions import BlockSpan, HeadingSpan, ItemSpan, SectionSpan, SourceLayout +from .validator.templates import Quote, block_quotes, drafting_note_blocks from .validator.units import FoundMarker, find_markers, is_include_only __all__ = [ @@ -46,6 +55,8 @@ "is_escaped", "format_value", "collect_source_directives", + "iter_document_directives", + "DirectiveLocation", # Markers (§5.7, §15.3) "Marker", "MARKER_RE", @@ -63,6 +74,10 @@ "text_fragments", "Quote", "block_quotes", + "quote_blocks", + "drafting_note_blocks", + "code_content", + "CodeContent", "list_items", "item_text", # Markdown @@ -75,4 +90,10 @@ "dedent", "indent_width", "strip_text", + # Where a parsed document's parts are in its file (``Document.layout``) + "SourceLayout", + "SectionSpan", + "HeadingSpan", + "BlockSpan", + "ItemSpan", ] diff --git a/src/legaldown/validator/templates.py b/src/legaldown/validator/templates.py index 1b54cb9..4774916 100644 --- a/src/legaldown/validator/templates.py +++ b/src/legaldown/validator/templates.py @@ -313,9 +313,9 @@ def drafting_note_blocks(quote: Block, *, depth: int = 0) -> list[Block]: return [Block(kind="paragraph", text=text)] if text.strip() else [] blocks = quote_blocks(quote, depth=depth) first = blocks[0] if blocks else None - if first is None or first.kind not in ("paragraph", "heading") or not first.text.upper().startswith(_DRAFTING_MARKER): + if first is None or first.kind not in ("paragraph", "heading") or not first.text.upper().startswith(DRAFTING_MARKER): return quote_blocks(Block(kind="quote", text=quote.text.partition("\n")[2]), depth=depth) - rest = first.text[len(_DRAFTING_MARKER):].lstrip() + rest = first.text[len(DRAFTING_MARKER):].lstrip() if not rest: return blocks[1:] first.text = rest diff --git a/tests/test_tooling_surface.py b/tests/test_tooling_surface.py index fbd88e8..ea407a8 100644 --- a/tests/test_tooling_surface.py +++ b/tests/test_tooling_surface.py @@ -34,13 +34,15 @@ SYNTAX = [ "lex", "Lexed", "Directive", "iter_directives", "is_escaped", "format_value", - "collect_source_directives", + "collect_source_directives", "iter_document_directives", "DirectiveLocation", "Marker", "MARKER_RE", "parse_marker", "format_marker", "is_look_alike", "FoundMarker", "find_markers", "is_include_only", "Fragment", "ListFragment", "block_fragments", "list_fragments", "text_fragments", - "Quote", "block_quotes", "list_items", "item_text", + "Quote", "block_quotes", "quote_blocks", "drafting_note_blocks", "code_content", "CodeContent", + "list_items", "item_text", "FRONTMATTER_RE", "LINE_ENDING_RE", "HTML_COMMENT_RE", "FENCE_OPEN_RE", "closes_fence", "fence_end", "dedent", "indent_width", "strip_text", + "SourceLayout", "SectionSpan", "HeadingSpan", "BlockSpan", "ItemSpan", ] #: Where each public name is implemented. @@ -114,6 +116,17 @@ "FoundMarker": "legaldown.validator.units", "find_markers": "legaldown.validator.units", "is_include_only": "legaldown.validator.units", + "DirectiveLocation": "legaldown.parser", + "iter_document_directives": "legaldown.parser", + "quote_blocks": "legaldown.parser", + "drafting_note_blocks": "legaldown.validator.templates", + "CodeContent": "legaldown.markdown", + "code_content": "legaldown.markdown", + "SourceLayout": "legaldown.positions", + "SectionSpan": "legaldown.positions", + "HeadingSpan": "legaldown.positions", + "BlockSpan": "legaldown.positions", + "ItemSpan": "legaldown.positions", } #: What PactTrack and legaldown-render import today from private modules: @@ -138,6 +151,8 @@ ("legaldown.models", "LIST_KINDS"): ("legaldown.grammar", "LIST_KINDS"), ("legaldown.parser", "FRONTMATTER_RE"): ("legaldown.syntax", "FRONTMATTER_RE"), ("legaldown.parser", "MAX_QUOTE_DEPTH"): ("legaldown.grammar", "MAX_QUOTE_DEPTH"), + ("legaldown.parser", "quote_content"): ("legaldown.syntax", "quote_blocks"), + ("legaldown.parser", "_layout"): ("legaldown.syntax", "SourceLayout"), # via Document.layout() ("legaldown.validator.patterns", "LEGALDOWN_EXTENSIONS"): ("legaldown.grammar", "LEGALDOWN_EXTENSIONS"), ("legaldown.validator.templates", "DECISION_QUESTION_TYPES"): ("legaldown.grammar", "DECISION_QUESTION_TYPES"), ("legaldown.validator.templates", "QUESTION_TYPES"): ("legaldown.grammar", "QUESTION_TYPES"), @@ -165,14 +180,6 @@ ("legaldown.validator", "is_valid_money_amount"): ("legaldown.grammar", "is_valid_money_amount"), } -#: TODO: downstream imports whose public home arrives in a later stage. Not -#: asserted yet; the stage that adds the home moves the entry into DOWNSTREAM. -DOWNSTREAM_TODO = { - ("legaldown.parser", "quote_content"): "syntax.quote_blocks (stage 3)", - ("legaldown.parser", "_layout"): "Document.layout() (stage 3)", -} - - def test_all_lists_are_the_expected_names(): assert sorted(grammar.__all__) == sorted(GRAMMAR) assert sorted(syntax.__all__) == sorted(SYNTAX) @@ -208,10 +215,13 @@ def test_a_downstream_import_has_a_public_home(deep, public): assert home is old -def test_the_downstream_todo_is_not_yet_a_public_home(): - # Remove an entry here when its stage lands, and add it to DOWNSTREAM. - for (module, name), home in DOWNSTREAM_TODO.items(): - assert hasattr(importlib.import_module(module), name), (module, name, home) +def test_the_source_layout_is_reached_from_the_document(): + # ``parser._layout`` gave body-relative spans; ``Document.layout()`` gives + # file lines, and ``Document.line_of`` one of them. + document = parse("---\ntitle: T\n---\n\n# A\n\nText.\n") + layout = document.layout() + assert isinstance(layout, syntax.SourceLayout) + assert layout.sections[0].blocks[0].start == document.line_of(0, 0) == 7 assert load_answers is not None From d42dc01822ec6e6d3f9106a2d9965795d3de19ad Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 05:09:15 +0000 Subject: [PATCH 5/5] Tooling API: address review findings - iter_document_directives / collect_source_directives: a ref, term or definition block built in code whose value holds a line break no longer raises; its directive is built directly (main's results again). - iter_document_directives also yields the {{def:}} lifted into a definition block (fragment None), as its docstring promised. - Blank: type is None when no occurrence has a usable type (invalid type, decision question's id); new `consistent` is False for mixed types or two currencies/units, and `fixed` is "" then. - syntax.body_layout(text): the layout of bare body text (a leading --- is a rule, not frontmatter), for tools that laid out editor fields with the private parser._layout. - syntax re-exports is_drafting_note; README mentions is_template and is_drafting_note again. - choose_problem raises ValueError for a directive that is not a well-formed {{choose:}}. - Docstrings: layout spans (end exclusive, blank lines between parts belong to none), quote helpers past MAX_QUOTE_DEPTH and depth=, SectionIndexEntry.alternative (its preceding sibling). - FINAL_CHECK_RULES moves to validator.patterns. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_013WmBAc5T7UCKVdpUmg9qxz --- README.md | 21 +++-- src/legaldown/grammar.py | 2 +- src/legaldown/models.py | 4 +- src/legaldown/parser.py | 56 +++++++++--- src/legaldown/positions.py | 71 ++++++++++----- src/legaldown/syntax.py | 9 +- src/legaldown/validator/core.py | 22 +++-- src/legaldown/validator/patterns.py | 4 + src/legaldown/validator/result.py | 32 ++++--- src/legaldown/validator/templates.py | 14 ++- tests/test_model_answers.py | 19 ++++ tests/test_result_answers.py | 127 ++++++++++++++++++++++++++- tests/test_tooling_surface.py | 84 ++++++++++++++++-- 13 files changed, 380 insertions(+), 85 deletions(-) diff --git a/README.md b/README.md index 062b64d..f6b62b2 100644 --- a/README.md +++ b/README.md @@ -299,13 +299,14 @@ itself is what was found, kept nowhere but in `diagnostics`: | `result.index.…` | Contents | |---|---| -| `sections`, `section_lookup` | Numbered section index; resolves `{{ref:}}` targets. Numbers count from the shallowest heading level, and a level a heading skips counts as 1 (`#`, `###`, `##` → 1, 1.1.1, 1.2), so no two sections share a number except alternatives and what they contain (§15.8); an entry's `alternative` is true for a section that is an alternative to the one before it and shares its number (§15.4) | +| `sections`, `section_lookup` | Numbered section index; resolves `{{ref:}}` targets. Numbers count from the shallowest heading level, and a level a heading skips counts as 1 (`#`, `###`, `##` → 1, 1.1.1, 1.2), so no two sections share a number except alternatives and what they contain (§15.8); an entry's `alternative` is true for a section that is an alternative to its preceding sibling — the section just before it at its level under the same parent, with the same identifier, never present together with it (§15.4) — and so shares its number (§15.8) | | `definition_lookup`, `party_lookup`, `side_lookup`, `attachment_lookup` | Resolved display text | | `values` | The field-spec values the checks met, as written, as `InlineValues`: `dates`, `money`, `durations`, `fields`, `placeholders` (those of the frontmatter too; one with malformed arguments is not among them) | -| `blanks` | The document's blanks, a template's or not, by placeholder id in order of first occurrence: `Blank(id, type, fixed, in_frontmatter)` — `type` the effective type (§10.7, §15.2; `text` when none is written), `fixed` the currency of a money blank or the unit of a duration blank that every occurrence fixes (`""` when one fixes none or they disagree), `in_frontmatter` whether one occurrence is in the frontmatter (§3.10) | +| `blanks` | The document's blanks, a template's or not, by placeholder id in order of first occurrence: `Blank(id, type, fixed, in_frontmatter, consistent)` — `type` the type the validator settled on: the effective type of the first occurrence with a valid one (§10.7, §15.2; `text` when none is written), or `None` when no occurrence has a usable type (each is written with an invalid one, or the id is a decision question's); `fixed` the currency of a money blank or the unit of a duration blank that every occurrence fixes (`""` when one fixes none, or the blank is not `consistent`); `consistent` false when its occurrences have different types or fix two currencies or units (placeholder-type-inconsistent); `in_frontmatter` whether one occurrence is in the frontmatter (§3.10) | | `is_template` | Whether the document is a template (§15.1): it declares `questions`, carries a condition, or holds a `{{choose:}}` | | `placed_markers` | The markers in body text that apply (§5.7, §15.3), in document order: `PlacedMarker(section, block, fragment, offset, source, identifier, condition, field, item, include_only, line)` — in fragment `fragment` of `block_fragments(block)`, at `offset`, which is the block's `field` (`text`, or `suffix` after a lifted `{{ref:}}`/`{{term:}}`); `item` is the list item it marks, counted in pre-order over all the list's items, nested and empty ones included, as `list_fragments` counts them; `identifier` is `""` where it does not apply (an include-only paragraph, §12.2). Identifiers and conditions are as written: check `is_valid` before relying on them | +`is_template(document)` (top-level) gives the template decision without validating (§15.1). A renderer builds from these decisions rather than re-deriving them, reading the source with the validator's own helpers ([Tooling API](#tooling-api)). [`legaldown-render`](https://github.com/ForLegalAI/legaldown-render) is built this way. @@ -456,7 +457,7 @@ slugify_identifier("日本語", fallback="") # "": nothing usable, as aga | `slugify_identifier(text, *, fallback="section")`, `format_section_number` | The §5.3 identifier of a heading or term, and `fallback` where the text yields none (`""` tells that case apart); a dotted section number | | `is_valid_iso_date`, `is_valid_numeric`, `is_valid_money_amount`, `is_positive_numeric` | Value checks (§3.10, §10) | | `parse_condition` → `Condition`, `condition_problem`, `exclusive`, `Presence`, `ALWAYS` | Conditions (§15.3, §15.4): parse one, tell why one is invalid, tell whether two units can never appear together | -| `choose_problem(directive, questions)` | Why a `{{choose:}}` is invalid (choose-invalid, §15.5): the first message `validate` records for it, or `None` | +| `choose_problem(directive, questions)` | Why a `{{choose:}}` is invalid (choose-invalid, §15.5): the first message `validate` records for it, or `None`; `ValueError` for a directive that is not a `{{choose:}}` or is malformed (`validate` reports that as directive-malformed) | **`legaldown.syntax`** is reading source text the way the validator does. @@ -475,12 +476,12 @@ for _section, _index, block in document.iter_blocks(): | Names | What they are | |---|---| | `lex`, `Lexed`, `Directive`, `iter_directives`, `is_escaped`, `format_value` | The directive lexer (§11.4) and its inverse. A `Directive`'s `positional_span` and `param_spans` are where each value is written, quotes included, so `text[start:end]` is what to replace to change it; `Lexed.literals` are the comments and code spans the lexer skipped, as `(kind, start, end)` | -| `iter_document_directives(document)` → `DirectiveLocation(section, block, fragment, directive)`, `collect_source_directives` | Every directive in a document's body, in order, read as `validate` reads it: `fragment` indexes `block_fragments(block)`, and is `None` for a `{{ref:}}`/`{{term:}}` the parser lifted into a block's fields; the `{{ref:}}` and `{{term:}}` targets of a document | +| `iter_document_directives(document)` → `DirectiveLocation(section, block, fragment, directive)`, `collect_source_directives` | Every directive in a document's body, in order, read as `validate` reads it: `fragment` indexes `block_fragments(block)`, and is `None` for a `{{ref:}}`, `{{term:}}` or `{{def:}}` the parser lifted into a block's fields (a definition's comes before its text); the `{{ref:}}` and `{{term:}}` targets of a document | | `Marker`, `MARKER_RE`, `parse_marker`, `format_marker`, `is_look_alike` | Anchor and condition markers (§5.7, §15.3) | | `find_markers(document)`, `FoundMarker`, `is_include_only` | Every marker and look-alike in a document's body, placed or not; `FoundMarker.placed(template)` says which apply (`validate` gives the placed ones as `result.index.placed_markers`) | | `Fragment`, `ListFragment`, `block_fragments`, `list_fragments`, `text_fragments`, `list_items`, `item_text` | Where a block's text is, and a list's items | -| `Quote`, `block_quotes` | The block quotes in a block and whether each is a drafting note | -| `quote_blocks(block)`, `drafting_note_blocks(block)` | The blocks a block quote holds, as the validator reads them, as copies you may change; a drafting note's without its `[!DRAFTING]` marker (§15.6) | +| `Quote`, `block_quotes`, `is_drafting_note(block)` | The block quotes in a block and whether each is a drafting note; whether a quote block is one (§15.6) | +| `quote_blocks(block, depth=0)`, `drafting_note_blocks(block, depth=0)` | The blocks a block quote holds, as the validator reads them, as copies you may change; a drafting note's without its `[!DRAFTING]` marker (§15.6). `depth` is how many list items and quotes the quote is in, the count before entering it (a quote in a list item is at 1); past `MAX_QUOTE_DEPTH` the validator reads a quote as one text, so you get one paragraph of its text as written | | `code_content(block)` → `CodeContent(info, text, fenced)` | A code block read as CommonMark reads it: the info string, and the code without its fences or indentation | | `FRONTMATTER_RE`, `LINE_ENDING_RE`, `HTML_COMMENT_RE`, `FENCE_OPEN_RE`, `closes_fence`, `fence_end`, `dedent`, `indent_width`, `strip_text` | The Markdown rules the reading is built on: frontmatter, line endings, comments, fenced code, indentation | @@ -491,7 +492,11 @@ exclusive; a list's `BlockSpan` has its `ItemSpan`s, each with the spans of its `document.line_of(section, block=None, item=None)` is one line: a section's heading, a top-level block (`section=None` for the preamble), or a list item's marker, items counted in pre-order as `PlacedMarker.item` counts them. These are the lines diagnostics name (§16.9). Both are `None` -for a document built in code, or changed since it was parsed. +for a document built in code, or changed since it was parsed. A span covers the lines +`[start, end)`; the blank lines between parts belong to none (but for those between a list's items, +which the item before holds). For text you hold apart from its file — an editor's field — +`body_layout(text)` is the same `SourceLayout` of that text read as a body alone, with lines counted +from 1 at its start and no frontmatter (a `---` first line is a thematic break there). ```python document = load("contract.lgd") @@ -518,7 +523,7 @@ If you import one of these from a private module, use its public home: | `legaldown.validator.templates`: `check_choose` | `legaldown.grammar.choose_problem` | | `legaldown.validator.units`: `find_markers`, `is_include_only` | `legaldown.syntax` | | `legaldown.parser`: `quote_content` | `legaldown.syntax.quote_blocks` (and `drafting_note_blocks`) | -| `legaldown.parser`: `_layout` | `Document.layout()` (file lines, not body-relative) | +| `legaldown.parser`: `_layout` | `legaldown.syntax.body_layout(text)` for bare body text (lines from 1 at its start); `Document.layout()` for a parsed document (file lines) | | your own fence stripping (`FENCE_OPEN_RE`, `closes_fence`, `dedent`) | `legaldown.syntax.code_content` | | `legaldown.cli`: `_read_answers` | `legaldown.load_answers` | diff --git a/src/legaldown/grammar.py b/src/legaldown/grammar.py index ef1502f..e9b8a1d 100644 --- a/src/legaldown/grammar.py +++ b/src/legaldown/grammar.py @@ -26,7 +26,6 @@ from .parser import MAX_LIST_DEPTH, MAX_QUOTE_DEPTH from .specification import SPEC_VERSION from .validator.conditions import ALWAYS, Condition, Presence, condition_problem, exclusive, parse_condition -from .validator.core import FINAL_CHECK_RULES from .validator.helpers import ( format_section_number, is_positive_numeric, @@ -37,6 +36,7 @@ ) from .validator.patterns import ( DURATION_UNITS, + FINAL_CHECK_RULES, IDENTIFIER_RE, KNOWN_CURRENCIES, LEGALDOWN_EXTENSIONS, diff --git a/src/legaldown/models.py b/src/legaldown/models.py index 7c7a7b6..29e0bbd 100644 --- a/src/legaldown/models.py +++ b/src/legaldown/models.py @@ -217,7 +217,9 @@ def iter_indexed_blocks(self) -> Iterator[tuple[int | None, int, Block]]: def layout(self) -> SourceLayout | None: """Where the document's frontmatter, headings, blocks and list items lie in the file it was parsed from (§16.9): file lines counted from - 1, each span ending where the next begins. None for a document + 1, each span the lines ``[start, end)`` (end exclusive; the blank + lines between parts belong to none, but for those between a list's + items). None for a document built in code, and for one changed since it was parsed so that it no longer fits its source: where its parts lie is not known then.""" from .positions import source_layout # positions describes documents: it builds on this module diff --git a/src/legaldown/parser.py b/src/legaldown/parser.py index 16b017b..af63305 100644 --- a/src/legaldown/parser.py +++ b/src/legaldown/parser.py @@ -1368,9 +1368,10 @@ def quote_blocks(block: Block, *, depth: int = 0) -> list[Block]: validator keeps its own reading of a quote's content. A heading in one is a ``heading`` block, not a section, and no directive is lifted into block fields (§4.1). *depth*: how many list items and quotes *block* is - in; a quote in ``MAX_QUOTE_DEPTH`` of them is read as the validator reads - it, as one text: a single paragraph, or no block when it holds none. - Raises ``ValueError`` for a block that is not a quote.""" + in, the count before entering it; a quote in ``MAX_QUOTE_DEPTH`` of them + is not read into blocks: it is one paragraph of its text as written + (the validator reads no directives in its fenced code there), or no + block when it holds none. Raises ``ValueError`` for a block that is not a quote.""" if block.kind != "quote": raise ValueError(f"not a block quote: {block.kind}") if depth >= MAX_QUOTE_DEPTH: @@ -1478,9 +1479,11 @@ class DirectiveLocation(NamedTuple): """A directive and where it is (``iter_document_directives``): in block *block* of the preamble (*section* None) or of section *section*, in fragment *fragment* of ``block_fragments(block)``, which *directive*'s - offsets are into. *fragment* is None for the ``{{ref:}}`` or - ``{{term:}}`` the parser lifted into the block's own fields: its offsets - are into the directive as the serializer writes it.""" + offsets are into. *fragment* is None for the ``{{ref:}}``, ``{{term:}}`` + or ``{{def:}}`` the parser lifted into the block's own fields: its + offsets are into the directive as the serializer writes it alone, and a + directive of a block built in code whose value holds a line break, which + no source can write, has none (offsets 0, no spans, empty source).""" section: int | None block: int @@ -1497,10 +1500,13 @@ def iter_document_directives(document: Document) -> Iterator[DirectiveLocation]: blocks and raw HTML hold none (§11.4), and a block quote's or a list's are those of the blocks it holds. A ``{{ref:}}`` or ``{{term:}}`` the parser lifted into a block's fields is among them, between the directives of the - block's text before it and after it. The frontmatter and the headings are - not read. + block's text before it and after it, and the ``{{def:}}`` it lifted into + a definition block's fields, before those of the definition's text. The + frontmatter and the headings are not read. """ for section, index, block in document.iter_indexed_blocks(): + if block.kind == "definition": + yield from _lifted_directive(section, index, block) lifted = block.kind in ("ref", "term") and bool(block.target) # The fragments before the lifted directive: its text, and its prefix. before = bool(block.text) + bool(block.prefix) @@ -1515,14 +1521,36 @@ def iter_document_directives(document: Document) -> Iterator[DirectiveLocation]: def _lifted_directive(section: int | None, index: int, block: Block) -> Iterator[DirectiveLocation]: - """The directive a ``ref`` or ``term`` block holds in its fields, as the - serializer writes it alone.""" + """The directive a ``ref``, ``term`` or ``definition`` block holds in its + fields, as the serializer writes it alone; built directly, with no + offsets, when the serializer refuses its value (a line break).""" from .serializer import render_block # the serializer builds on this module - source = render_block(Block(kind=block.kind, target=block.target, label=block.label)) - for directive in lex(source).directives: - yield DirectiveLocation(section, index, None, directive) - return + name = "def" if block.kind == "definition" else block.kind + try: + source = render_block( + Block( + kind=block.kind, + target=block.target, + label=block.label, + definition_id=block.definition_id, + term=block.term, + ) + ) + except ValueError: + source = "" + # The last: a term written before a ``{{def:}}`` may hold directives itself. + found = [directive for directive in lex(source).directives if directive.name == name] + if found: + directive = found[-1] + else: + positional = block.definition_id if block.kind == "definition" else block.target + params = {"label": block.label} if block.kind == "term" and block.label else {} + directive = Directive( + name=name, positional=positional or None, params=params, duplicates=(), malformed="", + start=0, end=0, source="", + ) + yield DirectiveLocation(section, index, None, directive) def collect_source_directives(document: Document) -> tuple[set[str], set[str]]: diff --git a/src/legaldown/positions.py b/src/legaldown/positions.py index 9ec299f..703677e 100644 --- a/src/legaldown/positions.py +++ b/src/legaldown/positions.py @@ -14,7 +14,7 @@ import yaml from .directives import is_escaped -from .markdown import paragraph_text +from .markdown import LINE_ENDING_RE, paragraph_text if TYPE_CHECKING: from .models import Block, Document @@ -129,10 +129,13 @@ class SectionSpan: @dataclass(frozen=True, slots=True) class SourceLayout: """Where a parsed document's parts lie in its file (``Document.layout``), - in lines counted from 1 and ending where the next begins: the frontmatter - as ``(start, end)`` with its ``---`` lines, or None without one; the - preamble's blocks (§4.4); each section's heading and blocks, in the - document's order, so that ``sections[n]`` is ``Document.sections[n]``'s.""" + in lines counted from 1: each span is the lines ``[start, end)``, its end + exclusive — the line after its last — and the blank lines between parts + belong to none, but for those between a list's items, which the item + before holds. The frontmatter is ``(start, end)`` with its ``---`` + lines, or None without one; then the preamble's blocks (§4.4); each + section's heading and blocks, in the document's order, so that + ``sections[n]`` is ``Document.sections[n]``'s.""" frontmatter: tuple[int, int] | None preamble: tuple[BlockSpan, ...] @@ -193,14 +196,6 @@ def block(self, section: int | None, index: int) -> int: def to_layout(self) -> SourceLayout: """``layout`` as the public ``SourceLayout``, in file lines.""" - base = self.body_start + 1 - - def block(span: Any) -> BlockSpan: - return BlockSpan(span.kind, base + span.start, base + span.end, tuple(item(each) for each in span.items)) - - def item(span: Any) -> ItemSpan: - return ItemSpan(base + span.start, base + span.end, tuple(block(each) for each in span.blocks)) - frontmatter = None if () in self.keys: # The frontmatter's lines, its closing ``---`` too, which the @@ -211,17 +206,7 @@ def item(span: Any) -> ItemSpan: if written is not None: text = written.group() frontmatter = (1, 1 + text.count("\n") + (0 if text.endswith("\n") else 1)) - return SourceLayout( - frontmatter, - tuple(block(each) for each in self.layout.preamble), - tuple( - SectionSpan( - HeadingSpan(base + heading.start, base + heading.end, base + heading.marker_line), - tuple(block(each) for each in blocks), - ) - for heading, blocks in self.layout.sections - ), - ) + return _public_layout(self.layout, self.body_start + 1, frontmatter) def item(self, section: int | None, index: int, item: int) -> int: """The line of list item *item* of top-level block *index* of a @@ -539,6 +524,44 @@ def _lifted(block: Block, own: list[str], first: int, lines: list[str], start: i return _Leaf(start, end, [joined], places, lifted if lifted >= 0 else None) +def _public_layout(layout: Any, base: int, frontmatter: tuple[int, int] | None) -> SourceLayout: + """The parser's *layout* (body lines from 0) as a ``SourceLayout`` whose + lines are counted from *base*, the line of the body's first.""" + + def block(span: Any) -> BlockSpan: + return BlockSpan(span.kind, base + span.start, base + span.end, tuple(item(each) for each in span.items)) + + def item(span: Any) -> ItemSpan: + return ItemSpan(base + span.start, base + span.end, tuple(block(each) for each in span.blocks)) + + return SourceLayout( + frontmatter, + tuple(block(each) for each in layout.preamble), + tuple( + SectionSpan( + HeadingSpan(base + heading.start, base + heading.end, base + heading.marker_line), + tuple(block(each) for each in blocks), + ) + for heading, blocks in layout.sections + ), + ) + + +def body_layout(text: str) -> SourceLayout: + """Where the parts of *text* lie, read as a document body alone: a + ``SourceLayout`` of lines counted from 1 at the start of *text*, with no + frontmatter — a ``---`` first line is a thematic break, not the opening + of one. For text a tool holds apart from the file it came from, such as + an editor's field; ``Document.layout`` is the layout of a parsed + document. Line endings are LF, CR or CRLF, as the parser reads them.""" + from .parser import _layout # the parser builds on this module + + lines = LINE_ENDING_RE.sub("\n", text).split("\n") + if lines[-1] == "": + lines.pop() # the last line's ending, not a line + return _public_layout(_layout(lines), 1, None) + + def source_layout(document: Document) -> SourceLayout | None: """``Document.layout``: where *document*'s parts lie in the source it was parsed from; None without a source map, or when the document no longer diff --git a/src/legaldown/syntax.py b/src/legaldown/syntax.py index 1bb894c..275a12c 100644 --- a/src/legaldown/syntax.py +++ b/src/legaldown/syntax.py @@ -42,8 +42,8 @@ iter_document_directives, quote_blocks, ) -from .positions import BlockSpan, HeadingSpan, ItemSpan, SectionSpan, SourceLayout -from .validator.templates import Quote, block_quotes, drafting_note_blocks +from .positions import BlockSpan, HeadingSpan, ItemSpan, SectionSpan, SourceLayout, body_layout +from .validator.templates import Quote, block_quotes, drafting_note_blocks, is_drafting_note from .validator.units import FoundMarker, find_markers, is_include_only __all__ = [ @@ -74,6 +74,7 @@ "text_fragments", "Quote", "block_quotes", + "is_drafting_note", "quote_blocks", "drafting_note_blocks", "code_content", @@ -90,7 +91,9 @@ "dedent", "indent_width", "strip_text", - # Where a parsed document's parts are in its file (``Document.layout``) + # Where a parsed document's parts are in its file (``Document.layout``), + # and where a body text's are (``body_layout``) + "body_layout", "SourceLayout", "SectionSpan", "HeadingSpan", diff --git a/src/legaldown/validator/core.py b/src/legaldown/validator/core.py index 9981cc7..b4b7538 100644 --- a/src/legaldown/validator/core.py +++ b/src/legaldown/validator/core.py @@ -40,6 +40,8 @@ slugify_identifier, ) from .patterns import ( + _CONSTRUCT_PRESENT, + _UNFILLED, DURATION_UNITS, IDENTIFIER_RE, KNOWN_CURRENCIES, @@ -354,6 +356,7 @@ def _check_placeholder( if blank.type is None: blank.type = ptype if blank.type != ptype: + blank.mixed = True result.error( "placeholder-type-inconsistent", f"Placeholder '{pid}' used with inconsistent types: " @@ -389,12 +392,6 @@ def _check_blank_codes(blanks: dict[str, _BlankState], result: _Recorder) -> Non ) -_UNFILLED = "placeholder-unfilled" -_CONSTRUCT_PRESENT = "template-construct-present" -#: The rules of the final check (§15.9): the only ones it reports. -FINAL_CHECK_RULES: frozenset[str] = frozenset({_UNFILLED, _CONSTRUCT_PRESENT}) - - def _check_final( placeholders: list[tuple[Directive, Line]], chooses: list[tuple[Directive, Line]], @@ -1567,15 +1564,16 @@ def definition_line(ref: Any) -> int | None: ) _check_blank_codes(blanks, result) - result.index.blanks = { - pid: Blank( + result.index.blanks = {} + for pid, state in blanks.items(): + consistent = not state.mixed and len(state.codes - {""}) < 2 + result.index.blanks[pid] = Blank( id=pid, - type=state.type or "text", - fixed=_fixed_by_all(state.codes) or "", + type=state.type, + fixed=(_fixed_by_all(state.codes) or "") if consistent else "", in_frontmatter=state.in_frontmatter, + consistent=consistent, ) - for pid, state in blanks.items() - } # ── Templates (§15) ── # Every condition in a condition position (§15.3): on what, as written, diff --git a/src/legaldown/validator/patterns.py b/src/legaldown/validator/patterns.py index f4f4f2a..c51c385 100644 --- a/src/legaldown/validator/patterns.py +++ b/src/legaldown/validator/patterns.py @@ -22,6 +22,10 @@ # In the order §10.5 lists them, for diagnostics. DURATION_UNITS: tuple[str, ...] = ("S", "MIN", "H", "D", "W", "MO", "Y") VALID_DURATION_UNITS: frozenset[str] = frozenset(DURATION_UNITS) +# The rules of the final check (§15.9): the only ones it reports. +_UNFILLED = "placeholder-unfilled" +_CONSTRUCT_PRESENT = "template-construct-present" +FINAL_CHECK_RULES: frozenset[str] = frozenset({_UNFILLED, _CONSTRUCT_PRESENT}) # §10.7 placeholder types — also the value question types of §15.2. VALID_PLACEHOLDER_TYPES: frozenset[str] = frozenset({"text", "date", "money", "duration"}) diff --git a/src/legaldown/validator/result.py b/src/legaldown/validator/result.py index 2f2ba1a..8429802 100644 --- a/src/legaldown/validator/result.py +++ b/src/legaldown/validator/result.py @@ -16,9 +16,10 @@ class SectionIndexEntry: as 1, so ``# A``, ``### B``, ``## C`` are 1, 1.1.1, 1.2: no two sections share a number, except alternatives and what they contain (§15.8). ``path`` joins the identifiers of the section and its ancestors. - ``alternative`` is True when the section is an alternative to the one - before it — the same identifier, never present together (§15.4) — and so - shares that section's number (§15.8).""" + ``alternative`` is True when the section is an alternative to its + preceding sibling — the section just before it at its level under the + same parent — which has the same identifier and is never present + together with it (§15.4), and so shares its number (§15.8).""" title: str identifier: str path: str @@ -90,16 +91,23 @@ class Blank: validating the document resolved it. A tool that asks for the answers (a form, an interview) reads the blank's type and what it fixes here. - ``type`` is the effective type of its first occurrence with a valid one - (§10.7, §15.2): ``text`` when none is written. ``fixed`` is the currency - of a money blank, or the unit of a duration blank, that every occurrence - fixes, and ``""`` when some occurrence fixes none or they disagree (the - disagreement is placeholder-type-inconsistent). ``in_frontmatter`` is - True when one of its occurrences is in the frontmatter (§3.10).""" + ``type`` is the type the validator settled on: the effective type of its + first occurrence with a valid one (§10.7, §15.2), ``text`` when none is + written. It is None when no occurrence has a usable type: each is written + with an invalid one (placeholder-type-invalid), or the id is a decision + question's (placeholder-question-mismatch). ``fixed`` is the currency of + a money blank, or the unit of a duration blank, that every occurrence + fixes, and ``""`` when some occurrence fixes none, or the blank is not + ``consistent``. ``consistent`` is False when its occurrences have + different types, or fix two currencies or two units (both are + placeholder-type-inconsistent); ``type`` is then the first one's. + ``in_frontmatter`` is True when one of its occurrences is in the + frontmatter (§3.10).""" id: str - type: str + type: str | None fixed: str = "" in_frontmatter: bool = False + consistent: bool = True #: A diagnostic's line (from 1), or a function giving it, called only when a @@ -152,7 +160,9 @@ class DocumentIndex: #: The document's blanks (``Blank``), by placeholder id, in the order of #: their first occurrence — frontmatter first. A document that is not a #: template has them too. A placeholder whose arguments or id are - #: malformed is not among them (the diagnostics report it). + #: malformed is not among them (the diagnostics report it); one with no + #: usable type is, with type None, and so is one that is not consistent + #: (``Blank``). blanks: dict[str, Blank] = field(default_factory=dict) diff --git a/src/legaldown/validator/templates.py b/src/legaldown/validator/templates.py index 4774916..53d1c96 100644 --- a/src/legaldown/validator/templates.py +++ b/src/legaldown/validator/templates.py @@ -54,6 +54,8 @@ class _BlankState: #: fixes none. Two codes are an Error (placeholder-type-inconsistent). codes: set[str] = field(default_factory=set) in_frontmatter: bool = False + #: Whether two occurrences have different types (placeholder-type-inconsistent). + mixed: bool = False #: The line of the first occurrence fixing each code (§16.9). code_lines: dict[str, Any] = field(default_factory=dict) # a ``result.Line`` each @@ -302,8 +304,9 @@ def drafting_note_blocks(quote: Block, *, depth: int = 0) -> list[Block]: The marker starts the first paragraph or heading, and is cut from it, the block going when nothing else is in it; where it does not (a marker line indented as code), the note is what is written after its first - line. *depth*: as in ``quote_blocks``. Raises ``ValueError`` for a block - that is not a drafting note (``is_drafting_note``).""" + line. *depth*: as in ``quote_blocks``; past it the note is one paragraph + of its text after the first line, as written. Raises ``ValueError`` for a + block that is not a drafting note (``is_drafting_note``).""" from ..parser import MAX_QUOTE_DEPTH, quote_blocks # see the import note in check_template_body if not is_drafting_note(quote): @@ -450,7 +453,12 @@ def choose_problem(directive: Directive, questions: Any) -> str | None: """Why the ``{{choose:}}`` *directive* is not valid for *questions* (the document's ``metadata.questions``): the first choose-invalid message (§15.5) the validator reports for it, or None when it names a declared - boolean or choice question and lists exactly its answers.""" + boolean or choice question and lists exactly its answers. Raises + ``ValueError`` for a directive that is not a ``{{choose:}}`` or is + malformed: the validator reports a malformed one as directive-malformed, + never as choose-invalid.""" + if directive.name != "choose" or directive.malformed: + raise ValueError(f"not a well-formed {{{{choose:}}}}: {directive.source!r}") problems = _choose_problems(directive, questions) return problems[0] if problems else None diff --git a/tests/test_model_answers.py b/tests/test_model_answers.py index 3671a18..53f9786 100644 --- a/tests/test_model_answers.py +++ b/tests/test_model_answers.py @@ -67,6 +67,16 @@ def test_a_quote_past_the_quote_depth_is_one_text(): assert quote_blocks(Block(kind="quote", text=""), depth=MAX_QUOTE_DEPTH) == [] +def test_a_quote_past_the_quote_depth_is_its_text_as_written_fenced_code_included(): + text = "> not read\n\n```\n{{ref: x}}\n```\n\nafter" + quote = Block(kind="quote", text=text) + assert _shown(quote_blocks(quote, depth=MAX_QUOTE_DEPTH)) == [("paragraph", text)] + note = Block(kind="quote", text="[!DRAFTING]\n" + text) + assert _shown(drafting_note_blocks(note, depth=MAX_QUOTE_DEPTH)) == [("paragraph", text)] + # Short of the depth the same text is read into blocks, the fence a code block. + assert [block.kind for block in quote_blocks(quote, depth=MAX_QUOTE_DEPTH - 1)][:3] == ["quote", "code", "paragraph"] + + @pytest.mark.parametrize("kind", ["paragraph", "code", "unordered_list", "rule"]) def test_quote_blocks_of_a_block_that_is_no_quote(kind): with pytest.raises(ValueError, match="quote"): @@ -229,6 +239,15 @@ def test_the_layout_gives_the_file_lines_of_each_part(): assert third.blocks == (BlockSpan("paragraph", 25, 26),) +def test_a_span_ends_after_its_last_line_and_the_blank_lines_between_parts_belong_to_none(): + layout = parse("# A\n\nOne\ntwo\n\n\nThree\n").layout() + first, second = layout.sections[0].blocks + assert (layout.sections[0].heading.start, layout.sections[0].heading.end) == (1, 2) + assert (first.start, first.end) == (3, 5) # lines 3 and 4 + assert (second.start, second.end) == (7, 8) # lines 5 and 6 are blank, in no span + assert first.end < second.start + + def test_the_layout_is_frozen(): layout = parse(_SOURCE).layout() assert layout is not None diff --git a/tests/test_result_answers.py b/tests/test_result_answers.py index a4fd2fb..456fc13 100644 --- a/tests/test_result_answers.py +++ b/tests/test_result_answers.py @@ -37,7 +37,8 @@ def test_blank_is_public_and_frozen(): assert "Blank" in legaldown.validator.__all__ assert legaldown.validator.Blank is Blank blank = Blank(id="a", type="text") - assert (blank.fixed, blank.in_frontmatter) == ("", False) + assert (blank.fixed, blank.in_frontmatter, blank.consistent) == ("", False, True) + assert [f.name for f in dataclasses.fields(Blank)] == ["id", "type", "fixed", "in_frontmatter", "consistent"] with pytest.raises(dataclasses.FrozenInstanceError): blank.type = "date" # type: ignore[misc] @@ -96,6 +97,59 @@ def test_a_blank_has_the_type_of_its_first_occurrence_with_a_valid_one(): assert blank.type == "date" +def test_a_plain_blank_is_text_and_consistent(): + blank = _blanks("# A\n\n{{placeholder: name}} {{placeholder: name}}")["name"] + assert (blank.type, blank.fixed, blank.consistent) == ("text", "", True) + + +def test_a_blank_whose_occurrences_disagree_on_type_is_not_consistent_and_fixes_nothing(): + result = validate( + parse( + _FRONTMATTER + + "# A\n\n{{placeholder: p, type=money, currency=EUR}} {{placeholder: p, type=date}}\n" + ) + ) + assert "placeholder-type-inconsistent" in result.rules("error") + blank = result.index.blanks["p"] + assert (blank.type, blank.fixed, blank.consistent) == ("money", "", False) + + +@pytest.mark.parametrize( + "occurrences", + [ + "{{placeholder: p, type=money, currency=EUR}} {{placeholder: p, type=money, currency=USD}}", + "{{placeholder: p, type=duration, unit=D}} {{placeholder: p, type=duration, unit=W}}", + ], +) +def test_a_blank_fixing_two_currencies_or_units_is_not_consistent_and_fixes_nothing(occurrences): + blank = _blanks(f"# A\n\n{occurrences}")["p"] + assert (blank.fixed, blank.consistent) == ("", False) + assert blank.type in ("money", "duration") + + +def test_a_blank_that_fixes_one_currency_or_none_is_consistent(): + blank = _blanks("# A\n\n{{placeholder: p, type=money, currency=EUR}} {{placeholder: p, type=money}}")["p"] + assert (blank.type, blank.fixed, blank.consistent) == ("money", "", True) + + +def test_a_blank_with_only_an_invalid_type_has_no_type(): + result = validate(parse(_FRONTMATTER + "# A\n\n{{placeholder: p, type=foo}}\n")) + assert "placeholder-type-invalid" in result.rules("error") + blank = result.index.blanks["p"] + assert (blank.type, blank.fixed, blank.consistent) == (None, "", True) + + +def test_a_blank_with_the_id_of_a_decision_question_has_no_type(): + result = validate( + parse( + _FRONTMATTER.replace("---\n\n", "questions:\n q:\n type: boolean\n---\n\n") + + "# A\n\n{{placeholder: q}}\n" + ) + ) + assert "placeholder-question-mismatch" in result.rules("error") + assert result.index.blanks["q"].type is None + + def test_a_blank_takes_its_type_from_its_question(): blanks = _blanks("# A\n\n{{placeholder: fee}}", "questions:\n fee:\n type: money\n") assert (blanks["fee"].type, blanks["fee"].fixed) == ("money", "") @@ -132,6 +186,16 @@ def test_a_section_sharing_its_predecessors_number_is_an_alternative(): assert [s.alternative for s in sections] == [False, False, True, False, False] +def test_an_alternative_is_to_its_preceding_sibling_not_to_any_earlier_section_with_its_identifier(): + body = ( + "# Disputes {#disputes when=forum:courts}\n\nText.\n\n# Notices\n\nText.\n\n" + "# Disputes {#disputes when=forum:arbitration}\n\nText.\n" + ) + result = validate(parse(f"---\ntitle: T\n{_QUESTIONS}---\n\n{body}\n")) + assert [s.alternative for s in result.index.sections] == [False, False, False] + assert [s.number for s in result.index.sections] == ["1", "2", "3"] + + def test_sections_with_one_identifier_that_can_appear_together_are_not_alternatives(): result = validate(parse(_FRONTMATTER + "# A {#same}\n\nText.\n\n# A {#same}\n\nText.\n")) assert not any(s.alternative for s in result.index.sections) @@ -400,6 +464,65 @@ def test_collect_source_directives_skips_malformed_and_empty_targets(): assert collect_source_directives(document) == ({"ok"}, set()) +def test_a_directive_a_document_built_in_code_holds_with_a_line_break_is_still_yielded(): + from legaldown import Block + + document = parse(_FRONTMATTER + "# A\n\nText.\n") + blocks = document.sections[0].blocks + blocks.append(Block(kind="ref", target="line\nbreak")) + assert collect_source_directives(document) == ({"line\nbreak"}, set()) + blocks.append(Block(kind="term", target="line\nbreak", label="la\nbel")) + assert collect_source_directives(document) == ({"line\nbreak"}, {"line\nbreak"}) + ref, term = [loc for loc in iter_document_directives(document) if loc.fragment is None] + assert (ref.directive.name, ref.directive.positional, ref.directive.params) == ("ref", "line\nbreak", {}) + assert (term.directive.name, term.directive.params) == ("term", {"label": "la\nbel"}) + for loc in (ref, term): + directive = loc.directive + assert (directive.start, directive.end, directive.source) == (0, 0, "") + assert directive.malformed == "" and directive.duplicates == () and directive.param_spans == {} + assert directive.positional_span is None + # Without a label, a term holds no parameter. + blocks[-1] = Block(kind="term", target="a\nb") + assert list(iter_document_directives(document))[-1].directive.params == {} + + +def test_the_definition_the_parser_lifted_is_yielded_before_its_text(): + from legaldown import block_fragments + + document = parse('# A\n\n"Services" {{def: services}} means work, see {{ref: a}}.\n') + block = document.sections[0].blocks[0] + assert block.kind == "definition" + found = list(iter_document_directives(document)) + assert [(loc.directive.name, loc.fragment) for loc in found] == [("def", None), ("ref", 0)] + lifted = found[0].directive + assert lifted.positional == "services" and not lifted.malformed + assert lifted.source == "{{def: services}}" + assert render_block(dataclasses.replace(block, text="")) == f'"Services" {lifted.source}' + text = block_fragments(block)[0].text + assert text[found[1].directive.start:found[1].directive.end] == "{{ref: a}}" + # The term is not text, and the definition is no ref or term. + assert collect_source_directives(document) == ({"a"}, set()) + + +def test_a_definition_without_an_id_yields_its_bare_directive(): + document = parse('# A\n\n"Services" {{def:}} means work\n') + (loc,) = iter_document_directives(document) + assert (loc.directive.name, loc.directive.positional, loc.fragment) == ("def", None, None) + + +def test_a_definition_built_in_code_with_a_line_break_in_its_id_is_still_yielded(): + from legaldown import Block + + document = parse(_FRONTMATTER + "# A\n\nText.\n") + document.sections[0].blocks.append(Block(kind="definition", definition_id="a\nb", term="T", text="x {{ref: r}}")) + found = list(iter_document_directives(document))[-2:] + assert [(loc.directive.name, loc.directive.positional, loc.fragment) for loc in found] == [ + ("def", "a\nb", None), + ("ref", "r", 0), + ] + assert found[0].directive.source == "" + + # ── Over the specification fixtures ─────────────────────────────── _FIXTURES = Path(os.environ.get("LEGALDOWN_FIXTURES_DIR", "")) @@ -463,7 +586,7 @@ def test_every_fixture_document_reads_consistently(path): assert blank.id == pid met = [ptype for met_id, ptype in placeholders if met_id == pid] assert met, pid - assert blank.type in met or blank.type == "text" + assert blank.type is None or blank.type in met if blank.fixed: assert blank.type in ("money", "duration") assert set(result.index.blanks) <= {pid for pid, _ptype in placeholders} diff --git a/tests/test_tooling_surface.py b/tests/test_tooling_surface.py index ea407a8..51b3afe 100644 --- a/tests/test_tooling_surface.py +++ b/tests/test_tooling_surface.py @@ -13,8 +13,8 @@ from legaldown import load_answers, parse, validate from legaldown.directives import lex from legaldown.syntax import find_markers -from legaldown.validator.core import FINAL_CHECK_RULES from legaldown.validator.helpers import generate_identifier +from legaldown.validator.patterns import FINAL_CHECK_RULES from legaldown.validator.templates import check_choose, choose_problem GRAMMAR = [ @@ -38,11 +38,11 @@ "Marker", "MARKER_RE", "parse_marker", "format_marker", "is_look_alike", "FoundMarker", "find_markers", "is_include_only", "Fragment", "ListFragment", "block_fragments", "list_fragments", "text_fragments", - "Quote", "block_quotes", "quote_blocks", "drafting_note_blocks", "code_content", "CodeContent", + "Quote", "block_quotes", "is_drafting_note", "quote_blocks", "drafting_note_blocks", "code_content", "CodeContent", "list_items", "item_text", "FRONTMATTER_RE", "LINE_ENDING_RE", "HTML_COMMENT_RE", "FENCE_OPEN_RE", "closes_fence", "fence_end", "dedent", "indent_width", "strip_text", - "SourceLayout", "SectionSpan", "HeadingSpan", "BlockSpan", "ItemSpan", + "body_layout", "SourceLayout", "SectionSpan", "HeadingSpan", "BlockSpan", "ItemSpan", ] #: Where each public name is implemented. @@ -89,14 +89,14 @@ "condition_problem": "legaldown.validator.conditions", "exclusive": "legaldown.validator.conditions", "parse_condition": "legaldown.validator.conditions", - "FINAL_CHECK_RULES": "legaldown.validator.core", - "format_section_number": "legaldown.validator.helpers", + "format_section_number": "legaldown.validator.helpers", "is_positive_numeric": "legaldown.validator.helpers", "is_valid_iso_date": "legaldown.validator.helpers", "is_valid_money_amount": "legaldown.validator.helpers", "is_valid_numeric": "legaldown.validator.helpers", "slugify_identifier": "legaldown.validator.helpers", "DURATION_UNITS": "legaldown.validator.patterns", + "FINAL_CHECK_RULES": "legaldown.validator.patterns", "IDENTIFIER_RE": "legaldown.validator.patterns", "KNOWN_CURRENCIES": "legaldown.validator.patterns", "LEGALDOWN_EXTENSIONS": "legaldown.validator.patterns", @@ -112,6 +112,7 @@ "Quote": "legaldown.validator.templates", "VALUE_QUESTION_TYPES": "legaldown.validator.templates", "block_quotes": "legaldown.validator.templates", + "is_drafting_note": "legaldown.validator.templates", "choose_problem": "legaldown.validator.templates", "FoundMarker": "legaldown.validator.units", "find_markers": "legaldown.validator.units", @@ -122,6 +123,7 @@ "drafting_note_blocks": "legaldown.validator.templates", "CodeContent": "legaldown.markdown", "code_content": "legaldown.markdown", + "body_layout": "legaldown.positions", "SourceLayout": "legaldown.positions", "SectionSpan": "legaldown.positions", "HeadingSpan": "legaldown.positions", @@ -152,7 +154,7 @@ ("legaldown.parser", "FRONTMATTER_RE"): ("legaldown.syntax", "FRONTMATTER_RE"), ("legaldown.parser", "MAX_QUOTE_DEPTH"): ("legaldown.grammar", "MAX_QUOTE_DEPTH"), ("legaldown.parser", "quote_content"): ("legaldown.syntax", "quote_blocks"), - ("legaldown.parser", "_layout"): ("legaldown.syntax", "SourceLayout"), # via Document.layout() + ("legaldown.parser", "_layout"): ("legaldown.syntax", "body_layout"), # for bare body text ("legaldown.validator.patterns", "LEGALDOWN_EXTENSIONS"): ("legaldown.grammar", "LEGALDOWN_EXTENSIONS"), ("legaldown.validator.templates", "DECISION_QUESTION_TYPES"): ("legaldown.grammar", "DECISION_QUESTION_TYPES"), ("legaldown.validator.templates", "QUESTION_TYPES"): ("legaldown.grammar", "QUESTION_TYPES"), @@ -225,6 +227,63 @@ def test_the_source_layout_is_reached_from_the_document(): assert load_answers is not None +def _spans(layout, shift=0): + """Every span of *layout* as ``(what, start, end)``, moved up by *shift* lines.""" + + def block(span): + yield span.kind, span.start - shift, span.end - shift + for item in span.items: + yield "item", item.start - shift, item.end - shift + for inner in item.blocks: + yield from block(inner) + + found = [each for span in layout.preamble for each in block(span)] + for section in layout.sections: + heading = section.heading + found.append(("heading", heading.start - shift, heading.end - shift)) + found.append(("marker", heading.marker_line - shift, 0)) + found.extend(each for span in section.blocks for each in block(span)) + return found + + +def test_body_layout_reads_text_as_a_body_with_no_frontmatter(): + layout = syntax.body_layout("---\ntitle: x\n---\nClause\n") + assert layout.frontmatter is None + # What the body parser reads: a thematic break, then a heading ("title: x" underlined) and a paragraph. + assert [(span.kind, span.start, span.end) for span in layout.preamble] == [("rule", 1, 2)] + (section,) = layout.sections + assert (section.heading.start, section.heading.end) == (2, 4) + assert [(span.kind, span.start, span.end) for span in section.blocks] == [("paragraph", 4, 5)] + + +def test_body_layout_agrees_with_the_layout_of_a_document_shifted_by_its_body_start(): + body = "Pre.\n\n# A\n\nText\nmore\n\n1. one\n - x\n - y\n2. two\n\n> q\n\n## B\n\n| h |\n|---|\n| c |\n\nEnd.\n" + source = "---\ntitle: T\nparties: []\n---\n\n" + body + document = parse(source) + layout = document.layout() + start = document.line_of(None, 0) - 1 + assert start == 5 + assert layout.frontmatter == (1, 5) + assert _spans(syntax.body_layout(body)) == _spans(layout, start) + assert syntax.body_layout(body).frontmatter is None + + +def test_body_layout_counts_lines_from_the_start_of_the_text_and_reads_every_line_ending(): + assert syntax.body_layout("A\r\n\r\n# H\r\n\rText\r") == syntax.body_layout("A\n\n# H\n\nText\n") + layout = syntax.body_layout("A\n\n# H\n\nText") + assert (layout.preamble[0].start, layout.sections[0].blocks[0].start) == (1, 5) + assert syntax.body_layout("") == syntax.SourceLayout(None, (), ()) + + +def test_is_drafting_note_is_the_validators_and_reads_a_quote_block(): + from legaldown import Block + + assert syntax.is_drafting_note is legaldown.is_drafting_note + assert syntax.is_drafting_note(Block(kind="quote", text="[!drafting]\nAdvise.")) + assert not syntax.is_drafting_note(Block(kind="quote", text="Advise.")) + assert not syntax.is_drafting_note(Block(kind="paragraph", text="[!DRAFTING]")) + + # ── The constants the validator now reads ───────────────────────── @@ -372,6 +431,19 @@ def test_choose_problem_without_questions(): assert "declared boolean or choice question" in (choose_problem(directive, None) or "") +@pytest.mark.parametrize( + "source", + ["{{ref: vat}}", "{{placeholder: vat}}", '{{choose: vat, true="oops}}', "{{choose: vat, true=a, false=b"], +) +def test_choose_problem_refuses_what_is_no_well_formed_choose(source): + [directive] = lex(source).directives[:1] + with pytest.raises(ValueError, match="choose"): + choose_problem(directive, _QUESTIONS) + # The validator never reports one as choose-invalid. + result = validate(parse(f"---\ntitle: T\nquestions:\n vat:\n type: boolean\n---\n\n# A\n\n{source}\n")) + assert "choose-invalid" not in result.rules() + + # ── find_markers ──────────────────────────────────────────────────