diff --git a/epythet/agentic_readme.py b/epythet/agentic_readme.py index 49f7929..530eb12 100644 --- a/epythet/agentic_readme.py +++ b/epythet/agentic_readme.py @@ -69,8 +69,11 @@ ) #: The kinds a check reports on, in display order. KINDS = ("skills", "subagents", "instruction_files", "agent_docs", "section") -#: The snippet names the section is rendered from. +#: The snippet names the section is rendered from. A project whose only agentic +#: aspect is its agent-readable documentation gets the shorter docs-only variant: +#: it ships no tooling, so the section must not say it does. SECTION_SNIPPET = "agentic-readme-section" +DOCS_ONLY_SECTION_SNIPPET = "agentic-readme-section-docs-only" HUMOR_SNIPPET = "agentic-readme-humor" INSTRUCTION_SNIPPET = "agentic-readme-instruction" #: The opener used when ``humor`` is off. @@ -599,7 +602,8 @@ def render_section( name = config.name if config is not None else artifacts.project_dir.name repo_stub = repo_stub_for(config.repo_url) if config is not None else "" site_url = site_url_for(config.repo_url) if config is not None else "" - template = snippets(SECTION_SNIPPET) + snippet_name = section_snippet_for(artifacts) + template = snippets(snippet_name) fields = dict( marker_start=MARKER_START, marker_end=MARKER_END, @@ -618,18 +622,39 @@ def render_section( rendered = template.format(**fields) except Exception as e: # str.format raises Key/Index/Value/Attribute/TypeError raise SectionError( - f"snippet {SECTION_SNIPPET!r} does not format: {e!r}; the fields are " + f"snippet {snippet_name!r} does not format: {e!r}; the fields are " f"{sorted(SECTION_FIELDS)} and literal braces must be doubled ({{{{ and }}}})" ) from e rendered = rendered.strip("\n") + "\n" if marker_span(rendered, strict=False) != (0, len(rendered)): raise SectionError( - f"snippet {SECTION_SNIPPET!r} must start with {{marker_start}} and end with " + f"snippet {snippet_name!r} must start with {{marker_start}} and end with " "{marker_end}, each on its own line, or the section cannot be updated in place" ) return rendered +def ships_tooling(artifacts: AIArtifacts) -> bool: + """Whether the project ships anything an agent installs or reads as instructions. + + Skills, subagents and instruction files count; published agent-readable + documentation (``llms.txt``, ``.md``) does not, because it is a + view of the docs rather than tooling. + """ + return bool( + artifacts.skills or artifacts.subagents or artifacts.instruction_files + ) + + +def section_snippet_for(artifacts: AIArtifacts) -> str: + """The name of the section snippet ``artifacts`` calls for. + + :data:`SECTION_SNIPPET` when the project ships tooling, else the shorter + :data:`DOCS_ONLY_SECTION_SNIPPET`, which makes no "ships tooling" claim. + """ + return SECTION_SNIPPET if ships_tooling(artifacts) else DOCS_ONLY_SECTION_SNIPPET + + def _for_humans_intro(name: str, policy: ReadmePolicy, pool_text: str) -> str: """The opener: a stable pick from the humour pool when ``humor`` is on, else neutral.""" lines = pool_lines(pool_text) if policy.humor else [] diff --git a/epythet/cli.py b/epythet/cli.py index 5db6853..b70f29e 100644 --- a/epythet/cli.py +++ b/epythet/cli.py @@ -180,6 +180,10 @@ def _resolve_repo_stub(repo): #: still become ``nargs="*"`` (``--ignore a b``), not a single value. CONVENTION = dataclasses.replace(cw.ARGH, resolve_hints=True) +#: ``--ignore a --ignore b`` accumulates on every command that takes it (argparse +#: would keep only the last flag); ``validate`` and ``repair`` declare the same. +quickstart._cw = {"params": {"ignore": {"action": "extend", "nargs": "*"}}} + def mk_epythet_parser(**parser_kwargs): """The full ``epythet`` parser: the flat commands, the tool commands, the ``ledger`` group.""" diff --git a/epythet/data/skills/epythet-agentic-readme/SKILL.md b/epythet/data/skills/epythet-agentic-readme/SKILL.md index 42eeee0..edfdcee 100644 --- a/epythet/data/skills/epythet-agentic-readme/SKILL.md +++ b/epythet/data/skills/epythet-agentic-readme/SKILL.md @@ -76,11 +76,12 @@ agentic_first = true ## Changing the wording: snippets -Three snippets render the section. The user's copy in `/snippets/.md` wins over the packaged default in `epythet/data/snippets/`: +Four snippets render the section. The user's copy in `/snippets/.md` wins over the packaged default in `epythet/data/snippets/`: | Snippet | What it is | |---|---| -| `agentic-readme-section` | the section template (`str.format` fields: `{marker_start}`, `{marker_end}`, `{heading}`, `{name}`, `{repo_stub}`, `{site_url}`, `{skills_block}`, `{subagents_block}`, `{instructions_block}`, `{docs_block}`, `{for_humans_intro}`, `{humans_link}`; literal braces doubled) | +| `agentic-readme-section` | the section template for a project that ships skills, subagents or instruction files (`str.format` fields: `{marker_start}`, `{marker_end}`, `{heading}`, `{name}`, `{repo_stub}`, `{site_url}`, `{skills_block}`, `{subagents_block}`, `{instructions_block}`, `{docs_block}`, `{for_humans_intro}`, `{humans_link}`; literal braces doubled) | +| `agentic-readme-section-docs-only` | the shorter template used when the project's only agentic aspect is its published agent-readable documentation (`llms.txt`, `.md`); same fields, no "ships tooling" claim | | `agentic-readme-humor` | the pool of openers for the "for humans" sentence, one per line, `#` comments allowed | | `agentic-readme-instruction` | what step 2 tells the agent when the policy is `add` | diff --git a/epythet/data/skills/epythet-docstring-style/SKILL.md b/epythet/data/skills/epythet-docstring-style/SKILL.md index aa23b75..41a198f 100644 --- a/epythet/data/skills/epythet-docstring-style/SKILL.md +++ b/epythet/data/skills/epythet-docstring-style/SKILL.md @@ -40,7 +40,7 @@ What epythet's normalizer repairs at build time (so existing code renders, not s **Tier 3, complex class**: everything in Tier 2 on the class docstring (not `__init__`); `Attributes:` with invariants; a lifecycle sketch (construct, configure, use, tear down) as a doctest; state invariants (what mutates, what is safe to reuse); methods at Tier 1 or 2. Target: 40 to 80 lines on the class. -**Module docstring** (required on every non-underscore module): one line on purpose; two to four sentences of intent and how the module relates to the package; a curated `Main entry points:` block naming the 2 to 5 things to start with (a curation, not an inventory); one minimal doctest. +**Module docstring** (required on every non-underscore module): one line on purpose; two to four sentences of intent and how the module relates to the package; a curated `Main entry points:` line followed by a blank line and a bullet list naming the 2 to 5 things to start with (a curation, not an inventory); one minimal doctest. The blank line matters: `Main entry points:` directly over an indented block is an RST definition list, which `epythet validate` reports as DR014. ```python """ @@ -48,12 +48,13 @@ What epythet's normalizer repairs at build time (so existing code renders, not s <2-4 sentences: the problem it solves, the mental model, its place in the package.> Main entry points: - : - : - >>> from pkg.module import main_thing - >>> main_thing([1, 2, 3]) - 6 +- ````: +- ````: + +>>> from pkg.module import main_thing +>>> main_thing([1, 2, 3]) +6 """ ``` diff --git a/epythet/data/snippets/agentic-readme-instruction.md b/epythet/data/snippets/agentic-readme-instruction.md index 47b71ba..595af4c 100644 --- a/epythet/data/snippets/agentic-readme-instruction.md +++ b/epythet/data/snippets/agentic-readme-instruction.md @@ -1,5 +1,5 @@ Add the agentic aspects to this README. With `agentic_first` on, agents get their section before the humans get theirs; otherwise it closes the README. -Do not hand-write the section. Run `epythet ai-readme-check --write` so it lands between epythet's marker comments and a later run updates it in place. To change the wording, change the snippets, not the README: `epythet snippets init`, then edit `agentic-readme-section.md` and `agentic-readme-humor.md` in your snippets folder. +Do not hand-write the section. Run `epythet ai-readme-check --write` so it lands between epythet's marker comments and a later run updates it in place. To change the wording, change the snippets, not the README: `epythet snippets init`, then edit `agentic-readme-section.md` (a project that ships skills, subagents or instruction files), `agentic-readme-section-docs-only.md` (a project whose only agentic aspect is its agent-readable documentation) and `agentic-readme-humor.md` in your snippets folder. Then read the result as a stranger would. The humour is light and stays on the writer's side of the joke. No em-dashes, no "not X but Y", no closing paragraph that restates the section. Every link resolves. If the README already covers the same artifacts under a heading of its own, keep that heading only if it says something the generated section does not; otherwise remove it, so the two never disagree. diff --git a/epythet/data/snippets/agentic-readme-section-docs-only.md b/epythet/data/snippets/agentic-readme-section-docs-only.md new file mode 100644 index 0000000..fc50601 --- /dev/null +++ b/epythet/data/snippets/agentic-readme-section-docs-only.md @@ -0,0 +1,7 @@ +{marker_start} +{heading} For AI agents + +`{name}` publishes its documentation in forms made for coding agents. If you are one, start here. +{docs_block} +{for_humans_intro}, the rest of this README is written for you, starting at {humans_link}. +{marker_end} diff --git a/epythet/ledger/rules/rendering/DR014.py b/epythet/ledger/rules/rendering/DR014.py index 0293bc3..0f00037 100644 --- a/epythet/ledger/rules/rendering/DR014.py +++ b/epythet/ledger/rules/rendering/DR014.py @@ -23,3 +23,22 @@ def good_real_definition_list(): # ok: DR014 term definition of the term """ + + +def bad_entry_points_over_indented_block(): # ruleid: DR014 + """Do a thing. + + Main entry points: + thing: the one to start with + other: the second one + """ + + +def good_entry_points_blank_line_then_list(): # ok: DR014 + """Do a thing. + + Main entry points: + + - ``thing``: the one to start with + - ``other``: the second one + """ diff --git a/epythet/normalizer.py b/epythet/normalizer.py index fe9a4ce..e0f2bb0 100644 --- a/epythet/normalizer.py +++ b/epythet/normalizer.py @@ -11,8 +11,24 @@ event, so the rendered site is right without editing any source. Each rule is a pure function ``list[str] -> list[str]`` and :data:`DEFAULT_RULES` is the ordered tuple that runs by default. The rules only touch prose: lines inside -doctest blocks, literal blocks and directive bodies are left byte-for-byte -alone, because doctests are executed and code is code. +doctest blocks, literal blocks, directive bodies and ASCII-art drawings are +left byte-for-byte alone, because doctests are executed and code is code. + +**The principle: rewrite only what is unambiguous; otherwise report.** A rule +fires when the line can mean one thing (a ``>>>`` glued to prose is a doctest; +a ```` ``` ```` fence is a fence) and stays out when the author's intent has two +readings. So a ``#`` line becomes a rubric only when it is shaped like a +Markdown heading and stands alone between blank lines, never when it could be +a code comment; ``text:`` followed by an indented block becomes a literal +block only when the block reads as code, never when it reads as a paragraph or +a definition; a section one-liner folds in the prose that wraps it, never a +field list that follows it; and nothing at all is rewritten inside the body of +a Google section (an argument called ``x:`` is not a literal block marker, an +argument called ``error:`` is not a section) or inside a drawing. What the +rules leave alone, ``epythet validate`` reports (DR014 for the accidental +definition list, DR002 for the bare section header, DR031 for the commented +doctest), so nothing is silently dropped: the same fixture that pins a rule's +silence pins the finding that replaces it. The rules, in order: @@ -82,12 +98,22 @@ _SECTION_ONE_LINER_RE = re.compile(r"^(\s*)([A-Z][A-Za-z ]+):\s+(\S.*)$") _LITERAL_SPAN_RE = re.compile(r"``.+?``|`[^`]+`") _STAR_WORD_RE = re.compile(r"(?|<-{2,}|={2,}>|<={2,}|\+-{2,}|-{2,}\+|^\s*[|+]\s*$|\|\s{2,}\|") + +BLANK, PROSE, DOCTEST, LITERAL, LIST, FIELD, FENCE, ART = ( + "blank", "prose", "doctest", "literal", "list", "field", "fence", "art", ) # fmt: skip -_CODE_CONTEXTS = frozenset({DOCTEST, LITERAL, FENCE}) +_CODE_CONTEXTS = frozenset({DOCTEST, LITERAL, FENCE, ART}) _PROSE_LIKE = frozenset({PROSE, LIST, FIELD}) +#: How many plain words make a line read as prose rather than code. +PROSE_WORDS = 4 +#: How many lines of a run must be drawing lines for the run to be a drawing. +ART_MIN_LINES = 2 def indent_of(line: str) -> int: @@ -100,17 +126,158 @@ def indent_of(line: str) -> int: def line_contexts(lines: Sequence[str]) -> list[str]: - """Classify every line as blank, prose, doctest, literal, list, field or fence. + """Classify every line as blank, prose, doctest, literal, list, field, fence or art. The classification is what keeps every rule away from code: a line inside a - doctest block, a ``::`` literal block, a directive body or a Markdown fence - is never rewritten. + doctest block, a ``::`` literal block, a directive body, a Markdown fence or + an ASCII-art drawing is never rewritten. >>> line_contexts(["Text:", "", " >>> 1", " 1", "", "- a", " b", "", ":param x: y"]) ['prose', 'blank', 'doctest', 'doctest', 'blank', 'list', 'list', 'blank', 'field'] >>> line_contexts([" >>> 1", " 1", "back to prose"]) ['doctest', 'doctest', 'prose'] + >>> line_contexts(["a --> b", " |", " v", "- c"]) + ['art', 'art', 'art', 'list'] + """ + contexts = _line_contexts(lines) + for start, end in _art_runs(lines): + contexts[start:end] = [ART] * (end - start) + return contexts + + +def is_art_line(line: str) -> bool: + """Whether ``line`` is a piece of a drawing: box characters, arrows, or mostly strokes. + + >>> is_art_line("│ 0 │ ──▶ │ 2 │"), is_art_line(" +----+"), is_art_line("- a bullet") + (True, True, False) + >>> is_art_line("func1 --> merge"), is_art_line("x = 1 # comment") + (True, False) + """ + if _DOCTEST_RE.match(line): + return False # a bare ``>>>`` prompt is all strokes, and never a drawing + return _is_stroke_line(line) or bool(_ASCII_ART_RE.search(line)) + + +def _is_stroke_line(line: str) -> bool: + """A line drawn with box characters or made at least half of strokes: art beyond doubt.""" + stripped = line.strip() + if not stripped: + return False + if _BOX_CHAR_RE.search(stripped): + return True + strokes = sum(ch in "-+|/\\_<>^v" for ch in stripped) + return len(stripped) >= 3 and strokes / len(stripped) >= 0.5 + + +def _art_runs(lines: Sequence[str]) -> list[tuple[int, int]]: + """``(start, end)`` of every drawing: a run of non-blank lines that is mostly art. + + Short label lines inside the run (``a``, ``v``, ``merge``) belong to the + drawing; a run with fewer than :data:`ART_MIN_LINES` art lines is prose. A + doctest, bullet or field line ends a run (``- a --> b`` is a list item that + happens to hold an arrow), unless it is itself drawn in strokes. + """ + runs: list[tuple[int, int]] = [] + i = 0 + while i < len(lines): + if not lines[i].strip() or _breaks_art(lines[i]): + i += 1 + continue + j = i + while j < len(lines) and lines[j].strip() and not _breaks_art(lines[j]): + j += 1 + art = [is_art_line(line) for line in lines[i:j]] + labels = [_is_label(line) for line in lines[i:j]] + if sum(art) >= ART_MIN_LINES and all(a or l for a, l in zip(art, labels)): + runs.append((i, j)) + i = j + return runs + + +def _breaks_art(line: str) -> bool: + if _DOCTEST_RE.match(line): + return True + return bool( + (_BULLET_RE.match(line) or _FIELD_RE.match(line)) and not _is_stroke_line(line) + ) + + +def _is_label(line: str) -> bool: + """A short caption inside a drawing: up to three tokens, no sentence punctuation.""" + tokens = line.split() + return 0 < len(tokens) <= 3 and not line.rstrip().endswith((".", "!", "?", ":")) + + +def google_section_bodies(lines: Sequence[str]) -> list[int | None]: + """For every line, the indentation of the Google section body it is in, else ``None``. + + A section is a known header (``Args:``, ``Returns:``, ...) on its own line + with an indented body below; the body ends at the first non-blank line + that is not deeper than the header. Rules use this to stay out of section + bodies, where ``x:`` is an argument and not a literal-block lead-in. + + >>> google_section_bodies(["Args:", " x: the x", " more", "", "Text."]) + [None, 4, 4, 4, None] """ + bodies: list[int | None] = [None] * len(lines) + i = 0 + while i < len(lines): + if not _is_section_header(lines[i]): + i += 1 + continue + header_indent = indent_of(lines[i]) + first = _next_nonblank(lines, i + 1) + if first is None or indent_of(lines[first]) <= header_indent: + i += 1 + continue + body_indent = indent_of(lines[first]) + j = i + 1 + while j < len(lines) and ( + not lines[j].strip() or indent_of(lines[j]) > header_indent + ): + bodies[j] = body_indent + j += 1 + i = j + return bodies + + +def _looks_like_prose(line: str) -> bool: + """Whether a line reads as a sentence rather than a command or an expression. + + At least :data:`PROSE_WORDS` plain words and a sentence's punctuation at the + end (a wrapped line ends in a comma as often as a period), or half again + as many plain words with no punctuation at all. + + >>> _looks_like_prose("which will just return the (args,"), _looks_like_prose("x = f(a, b)") + (True, False) + >>> _looks_like_prose("pip install foo bar"), _looks_like_prose("the name of the thing to do") + (False, True) + """ + words = sum(bool(_PLAIN_WORD_RE.match(t)) for t in line.split()) + punctuated = line.rstrip().endswith((".", ",", ";", ":", "!", "?")) + return words >= PROSE_WORDS and (punctuated or words >= PROSE_WORDS * 3 // 2) + + +_CODE_SIGNAL_RE = re.compile(r"[=(){}\[\]<>$@#|/\\*]|::|--|\.py\b") + + +def _looks_like_code(block: Sequence[str]) -> bool: + """Whether an indented block is unmistakably code: no prose line, and a code signal somewhere. + + >>> _looks_like_code(["python run.py # top 12"]), _looks_like_code(["the thing to do."]) + (True, False) + >>> _looks_like_code(["fast"]), _looks_like_code(["{'a': 1}"]) + (False, True) + """ + nonblank = [line for line in block if line.strip()] + return ( + bool(nonblank) + and not any(_looks_like_prose(line) for line in nonblank) + and any(_CODE_SIGNAL_RE.search(line) for line in nonblank) + ) + + +def _line_contexts(lines: Sequence[str]) -> list[str]: contexts: list[str] = [] block: str | None = None # doctest / list / field, reset on a blank line block_indent = 0 # indentation of the >>> that opened a doctest block @@ -181,6 +348,29 @@ def _is_section_header(line: str) -> bool: return bool(m) and m.group(2).lower() in GOOGLE_SECTIONS +_HEADING_CODE_RE = re.compile(r"[=(){}\[\]]|>>>|\|") + + +def _is_heading_title(title: str) -> bool: + """Whether ``title`` reads as a Markdown heading rather than a comment. + + >>> _is_heading_title("Making a signature"), _is_heading_title("x=1, y=(2, 3)") + (True, False) + >>> _is_heading_title("Can infer types:"), _is_heading_title("TODO: fix"), _is_heading_title("true") + (False, False, False) + >>> _is_heading_title("3 ways to do it"), _is_heading_title("`Sig` basics") + (True, True) + """ + title = title.strip() + return ( + bool(title) + and (title[0].isupper() or title[0].isdigit() or title[0] == "`") + and not title.endswith((":", ".", ",", ";")) + and not _HEADING_CODE_RE.search(title) + and not re.match(r"[A-Z]{2,}:", title) # TODO: / FIXME: / NOTE: tags + ) + + def _next_nonblank(lines: Sequence[str], start: int) -> int | None: for j in range(start, len(lines)): if lines[j].strip(): @@ -255,21 +445,29 @@ def fix_short_underlines(lines: list[str]) -> list[str]: def google_one_liners(lines: list[str]) -> list[str]: """Expand ``Returns: text`` (and other one-line sections) into real sections. - Continuation lines at the same indentation are folded into the section body; - a Markdown heading or another section ends it. + Prose lines that wrap the sentence at the same indentation are folded into + the section body; a field list, a bullet list, a Markdown heading or another + section ends it. Inside the body of another section the line is an entry + (an argument called ``error``), not a header, and is left alone. >>> normalize_text("Returns: a thing that\\nspans two lines.\\n\\nNext.", rules=[google_one_liners]) 'Returns:\\n a thing that\\n spans two lines.\\n\\nNext.' >>> normalize_text("Returns: a thing.\\n## Notes\\nText.", rules=[google_one_liners]) 'Returns:\\n a thing.\\n\\n## Notes\\nText.' + >>> normalize_text("Note: be careful.\\n:return: the thing", rules=[google_one_liners]) + 'Note:\\n be careful.\\n\\n:return: the thing' """ contexts = line_contexts(lines) + bodies = google_section_bodies(lines) out: list[str] = [] i = 0 while i < len(lines): m = _SECTION_ONE_LINER_RE.match(lines[i]) if not ( - m and contexts[i] in _PROSE_LIKE and m.group(2).lower() in GOOGLE_SECTIONS + m + and contexts[i] in _PROSE_LIKE + and bodies[i] is None + and m.group(2).lower() in GOOGLE_SECTIONS ): out.append(lines[i]) i += 1 @@ -281,7 +479,9 @@ def google_one_liners(lines: list[str]) -> list[str]: j = i + 1 while ( j < len(lines) - and contexts[j] in _PROSE_LIKE + and contexts[j] in _PROSE_LIKE # a wrapped sentence, wherever it sits + and not _BULLET_RE.match(lines[j]) # ... but a new bullet ends it + and not _FIELD_RE.match(lines[j]) # ... and so does a field and indent_of(lines[j]) == len(indent) and not _SECTION_ONE_LINER_RE.match(lines[j]) and not _is_section_header(lines[j]) @@ -323,22 +523,43 @@ def bare_headers_to_rubrics(lines: list[str]) -> list[str]: def markdown_headings_to_rubrics(lines: list[str]) -> list[str]: """Render ``## Heading`` as a rubric instead of a literal ``##``. - A ``#`` line right after code is left alone: it is most likely a comment - that fell out of a doctest. + A ``#`` line is also how a code comment, a commented-out doctest and its + output (``# True``) or a commented-out paragraph look, so the rule wants a + heading shape: ``#`` marks, a space, then a title that starts with a + capital letter, a digit or a backtick, holds no code (``=``, ``(``, + ``>>>``) and no trailing ``:`` or ``.``, and is not a ``TODO:`` tag. A + single ``#`` must also follow a blank line (or open the docstring); + ``##`` and deeper may sit against prose. No heading of any level sits + against another ``#`` line (that is a commented-out paragraph or doctest) + or right after code. >>> normalize_text("Intro.\\n## Usage\\nText.", rules=[markdown_headings_to_rubrics]) 'Intro.\\n\\n.. rubric:: Usage\\n\\nText.' + >>> normalize_text("# >>> f()\\n# True", rules=[markdown_headings_to_rubrics]) + '# >>> f()\\n# True' + >>> normalize_text("# Making a signature\\nText.", rules=[markdown_headings_to_rubrics]) + '.. rubric:: Making a signature\\n\\nText.' + >>> normalize_text("Intro.\\n# Not a heading\\n\\nText.", rules=[markdown_headings_to_rubrics]) + 'Intro.\\n# Not a heading\\n\\nText.' """ contexts = line_contexts(lines) out: list[str] = [] for i, line in enumerate(lines): m = _MD_HEADING_RE.match(line) - if not m or contexts[i] != PROSE or not m.group(2)[0].isalnum(): + if not (m and contexts[i] == PROSE and _is_heading_title(m.group(2))): out.append(line) continue if i > 0 and contexts[i - 1] in _CODE_CONTEXTS: out.append(line) continue + neighbours = [lines[k] for k in (i - 1, i + 1) if 0 <= k < len(lines)] + if any(n.lstrip().startswith("#") for n in neighbours): + out.append(line) # one of a run of comment lines + continue + single = line.lstrip().startswith("# ") + if single and i > 0 and lines[i - 1].strip(): + out.append(line) # a lone ``#`` glued to the text above is a comment + continue _ensure_trailing_blank(out) out.append(f"{m.group(1)}.. rubric:: {m.group(2)}") if i + 1 < len(lines) and lines[i + 1].strip(): @@ -349,10 +570,24 @@ def markdown_headings_to_rubrics(lines: list[str]) -> list[str]: def literal_block_after_colon(lines: list[str]) -> list[str]: """Make ``text:`` followed by an indented block a proper ``::`` literal block. + Only when the block is unmistakably code (:func:`_looks_like_code`: no + line reads as prose and some line carries a code signal such as ``=``, + ``(`` or ``#``), and never inside a Google section body, where ``x:`` is an + argument. A lead-in over an indented paragraph is a definition list the + author may have meant; it is left alone and DR014 reports it. A lone + ``Usage:`` or ``Output:`` over a command or a value is code all the same. + >>> normalize_text("For example:\\n x = f(1)\\nThen more.", rules=[literal_block_after_colon]) 'For example::\\n\\n x = f(1)\\n\\nThen more.' + >>> normalize_text("Usage:\\n python run.py # top 12", rules=[literal_block_after_colon]) + 'Usage::\\n\\n python run.py # top 12' + >>> normalize_text("specifying:\\n the name of the thing to do.", rules=[literal_block_after_colon]) + 'specifying:\\n the name of the thing to do.' + >>> normalize_text("Args:\\n x:\\n The x.", rules=[literal_block_after_colon]) + 'Args:\\n x:\\n The x.' """ contexts = line_contexts(lines) + bodies = google_section_bodies(lines) out: list[str] = [] i = 0 while i < len(lines): @@ -362,6 +597,7 @@ def literal_block_after_colon(lines: list[str]) -> list[str]: j = i + 1 eligible = ( contexts[i] == PROSE + and bodies[i] is None and stripped.endswith(":") and not stripped.endswith("::") and not _is_section_header(line) @@ -369,23 +605,27 @@ def literal_block_after_colon(lines: list[str]) -> list[str]: and j < len(lines) and lines[j].strip() and indent_of(lines[j]) > indent_of(line) - and contexts[j] not in (DOCTEST, LIST, FIELD) + and contexts[j] not in (DOCTEST, LIST, FIELD, ART) and not _BULLET_RE.match(lines[j]) ) if not eligible: i += 1 continue - out[-1] = stripped + ":" - out.append("") block_indent = indent_of(line) - while j < len(lines) and ( - not lines[j].strip() or indent_of(lines[j]) > block_indent + end = j + while end < len(lines) and ( + not lines[end].strip() or indent_of(lines[end]) > block_indent ): - out.append(lines[j]) - j += 1 - if j < len(lines) and out[-1].strip(): + end += 1 + if not _looks_like_code(lines[j:end]): + i += 1 + continue # an indented paragraph or a definition, not code: report, do not rewrite + out[-1] = stripped + ":" + out.append("") + out.extend(lines[j:end]) + if end < len(lines) and out[-1].strip(): out.append("") - i = j + i = end return out @@ -402,23 +642,40 @@ def blank_lines_between_blocks(lines: list[str]) -> list[str]: 'Prose\\n\\n:param x: y\\n more\\n:param z: w' >>> normalize_text("Text:\\n indented\\nback", rules=[blank_lines_between_blocks]) 'Text:\\n indented\\n\\nback' + + Inside a Google section body the entries are a definition list, where + consecutive terms need no blank line between them, so a dedent from a + wrapped ``Args:`` entry to the next entry is left as written: + + >>> normalize_text("Args:\\n a: one that\\n wraps.\\n b: two.", rules=[blank_lines_between_blocks]) + 'Args:\\n a: one that\\n wraps.\\n b: two.' """ contexts = line_contexts(lines) + bodies = google_section_bodies(lines) out: list[str] = [] for i, line in enumerate(lines): if i > 0 and lines[i - 1].strip() and line.strip(): prev_ctx, ctx = contexts[i - 1], contexts[i] + # A blank line *before* a block that follows a drawing is outside the + # drawing, and a doctest glued to one still has to be separated to run. starts_block = ctx in (DOCTEST, LIST, FIELD) and prev_ctx not in ( ctx, DOCTEST, FENCE, LITERAL, ) # fmt: skip next_is_deeper = i + 1 < len(lines) and indent_of(lines[i + 1]) > indent_of( line ) + next_entry = ( + bodies[i] is not None + and indent_of(line) == bodies[i] + and prev_ctx == PROSE + and ctx == PROSE + ) dedents = ( indent_of(line) < indent_of(lines[i - 1]) and ctx not in _CODE_CONTEXTS and not (ctx == prev_ctx and ctx in (LIST, FIELD)) and not next_is_deeper # a new definition-list term, not a return to prose + and not next_entry # the next ``Args:`` entry after a wrapped one ) if starts_block or dedents: out.append("") diff --git a/epythet/repair.py b/epythet/repair.py index 1c81088..aa1842f 100644 --- a/epythet/repair.py +++ b/epythet/repair.py @@ -18,7 +18,15 @@ prose ``*args`` is *not* source-safe (it changes what the author wrote, and a later reader may not know why the backslash is there), so it stays a diagnostic (DR010), like unmatched backticks and every other artifact the - normalizer cannot fix. + normalizer cannot fix. Nor is turning a bare ``Examples:`` header into a + rubric: napoleon renders it as that rubric already, so the rewrite would + churn the source for no change on the page (:data:`UNSAFE_RULES` lists both + with the reason). +- The normalizer's own rule applies twice over here: **rewrite only what is + unambiguous, otherwise report.** A ``#`` line that could be a comment, a + ``term:`` over an indented paragraph, an entry inside an ``Args:`` body, a + drawing made of arrows: none is touched, and the author's blank lines + before the closing quotes are kept as written. - Every doctest keeps its source lines byte for byte (checked with :mod:`doctest`'s own parser); a rewrite that would change one is skipped. - Every rewritten docstring is re-validated at level 0.5: a rewrite that @@ -63,14 +71,15 @@ from epythet.validation.docstrings import Docstring, iter_python_files from epythet.validation.model import Finding -#: The normalizer rules whose rewrite is safe to commit to source, in normalizer order. -SOURCE_SAFE_RULES: tuple[N.Rule, ...] = tuple( - rule for rule in N.DEFAULT_RULES if rule is not N.escape_unmatched_stars -) #: Normalizer rules that stay build-time only, and why. UNSAFE_RULES: dict[str, str] = { N.escape_unmatched_stars.__name__: "escaping *args in prose changes what the author wrote; reported as DR010 instead", + N.bare_headers_to_rubrics.__name__: "napoleon already renders a bare Examples: header as that rubric, so the rewrite changes the source without changing the page; a bare Note: is ambiguous and reported as DR002 instead", } +#: The normalizer rules whose rewrite is safe to commit to source, in normalizer order. +SOURCE_SAFE_RULES: tuple[N.Rule, ...] = tuple( + rule for rule in N.DEFAULT_RULES if rule.__name__ not in UNSAFE_RULES +) FENCE_STYLES = ("code-block", "literal") _PREFIX_CHARS = "rRbBuUfF" _DOCTEST_FAILURES_RE = re.compile(r"\*\*\*Test Failed\*\*\* (\d+) failure") @@ -188,6 +197,20 @@ def _body_indented(lines: Sequence[str]) -> bool: return bool(rest) and all(N.indent_of(line) > 0 for line in rest) +def _trailing_blank_count(lines: Sequence[str]) -> int: + """How many blank lines end ``lines``. + + >>> _trailing_blank_count(["a", "", ""]), _trailing_blank_count(["a"]) + (2, 0) + """ + count = 0 + for line in reversed(lines): + if line.strip(): + break + count += 1 + return count + + def _doctest_sources(text: str) -> list[str] | None: """Doctest example sources, or ``None`` when :mod:`doctest` cannot parse the text.""" try: @@ -229,8 +252,12 @@ def rewrite_docstring_literal( if before_sources is None: return segment, "doctest could not parse the docstring; fix the doctest first" normalized = N.normalize_docstring(dedented, rules=rules) - while normalized and not normalized[-1].strip() and trailing is not None: - normalized.pop() + if trailing is not None: + # A rule may leave a blank line at the end; the author's own blank lines + # before the closing quotes are kept exactly as they were. + blank_tail = _trailing_blank_count(dedented) + while _trailing_blank_count(normalized) > blank_tail and len(normalized) > 1: + normalized.pop() if normalized == dedented: return segment, None if len(quote) == 1 and (len(normalized) > 1 or "\n" in normalized[0]): @@ -771,7 +798,7 @@ def repair_command( :param path: A .py file, a package directory, or a project root. :param write: Apply the changes (after re-validating each docstring and re-running doctests). :param fence_style: What a Markdown fence becomes: code-block or literal. - :param ignore: Skip files whose path contains this string (repeat -i for several). + :param ignore: Skip files whose path contains any of these strings (several after one -i, or -i repeated). :param ledger: Directory of extra rule YAML files overlaid on the bundled ledger. :param no_napoleon: Re-validate without napoleon's Google/NumPy pre-processing. :param no_doctests: Do not run each touched file's doctests before and after writing. @@ -809,6 +836,10 @@ def repair_command( raise cw.CommandError("a written file had to be restored; see above", code=3) +#: ``-i a -i b`` accumulates, as for ``epythet validate``. +repair_command._cw = {"params": {"ignore": {"action": "extend", "nargs": "*"}}} + + def render_repair(report: RepairReport, *, diff: bool = True) -> str: """The human report: the diff (dry run) or what was written, then the refusals.""" lines: list[str] = [] diff --git a/epythet/validation/cli.py b/epythet/validation/cli.py index 6d5cae7..1085b4e 100644 --- a/epythet/validation/cli.py +++ b/epythet/validation/cli.py @@ -58,7 +58,7 @@ def validate( :param ledger: Directory of extra rule YAML files overlaid on the bundled ledger. :param style: Docstring convention for the linters: google, numpy, or sphinx. :param no_napoleon: Parse docstrings without napoleon's Google/NumPy pre-processing. - :param ignore: Skip files whose path contains this string (repeat -i for several). + :param ignore: Skip files whose path contains any of these strings (several after one -i, or -i repeated). :param docsrc: Sphinx source directory for level 2 (default: /docsrc). :param no_observe: Do not append findings to the ledger's observations file. :param no_linters: Level 0 without ruff and pydoclint (coverage detectors only). @@ -140,4 +140,8 @@ def validate( ) +#: ``-i a -i b`` accumulates (argparse would keep only the last ``-i``); cw reads +#: this attribute as per-parameter ``add_argument`` particulars. +validate._cw = {"params": {"ignore": {"action": "extend", "nargs": "*"}}} + COMMANDS = [validate] diff --git a/epythet/validation/core.py b/epythet/validation/core.py index 64b4fe2..66ee1d5 100644 --- a/epythet/validation/core.py +++ b/epythet/validation/core.py @@ -261,7 +261,10 @@ def validate( with Timer(report.durations, "0"): if linters: found, notes = run_lint_level( - resolved.package_dir, project_dir=resolved.project_dir, style=style + resolved.package_dir, + project_dir=resolved.project_dir, + style=style, + ignore=ignore, ) findings += found report.notes += notes diff --git a/epythet/validation/docstrings.py b/epythet/validation/docstrings.py index a477215..0634936 100644 --- a/epythet/validation/docstrings.py +++ b/epythet/validation/docstrings.py @@ -163,6 +163,19 @@ def _in_package(path: Path, package_dir: Path) -> bool: return True +def is_ignored(path: Path | str, ignore: Iterable[str]) -> bool: + """Whether ``path`` matches the ``--ignore`` list: any token is a substring of its POSIX form. + + The one predicate every level uses, so a file the parse level skips is + also absent from the lint, coverage and repair results. + + >>> is_ignored("/p/pkg/tests/test_x.py", ["tests/"]), is_ignored("/p/pkg/x.py", ["tests/"]) + (True, False) + """ + posix = Path(path).as_posix() + return any(token in posix for token in ignore) + + def iter_python_files( package_dir: Path, *, ignore: Iterable[str] = () ) -> Iterator[Path]: @@ -170,10 +183,9 @@ def iter_python_files( ignore = tuple(ignore) package_dir = Path(package_dir) for path in sorted(package_dir.rglob("*.py")): - posix = path.as_posix() if "__pycache__" in path.parts or not _in_package(path, package_dir): continue - if any(token in posix for token in ignore): + if is_ignored(path, ignore): continue yield path diff --git a/epythet/validation/lint.py b/epythet/validation/lint.py index 49fc604..0ff4b03 100644 --- a/epythet/validation/lint.py +++ b/epythet/validation/lint.py @@ -18,7 +18,7 @@ import shutil import subprocess from pathlib import Path -from typing import Iterator +from typing import Iterable, Iterator from epythet.validation.model import Finding @@ -172,13 +172,28 @@ def run_pydoclint( def run_lint_level( - package_dir: Path, *, project_dir: Path, style: str = "google" + package_dir: Path, + *, + project_dir: Path, + style: str = "google", + ignore: Iterable[str] = (), ) -> tuple[list[Finding], list[str]]: - """Level 0: ruff D plus pydoclint, with notes for anything skipped.""" + """Level 0: ruff D plus pydoclint, with notes for anything skipped. + + ``ignore`` is the ``--ignore`` list every other level applies at file + discovery; the linters walk the package themselves, so their findings are + filtered by the same predicate (:func:`~epythet.validation.docstrings.is_ignored`) + on the file's full path. + """ + from epythet.validation.docstrings import is_ignored + + ignore = tuple(ignore) findings: list[Finding] = [] notes: list[str] = [] for runner in (run_ruff, run_pydoclint): found, noted = runner(package_dir, project_dir=project_dir, style=style) - findings.extend(found) + findings.extend( + f for f in found if not (f.file and is_ignored(project_dir / f.file, ignore)) + ) notes.extend(noted) return findings, notes diff --git a/tests/normalizer_fixtures/arrows_then_doctest.in b/tests/normalizer_fixtures/arrows_then_doctest.in new file mode 100644 index 0000000..c2887ec --- /dev/null +++ b/tests/normalizer_fixtures/arrows_then_doctest.in @@ -0,0 +1,4 @@ +Maps a --> b +and b --> c +>>> f() +1 diff --git a/tests/normalizer_fixtures/arrows_then_doctest.out b/tests/normalizer_fixtures/arrows_then_doctest.out new file mode 100644 index 0000000..2e189fb --- /dev/null +++ b/tests/normalizer_fixtures/arrows_then_doctest.out @@ -0,0 +1,5 @@ +Maps a --> b +and b --> c + +>>> f() +1 diff --git a/tests/normalizer_fixtures/ascii_art_untouched.in b/tests/normalizer_fixtures/ascii_art_untouched.in new file mode 100644 index 0000000..b0eae87 --- /dev/null +++ b/tests/normalizer_fixtures/ascii_art_untouched.in @@ -0,0 +1,19 @@ +The pipeline, as a drawing: + + ┌───┐ ┌───┐ + │ 0 │ ──▶ │ 2 │ + └───┘ └───┘ + │ ▲ + ▼ │ + ┌───┐ │ + │ 4 │ ──────┘ + └───┘ + +and in plain characters: + + func1 --+ + +--> merge + func2 --+ + - - - - - - - - + +Back to prose. diff --git a/tests/normalizer_fixtures/ascii_art_untouched.out b/tests/normalizer_fixtures/ascii_art_untouched.out new file mode 100644 index 0000000..b0eae87 --- /dev/null +++ b/tests/normalizer_fixtures/ascii_art_untouched.out @@ -0,0 +1,19 @@ +The pipeline, as a drawing: + + ┌───┐ ┌───┐ + │ 0 │ ──▶ │ 2 │ + └───┘ └───┘ + │ ▲ + ▼ │ + ┌───┐ │ + │ 4 │ ──────┘ + └───┘ + +and in plain characters: + + func1 --+ + +--> merge + func2 --+ + - - - - - - - - + +Back to prose. diff --git a/tests/normalizer_fixtures/bare_prompts_untouched.in b/tests/normalizer_fixtures/bare_prompts_untouched.in new file mode 100644 index 0000000..1bbf2bd --- /dev/null +++ b/tests/normalizer_fixtures/bare_prompts_untouched.in @@ -0,0 +1,9 @@ +Set it up. + +>>> from collections import UserDict +>>> +>>> +>>> class LoggedCache(UserDict): +... name = 'cache' +>>> LoggedCache().name +'cache' diff --git a/tests/normalizer_fixtures/bare_prompts_untouched.out b/tests/normalizer_fixtures/bare_prompts_untouched.out new file mode 100644 index 0000000..1bbf2bd --- /dev/null +++ b/tests/normalizer_fixtures/bare_prompts_untouched.out @@ -0,0 +1,9 @@ +Set it up. + +>>> from collections import UserDict +>>> +>>> +>>> class LoggedCache(UserDict): +... name = 'cache' +>>> LoggedCache().name +'cache' diff --git a/tests/normalizer_fixtures/bullets_with_arrows.in b/tests/normalizer_fixtures/bullets_with_arrows.in new file mode 100644 index 0000000..59bcbb1 --- /dev/null +++ b/tests/normalizer_fixtures/bullets_with_arrows.in @@ -0,0 +1,3 @@ +Two mappings: +- a --> b takes *args +- c --> d, see [x](https://x.org/a) diff --git a/tests/normalizer_fixtures/bullets_with_arrows.out b/tests/normalizer_fixtures/bullets_with_arrows.out new file mode 100644 index 0000000..a1e7cc0 --- /dev/null +++ b/tests/normalizer_fixtures/bullets_with_arrows.out @@ -0,0 +1,4 @@ +Two mappings: + +- a --> b takes \*args +- c --> d, see `x `_ diff --git a/tests/normalizer_fixtures/comment_lines_untouched.in b/tests/normalizer_fixtures/comment_lines_untouched.in new file mode 100644 index 0000000..8dbb9fe --- /dev/null +++ b/tests/normalizer_fixtures/comment_lines_untouched.in @@ -0,0 +1,17 @@ +Check two defaults. + +# >>> same(1, 1) +# True +# >>> same(1, 2) +# False + +# Further, know that within the context's scope, an instance +# will have the context managers it contains available. + +# TODO: Make this work! +# Right now raises: TypeError + +# Can infer types from annotations: + +>>> f() +1 diff --git a/tests/normalizer_fixtures/comment_lines_untouched.out b/tests/normalizer_fixtures/comment_lines_untouched.out new file mode 100644 index 0000000..8dbb9fe --- /dev/null +++ b/tests/normalizer_fixtures/comment_lines_untouched.out @@ -0,0 +1,17 @@ +Check two defaults. + +# >>> same(1, 1) +# True +# >>> same(1, 2) +# False + +# Further, know that within the context's scope, an instance +# will have the context managers it contains available. + +# TODO: Make this work! +# Right now raises: TypeError + +# Can infer types from annotations: + +>>> f() +1 diff --git a/tests/normalizer_fixtures/google_args_bare_name_untouched.in b/tests/normalizer_fixtures/google_args_bare_name_untouched.in new file mode 100644 index 0000000..25522ed --- /dev/null +++ b/tests/normalizer_fixtures/google_args_bare_name_untouched.in @@ -0,0 +1,9 @@ +Do a thing. + +Args: + x: + The x, described on the next line. + error: + Not a section, an argument called error. + y (int): + The y. diff --git a/tests/normalizer_fixtures/google_args_bare_name_untouched.out b/tests/normalizer_fixtures/google_args_bare_name_untouched.out new file mode 100644 index 0000000..25522ed --- /dev/null +++ b/tests/normalizer_fixtures/google_args_bare_name_untouched.out @@ -0,0 +1,9 @@ +Do a thing. + +Args: + x: + The x, described on the next line. + error: + Not a section, an argument called error. + y (int): + The y. diff --git a/tests/normalizer_fixtures/google_args_wrapped_entries_untouched.in b/tests/normalizer_fixtures/google_args_wrapped_entries_untouched.in new file mode 100644 index 0000000..2886fc6 --- /dev/null +++ b/tests/normalizer_fixtures/google_args_wrapped_entries_untouched.in @@ -0,0 +1,11 @@ +Do a thing. + +Args: + func: The method to be decorated. If not provided, a partially applied + decorator will be returned for later application. + maxsize: The maximum size of the cache. + typed: If True, cache entries will be different based on argument types, + such as distinguishing between ``1`` and ``1.0``. + +Returns: + The decorated method. diff --git a/tests/normalizer_fixtures/google_args_wrapped_entries_untouched.out b/tests/normalizer_fixtures/google_args_wrapped_entries_untouched.out new file mode 100644 index 0000000..2886fc6 --- /dev/null +++ b/tests/normalizer_fixtures/google_args_wrapped_entries_untouched.out @@ -0,0 +1,11 @@ +Do a thing. + +Args: + func: The method to be decorated. If not provided, a partially applied + decorator will be returned for later application. + maxsize: The maximum size of the cache. + typed: If True, cache entries will be different based on argument types, + such as distinguishing between ``1`` and ``1.0``. + +Returns: + The decorated method. diff --git a/tests/normalizer_fixtures/markdown_h1_glued_below.in b/tests/normalizer_fixtures/markdown_h1_glued_below.in new file mode 100644 index 0000000..55d80a8 --- /dev/null +++ b/tests/normalizer_fixtures/markdown_h1_glued_below.in @@ -0,0 +1,4 @@ +Intro. + +# FIRST EXAMPLE +We make an Ops class here. diff --git a/tests/normalizer_fixtures/markdown_h1_glued_below.out b/tests/normalizer_fixtures/markdown_h1_glued_below.out new file mode 100644 index 0000000..d86d4d1 --- /dev/null +++ b/tests/normalizer_fixtures/markdown_h1_glued_below.out @@ -0,0 +1,5 @@ +Intro. + +.. rubric:: FIRST EXAMPLE + +We make an Ops class here. diff --git a/tests/normalizer_fixtures/markdown_h1_heading.in b/tests/normalizer_fixtures/markdown_h1_heading.in new file mode 100644 index 0000000..cfff3e5 --- /dev/null +++ b/tests/normalizer_fixtures/markdown_h1_heading.in @@ -0,0 +1,5 @@ +Intro paragraph. + +# Making a signature + +You can construct one from a callable. diff --git a/tests/normalizer_fixtures/markdown_h1_heading.out b/tests/normalizer_fixtures/markdown_h1_heading.out new file mode 100644 index 0000000..22e6243 --- /dev/null +++ b/tests/normalizer_fixtures/markdown_h1_heading.out @@ -0,0 +1,5 @@ +Intro paragraph. + +.. rubric:: Making a signature + +You can construct one from a callable. diff --git a/tests/normalizer_fixtures/note_one_liner_then_field_list.in b/tests/normalizer_fixtures/note_one_liner_then_field_list.in new file mode 100644 index 0000000..faab3fb --- /dev/null +++ b/tests/normalizer_fixtures/note_one_liner_then_field_list.in @@ -0,0 +1,6 @@ +Do it. + +Note: be careful, this +wraps. +:return: the thing +:rtype: int diff --git a/tests/normalizer_fixtures/note_one_liner_then_field_list.out b/tests/normalizer_fixtures/note_one_liner_then_field_list.out new file mode 100644 index 0000000..cf12098 --- /dev/null +++ b/tests/normalizer_fixtures/note_one_liner_then_field_list.out @@ -0,0 +1,8 @@ +Do it. + +Note: + be careful, this + wraps. + +:return: the thing +:rtype: int diff --git a/tests/normalizer_fixtures/term_then_indented_prose_untouched.in b/tests/normalizer_fixtures/term_then_indented_prose_untouched.in new file mode 100644 index 0000000..3299081 --- /dev/null +++ b/tests/normalizer_fixtures/term_then_indented_prose_untouched.in @@ -0,0 +1,6 @@ +You have two ways to do that, specifying: + `remainder='ignore'`, which will just return the (args, + kwargs) tuple and drop the rest. + +specifying: + the name of the thing to do. diff --git a/tests/normalizer_fixtures/term_then_indented_prose_untouched.out b/tests/normalizer_fixtures/term_then_indented_prose_untouched.out new file mode 100644 index 0000000..3299081 --- /dev/null +++ b/tests/normalizer_fixtures/term_then_indented_prose_untouched.out @@ -0,0 +1,6 @@ +You have two ways to do that, specifying: + `remainder='ignore'`, which will just return the (args, + kwargs) tuple and drop the rest. + +specifying: + the name of the thing to do. diff --git a/tests/normalizer_fixtures/usage_lead_in_literal.in b/tests/normalizer_fixtures/usage_lead_in_literal.in new file mode 100644 index 0000000..f55bcfd --- /dev/null +++ b/tests/normalizer_fixtures/usage_lead_in_literal.in @@ -0,0 +1,7 @@ +Run the dependents. + +Usage: + python run_dependent_tests.py # top 12 by importance + +Output: + {'a': 1} diff --git a/tests/normalizer_fixtures/usage_lead_in_literal.out b/tests/normalizer_fixtures/usage_lead_in_literal.out new file mode 100644 index 0000000..f6cf2f1 --- /dev/null +++ b/tests/normalizer_fixtures/usage_lead_in_literal.out @@ -0,0 +1,9 @@ +Run the dependents. + +Usage:: + + python run_dependent_tests.py # top 12 by importance + +Output:: + + {'a': 1} diff --git a/tests/test_agentic_readme.py b/tests/test_agentic_readme.py index 36b0e99..4c6522e 100644 --- a/tests/test_agentic_readme.py +++ b/tests/test_agentic_readme.py @@ -699,6 +699,45 @@ def test_broken_pyproject_is_an_error_not_a_silent_fallback( assert code == 2 and "[tool.epythet.readme] humor" in capsys.readouterr().err +def test_docs_only_project_gets_the_shorter_section(make_project, config_dir): + """No skills, subagents or instruction files: the section must not claim tooling (#27, item 3).""" + from epythet.agentic_readme import ( + DOCS_ONLY_SECTION_SNIPPET, + SECTION_SNIPPET, + section_snippet_for, + ) + + root = make_project("onlydocs", {"core.py": '"""Core."""\n'}) + (root / "pyproject.toml").write_text( + '[project]\nname = "onlydocs"\nversion = "0.0.1"\n' + '[project.urls]\nHomepage = "https://github.com/org/onlydocs"\n' + ) + config = load_config(root) + artifacts = discover_artifacts(root, package_dir=config.package_dir) + assert section_snippet_for(artifacts) == DOCS_ONLY_SECTION_SNIPPET + text = render_section(artifacts, config, policy=ReadmePolicy(), level=2) + assert text.startswith(MARKER_START) and text.rstrip().endswith(MARKER_END) + assert "ships tooling" not in text + assert "publishes its documentation in forms made for coding agents" in text + assert "[`llms.txt`](https://org.github.io/onlydocs/llms.txt)" in text + assert "gh skill install" not in text and "Subagents" not in text + assert "If you are a human, the rest of this README" in text + # the docs-only check still passes as a "section" to document, so --write works + report = check_readme(root, user_config=_policy()) + assert _statuses(report)["agent_docs"] == "warn" + + +def test_project_with_tooling_keeps_the_full_section(project, config_dir): + from epythet.agentic_readme import SECTION_SNIPPET, section_snippet_for + + config = load_config(project) + artifacts = discover_artifacts(project, package_dir=config.package_dir) + assert section_snippet_for(artifacts) == SECTION_SNIPPET + assert "ships tooling for coding agents" in render_section( + artifacts, config, policy=ReadmePolicy() + ) + + def test_write_refuses_when_nothing_is_agentic(make_project, config_dir): root = make_project("bare", {"m.py": '"""M."""\n'}) (root / "README.md").write_text("# bare\n") diff --git a/tests/test_cli.py b/tests/test_cli.py index 6320c38..9bfaa0c 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -340,6 +340,24 @@ def test_unknown_command_exits_two(): assert "invalid choice" in result.stderr +@pytest.mark.parametrize("command,positional", [("validate", "package"), ("repair", "path")]) +@pytest.mark.parametrize( + "shape", + [ + ["{cmd}", ".", "-i", "tests/", "scrap/"], + ["{cmd}", "-i", "tests/", "scrap/", "--", "."], + ["{cmd}", ".", "-i", "tests/", "-i", "scrap/"], + ["{cmd}", ".", "--ignore", "tests/", "--ignore", "scrap/"], + ], +) +def test_ignore_parses_the_same_in_every_argument_order(command, positional, shape): + """``-i`` after or before the positional, several values or repeated: one list (#27, item 8).""" + argv = [part.format(cmd=command) for part in shape] + namespace = _parser().parse_args(argv) + assert getattr(namespace, positional) == "." + assert namespace.ignore == ["tests/", "scrap/"] + + def test_missing_required_argument_exits_two(): result = _run_cli("quickstart") assert result.returncode == 2 diff --git a/tests/test_repair.py b/tests/test_repair.py index 65d26ae..988fdfd 100644 --- a/tests/test_repair.py +++ b/tests/test_repair.py @@ -126,10 +126,24 @@ def m(self): ''' -def test_source_safe_rules_exclude_only_the_star_escaper(): - assert N.escape_unmatched_stars not in SOURCE_SAFE_RULES - assert set(N.DEFAULT_RULES) - set(SOURCE_SAFE_RULES) == {N.escape_unmatched_stars} - assert "escape_unmatched_stars" in UNSAFE_RULES +def test_source_safe_rules_exclude_the_star_escaper_and_the_bare_header_rubric(): + """The two build-time-only rules: one changes the author's text, one changes nothing on the page.""" + excluded = {N.escape_unmatched_stars, N.bare_headers_to_rubrics} + assert set(N.DEFAULT_RULES) - set(SOURCE_SAFE_RULES) == excluded + assert set(UNSAFE_RULES) == {rule.__name__ for rule in excluded} + # napoleon renders a bare ``Examples:`` as that rubric already: the source keeps its header + literal = '"""Do it.\n\n Examples:\n\n >>> f()\n 1\n """' + assert rewrite_docstring_literal(literal) == (literal, None) + + +def test_trailing_blank_lines_before_the_closing_quotes_are_kept(): + """A docstring ending in a blank line is not "rewritten" to drop it.""" + literal = '"""Do it.\n\n >>> f()\n 1\n\n """' + assert rewrite_docstring_literal(literal) == (literal, None) + # and a real fix keeps the author's trailing blank line too + glued = '"""Do it.\n >>> f()\n 1\n\n """' + new, reason = rewrite_docstring_literal(glued) + assert reason is None and new == '"""Do it.\n\n >>> f()\n 1\n\n """' def test_split_literal_shapes(): diff --git a/tests/test_userconfig.py b/tests/test_userconfig.py index 0811fd6..69486ab 100644 --- a/tests/test_userconfig.py +++ b/tests/test_userconfig.py @@ -33,6 +33,7 @@ EXPECTED_SNIPPETS = { "agentic-readme-section", + "agentic-readme-section-docs-only", "agentic-readme-humor", "agentic-readme-instruction", } diff --git a/tests/test_validation_core.py b/tests/test_validation_core.py index f1cc57f..c7cd434 100644 --- a/tests/test_validation_core.py +++ b/tests/test_validation_core.py @@ -157,6 +157,25 @@ def test_renderers_agree(tmp_path, observations): } +def test_ignore_applies_to_the_linters_at_level_0(tmp_path, monkeypatch): + """``--ignore tests/`` drops ruff and pydoclint findings under ``tests/`` too (#27, item 8).""" + import shutil + + if shutil.which("ruff") is None: + pytest.skip("ruff not installed") + project = _project(tmp_path, "ignpkg", CLEAN_MODULE) + tests = project / "ignpkg" / "tests" + tests.mkdir() + (tests / "__init__.py").write_text("") + (tests / "test_x.py").write_text("def undocumented_public():\n return 1\n") + monkeypatch.setenv("EPYTHET_DATA_DIR", str(tmp_path / "data")) + with_tests = validate(project, level=0, observe=False) + assert any("tests/" in f.file for f in with_tests.findings), "the fixture must bite" + ignored = validate(project, level=0, ignore=["tests/"], observe=False) + assert not any("tests/" in f.file for f in ignored.findings) + assert ignored.objects_checked < with_tests.objects_checked + + def test_cli_grammar(): parser = cw.mk_parser(validate_command, prog="epythet-validate") options = {opt for action in parser._actions for opt in action.option_strings}