diff --git a/clients/viewer/src/components/pipeline-viewer/StageTabs.tsx b/clients/viewer/src/components/pipeline-viewer/StageTabs.tsx index d2bf4dc..baf50e5 100644 --- a/clients/viewer/src/components/pipeline-viewer/StageTabs.tsx +++ b/clients/viewer/src/components/pipeline-viewer/StageTabs.tsx @@ -141,7 +141,7 @@ function isProcessingInStage( extraction: ['Docling Extraction', 'OCR Re-extraction', 'Extraction'], analysis: ['PDF Classification', 'Structure Analysis'], headings: ['Heading Levels', 'Heading Reconciliation'], - translation: ['Page Content Corrections', 'Code Block Languages'], + translation: ['Page Content Corrections', 'Code Block Languages', 'Form Fields'], assembly: ['Cross-Page Fixes', 'Final Cleanup'], }; const names = nameMap[stage.name]; diff --git a/clients/viewer/src/types/pipeline-viewer.ts b/clients/viewer/src/types/pipeline-viewer.ts index e539880..eda1baf 100644 --- a/clients/viewer/src/types/pipeline-viewer.ts +++ b/clients/viewer/src/types/pipeline-viewer.ts @@ -45,7 +45,7 @@ export const PIPELINE_STAGES: StageDefinition[] = [ { name: 'extraction', label: 'Extraction', steps: ['docling', 'docling_ocr'] }, { name: 'analysis', label: 'Analysis', steps: ['classification', 'structure'] }, { name: 'headings', label: 'Headings', steps: ['heading_levels', 'heading_reconciliation'] }, - { name: 'translation', label: 'Translation', steps: ['page_content', 'code_blocks'] }, + { name: 'translation', label: 'Translation', steps: ['page_content', 'code_blocks', 'form_fields'] }, { name: 'assembly', label: 'Assembly', steps: ['boundaries', 'cleanup'] }, ]; diff --git a/docs/reference/pipeline-phases.md b/docs/reference/pipeline-phases.md index 2db405f..8a1f1e2 100644 --- a/docs/reference/pipeline-phases.md +++ b/docs/reference/pipeline-phases.md @@ -12,14 +12,14 @@ The pipeline has **5 versioned conversion phases** plus a **PII Review** gate th | 1. **Extraction** | `docling`, `docling_ocr` (conditional) | No | PDF → markdown + page images via IBM Docling. `docling_ocr` only fires when the classifier flags a scanned document. | | 2. **Analysis** | `classification`, `structure` | Yes | `classification` tags the document as digital / scanned / malformed. `structure` identifies headings, footnotes, code blocks, per-page layout attributes. | | 3. **Headings** | `heading_reconciliation`, `heading_levels` | Yes | `heading_reconciliation` reconciles per-page heading candidates against the global outline. `heading_levels` normalises the hierarchy (H1 → H2 → H3, no skips). | -| 4. **Translation** | `page_content`, `code_blocks` | Yes | `page_content` does per-page accessibility corrections (invokes image / table / list subagents). `code_blocks` tags fenced blocks with detected programming language. | +| 4. **Translation** | `page_content`, `code_blocks`, `form_fields` | Yes | `page_content` does per-page accessibility corrections (invokes image / table / list subagents). `code_blocks` tags fenced blocks with detected programming language. `form_fields` detects form controls from each page image and injects accessible HTML (labelled inputs, `
`/`` option groups). | | 5. **Assembly** | `boundaries`, `cleanup` | Mixed | `boundaries` rejoins cross-page split content and relocates footnotes (AI). `cleanup` normalises whitespace and lints the markdown (deterministic). | The viewer also shows a dynamic **Review** stage that catches any orphan steps (`revision_*`, `feedback_*`, custom steps) not listed above. ## Internal step → `_step_*` method map -Nine methods in `src/services/pipeline_viewer.py`, plus `pii_scan` which runs inline in `src/api/pipeline_viewer.py` (no `_step_*` method — it's pre-extraction gate logic, not a PipelineViewerService method): +Ten methods in `src/services/pipeline_viewer.py`, plus `pii_scan` which runs inline in `src/api/pipeline_viewer.py` (no `_step_*` method — it's pre-extraction gate logic, not a PipelineViewerService method): | Step name | Method | Phase | Deterministic / AI | |---|---|---|---| @@ -32,10 +32,11 @@ Nine methods in `src/services/pipeline_viewer.py`, plus `pii_scan` which runs in | `heading_levels` | `_step_heading_levels` | Headings | AI | | `page_content` | `_step_page_content` | Translation | AI + subagents | | `code_blocks` | `_step_code_blocks` | Translation | AI | +| `form_fields` | `_step_form_fields` | Translation | AI (vision) + deterministic injection | | `boundaries` | `_step_boundaries` | Assembly | AI + subagent | | `cleanup` | `_step_cleanup` | Assembly | Deterministic | -`classification` is not its own `_step_*` method — it's a `StepResult` emitted from within `_step_docling` when the classifier runs. Up to 10 named step results can appear in one run. +`classification` is not its own `_step_*` method — it's a `StepResult` emitted from within `_step_docling` when the classifier runs. Up to 11 named step results can appear in one run. ## Subagents diff --git a/src/agents/prompts/form_fields.py b/src/agents/prompts/form_fields.py new file mode 100644 index 0000000..ead1a99 --- /dev/null +++ b/src/agents/prompts/form_fields.py @@ -0,0 +1,101 @@ +"""System prompt and user message template for the Form Fields agent. + +This agent examines one page of a PDF at a time, comparing the page image +(visual ground truth) against its extracted markdown, and reports any form +fields it finds. It does NOT modify the markdown — a deterministic injector +turns each reported field into accessible HTML and splices it in afterwards. +""" + +FORM_FIELDS_SYSTEM_PROMPT = """\ +You are a form accessibility analyst. You examine one page of a PDF at a \ +time, comparing the page image (visual ground truth) against its extracted \ +markdown, and you identify form fields a person would be expected to fill in. + +Your job is strictly detection — you do NOT modify the markdown. You report \ +what you find so that a later deterministic step can replace each field with \ +accessible HTML. + +## What counts as a form field + +Look at the page image for anything a person fills in by hand or on screen: + +- **Text fields** — a label followed by a blank line or box to write in, e.g. \ +"Name: ____________", "Email", an empty ruled box. +- **Textareas** — large multi-line blank areas for long answers, comments, \ +or essays. +- **Checkboxes** — a single square/box to tick (☐, [ ], a small empty box) \ +next to a statement, e.g. "☐ I agree to the terms". +- **Checkbox groups** — several checkboxes under one prompt where more than \ +one may be ticked, e.g. "Select all that apply". +- **Radio groups** — several mutually-exclusive options under one prompt \ +where exactly one is chosen, e.g. "Marital status: ☐ Single ☐ Married". +- **Select / dropdowns** — a labelled choice list (often a box with a chevron). +- **Date fields** — a labelled blank for a date (e.g. "Date of birth: __/__/____"). +- **Signature lines** — a ruled line labelled "Signature". + +Decorative rules, table borders, and underlines used purely for emphasis are \ +NOT form fields. Only report something a person is meant to complete. + +## What to report for each field + +- **field_type**: one of text, textarea, checkbox, checkbox_group, \ +radio_group, select, date, signature. +- **label**: the accessible label, read from the IMAGE (not the markdown, \ +which may have OCR errors). For a single field this is the prompt next to \ +the blank (e.g. "Full name"). For a grouped field (radio_group, \ +checkbox_group, select) this is the group prompt/legend (e.g. "Marital \ +status"). If a field has no visible label, write a short descriptive one. +- **anchor_text**: copy, VERBATIM, the exact text in the extracted markdown \ +that represents this field — for example the line "Name: ____________" or \ +"☐ I agree". This is used to locate and replace the field, so it must match \ +the markdown character-for-character. If the field is visible in the image \ +but absent from the markdown, set anchor_text to the nearest preceding \ +markdown line (e.g. the section heading) so the field can be inserted after it. +- **options**: for radio_group, checkbox_group, and select only — the list of \ +choices, each with its visible label and whether it appears pre-ticked. \ +Empty for all other types. +- **required**: true if the field is visibly marked required (an asterisk, \ +the word "required", bold "must"). +- **reasoning**: one sentence on how you determined the type and label. + +## Rules + +- Report fields in the order they appear top-to-bottom on the page. +- A page with no form fields is a valid result — return an empty list. +- Never invent options or labels that are not visible in the image. +- Do not report the same field twice. +""" + + +def build_form_fields_user_message( + page_markdown: str, + page_number: int, + total_pages: int, +) -> str: + """Build the text portion of the user message for one page. + + The page image is passed separately as a binary content part. + + Args: + page_markdown: Extracted markdown for this page (newest version). + page_number: Current page number (1-indexed). + total_pages: Total number of pages in the document. + + Returns: + Text portion of the user message. + """ + parts: list[str] = [] + parts.append(f"## Page {page_number} of {total_pages}") + parts.append("") + parts.append("### Extracted markdown for this page") + parts.append("") + parts.append("```markdown") + parts.append(page_markdown) + parts.append("```") + parts.append("") + parts.append( + "The page image is attached. Compare the image (ground truth) against " + "the markdown above and report every form field you find. Copy each " + "field's anchor_text verbatim from the markdown." + ) + return "\n".join(parts) diff --git a/src/services/pipeline_viewer.py b/src/services/pipeline_viewer.py index 10188ee..d11fea1 100644 --- a/src/services/pipeline_viewer.py +++ b/src/services/pipeline_viewer.py @@ -13,6 +13,7 @@ import asyncio import base64 import difflib +import html import logging import re import time @@ -26,6 +27,9 @@ DocumentChange, FigureData, FootnoteInfo, + FormFieldInfo, + FormFieldsPageOutput, + FormFieldType, HeadingReconciliationOutput, ImageDescriptionResult, LayoutType, @@ -273,6 +277,126 @@ def _apply_code_block_fence( ) +def _render_form_field_html(field: FormFieldInfo, field_id: str) -> str: + """Render a detected form field as an accessible, static HTML control. + + The output is a *representation* of the form for accessibility, not a + working form: there is no surrounding ``
`` and inputs are marked + ``readonly`` / ``disabled`` so they advertise their role and label to a + screen reader without implying the document is fillable. + + Every control is paired with a programmatic label: + - text / textarea / date / select use ``