From 41998611759f3f284e667f4ea07a0545ef704356 Mon Sep 17 00:00:00 2001 From: Blake Bertuccelli-Booth <46652+bbertucc@users.noreply.github.com> Date: Thu, 1 Oct 2026 16:46:26 -0400 Subject: [PATCH 1/2] Retag an already-tagged PDF with --retag A tagged PDF is still refused with already_tagged, now pointing at --retag, so a caller can confirm with the user and run again. --retag drops the old tree; on our own output it also restores the original content, so overlays do not stack. Co-Authored-By: Claude Opus 5.5 --- README.md | 3 ++- src/cli.ts | 6 +++--- src/pdf/document.ts | 30 +++++++++++++++++++++++++++--- src/tag.ts | 4 +++- test/cli.test.ts | 11 +++++++++++ test/document.test.ts | 33 ++++++++++++++++++++++++++++----- 6 files changed, 74 insertions(+), 13 deletions(-) diff --git a/README.md b/README.md index 35a52b9..31ebb10 100644 --- a/README.md +++ b/README.md @@ -39,6 +39,7 @@ iris-pdf review --pdf out.pdf [--report review.json] # optional AI review, bel | `--flatten` | Draw the field values into the page and remove the fields. | | `--password` | Open an encrypted PDF. The output keeps its encryption. | | `--allow-signed` | Tag a signed PDF. This breaks the signature, and the report says so. | +| `--retag` | Tag a PDF that is already tagged, replacing its tags. Without it, such a PDF is refused with `already_tagged`. | | `--partial` | Leave a page untagged, instead of failing, when it has no way to place text. | | `--strict` | Fail on any warning that means content went untagged or unmatched, or that the file was `repaired`. | @@ -92,7 +93,7 @@ The output declares PDF/UA-1 only when it has a title, every page is tagged, eve | Exit | When | |---|---| | 0 | Done. | -| 1 | Refused: `encrypted` (no or wrong password), `permissions_denied`, `too_many_pages` (over 25), `too_many_words` (over 4000 on a page), `already_tagged`, `xfa` (dynamic form), `signed`, `no_acroform_field`, `no_text_positions`, `strict`. From `review`: `review_failed` (the model or its API failed on a page). | +| 1 | Refused: `encrypted` (no or wrong password), `permissions_denied`, `too_many_pages` (over 25), `too_many_words` (over 4000 on a page), `already_tagged` (see `--retag`), `xfa` (dynamic form), `signed`, `no_acroform_field`, `no_text_positions`, `strict`. From `review`: `review_failed` (the model or its API failed on a page). | | 2 | A check failed: `pixels_changed`, `text_lost`. | | 3 | Bad input: `unreadable`, `bad_pages`, `no_document_language`, `bad_value`, `field_not_settable`, `bad_arguments`. From `review`: `not_tagged`, `no_readable_structure` (not tagged by this tool), `bad_structure` (nested over 64 levels), `no_credentials`. | diff --git a/src/cli.ts b/src/cli.ts index 9070b89..19dfdfa 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -12,7 +12,7 @@ const USAGE = `iris-pdf ${VERSION} iris-pdf tag --pdf --pages --out [--values ] [--report ] [--lang ] [--title ] [--ocr auto|off|required] [--verify pixels,text|off] [--verify-dpi 150] - [--flatten] [--password ] [--allow-signed] [--partial] [--strict] + [--flatten] [--password ] [--allow-signed] [--retag] [--partial] [--strict] iris-pdf fields --pdf [--json] [--password ] iris-pdf check --pdf iris-pdf review --pdf [--report ] [--provider anthropic|bedrock] [--model ] [--password ]`; @@ -21,7 +21,7 @@ const OPTIONS = { pdf: { type: "string" }, pages: { type: "string" }, values: { type: "string" }, out: { type: "string" }, report: { type: "string" }, lang: { type: "string" }, title: { type: "string" }, ocr: { type: "string" }, verify: { type: "string" }, "verify-dpi": { type: "string" }, flatten: { type: "boolean" }, - password: { type: "string" }, "allow-signed": { type: "boolean" }, partial: { type: "boolean" }, + password: { type: "string" }, "allow-signed": { type: "boolean" }, retag: { type: "boolean" }, partial: { type: "boolean" }, strict: { type: "boolean" }, provider: { type: "string" }, model: { type: "string" }, json: { type: "boolean" }, help: { type: "boolean", short: "h" }, } as const; @@ -98,7 +98,7 @@ async function main(argv: string[]): Promise { values: args.values ? (readJson(args.values, "values") as TagOptions["values"]) : undefined, lang: args.lang, title: args.title, ocr: ocr as TagOptions["ocr"], verify: verify !== "off", verifyDpi: dpi, flatten: args.flatten, password: args.password, - allowSigned: args["allow-signed"], partial: args.partial, strict: args.strict, + allowSigned: args["allow-signed"], retag: args.retag, partial: args.partial, strict: args.strict, }; const report = newReport(); const pdf = readPdf(args.pdf); diff --git a/src/pdf/document.ts b/src/pdf/document.ts index 5ed2283..7b10382 100644 --- a/src/pdf/document.ts +++ b/src/pdf/document.ts @@ -11,12 +11,14 @@ export type Source = { acroform: boolean; xfa: boolean; repaired: boolean; // damaged: saved as a full rewrite, not an update + restored: boolean; // retagging our own output: pages got their original content back warnings: Warning[]; }; // readOnly: only reading (listing fields), so the refusals that protect the // file from changes do not apply. -export type OpenOptions = { password?: string; allowSigned?: boolean; readOnly?: boolean }; +// retag: replace the tags of a PDF that has them. +export type OpenOptions = { password?: string; allowSigned?: boolean; retag?: boolean; readOnly?: boolean }; export function openPdf(bytes: Uint8Array, opts: OpenOptions = {}): Source { let doc: mupdf.PDFDocument; @@ -39,7 +41,7 @@ export function openPdf(bytes: Uint8Array, opts: OpenOptions = {}): Source { if (inherited(field, "FT")?.asName() === "Sig" && inherited(field, "V")) signed = true; }); const repaired = doc.wasRepaired() || !doc.canBeSavedIncrementally(); - const source = { doc, encrypted, signed, acroform: !acroform.isNull(), xfa, repaired, warnings }; + const source = { doc, encrypted, signed, acroform: !acroform.isNull(), xfa, repaired, restored: false, warnings }; if (opts.readOnly) return source; // An owner password can forbid changes. We do not work around it. @@ -54,7 +56,9 @@ export function openPdf(bytes: Uint8Array, opts: OpenOptions = {}): Source { throw new IrisPdfError("too_many_pages", `The PDF has ${pages} pages; the limit is ${MAX_PAGES}.`); } if (!root.get("StructTreeRoot").isNull()) { - throw new IrisPdfError("already_tagged", "The PDF is already tagged. Retagging it is not supported."); + if (!opts.retag) throw new IrisPdfError("already_tagged", "The PDF is already tagged. Pass --retag to replace its tags."); + source.restored = untag(doc); + warnings.push({ code: "retagged", detail: "The PDF's existing tags were removed and replaced." }); } // Dynamic XFA draws the form when it opens; its pages are not in the file. if (xfa && root.get("NeedsRendering").valueOf() === true) { @@ -68,6 +72,26 @@ export function openPdf(bytes: Uint8Array, opts: OpenOptions = {}): Source { return source; } +// Drop the structure tree and what points into it. A page we tagged gets its +// original content back, so retagging our own output does not stack overlays. +function untag(doc: mupdf.PDFDocument): boolean { + let restored = false; + doc.getTrailer().get("Root").delete("StructTreeRoot"); + for (let i = 0; i < doc.countPages(); i++) { + const page = doc.findPage(i), contents = page.get("Contents"); + page.delete("StructParents"); + page.get("Annots").forEach((a) => { if (a.isDictionary()) a.delete("StructParent"); }); + if (!contents.isArray() || contents.length < 4) continue; + const body = (j: number) => { try { return contents.get(j).readStream().asString(); } catch { return null; } }; + if (body(0) !== "/Artifact BMC q\n" || body(contents.length - 2) !== "\nQ EMC\n") continue; + const original: mupdf.PDFObject[] = []; + for (let j = 1; j < contents.length - 2; j++) original.push(contents.get(j)); + page.put("Contents", original); + restored = true; + } + return restored; +} + // Every terminal field in the AcroForm tree. export function forEachField(doc: mupdf.PDFDocument, fn: (field: mupdf.PDFObject) => void) { const visit = (field: mupdf.PDFObject, depth: number) => { diff --git a/src/tag.ts b/src/tag.ts index c93d8fd..667e660 100644 --- a/src/tag.ts +++ b/src/tag.ts @@ -38,6 +38,8 @@ const STRICT = ["no_title", "page_not_in_html", "unmatched_text", "missing_glyph export function tag(pdf: Uint8Array, input: PagesInput, opts: TagOptions = {}, report: Report = newReport()): Uint8Array { const src = openPdf(pdf, opts); const { doc } = src; + // The old overlay's text is gone on purpose, so the checks compare against the source without it. + const baseline = src.restored ? save(doc, true) : pdf; const warn = (w: Warning) => report.warnings.push(w); src.warnings.forEach(warn); const pageCount = doc.countPages(); @@ -117,7 +119,7 @@ export function tag(pdf: Uint8Array, input: PagesInput, opts: TagOptions = {}, r const out = save(doc, src.repaired); report.sizeIncreaseBytes = out.length - pdf.length; - if (opts.verify !== false) verify(pdf, out, opts, filled.changed, overlayText, report); + if (opts.verify !== false) verify(baseline, out, opts, filled.changed, overlayText, report); const strict = report.warnings.filter((w) => STRICT.includes(w.code)); if (opts.strict && strict.length) { throw new IrisPdfError("strict", `--strict: ${[...new Set(strict.map((w) => w.code))].join(", ")}`); diff --git a/test/cli.test.ts b/test/cli.test.ts index a424e15..c126e4f 100644 --- a/test/cli.test.ts +++ b/test/cli.test.ts @@ -34,6 +34,17 @@ test("a refusal exits 1, writes no PDF, and still writes the report", () => { assert.equal(JSON.parse(readFileSync(report, "utf8")).error.code, "signed"); }); +test("a tagged PDF is refused as already_tagged, and --retag tags it", () => { + const once = join(dir, "once.pdf"), twice = join(dir, "twice.pdf"); + assert.equal(run(...tagArgs("text-simple", once)).code, 0); + const again = ["tag", "--pdf", once, "--pages", fixture("text-simple.pages.json"), "--out", twice]; + const r = run(...again); + assert.equal(r.code, 1); + assert.match(r.err, /^iris-pdf: already_tagged: .*--retag/); + assert.equal(run(...again, "--retag").code, 0); + assert.ok(existsSync(twice)); +}); + test("a failed verification exits 2 and writes no PDF", () => { // A stream ending inside a string swallows the overlay (see tag.test.ts). const doc = new mupdf.PDFDocument(readFileSync(fixture("text-simple.pdf"))); diff --git a/test/document.test.ts b/test/document.test.ts index bbc69b6..b36e5cb 100644 --- a/test/document.test.ts +++ b/test/document.test.ts @@ -7,11 +7,11 @@ import { readFixture, pagesOf, tagFixture } from "./helpers.ts"; const simple: PagesInput = pagesOf("text-simple"); // Runs tag and returns the error code it refused with. -function refusal(pdf: Uint8Array, opts: TagOptions = {}, pages: PagesInput = simple): { code: string; exit: number } { +function refusal(pdf: Uint8Array, opts: TagOptions = {}, pages: PagesInput = simple): { code: string; exit: number; message: string } { try { tag(pdf, pages, opts); } catch (e) { - return e as { code: string; exit: number }; + return e as { code: string; exit: number; message: string }; } assert.fail("tag did not refuse"); } @@ -72,9 +72,32 @@ test("--allow-signed tags a signed PDF and says the signature is gone", () => { assert.equal(report.source.signed, true); }); -test("refuses a PDF that is already tagged", () => { - const { out } = tagFixture("text-simple"); - assert.equal(refusal(out).code, "already_tagged"); +test("refuses a PDF that is already tagged, and retags it when asked", () => { + const first = tagFixture("text-simple"); + assert.deepEqual([refusal(first.out).code, refusal(first.out).message], ["already_tagged", "The PDF is already tagged. Pass --retag to replace its tags."]); + const report = newReport(); + const doc = new mupdf.PDFDocument(tag(first.out, simple, { retag: true }, report)); + assert.ok(report.warnings.some((w) => w.code === "retagged")); + // Our own output gets its original content back: one overlay, not two, and the same tags. + assert.ok(!report.warnings.some((w) => w.code === "source_marked_content")); + assert.equal(doc.findPage(0).get("Contents").length, first.doc.findPage(0).get("Contents").length); + assert.deepEqual(report.structure, first.report.structure); + assert.equal(report.verification.textPreserved, true); + assert.equal(report.verification.differingPixels, 0); +}); + +test("--retag drops a foreign tag tree and what points into it", () => { + const pdf = blankPdf(1, (doc) => { + const page = doc.findPage(0); + doc.getTrailer().get("Root").put("StructTreeRoot", doc.addObject({ Type: "StructTreeRoot", K: [] })); + page.put("StructParents", 0); + page.put("Annots", [doc.addObject({ Type: "Annot", Subtype: "Text", Rect: [0, 0, 10, 10], StructParent: 1 })]); + }); + const doc = new mupdf.PDFDocument(tag(pdf, { lang: "en", pages: [] }, { retag: true })); + const page = doc.findPage(0); + assert.ok(page.get("StructParents").isNull()); + assert.ok(page.get("Annots").get(0).get("StructParent").isNull()); + assert.equal(doc.getTrailer().get("Root", "StructTreeRoot", "K").length, 0, "a new, empty tree"); }); test("XFA: a dynamic form is refused, a hybrid form loses only its XFA", () => { From 3f7e86b5a21b63e82e66c15deffc7905a4492539 Mon Sep 17 00:00:00 2001 From: Blake Bertuccelli-Booth <46652+bbertucc@users.noreply.github.com> Date: Thu, 1 Oct 2026 16:58:08 -0400 Subject: [PATCH 2/2] Retag rewrites link descriptions we wrote; test the foreign-PDF path Link /Contents written by this tool is marked, so a retag drops it and describes the link from the new HTML. An author's description stays. Co-Authored-By: Claude Opus 5.5 --- README.md | 2 +- src/pdf/document.ts | 14 +++++++++++++- src/tag.ts | 6 +++--- test/document.test.ts | 25 ++++++++++++++++++++++++- 4 files changed, 41 insertions(+), 6 deletions(-) diff --git a/README.md b/README.md index 31ebb10..8861f3a 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ iris-pdf review --pdf out.pdf [--report review.json] # optional AI review, bel | `--flatten` | Draw the field values into the page and remove the fields. | | `--password` | Open an encrypted PDF. The output keeps its encryption. | | `--allow-signed` | Tag a signed PDF. This breaks the signature, and the report says so. | -| `--retag` | Tag a PDF that is already tagged, replacing its tags. Without it, such a PDF is refused with `already_tagged`. | +| `--retag` | Tag a PDF that is already tagged, replacing its tags. Without it, such a PDF is refused with `already_tagged`. The output is an update of the input, so each retag adds to the file's size. | | `--partial` | Leave a page untagged, instead of failing, when it has no way to place text. | | `--strict` | Fail on any warning that means content went untagged or unmatched, or that the file was `repaired`. | diff --git a/src/pdf/document.ts b/src/pdf/document.ts index 7b10382..28b2b3f 100644 --- a/src/pdf/document.ts +++ b/src/pdf/document.ts @@ -72,6 +72,13 @@ export function openPdf(bytes: Uint8Array, opts: OpenOptions = {}): Source { return source; } +// Marks an annotation /Contents that this tool wrote, so a retag can tell it from the author's. +const DESCRIBED = "IrisPdfContents"; +export function describe(doc: mupdf.PDFDocument, annot: mupdf.PDFObject, text: string) { + annot.put("Contents", doc.newString(text)); + annot.put(DESCRIBED, true); +} + // Drop the structure tree and what points into it. A page we tagged gets its // original content back, so retagging our own output does not stack overlays. function untag(doc: mupdf.PDFDocument): boolean { @@ -80,7 +87,12 @@ function untag(doc: mupdf.PDFDocument): boolean { for (let i = 0; i < doc.countPages(); i++) { const page = doc.findPage(i), contents = page.get("Contents"); page.delete("StructParents"); - page.get("Annots").forEach((a) => { if (a.isDictionary()) a.delete("StructParent"); }); + page.get("Annots").forEach((a) => { + if (!a.isDictionary()) return; + a.delete("StructParent"); + // A description we wrote is rewritten from the new HTML; the author's stays. + if (a.get(DESCRIBED).isBoolean()) { a.delete("Contents"); a.delete(DESCRIBED); } + }); if (!contents.isArray() || contents.length < 4) continue; const body = (j: number) => { try { return contents.get(j).readStream().asString(); } catch { return null; } }; if (body(0) !== "/Artifact BMC q\n" || body(contents.length - 2) !== "\nQ EMC\n") continue; diff --git a/src/tag.ts b/src/tag.ts index 667e660..d9faf4a 100644 --- a/src/tag.ts +++ b/src/tag.ts @@ -1,7 +1,7 @@ // tag(): the original PDF plus Iris's HTML -> the same PDF, tagged, with any // form values filled in, checked to render identically (spec ยง7). import * as mupdf from "mupdf"; -import { inherited, openPdf, save, type OpenOptions } from "./pdf/document.ts"; +import { describe, inherited, openPdf, save, type OpenOptions } from "./pdf/document.ts"; import { artifactStreams, drawsNothing, Overlay, pagesWithMcids } from "./pdf/content.ts"; import { FontSet, unembeddedFonts } from "./pdf/fonts.ts"; import { StructTree } from "./pdf/struct.ts"; @@ -234,7 +234,7 @@ function tagPage(page: mupdf.PDFPage, i: number, html: string, ctx: PageCtx): { const generic = !l.obj.get("Contents").isString() && !under && !l.uri; const elem = ctx.struct.add(ctx.struct.top, "Link", generic && !ctx.lang.startsWith("en") ? { Lang: ctx.doc.newString("en") } : {}); ctx.struct.objr(elem, pageObj, l.obj); - if (l.obj.get("Contents").isNull()) l.obj.put("Contents", ctx.doc.newString(under || l.uri || "Link to another part of this document")); + if (l.obj.get("Contents").isNull()) describe(ctx.doc, l.obj, under || l.uri || "Link to another part of this document"); warn({ code: "unmatched_link", detail: l.uri || "internal link" }); } for (const w of ctx.widgets.filter((w) => !w.used && !ctx.flatten)) { @@ -325,7 +325,7 @@ function linkAnnotation(n: Node, elem: mupdf.PDFObject, e: Emitter) { link.used = true; e.struct.objr(elem, e.pageObj, link.obj); const text = n.kids.flatMap((k) => (isRun(k) ? k.words.map((w) => w.text) : [])).join(" "); - if (link.obj.get("Contents").isNull() && text) link.obj.put("Contents", e.doc.newString(text)); + if (link.obj.get("Contents").isNull() && text) describe(e.doc, link.obj, text); } // An : a Form element owning its widgets, which get the label as /TU. diff --git a/test/document.test.ts b/test/document.test.ts index b36e5cb..c4b54dc 100644 --- a/test/document.test.ts +++ b/test/document.test.ts @@ -97,7 +97,30 @@ test("--retag drops a foreign tag tree and what points into it", () => { const page = doc.findPage(0); assert.ok(page.get("StructParents").isNull()); assert.ok(page.get("Annots").get(0).get("StructParent").isNull()); - assert.equal(doc.getTrailer().get("Root", "StructTreeRoot", "K").length, 0, "a new, empty tree"); + assert.equal(doc.getTrailer().get("Root", "StructTreeRoot", "K", "S").asName(), "Document", "a new tree"); +}); + +test("--retag rewrites link descriptions we wrote, and keeps the author's", () => { + const html = (a: string) => ({ ...pagesOf("links"), pages: [{ sourcePage: 1, html: `

Contact

Apply online at ${a}.

` }] }); + const contents = (pdf: Uint8Array) => new mupdf.PDFDocument(pdf).findPage(0).get("Annots").get(0).get("Contents").asString(); + const once = tag(readFixture("links.pdf"), pagesOf("links")); + assert.equal(contents(once), "the city website"); + assert.equal(contents(tag(once, html(`the city website`), { retag: true })), "website"); + const authored = new mupdf.PDFDocument(readFixture("links.pdf")); + authored.findPage(0).get("Annots").get(0).put("Contents", authored.newString("Permits")); + const own = tag(tag(authored.saveToBuffer("").asUint8Array().slice(), pagesOf("links")), html(`the city website`), { retag: true }); + assert.equal(contents(own), "Permits"); +}); + +test("--retag on a foreign tagged PDF keeps its marked content, so it claims no PDF/UA", () => { + const doc = new mupdf.PDFDocument(readFixture("text-embedded.pdf")); + const page = doc.findPage(0); + page.put("Contents", doc.addStream("/P <> BDC " + page.get("Contents").readStream().asString() + " EMC", {})); + doc.getTrailer().get("Root").put("StructTreeRoot", doc.addObject({ Type: "StructTreeRoot" })); + const report = newReport(); + const out = new mupdf.PDFDocument(tag(doc.saveToBuffer("").asUint8Array().slice(), pagesOf("text-embedded"), { retag: true }, report)); + assert.deepEqual(["retagged", "source_marked_content"].map((c) => report.warnings.some((w) => w.code === c)), [true, true]); + assert.doesNotMatch(out.getTrailer().get("Root", "Metadata").readStream().asString(), /pdfuaid:part/); }); test("XFA: a dynamic form is refused, a hybrid form loses only its XFA", () => {