Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,7 @@ iris-pdf review --pdf out.pdf [--report review.json] # optional AI review, bel
| `--flatten` | Draw the field values into the page and remove the fields. |
| `--password` | Open an encrypted PDF. The output keeps its encryption. |
| `--allow-signed` | Tag a signed PDF. This breaks the signature, and the report says so. |
| `--retag` | Tag a PDF that is already tagged, replacing its tags. Without it, such a PDF is refused with `already_tagged`. The output is an update of the input, so each retag adds to the file's size. |
| `--partial` | Leave a page untagged, instead of failing, when it has no way to place text. |
| `--strict` | Fail on any warning that means content went untagged or unmatched, or that the file was `repaired`. |

Expand Down Expand Up @@ -92,7 +93,7 @@ The output declares PDF/UA-1 only when it has a title, every page is tagged, eve
| Exit | When |
|---|---|
| 0 | Done. |
| 1 | Refused: `encrypted` (no or wrong password), `permissions_denied`, `too_many_pages` (over 25), `too_many_words` (over 4000 on a page), `already_tagged`, `xfa` (dynamic form), `signed`, `no_acroform_field`, `no_text_positions`, `strict`. From `review`: `review_failed` (the model or its API failed on a page). |
| 1 | Refused: `encrypted` (no or wrong password), `permissions_denied`, `too_many_pages` (over 25), `too_many_words` (over 4000 on a page), `already_tagged` (see `--retag`), `xfa` (dynamic form), `signed`, `no_acroform_field`, `no_text_positions`, `strict`. From `review`: `review_failed` (the model or its API failed on a page). |
| 2 | A check failed: `pixels_changed`, `text_lost`. |
| 3 | Bad input: `unreadable`, `bad_pages`, `no_document_language`, `bad_value`, `field_not_settable`, `bad_arguments`. From `review`: `not_tagged`, `no_readable_structure` (not tagged by this tool), `bad_structure` (nested over 64 levels), `no_credentials`. |

Expand Down
6 changes: 3 additions & 3 deletions src/cli.ts
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@ const USAGE = `iris-pdf ${VERSION}
iris-pdf tag --pdf <in.pdf> --pages <pages.json> --out <out.pdf>
[--values <values.json>] [--report <report.json>] [--lang <bcp47>] [--title <text>]
[--ocr auto|off|required] [--verify pixels,text|off] [--verify-dpi 150]
[--flatten] [--password <pw>] [--allow-signed] [--partial] [--strict]
[--flatten] [--password <pw>] [--allow-signed] [--retag] [--partial] [--strict]
iris-pdf fields --pdf <in.pdf> [--json] [--password <pw>]
iris-pdf check --pdf <in.pdf>
iris-pdf review --pdf <tagged.pdf> [--report <review.json>] [--provider anthropic|bedrock] [--model <id>] [--password <pw>]`;
Expand All @@ -21,7 +21,7 @@ const OPTIONS = {
pdf: { type: "string" }, pages: { type: "string" }, values: { type: "string" }, out: { type: "string" },
report: { type: "string" }, lang: { type: "string" }, title: { type: "string" }, ocr: { type: "string" },
verify: { type: "string" }, "verify-dpi": { type: "string" }, flatten: { type: "boolean" },
password: { type: "string" }, "allow-signed": { type: "boolean" }, partial: { type: "boolean" },
password: { type: "string" }, "allow-signed": { type: "boolean" }, retag: { type: "boolean" }, partial: { type: "boolean" },
strict: { type: "boolean" }, provider: { type: "string" }, model: { type: "string" }, json: { type: "boolean" }, help: { type: "boolean", short: "h" },
} as const;

Expand Down Expand Up @@ -98,7 +98,7 @@ async function main(argv: string[]): Promise<number> {
values: args.values ? (readJson(args.values, "values") as TagOptions["values"]) : undefined,
lang: args.lang, title: args.title, ocr: ocr as TagOptions["ocr"],
verify: verify !== "off", verifyDpi: dpi, flatten: args.flatten, password: args.password,
allowSigned: args["allow-signed"], partial: args.partial, strict: args.strict,
allowSigned: args["allow-signed"], retag: args.retag, partial: args.partial, strict: args.strict,
};
const report = newReport();
const pdf = readPdf(args.pdf);
Expand Down
42 changes: 39 additions & 3 deletions src/pdf/document.ts
Original file line number Diff line number Diff line change
Expand Up @@ -11,12 +11,14 @@ export type Source = {
acroform: boolean;
xfa: boolean;
repaired: boolean; // damaged: saved as a full rewrite, not an update
restored: boolean; // retagging our own output: pages got their original content back
warnings: Warning[];
};

// readOnly: only reading (listing fields), so the refusals that protect the
// file from changes do not apply.
export type OpenOptions = { password?: string; allowSigned?: boolean; readOnly?: boolean };
// retag: replace the tags of a PDF that has them.
export type OpenOptions = { password?: string; allowSigned?: boolean; retag?: boolean; readOnly?: boolean };

export function openPdf(bytes: Uint8Array, opts: OpenOptions = {}): Source {
let doc: mupdf.PDFDocument;
Expand All @@ -39,7 +41,7 @@ export function openPdf(bytes: Uint8Array, opts: OpenOptions = {}): Source {
if (inherited(field, "FT")?.asName() === "Sig" && inherited(field, "V")) signed = true;
});
const repaired = doc.wasRepaired() || !doc.canBeSavedIncrementally();
const source = { doc, encrypted, signed, acroform: !acroform.isNull(), xfa, repaired, warnings };
const source = { doc, encrypted, signed, acroform: !acroform.isNull(), xfa, repaired, restored: false, warnings };
if (opts.readOnly) return source;

// An owner password can forbid changes. We do not work around it.
Expand All @@ -54,7 +56,9 @@ export function openPdf(bytes: Uint8Array, opts: OpenOptions = {}): Source {
throw new IrisPdfError("too_many_pages", `The PDF has ${pages} pages; the limit is ${MAX_PAGES}.`);
}
if (!root.get("StructTreeRoot").isNull()) {
throw new IrisPdfError("already_tagged", "The PDF is already tagged. Retagging it is not supported.");
if (!opts.retag) throw new IrisPdfError("already_tagged", "The PDF is already tagged. Pass --retag to replace its tags.");
source.restored = untag(doc);
warnings.push({ code: "retagged", detail: "The PDF's existing tags were removed and replaced." });
}
// Dynamic XFA draws the form when it opens; its pages are not in the file.
if (xfa && root.get("NeedsRendering").valueOf() === true) {
Expand All @@ -68,6 +72,38 @@ export function openPdf(bytes: Uint8Array, opts: OpenOptions = {}): Source {
return source;
}

// Marks an annotation /Contents that this tool wrote, so a retag can tell it from the author's.
const DESCRIBED = "IrisPdfContents";
export function describe(doc: mupdf.PDFDocument, annot: mupdf.PDFObject, text: string) {
annot.put("Contents", doc.newString(text));
annot.put(DESCRIBED, true);
}

// Drop the structure tree and what points into it. A page we tagged gets its
// original content back, so retagging our own output does not stack overlays.
function untag(doc: mupdf.PDFDocument): boolean {
let restored = false;
doc.getTrailer().get("Root").delete("StructTreeRoot");
for (let i = 0; i < doc.countPages(); i++) {
const page = doc.findPage(i), contents = page.get("Contents");
page.delete("StructParents");
page.get("Annots").forEach((a) => {
if (!a.isDictionary()) return;
a.delete("StructParent");
// A description we wrote is rewritten from the new HTML; the author's stays.
if (a.get(DESCRIBED).isBoolean()) { a.delete("Contents"); a.delete(DESCRIBED); }
});
if (!contents.isArray() || contents.length < 4) continue;
const body = (j: number) => { try { return contents.get(j).readStream().asString(); } catch { return null; } };
if (body(0) !== "/Artifact BMC q\n" || body(contents.length - 2) !== "\nQ EMC\n") continue;
const original: mupdf.PDFObject[] = [];
for (let j = 1; j < contents.length - 2; j++) original.push(contents.get(j));
page.put("Contents", original);
restored = true;
}
return restored;
}

// Every terminal field in the AcroForm tree.
export function forEachField(doc: mupdf.PDFDocument, fn: (field: mupdf.PDFObject) => void) {
const visit = (field: mupdf.PDFObject, depth: number) => {
Expand Down
10 changes: 6 additions & 4 deletions src/tag.ts
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
// tag(): the original PDF plus Iris's HTML -> the same PDF, tagged, with any
// form values filled in, checked to render identically (spec 搂7).
import * as mupdf from "mupdf";
import { inherited, openPdf, save, type OpenOptions } from "./pdf/document.ts";
import { describe, inherited, openPdf, save, type OpenOptions } from "./pdf/document.ts";
import { artifactStreams, drawsNothing, Overlay, pagesWithMcids } from "./pdf/content.ts";
import { FontSet, unembeddedFonts } from "./pdf/fonts.ts";
import { StructTree } from "./pdf/struct.ts";
Expand Down Expand Up @@ -38,6 +38,8 @@ const STRICT = ["no_title", "page_not_in_html", "unmatched_text", "missing_glyph
export function tag(pdf: Uint8Array, input: PagesInput, opts: TagOptions = {}, report: Report = newReport()): Uint8Array {
const src = openPdf(pdf, opts);
const { doc } = src;
// The old overlay's text is gone on purpose, so the checks compare against the source without it.
const baseline = src.restored ? save(doc, true) : pdf;
const warn = (w: Warning) => report.warnings.push(w);
src.warnings.forEach(warn);
const pageCount = doc.countPages();
Expand Down Expand Up @@ -117,7 +119,7 @@ export function tag(pdf: Uint8Array, input: PagesInput, opts: TagOptions = {}, r

const out = save(doc, src.repaired);
report.sizeIncreaseBytes = out.length - pdf.length;
if (opts.verify !== false) verify(pdf, out, opts, filled.changed, overlayText, report);
if (opts.verify !== false) verify(baseline, out, opts, filled.changed, overlayText, report);
const strict = report.warnings.filter((w) => STRICT.includes(w.code));
if (opts.strict && strict.length) {
throw new IrisPdfError("strict", `--strict: ${[...new Set(strict.map((w) => w.code))].join(", ")}`);
Expand Down Expand Up @@ -232,7 +234,7 @@ function tagPage(page: mupdf.PDFPage, i: number, html: string, ctx: PageCtx): {
const generic = !l.obj.get("Contents").isString() && !under && !l.uri;
const elem = ctx.struct.add(ctx.struct.top, "Link", generic && !ctx.lang.startsWith("en") ? { Lang: ctx.doc.newString("en") } : {});
ctx.struct.objr(elem, pageObj, l.obj);
if (l.obj.get("Contents").isNull()) l.obj.put("Contents", ctx.doc.newString(under || l.uri || "Link to another part of this document"));
if (l.obj.get("Contents").isNull()) describe(ctx.doc, l.obj, under || l.uri || "Link to another part of this document");
warn({ code: "unmatched_link", detail: l.uri || "internal link" });
}
for (const w of ctx.widgets.filter((w) => !w.used && !ctx.flatten)) {
Expand Down Expand Up @@ -323,7 +325,7 @@ function linkAnnotation(n: Node, elem: mupdf.PDFObject, e: Emitter) {
link.used = true;
e.struct.objr(elem, e.pageObj, link.obj);
const text = n.kids.flatMap((k) => (isRun(k) ? k.words.map((w) => w.text) : [])).join(" ");
if (link.obj.get("Contents").isNull() && text) link.obj.put("Contents", e.doc.newString(text));
if (link.obj.get("Contents").isNull() && text) describe(e.doc, link.obj, text);
}

// An <input>: a Form element owning its widgets, which get the label as /TU.
Expand Down
11 changes: 11 additions & 0 deletions test/cli.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,17 @@ test("a refusal exits 1, writes no PDF, and still writes the report", () => {
assert.equal(JSON.parse(readFileSync(report, "utf8")).error.code, "signed");
});

test("a tagged PDF is refused as already_tagged, and --retag tags it", () => {
const once = join(dir, "once.pdf"), twice = join(dir, "twice.pdf");
assert.equal(run(...tagArgs("text-simple", once)).code, 0);
const again = ["tag", "--pdf", once, "--pages", fixture("text-simple.pages.json"), "--out", twice];
const r = run(...again);
assert.equal(r.code, 1);
assert.match(r.err, /^iris-pdf: already_tagged: .*--retag/);
assert.equal(run(...again, "--retag").code, 0);
assert.ok(existsSync(twice));
});

test("a failed verification exits 2 and writes no PDF", () => {
// A stream ending inside a string swallows the overlay (see tag.test.ts).
const doc = new mupdf.PDFDocument(readFileSync(fixture("text-simple.pdf")));
Expand Down
56 changes: 51 additions & 5 deletions test/document.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,11 +7,11 @@ import { readFixture, pagesOf, tagFixture } from "./helpers.ts";
const simple: PagesInput = pagesOf("text-simple");

// Runs tag and returns the error code it refused with.
function refusal(pdf: Uint8Array, opts: TagOptions = {}, pages: PagesInput = simple): { code: string; exit: number } {
function refusal(pdf: Uint8Array, opts: TagOptions = {}, pages: PagesInput = simple): { code: string; exit: number; message: string } {
try {
tag(pdf, pages, opts);
} catch (e) {
return e as { code: string; exit: number };
return e as { code: string; exit: number; message: string };
}
assert.fail("tag did not refuse");
}
Expand Down Expand Up @@ -72,9 +72,55 @@ test("--allow-signed tags a signed PDF and says the signature is gone", () => {
assert.equal(report.source.signed, true);
});

test("refuses a PDF that is already tagged", () => {
const { out } = tagFixture("text-simple");
assert.equal(refusal(out).code, "already_tagged");
test("refuses a PDF that is already tagged, and retags it when asked", () => {
const first = tagFixture("text-simple");
assert.deepEqual([refusal(first.out).code, refusal(first.out).message], ["already_tagged", "The PDF is already tagged. Pass --retag to replace its tags."]);
const report = newReport();
const doc = new mupdf.PDFDocument(tag(first.out, simple, { retag: true }, report));
assert.ok(report.warnings.some((w) => w.code === "retagged"));
// Our own output gets its original content back: one overlay, not two, and the same tags.
assert.ok(!report.warnings.some((w) => w.code === "source_marked_content"));
assert.equal(doc.findPage(0).get("Contents").length, first.doc.findPage(0).get("Contents").length);
assert.deepEqual(report.structure, first.report.structure);
assert.equal(report.verification.textPreserved, true);
assert.equal(report.verification.differingPixels, 0);
});

test("--retag drops a foreign tag tree and what points into it", () => {
const pdf = blankPdf(1, (doc) => {
const page = doc.findPage(0);
doc.getTrailer().get("Root").put("StructTreeRoot", doc.addObject({ Type: "StructTreeRoot", K: [] }));
page.put("StructParents", 0);
page.put("Annots", [doc.addObject({ Type: "Annot", Subtype: "Text", Rect: [0, 0, 10, 10], StructParent: 1 })]);
});
const doc = new mupdf.PDFDocument(tag(pdf, { lang: "en", pages: [] }, { retag: true }));
const page = doc.findPage(0);
assert.ok(page.get("StructParents").isNull());
assert.ok(page.get("Annots").get(0).get("StructParent").isNull());
assert.equal(doc.getTrailer().get("Root", "StructTreeRoot", "K", "S").asName(), "Document", "a new tree");
});

test("--retag rewrites link descriptions we wrote, and keeps the author's", () => {
const html = (a: string) => ({ ...pagesOf("links"), pages: [{ sourcePage: 1, html: `<h1>Contact</h1><p>Apply online at ${a}.</p>` }] });
const contents = (pdf: Uint8Array) => new mupdf.PDFDocument(pdf).findPage(0).get("Annots").get(0).get("Contents").asString();
const once = tag(readFixture("links.pdf"), pagesOf("links"));
assert.equal(contents(once), "the city website");
assert.equal(contents(tag(once, html(`the city <a href="https://example.org/permits">website</a>`), { retag: true })), "website");
const authored = new mupdf.PDFDocument(readFixture("links.pdf"));
authored.findPage(0).get("Annots").get(0).put("Contents", authored.newString("Permits"));
const own = tag(tag(authored.saveToBuffer("").asUint8Array().slice(), pagesOf("links")), html(`the city <a href="https://example.org/permits">website</a>`), { retag: true });
assert.equal(contents(own), "Permits");
});

test("--retag on a foreign tagged PDF keeps its marked content, so it claims no PDF/UA", () => {
const doc = new mupdf.PDFDocument(readFixture("text-embedded.pdf"));
const page = doc.findPage(0);
page.put("Contents", doc.addStream("/P <</MCID 0>> BDC " + page.get("Contents").readStream().asString() + " EMC", {}));
doc.getTrailer().get("Root").put("StructTreeRoot", doc.addObject({ Type: "StructTreeRoot" }));
const report = newReport();
const out = new mupdf.PDFDocument(tag(doc.saveToBuffer("").asUint8Array().slice(), pagesOf("text-embedded"), { retag: true }, report));
assert.deepEqual(["retagged", "source_marked_content"].map((c) => report.warnings.some((w) => w.code === c)), [true, true]);
assert.doesNotMatch(out.getTrailer().get("Root", "Metadata").readStream().asString(), /pdfuaid:part/);
});

test("XFA: a dynamic form is refused, a hybrid form loses only its XFA", () => {
Expand Down
Loading