Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 1 addition & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -40,7 +40,6 @@ iris-pdf review --pdf out.pdf [--report review.json] # optional AI review, bel
| `--password` | Open an encrypted PDF. The output keeps its encryption. |
| `--allow-signed` | Tag a signed PDF. This breaks the signature, and the report says so. |
| `--partial` | Leave a page untagged, instead of failing, when it has no way to place text. |
| `--strict` | Fail on any warning that means content went untagged or unmatched, or that the file was `repaired`. |

### pages.json

Expand Down Expand Up @@ -92,7 +91,7 @@ The output declares PDF/UA-1 only when it has a title, every page is tagged, eve
| Exit | When |
|---|---|
| 0 | Done. |
| 1 | Refused: `encrypted` (no or wrong password), `permissions_denied`, `too_many_pages` (over 25), `too_many_words` (over 4000 on a page), `xfa` (dynamic form), `signed`, `no_acroform_field`, `no_text_positions`, `strict`. From `review`: `review_failed` (the model or its API failed on a page). |
| 1 | Refused: `encrypted` (no or wrong password), `permissions_denied`, `too_many_pages` (over 25), `too_many_words` (over 4000 on a page), `xfa` (dynamic form), `signed`, `no_acroform_field`, `no_text_positions`. From `review`: `review_failed` (the model or its API failed on a page). |
| 2 | A check failed: `pixels_changed`, `text_lost`. |
| 3 | Bad input: `unreadable`, `bad_pages`, `no_document_language`, `bad_value`, `field_not_settable`, `bad_arguments`. From `review`: `not_tagged`, `no_readable_structure` (not tagged by this tool), `bad_structure` (nested over 64 levels), `no_credentials`. |

Expand Down
6 changes: 3 additions & 3 deletions src/cli.ts
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@ const USAGE = `iris-pdf ${VERSION}
iris-pdf tag --pdf <in.pdf> --pages <pages.json> --out <out.pdf>
[--values <values.json>] [--report <report.json>] [--lang <bcp47>] [--title <text>]
[--ocr auto|off|required] [--verify pixels,text|off] [--verify-dpi 150]
[--flatten] [--password <pw>] [--allow-signed] [--partial] [--strict]
[--flatten] [--password <pw>] [--allow-signed] [--partial]
iris-pdf fields --pdf <in.pdf> [--json] [--password <pw>]
iris-pdf check --pdf <in.pdf>
iris-pdf review --pdf <tagged.pdf> [--report <review.json>] [--provider anthropic|bedrock] [--model <id>] [--password <pw>]`;
Expand All @@ -23,7 +23,7 @@ const OPTIONS = {
verify: { type: "string" }, "verify-dpi": { type: "string" }, flatten: { type: "boolean" },
password: { type: "string" }, "allow-signed": { type: "boolean" }, partial: { type: "boolean" },
retag: { type: "boolean" }, // ignored, for older callers: tagging always replaces old tags
strict: { type: "boolean" }, provider: { type: "string" }, model: { type: "string" }, json: { type: "boolean" }, help: { type: "boolean", short: "h" },
provider: { type: "string" }, model: { type: "string" }, json: { type: "boolean" }, help: { type: "boolean", short: "h" },
} as const;

function badArgs(message: string): never {
Expand Down Expand Up @@ -99,7 +99,7 @@ async function main(argv: string[]): Promise<number> {
values: args.values ? (readJson(args.values, "values") as TagOptions["values"]) : undefined,
lang: args.lang, title: args.title, ocr: ocr as TagOptions["ocr"],
verify: verify !== "off", verifyDpi: dpi, flatten: args.flatten, password: args.password,
allowSigned: args["allow-signed"], partial: args.partial, strict: args.strict,
allowSigned: args["allow-signed"], partial: args.partial,
};
const report = newReport();
const pdf = readPdf(args.pdf);
Expand Down
8 changes: 0 additions & 8 deletions src/tag.ts
Original file line number Diff line number Diff line change
Expand Up @@ -28,12 +28,8 @@ export type TagOptions = OpenOptions & {
verifyDpi?: number;
flatten?: boolean;
partial?: boolean;
strict?: boolean;
};

// With --strict these fail the run instead of only being reported.
const STRICT = ["no_title", "page_not_in_html", "unmatched_text", "missing_glyph", "missing_alt", "unmapped_element", "field_not_in_html", "field_not_in_pdf", "unmatched_link", "alignment_incomplete", "page_not_tagged", "repaired"];

// Throws IrisPdfError. `report` is filled in as far as the run got, either way.
export function tag(pdf: Uint8Array, input: PagesInput, opts: TagOptions = {}, report: Report = newReport()): Uint8Array {
const src = openPdf(pdf, opts);
Expand Down Expand Up @@ -121,10 +117,6 @@ export function tag(pdf: Uint8Array, input: PagesInput, opts: TagOptions = {}, r
const out = save(doc, src.repaired);
report.sizeIncreaseBytes = out.length - pdf.length;
if (opts.verify !== false) verify(pdf, baseline, out, opts, filled.changed, overlayText, report);
const strict = report.warnings.filter((w) => STRICT.includes(w.code));
if (opts.strict && strict.length) {
throw new IrisPdfError("strict", `--strict: ${[...new Set(strict.map((w) => w.code))].join(", ")}`);
}
return out;
}

Expand Down
1 change: 0 additions & 1 deletion test/document.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -59,7 +59,6 @@ test("a damaged PDF is rewritten from its repair, checked, and keeps its encrypt
assert.equal(report.verification.textPreserved, true);
assert.equal(report.verification.differingPixels, 0);
assert.ok(!new mupdf.PDFDocument(out).wasRepaired());
assert.throws(() => tag(damage(readFixture("text-simple.pdf")), simple, { strict: true }), { code: "strict", message: /repaired/ });
const locked = tag(damage(readFixture("encrypted.pdf")), simple, { password: "open" });
const back = new mupdf.PDFDocument(locked);
assert.ok(back.needsPassword() && back.authenticatePassword("open"));
Expand Down
10 changes: 3 additions & 7 deletions test/tag.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -138,7 +138,6 @@ test("text Iris left out is kept and reported, not dropped", () => {
assert.ok(report.pages[0].lost > 0);
assert.ok(report.warnings.some((w) => w.code === "unmatched_text"));
assert.match(readingOrder(structTree(doc)), /^Parking Permit Residents may apply/);
assert.throws(() => tag(readFixture("text-simple.pdf"), pages, { strict: true }), { code: "strict" });
});

test("the verification gate sees changed pixels and lost text", () => {
Expand Down Expand Up @@ -308,7 +307,6 @@ test("a page missing from pages.json is left untouched, warned, and stops the PD
assert.ok(page.get("StructParents").isNull());
assert.ok(report.warnings.some((w) => w.code === "page_not_in_html" && w.page === 2));
assert.doesNotMatch(doc.getTrailer().get("Root", "Metadata").readStream().asString(), /pdfuaid:part/);
assert.throws(() => tag(pdf, pagesOf("text-simple"), { strict: true }), { code: "strict" });
});

test("a page whose HTML holds nothing to tag is left as it was, not hidden as an artifact", () => {
Expand All @@ -320,11 +318,10 @@ test("a page whose HTML holds nothing to tag is left as it was, not hidden as an
assert.equal(doc.findPage(0).get("Contents").readStream().asString(), before);
assert.ok(report.warnings.some((w) => w.code === "page_not_tagged"));
assert.doesNotMatch(doc.getTrailer().get("Root", "Metadata").readStream().asString(), /pdfuaid:part/);
assert.throws(() => tag(pdf, pages, { strict: true }), { code: "strict" });
});

test("a blank page needs no HTML and does not cost the PDF/UA claim", () => {
const { doc, report } = tagFixture("blank-page", { strict: true });
const { doc, report } = tagFixture("blank-page");
assert.ok(!report.warnings.some((w) => w.code === "page_not_in_html"));
assert.match(doc.getTrailer().get("Root", "Metadata").readStream().asString(), /<pdfuaid:part>1/);

Expand All @@ -350,7 +347,7 @@ test("a blank page needs no HTML and does not cost the PDF/UA claim", () => {
edit(annots.get(0), d);
annots.push(null);
const quiet = newReport();
tag(d.saveToBuffer("").asUint8Array().slice(), pagesOf("blank-page"), { strict: true }, quiet);
tag(d.saveToBuffer("").asUint8Array().slice(), pagesOf("blank-page"), {}, quiet);
assert.ok(!quiet.warnings.some((w) => w.code === "page_not_in_html"), name);
}
});
Expand Down Expand Up @@ -383,15 +380,14 @@ test("an unmatched link is described by the words under it", () => {
assert.equal(link.objr[0].get("Contents").asString(), "Fees");
});

test("with no title there is no PDF/UA claim, no empty title, and --strict fails", () => {
test("with no title there is no PDF/UA claim, and no empty title", () => {
const pages = { ...pagesOf("text-simple"), title: undefined };
const report = newReport();
const doc = new mupdf.PDFDocument(tag(readFixture("text-simple.pdf"), pages, {}, report));
const root = doc.getTrailer().get("Root");
assert.ok(report.warnings.some((w) => w.code === "no_title"));
assert.ok(root.get("ViewerPreferences", "DisplayDocTitle").isNull());
assert.ok(root.get("Metadata").isNull(), "no packet to write");
assert.throws(() => tag(readFixture("text-simple.pdf"), pages, { strict: true }), { code: "strict" });
const packet = xmp('<x:xmpmeta><rdf:RDF><rdf:Description><dc:title>Old</dc:title><pdfuaid:part>1</pdfuaid:part></rdf:Description></rdf:RDF></x:xmpmeta>', "", false);
assert.doesNotMatch(packet, /pdfuaid:part>|<dc:title><rdf:Alt>/);
assert.match(packet, /Old/, "an old title stays when there is no new one");
Expand Down
Loading