From 3bf2de1ee94318bfe5a67e056c8db413e3d9a028 Mon Sep 17 00:00:00 2001 From: Blake Bertuccelli-Booth <46652+bbertucc@users.noreply.github.com> Date: Sun, 4 Oct 2026 09:40:55 -0400 Subject: [PATCH 1/4] Drop --ocr, --verify, --verify-dpi, --partial and --retag OCR is always auto, the checks always run, and a page with no way to place text is left untagged with a warning. Co-Authored-By: Claude Opus 5.5 --- README.md | 7 ++----- src/cli.ts | 20 +++++--------------- src/tag.ts | 22 ++++++---------------- test/cli.test.ts | 20 +++++++++++++------- test/network.test.ts | 2 +- test/review.test.ts | 2 +- test/tag.test.ts | 9 --------- 7 files changed, 28 insertions(+), 54 deletions(-) diff --git a/README.md b/README.md index d20bc14..91bcacf 100644 --- a/README.md +++ b/README.md @@ -34,12 +34,9 @@ iris-pdf review --pdf out.pdf [--report review.json] # optional AI review, bel |---|---| | `--values values.json` | Fill form fields (below). | | `--lang`, `--title` | Used when `pages.json` has none. A language is required. | -| `--ocr auto\|off\|required` | Use Tesseract for pages with no text layer. Default `auto`. | -| `--verify pixels,text\|off` | The checks below. On by default. `--verify-dpi` sets the render resolution (36–600, default 150). | | `--flatten` | Draw the field values into the page and remove the fields. | | `--password` | Open an encrypted PDF. The output keeps its encryption. | | `--allow-signed` | Tag a signed PDF. This breaks the signature, and the report says so. | -| `--partial` | Leave a page untagged, instead of failing, when it has no way to place text. | ### pages.json @@ -74,7 +71,7 @@ Then two checks run, and if either fails nothing is written (exit 2): ## The report -`--report` writes JSON: per page, where the text came from and how many words matched; the structure written; fields set and skipped; the check results; and every warning. Warnings name what could not be done, for example `unmatched_text` (page text missing from the HTML, kept as a paragraph), `missing_alt`, `field_not_in_html`, `unmatched_link`, `duplicate_text_layer`, `page_not_in_html` and `page_not_tagged` (the page is left as it was; a blank page needs no HTML and is not warned), `no_title`, `font_not_embedded` (a source font has no embedded program, which PDF/UA-1 requires; the source drawing is not changed), `source_marked_content` (the page drawing has marked-content ids left from an earlier tag tree), `alignment_incomplete` (the page and the HTML differ too much to match every word in time; the rest is kept as unmatched text). +`--report` writes JSON: per page, where the text came from and how many words matched; the structure written; fields set and skipped; the check results; and every warning. Warnings name what could not be done, for example `unmatched_text` (page text missing from the HTML, kept as a paragraph), `missing_alt`, `field_not_in_html`, `unmatched_link`, `duplicate_text_layer`, `page_not_in_html`, `page_not_tagged` and `no_text_positions` (a scan, with Tesseract not installed; the page is left as it was; a blank page needs no HTML and is not warned), `no_title`, `font_not_embedded` (a source font has no embedded program, which PDF/UA-1 requires; the source drawing is not changed), `source_marked_content` (the page drawing has marked-content ids left from an earlier tag tree), `alignment_incomplete` (the page and the HTML differ too much to match every word in time; the rest is kept as unmatched text). ## Review @@ -91,7 +88,7 @@ The output declares PDF/UA-1 only when it has a title, every page is tagged, eve | Exit | When | |---|---| | 0 | Done. | -| 1 | Refused: `encrypted` (no or wrong password), `permissions_denied`, `too_many_pages` (over 25), `too_many_words` (over 4000 on a page), `xfa` (dynamic form), `signed`, `no_acroform_field`, `no_text_positions`. From `review`: `review_failed` (the model or its API failed on a page). | +| 1 | Refused: `encrypted` (no or wrong password), `permissions_denied`, `too_many_pages` (over 25), `too_many_words` (over 4000 on a page), `xfa` (dynamic form), `signed`, `no_acroform_field`. From `review`: `review_failed` (the model or its API failed on a page). | | 2 | A check failed: `pixels_changed`, `text_lost`. | | 3 | Bad input: `unreadable`, `bad_pages`, `no_document_language`, `bad_value`, `field_not_settable`, `bad_arguments`. From `review`: `not_tagged`, `no_readable_structure` (not tagged by this tool), `bad_structure` (nested over 64 levels), `no_credentials`. | diff --git a/src/cli.ts b/src/cli.ts index 16157c0..52f9b00 100644 --- a/src/cli.ts +++ b/src/cli.ts @@ -11,18 +11,15 @@ const USAGE = `iris-pdf ${VERSION} iris-pdf tag --pdf --pages --out [--values ] [--report ] [--lang ] [--title ] - [--ocr auto|off|required] [--verify pixels,text|off] [--verify-dpi 150] - [--flatten] [--password ] [--allow-signed] [--partial] + [--flatten] [--password ] [--allow-signed] iris-pdf fields --pdf [--json] [--password ] iris-pdf check --pdf iris-pdf review --pdf [--report ] [--provider anthropic|bedrock] [--model ] [--password ]`; const OPTIONS = { pdf: { type: "string" }, pages: { type: "string" }, values: { type: "string" }, out: { type: "string" }, - report: { type: "string" }, lang: { type: "string" }, title: { type: "string" }, ocr: { type: "string" }, - verify: { type: "string" }, "verify-dpi": { type: "string" }, flatten: { type: "boolean" }, - password: { type: "string" }, "allow-signed": { type: "boolean" }, partial: { type: "boolean" }, - retag: { type: "boolean" }, // ignored, for older callers: tagging always replaces old tags + report: { type: "string" }, lang: { type: "string" }, title: { type: "string" }, flatten: { type: "boolean" }, + password: { type: "string" }, "allow-signed": { type: "boolean" }, provider: { type: "string" }, model: { type: "string" }, json: { type: "boolean" }, help: { type: "boolean", short: "h" }, } as const; @@ -88,18 +85,11 @@ async function main(argv: string[]): Promise { if (command !== "tag") badArgs(`Unknown command "${command}".\n${USAGE}`); if (!args.pages || !args.out) badArgs("tag needs --pdf, --pages and --out."); - const ocr = args.ocr ?? "auto"; - if (!["auto", "off", "required"].includes(ocr)) badArgs("--ocr is auto, off or required."); - const verify = args.verify ?? "pixels,text"; - if (verify !== "off" && verify !== "pixels,text") badArgs("--verify is pixels,text or off."); - const dpi = Number(args["verify-dpi"] ?? 150); - if (!(dpi >= 36 && dpi <= 600)) badArgs("--verify-dpi is between 36 and 600."); const opts: TagOptions = { values: args.values ? (readJson(args.values, "values") as TagOptions["values"]) : undefined, - lang: args.lang, title: args.title, ocr: ocr as TagOptions["ocr"], - verify: verify !== "off", verifyDpi: dpi, flatten: args.flatten, password: args.password, - allowSigned: args["allow-signed"], partial: args.partial, + lang: args.lang, title: args.title, flatten: args.flatten, password: args.password, + allowSigned: args["allow-signed"], }; const report = newReport(); const pdf = readPdf(args.pdf); diff --git a/src/tag.ts b/src/tag.ts index ca14708..2dad141 100644 --- a/src/tag.ts +++ b/src/tag.ts @@ -12,7 +12,7 @@ import { buildPage, isRun, wordsInOrder, type Node, type Placed, type Run, type import { joinHyphenated, type Box, type PageWord } from "./align/words.ts"; import { align, MAX_WORDS } from "./align/align.ts"; import { isFurniture } from "./align/classify.ts"; -import { ocrWords, tesseractInstalled } from "./ocr/tesseract.ts"; +import { ocrWords } from "./ocr/tesseract.ts"; import { comparePixels } from "./verify/pixels.ts"; import { compareText } from "./verify/text.ts"; import { EXIT, IrisPdfError, newReport, type PageReport, type Report, type Warning } from "./report.ts"; @@ -23,11 +23,7 @@ export type TagOptions = OpenOptions & { values?: Record; lang?: string; title?: string; - ocr?: "auto" | "off" | "required"; - verify?: boolean; - verifyDpi?: number; flatten?: boolean; - partial?: boolean; }; // Throws IrisPdfError. `report` is filled in as far as the run got, either way. @@ -52,8 +48,6 @@ export function tag(pdf: Uint8Array, input: PagesInput, opts: TagOptions = {}, r if (!lang) throw new IrisPdfError("no_document_language", "No document language. Iris gave none; pass --lang.", EXIT.badInput); const title = input.title || opts.title || doc.getMetaData("info:Title"); if (!title) warn({ code: "no_title", detail: "No document title. Pass --title." }); - const ocr = opts.ocr ?? "auto"; - if (ocr === "required" && !tesseractInstalled()) throw new IrisPdfError("no_text_positions", "Tesseract is not installed, and --ocr required was given."); // Forms: check every value, then set them, then flatten if asked. const fields = inventory(doc); @@ -86,7 +80,7 @@ export function tag(pdf: Uint8Array, input: PagesInput, opts: TagOptions = {}, r continue; } const r = tagPage(page, i, html.get(i)!, { - doc, struct, fonts, lang, ocr, partial: !!opts.partial, flatten: !!opts.flatten, warn, + doc, struct, fonts, lang, flatten: !!opts.flatten, warn, widgets: widgets.filter((w) => w.page === i), allWidgets: widgets, values, fields, }); report.pages.push(r.report); @@ -116,7 +110,7 @@ export function tag(pdf: Uint8Array, input: PagesInput, opts: TagOptions = {}, r const out = save(doc, src.repaired); report.sizeIncreaseBytes = out.length - pdf.length; - if (opts.verify !== false) verify(pdf, baseline, out, opts, filled.changed, overlayText, report); + verify(pdf, baseline, out, opts, filled.changed, overlayText, report); return out; } @@ -136,8 +130,6 @@ type PageCtx = { struct: StructTree; fonts: FontSet; lang: string; - ocr: "auto" | "off" | "required"; - partial: boolean; flatten: boolean; warn: (w: Warning) => void; widgets: (Widget & { box: Box; used: boolean })[]; // this page's @@ -158,11 +150,9 @@ function tagPage(page: mupdf.PDFPage, i: number, html: string, ctx: PageCtx): { let words: PageWord[] = textLayerWords(page); if (words.length) report.textSource = "pdf-text"; else if (ordered.length) { - const ocr = ctx.ocr === "off" ? null : ocrWords(page); + const ocr = ocrWords(page); if (!ocr) { - const why = ctx.ocr === "off" ? "OCR is off" : "Tesseract is not installed"; - if (!ctx.partial) throw new IrisPdfError("no_text_positions", `Page ${n} has no text layer and ${why}.`); - warn({ code: "no_text_positions", detail: `${why}; the page was left untagged.` }); + warn({ code: "no_text_positions", detail: "The page has no text layer and Tesseract is not installed; it was left untagged." }); return { report, overlay: null, untagged: true }; } words = ocr; @@ -431,7 +421,7 @@ function verify(pdf: Uint8Array, baseline: Uint8Array, out: Uint8Array, opts: Ta return d; }; const after = open(out); - const pixels = comparePixels(open(pdf), after, opts.verifyDpi ?? 150, changed); + const pixels = comparePixels(open(pdf), after, 150, changed); const text = compareText(open(baseline), after, added); report.verification = { pixels: pixels.length ? "failed" : "identical-outside-fields", diff --git a/test/cli.test.ts b/test/cli.test.ts index 26a6b47..f7ecb24 100644 --- a/test/cli.test.ts +++ b/test/cli.test.ts @@ -3,7 +3,7 @@ import assert from "node:assert/strict"; import { spawnSync } from "node:child_process"; import { existsSync, mkdtempSync, readFileSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; -import { join } from "node:path"; +import { dirname, join } from "node:path"; import * as mupdf from "mupdf"; import { fixture } from "./helpers.ts"; @@ -34,14 +34,21 @@ test("a refusal exits 1, writes no PDF, and still writes the report", () => { assert.equal(JSON.parse(readFileSync(report, "utf8")).error.code, "signed"); }); -test("a tagged PDF is retagged, and --retag is still accepted", () => { +test("a tagged PDF is retagged", () => { const once = join(dir, "once.pdf"), twice = join(dir, "twice.pdf"); assert.equal(run(...tagArgs("text-simple", once)).code, 0); const again = ["tag", "--pdf", once, "--pages", fixture("text-simple.pages.json"), "--out", twice]; assert.equal(run(...again).code, 0); - const plain = readFileSync(twice); - assert.equal(run(...again, "--retag").code, 0); - assert.ok(readFileSync(twice).equals(plain), "--retag changes nothing"); + assert.ok(existsSync(twice)); +}); + +test("without Tesseract, a scan is left untagged with a warning", () => { + const out = join(dir, "mixed.pdf"), report = join(dir, "mixed.json"); + const r = spawnSync(process.execPath, [cli, ...tagArgs("mixed", out, "--report", report)], { encoding: "utf8", env: { PATH: dirname(process.execPath) } }); + assert.equal(r.status, 0, r.stderr); + const json = JSON.parse(readFileSync(report, "utf8")); + assert.deepEqual(json.pages.map((p: { textSource: string }) => p.textSource), ["pdf-text", "none"]); + assert.ok(json.warnings.some((w: { code: string; page?: number }) => w.code === "no_text_positions" && w.page === 2)); }); test("a failed verification exits 2 and writes no PDF", () => { @@ -60,8 +67,7 @@ test("a failed verification exits 2 and writes no PDF", () => { test("bad arguments and bad values exit 3; values never appear in output", () => { assert.equal(run().code, 3); assert.equal(run("tag", "--nope").code, 3); - assert.equal(run(...tagArgs("text-simple", join(dir, "x.pdf"), "--ocr", "maybe")).code, 3); - assert.equal(run(...tagArgs("text-simple", join(dir, "x.pdf"), "--verify-dpi", "5")).code, 3); + for (const gone of ["--ocr", "--verify", "--verify-dpi", "--partial", "--retag", "--strict"]) assert.equal(run(...tagArgs("text-simple", join(dir, "x.pdf"), gone)).code, 3, gone); assert.equal(run("frobnicate").code, 3); const secret = "SSN-123-45-6789-" + "x".repeat(30); diff --git a/test/network.test.ts b/test/network.test.ts index d46e7e9..b41cd0d 100644 --- a/test/network.test.ts +++ b/test/network.test.ts @@ -23,6 +23,6 @@ globalThis.fetch = blocked; test("tags every fixture with the network blocked", () => { for (const name of ["text-simple", "text-two-column", "links", "form-acroform", "cjk", "blank-page", "mixed", "scan-300dpi"]) { - assert.ok(tagFixture(name, { partial: true }).out.length, name); // partial: scans pass without Tesseract too + assert.ok(tagFixture(name).out.length, name); } }); diff --git a/test/review.test.ts b/test/review.test.ts index 3417cc0..8398c23 100644 --- a/test/review.test.ts +++ b/test/review.test.ts @@ -351,7 +351,7 @@ test("an internal link shows its target page, or its named destination", () => { test("the outline shows merged table cells", () => { const html = '
ZoneFees
North1020
'; - const out = tag(readFixture("text-simple.pdf"), { pages: [{ sourcePage: 1, html }], lang: "en" }, { partial: true }); + const out = tag(readFixture("text-simple.pdf"), { pages: [{ sourcePage: 1, html }], lang: "en" }); const outline = pageOutline(structTree(mupdf.PDFDocument.openDocument(out, "application/pdf") as mupdf.PDFDocument), 0); assert.match(outline, /^ {6}TH ID="p1-th2" Scope=Column ColSpan=2 "Fees"$/m); }); diff --git a/test/tag.test.ts b/test/tag.test.ts index f81821e..4d5f7af 100644 --- a/test/tag.test.ts +++ b/test/tag.test.ts @@ -122,15 +122,6 @@ test("cjk: every character is recoverable through ToUnicode", () => { assert.equal(cid.get("CIDToGIDMap").asName(), "Identity"); }); -test("a scan with OCR off is refused, or left untagged with --partial", () => { - const pdf = readFixture("mixed.pdf"), pages = pagesOf("mixed"); - assert.throws(() => tag(pdf, pages, { ocr: "off" }), { code: "no_text_positions" }); - const report = newReport(); - tag(pdf, pages, { ocr: "off", partial: true }, report); - assert.deepEqual(report.pages.map((p) => p.textSource), ["pdf-text", "none"]); - assert.ok(report.warnings.some((w) => w.code === "no_text_positions" && w.page === 2)); -}); - test("text Iris left out is kept and reported, not dropped", () => { const pages = { lang: "en", pages: [{ sourcePage: 1, html: "

Parking Permit

" }] }; const report = newReport(); From bf45a0d90c845d1a3f731e1b571fb9fb4687b171 Mon Sep 17 00:00:00 2001 From: Blake Bertuccelli-Booth <46652+bbertucc@users.noreply.github.com> Date: Sun, 4 Oct 2026 09:47:17 -0400 Subject: [PATCH 2/4] Leave a scan untagged when Tesseract fails, too; hide Tesseract reliably in the test Co-Authored-By: Claude Opus 5.5 --- README.md | 4 +++- src/ocr/tesseract.ts | 8 ++++---- src/tag.ts | 4 ++-- test/cli.test.ts | 21 ++++++++++++--------- 4 files changed, 21 insertions(+), 16 deletions(-) diff --git a/README.md b/README.md index 91bcacf..6375d4c 100644 --- a/README.md +++ b/README.md @@ -71,7 +71,9 @@ Then two checks run, and if either fails nothing is written (exit 2): ## The report -`--report` writes JSON: per page, where the text came from and how many words matched; the structure written; fields set and skipped; the check results; and every warning. Warnings name what could not be done, for example `unmatched_text` (page text missing from the HTML, kept as a paragraph), `missing_alt`, `field_not_in_html`, `unmatched_link`, `duplicate_text_layer`, `page_not_in_html`, `page_not_tagged` and `no_text_positions` (a scan, with Tesseract not installed; the page is left as it was; a blank page needs no HTML and is not warned), `no_title`, `font_not_embedded` (a source font has no embedded program, which PDF/UA-1 requires; the source drawing is not changed), `source_marked_content` (the page drawing has marked-content ids left from an earlier tag tree), `alignment_incomplete` (the page and the HTML differ too much to match every word in time; the rest is kept as unmatched text). +A page that could not be tagged is left as it was, with a warning, and the run still exits 0. Read the report to catch it; the output then makes no PDF/UA-1 claim. + +`--report` writes JSON: per page, where the text came from and how many words matched; the structure written; fields set and skipped; the check results; and every warning. Warnings name what could not be done, for example `unmatched_text` (page text missing from the HTML, kept as a paragraph), `missing_alt`, `field_not_in_html`, `unmatched_link`, `duplicate_text_layer`, `page_not_in_html` (a blank page needs no HTML and is not warned), `page_not_tagged`, `no_text_positions` (a scan, with Tesseract missing or failing), `no_title`, `font_not_embedded` (a source font has no embedded program, which PDF/UA-1 requires; the source drawing is not changed), `source_marked_content` (the page drawing has marked-content ids left from an earlier tag tree), `alignment_incomplete` (the page and the HTML differ too much to match every word in time; the rest is kept as unmatched text). ## Review diff --git a/src/ocr/tesseract.ts b/src/ocr/tesseract.ts index 42b5525..2117ccf 100644 --- a/src/ocr/tesseract.ts +++ b/src/ocr/tesseract.ts @@ -12,13 +12,13 @@ export function tesseractInstalled(): boolean { return installed; } -// null when Tesseract is not installed. -export function ocrWords(page: mupdf.PDFPage, lang = "eng"): PageWord[] | null { - if (!tesseractInstalled()) return null; +// A string says why there are no words: Tesseract is missing or failed. +export function ocrWords(page: mupdf.PDFPage, lang = "eng"): PageWord[] | string { + if (!tesseractInstalled()) return "Tesseract is not installed"; const s = DPI / 72; const png = page.toPixmap(mupdf.Matrix.scale(s, s), mupdf.ColorSpace.DeviceGray, false).asPNG(); const run = spawnSync("tesseract", ["stdin", "stdout", "-l", lang, "tsv"], { input: png, maxBuffer: 64 << 20 }); - if (run.status !== 0) throw new Error(`tesseract failed: ${run.stderr.toString().trim()}`); + if (run.status !== 0) return `Tesseract failed: ${run.stderr?.toString().trim() || run.error?.message}`; return parseTsv(run.stdout.toString(), s); } diff --git a/src/tag.ts b/src/tag.ts index 2dad141..4bb29bb 100644 --- a/src/tag.ts +++ b/src/tag.ts @@ -151,8 +151,8 @@ function tagPage(page: mupdf.PDFPage, i: number, html: string, ctx: PageCtx): { if (words.length) report.textSource = "pdf-text"; else if (ordered.length) { const ocr = ocrWords(page); - if (!ocr) { - warn({ code: "no_text_positions", detail: "The page has no text layer and Tesseract is not installed; it was left untagged." }); + if (typeof ocr === "string") { + warn({ code: "no_text_positions", detail: `The page has no text layer and ${ocr}; it was left untagged.` }); return { report, overlay: null, untagged: true }; } words = ocr; diff --git a/test/cli.test.ts b/test/cli.test.ts index f7ecb24..4a47770 100644 --- a/test/cli.test.ts +++ b/test/cli.test.ts @@ -3,7 +3,7 @@ import assert from "node:assert/strict"; import { spawnSync } from "node:child_process"; import { existsSync, mkdtempSync, readFileSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; -import { dirname, join } from "node:path"; +import { join } from "node:path"; import * as mupdf from "mupdf"; import { fixture } from "./helpers.ts"; @@ -39,16 +39,19 @@ test("a tagged PDF is retagged", () => { assert.equal(run(...tagArgs("text-simple", once)).code, 0); const again = ["tag", "--pdf", once, "--pages", fixture("text-simple.pages.json"), "--out", twice]; assert.equal(run(...again).code, 0); - assert.ok(existsSync(twice)); }); -test("without Tesseract, a scan is left untagged with a warning", () => { - const out = join(dir, "mixed.pdf"), report = join(dir, "mixed.json"); - const r = spawnSync(process.execPath, [cli, ...tagArgs("mixed", out, "--report", report)], { encoding: "utf8", env: { PATH: dirname(process.execPath) } }); - assert.equal(r.status, 0, r.stderr); - const json = JSON.parse(readFileSync(report, "utf8")); - assert.deepEqual(json.pages.map((p: { textSource: string }) => p.textSource), ["pdf-text", "none"]); - assert.ok(json.warnings.some((w: { code: string; page?: number }) => w.code === "no_text_positions" && w.page === 2)); +test("with Tesseract missing or failing, a scan is left untagged with a warning", () => { + const failing = mkdtempSync(join(tmpdir(), "iris-pdf-bin-")); + writeFileSync(join(failing, "tesseract"), '#!/bin/sh\n[ "$1" = --version ] && exit 0\necho no eng >&2; exit 1\n', { mode: 0o755 }); + for (const [PATH, why] of [["", /not installed/], [failing, /failed: no eng/]] as const) { + const out = join(dir, "mixed.pdf"), report = join(dir, "mixed.json"); + const r = spawnSync(process.execPath, [cli, ...tagArgs("mixed", out, "--report", report)], { encoding: "utf8", env: { PATH } }); + assert.equal(r.status, 0, r.stderr); + const json = JSON.parse(readFileSync(report, "utf8")); + assert.deepEqual(json.pages.map((p: { textSource: string }) => p.textSource), ["pdf-text", "none"]); + assert.ok(json.warnings.some((w: { code: string; page?: number; detail: string }) => w.code === "no_text_positions" && w.page === 2 && why.test(w.detail))); + } }); test("a failed verification exits 2 and writes no PDF", () => { From 3fb4e0bae405605372ccdc96efee1cb741c79ce8 Mon Sep 17 00:00:00 2001 From: Blake Bertuccelli-Booth <46652+bbertucc@users.noreply.github.com> Date: Sun, 4 Oct 2026 09:51:53 -0400 Subject: [PATCH 3/4] Name the exit status when Tesseract fails silently Co-Authored-By: Claude Opus 5.5 --- src/ocr/tesseract.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocr/tesseract.ts b/src/ocr/tesseract.ts index 2117ccf..63a8262 100644 --- a/src/ocr/tesseract.ts +++ b/src/ocr/tesseract.ts @@ -18,7 +18,7 @@ export function ocrWords(page: mupdf.PDFPage, lang = "eng"): PageWord[] | string const s = DPI / 72; const png = page.toPixmap(mupdf.Matrix.scale(s, s), mupdf.ColorSpace.DeviceGray, false).asPNG(); const run = spawnSync("tesseract", ["stdin", "stdout", "-l", lang, "tsv"], { input: png, maxBuffer: 64 << 20 }); - if (run.status !== 0) return `Tesseract failed: ${run.stderr?.toString().trim() || run.error?.message}`; + if (run.status !== 0) return `Tesseract failed: ${run.stderr?.toString().trim() || run.error?.message || `exit ${run.status}`}`; return parseTsv(run.stdout.toString(), s); } From a6c7d4795237f4c0315783be64b6a555496b8e1d Mon Sep 17 00:00:00 2001 From: Blake Bertuccelli-Booth <46652+bbertucc@users.noreply.github.com> Date: Sun, 4 Oct 2026 09:55:19 -0400 Subject: [PATCH 4/4] Name the signal when Tesseract is killed Co-Authored-By: Claude Opus 5.5 --- src/ocr/tesseract.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ocr/tesseract.ts b/src/ocr/tesseract.ts index 63a8262..cc3f12c 100644 --- a/src/ocr/tesseract.ts +++ b/src/ocr/tesseract.ts @@ -18,7 +18,7 @@ export function ocrWords(page: mupdf.PDFPage, lang = "eng"): PageWord[] | string const s = DPI / 72; const png = page.toPixmap(mupdf.Matrix.scale(s, s), mupdf.ColorSpace.DeviceGray, false).asPNG(); const run = spawnSync("tesseract", ["stdin", "stdout", "-l", lang, "tsv"], { input: png, maxBuffer: 64 << 20 }); - if (run.status !== 0) return `Tesseract failed: ${run.stderr?.toString().trim() || run.error?.message || `exit ${run.status}`}`; + if (run.status !== 0) return `Tesseract failed: ${run.stderr?.toString().trim() || run.error?.message || run.signal || `exit ${run.status}`}`; return parseTsv(run.stdout.toString(), s); }