From ef780ff1bf82213420f03d8564d46bdab0ae6009 Mon Sep 17 00:00:00 2001 From: rfntnms Date: Fri, 25 Sep 2026 10:52:09 +0700 Subject: [PATCH 1/5] Add optional Unlimited-OCR backend for scanned PDFs Adds --ocr-engine unlimited, which renders each page with pypdfium2 and parses it with Baidu's Unlimited-OCR model via transformers. Detection tags are stripped and image blocks are cropped into images/ so the existing EPUB step works unchanged. marker remains the default. Measured on an RTX 4060 (8 GB): ~60 s per dense page, peak ~7 GB VRAM, about 4x slower than marker. The advertised speed needs vLLM/SGLang. Co-Authored-By: Claude Opus 5.5 --- README.md | 24 +++++ main.py | 13 ++- modules/unlimited_ocr.py | 207 +++++++++++++++++++++++++++++++++++++++ 3 files changed, 243 insertions(+), 1 deletion(-) create mode 100644 modules/unlimited_ocr.py diff --git a/README.md b/README.md index 16c6b4a..ad61795 100644 --- a/README.md +++ b/README.md @@ -133,6 +133,7 @@ python main.py [input_path] [output_path] [options] Options: --max-pages INT Maximum number of pages to process --start-page INT Page number to start from + --ocr-engine ENGINE marker (default) or unlimited --skip-epub Skip EPUB generation, only create markdown --skip-md Skip markdown generation, use existing markdown files ``` @@ -151,6 +152,29 @@ Convert to markdown only: python main.py thesis.pdf --skip-epub ``` +### Alternative OCR engine: Unlimited-OCR + +For scanned PDFs, `--ocr-engine unlimited` uses Baidu's +[Unlimited-OCR](https://github.com/baidu/Unlimited-OCR) vision-language model +instead of marker. Every page is rendered and parsed by the model, so it is not +worth using on digital PDFs, where marker reads the embedded text directly. + +- Requires an NVIDIA GPU with a CUDA build of PyTorch (no CPU or MPS support, + so it does not work in the CPU Docker image). Peak VRAM is about 7 GB, so an + 8 GB card is tight; close other GPU-heavy apps. +- It is slow through transformers: about 60 s per dense page on an RTX 4060, + roughly 4x slower than marker. The speed Baidu advertises comes from serving + the model with vLLM or SGLang on Linux, which is not wired in here. +- Needs extra packages: `pip install addict easydict matplotlib` +- The model (~6.7 GB) is downloaded from HuggingFace on first run. It is + loaded with `trust_remote_code=True`, which runs Python code from that + HuggingFace repository. +- Images are cropped from the page render at 200 DPI and saved to `images/`. + +```bash +python main.py scanned_book.pdf --ocr-engine unlimited +``` + ### Output Structure ``` diff --git a/main.py b/main.py index 14a0f74..627513f 100755 --- a/main.py +++ b/main.py @@ -43,6 +43,13 @@ def main(): default=None, help='Page number to start from' ) + parser.add_argument( + '--ocr-engine', + choices=['marker', 'unlimited'], + default='marker', + help='OCR backend: marker (default) or unlimited (Baidu Unlimited-OCR, ' + 'NVIDIA GPU only, meant for scanned PDFs)' + ) parser.add_argument( '--skip-epub', action='store_true', @@ -88,7 +95,11 @@ def main(): # Convert PDF to Markdown unless skipped if not args.skip_md: print("Converting PDF to Markdown...") - pdf2md.convert_pdf( + if args.ocr_engine == 'unlimited': + import modules.unlimited_ocr as converter + else: + converter = pdf2md + converter.convert_pdf( str(pdf_path), markdown_dir, args.max_pages, diff --git a/modules/unlimited_ocr.py b/modules/unlimited_ocr.py new file mode 100644 index 0000000..dd82a76 --- /dev/null +++ b/modules/unlimited_ocr.py @@ -0,0 +1,207 @@ +""" +Alternative OCR backend using Baidu's Unlimited-OCR vision-language model +(https://github.com/baidu/Unlimited-OCR). + +Every page is rendered to an image and parsed by the model, so this backend is +aimed at scanned PDFs. It requires an NVIDIA GPU with CUDA; the model weights +(~6.7 GB, bfloat16) are downloaded from HuggingFace on first use. + +This runs the model through plain transformers, which is slow: about 60 s per +dense page on an 8 GB RTX 4060. The speed Baidu advertises comes from serving +the model with vLLM or SGLang on Linux, which this module does not do. +""" +from pathlib import Path +import json +import re +import sys +import tempfile +import time + +MODEL_NAME = "baidu/Unlimited-OCR" + +# Rendering resolution for scanned pages. The model resizes internally +# (base_size=1024 with 640px crops), so higher DPI mostly helps image crops. +RENDER_DPI = 200 + +# One detected block: "<|det|>category [x1, y1, x2, y2]<|/det|>content". +# Coordinates are normalised to 0-999 relative to the page image. +DET_RE = re.compile( + r"<\|det\|>\s*([A-Za-z_][\w-]*)\s*(\[[^\]]*\])?\s*<\|/det\|>(.*)", re.DOTALL +) +REF_RE = re.compile(r"<\|/?ref\|>") +BBOX_RE = re.compile(r"-?\d+(?:\.\d+)?") + +_model = None +_tokenizer = None + + +def _load_model(): + """Load the model once and reuse it for every PDF in the queue.""" + global _model, _tokenizer + if _model is not None: + return _model, _tokenizer + + import torch + from transformers import AutoModel, AutoTokenizer + + if not torch.cuda.is_available(): + raise RuntimeError( + "Unlimited-OCR requires an NVIDIA GPU with CUDA; " + "use the default marker engine on CPU." + ) + + print(f"Loading {MODEL_NAME} (downloaded on first run)...") + _tokenizer = AutoTokenizer.from_pretrained(MODEL_NAME, trust_remote_code=True) + _model = AutoModel.from_pretrained( + MODEL_NAME, + trust_remote_code=True, + use_safetensors=True, + dtype=torch.bfloat16, + ) + _model = _model.eval().cuda() + return _model, _tokenizer + + +def _page_indices(page_count: int, max_pages: int = None, start_page: int = None) -> list[int]: + start = start_page or 0 + end = page_count if max_pages is None else min(page_count, start + max_pages) + return list(range(start, end)) + + +def _parse_bbox(raw: str, width: int, height: int): + nums = [float(n) for n in BBOX_RE.findall(raw or "")] + if len(nums) < 4: + return None + x1, y1, x2, y2 = nums[:4] + box = ( + int(x1 / 999 * width), + int(y1 / 999 * height), + int(x2 / 999 * width), + int(y2 / 999 * height), + ) + if box[2] <= box[0] or box[3] <= box[1]: + return None + return box + + +def page_output_to_markdown(raw: str, page_image, image_dir: Path, page_no: int) -> str: + """ + Turn the model's tagged output for one page into plain markdown. + + Detection tags are stripped; blocks labelled "image" are cropped from the + page render, saved to image_dir and referenced as images/.jpg. + """ + raw = raw.replace("<|end▁of▁sentence|>", "") + width, height = page_image.size + blocks = [] + cur = None + img_idx = 0 + + for line in raw.splitlines(): + line = REF_RE.sub("", line).rstrip() + if not line: + continue + m = DET_RE.match(line) + if m: + category, bbox, content = m.group(1), m.group(2), m.group(3).strip() + if cur is not None: + blocks.append(cur) + cur = None + if category == "image": + box = _parse_bbox(bbox, width, height) + if box: + image_dir.mkdir(parents=True, exist_ok=True) + name = f"page{page_no:04d}_img{img_idx}.jpg" + page_image.crop(box).convert("RGB").save(image_dir / name, quality=90) + blocks.append([f"![](images/{name})"]) + img_idx += 1 + continue + cur = [content] if content else [] + continue + if cur is None: + cur = [] + cur.append(line) + if cur is not None: + blocks.append(cur) + + text = "\n\n".join("\n".join(b) for b in blocks if b) + return text.replace("\\coloneqq", ":=").replace("\\eqqcolon", "=:").strip() + + +def convert_pdf( + input_path: str, + output_dir: Path, + max_pages: int = None, + start_page: int = None, +) -> None: + """Convert a PDF to markdown page by page with Unlimited-OCR.""" + import pypdfium2 as pdfium + + try: + model, tokenizer = _load_model() + + output_dir.mkdir(parents=True, exist_ok=True) + image_dir = output_dir / "images" + stem = Path(input_path).stem + + pdf = pdfium.PdfDocument(input_path) + pages = _page_indices(len(pdf), max_pages, start_page) + page_texts = [] + started = time.time() + + with tempfile.TemporaryDirectory(prefix="unlimited_ocr_") as tmp: + tmp_dir = Path(tmp) + for n, idx in enumerate(pages, 1): + page_started = time.time() + page = pdf[idx] + page_image = page.render(scale=RENDER_DPI / 72).to_pil() + page.close() + page_path = tmp_dir / f"page_{idx + 1:04d}.png" + page_image.save(page_path) + + # eval_mode=True returns the raw text instead of streaming it + # to stdout and writing result.md into output_path. + raw = model.infer( + tokenizer, + prompt="document parsing.", + image_file=str(page_path), + output_path=str(tmp_dir / "infer"), + base_size=1024, + image_size=640, + crop_mode=True, + max_length=32768, + no_repeat_ngram_size=35, + ngram_window=128, + eval_mode=True, + ) + page_texts.append( + page_output_to_markdown(raw or "", page_image, image_dir, idx + 1) + ) + page_image.close() + print( + f" page {idx + 1} ({n}/{len(pages)}) " + f"in {time.time() - page_started:.1f}s" + ) + + pdf.close() + + md_output = output_dir / f"{stem}.md" + md_output.write_text("\n\n".join(t for t in page_texts if t), encoding="utf-8") + print(f"Markdown saved to: {md_output}") + + elapsed = time.time() - started + metadata = { + "ocr_engine": MODEL_NAME, + "pages": [i + 1 for i in pages], + "seconds_total": round(elapsed, 1), + "seconds_per_page": round(elapsed / max(len(pages), 1), 2), + } + meta_output = output_dir / f"{stem}_metadata.json" + with open(meta_output, "w", encoding="utf-8") as f: + json.dump(metadata, f, indent=2) + print(f"Metadata saved to: {meta_output}") + print(f"OCR finished: {len(pages)} pages in {elapsed:.1f}s") + + except Exception as e: + print(f"Error converting {input_path}: {str(e)}", file=sys.stderr) + raise From da1217c011c75d10d05a6b73697e5a252ab61828 Mon Sep 17 00:00:00 2001 From: rfntnms Date: Fri, 25 Sep 2026 10:56:12 +0700 Subject: [PATCH 2/5] Add handover notes for continuing on Fedora with vLLM Co-Authored-By: Claude Opus 5.5 --- HANDOVER.md | 201 ++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 201 insertions(+) create mode 100644 HANDOVER.md diff --git a/HANDOVER.md b/HANDOVER.md new file mode 100644 index 0000000..a004899 --- /dev/null +++ b/HANDOVER.md @@ -0,0 +1,201 @@ +# Handover: Unlimited-OCR on Fedora with vLLM + +Written 2026-09-25 when the work moved from Windows to a Fedora machine. +Claude Code on Fedora should read this whole file first, then do the tasks in +order. Ask the user before anything that needs `sudo`, a reboot, a push, or a +merge into `main`. + +## Why this exists + +The user converts **scanned** books with this project. The slow step is +marker's "Recognizing Text" (Surya OCR). They want to try Baidu's +[Unlimited-OCR](https://github.com/baidu/Unlimited-OCR) because it is said to +be faster. + +On Windows it was not faster. Baidu's speed claims come from serving the model +with **vLLM** (or SGLang), which only runs on Linux. The goal on Fedora is to +serve the model with vLLM and see whether that beats marker. + +## Current state (branch `unlimited-ocr-backend`) + +- `main.py` has `--ocr-engine marker|unlimited`. The default is `marker`, and + it is unchanged. +- `modules/unlimited_ocr.py` runs the model in-process through + `transformers` with `trust_remote_code=True`: + - it renders each page with `pypdfium2`, which marker already installs, at + 200 DPI + - it calls `model.infer(..., eval_mode=True)` once per page, which returns + the raw text + - `page_output_to_markdown()` removes the `<|det|>category [x1,y1,x2,y2]<|/det|>` + tags. Coordinates are normalised to 0–999. It crops `image` blocks into + `images/pageNNNN_imgK.jpg` and writes a markdown image link for each. + - it writes `.md` and `_metadata.json` with the timings, laid + out so the existing EPUB step (`modules/mark2epub.py`) works unchanged +- `README.md` documents the engine. + +### Measured on Windows (RTX 4060, 8 GB VRAM) + +These numbers come from a synthetic 3-page scanned PDF with Indonesian text. + +| Engine | Time per page | Notes | +|---|---|---| +| marker (Surya) | ~15 s | 44 s total for 3 pages, including model load | +| Unlimited-OCR via transformers | ~62 s | ~13 tokens/s, peak 7.05 GiB VRAM, ~1,400 output tokens for a dense page | + +The OCR output was good. It read the text correctly, found the heading and the +page number, and did not loop. The speed was the only problem. + +## Tasks + +### 1. Check the machine + +```bash +cat /etc/fedora-release; uname -r +nvidia-smi # driver present? GPU model and VRAM? +python3 --version; which uv podman docker +``` + +If `nvidia-smi` is missing, the NVIDIA driver is not installed. On Fedora it +comes from RPM Fusion (`akmod-nvidia`, `xorg-x11-drv-nvidia-cuda`). It needs +`sudo`, a kernel module build and a reboot. Give the user the commands and +wait for them to run them. Do not run them yourself. + +### 2. Python environment for this repo + +`requirements.txt` recommends Python 3.13. marker-pdf pins `Pillow<11`, and +there are no Pillow wheels for 3.14. Recent Fedora releases default to 3.14, +so use `uv` to get 3.13: + +```bash +uv venv --python 3.13 .venv && source .venv/bin/activate +uv pip install -r requirements.txt +uv pip install addict easydict matplotlib # needed by the model's remote code +python -c "import torch; print(torch.__version__, torch.cuda.is_available())" +``` + +If CUDA shows `False`, reinstall torch from the CUDA index that matches the +driver (see pytorch.org). Keep the torch version compatible with marker-pdf. + +### 3. Baseline tests (before vLLM) + +Ask the user for a real scanned PDF. If there is none, make a synthetic one: +render a few A4 pages of text and a picture with Pillow at 200 DPI, add a +slight rotation and noise, and save them as an image-only PDF. + +Always pass `--skip-epub`. The EPUB step asks for metadata interactively and +fails with `EOFError` when there is no terminal. + +```bash +time python main.py scan.pdf out_marker --skip-epub --max-pages 3 +time python main.py scan.pdf out_unlimited --skip-epub --max-pages 3 --ocr-engine unlimited +ruff check --select E9,F . # this is the only check CI runs; there is no test suite +``` + +Record the time per page for each engine. The unlimited engine prints a line +per page and saves the timings in `_metadata.json`. + +### 4. Serve Unlimited-OCR with vLLM + +The architecture is not in the stable vLLM wheel. Use the dedicated image +`vllm/vllm-openai:unlimited-ocr`. The official recipe is at +https://recipes.vllm.ai/baidu/Unlimited-OCR, and it uses Docker: + +```bash +docker run --rm --gpus all --network host --ipc host \ + vllm/vllm-openai:unlimited-ocr \ + baidu/Unlimited-OCR \ + --trust-remote-code \ + --logits_processors vllm.model_executor.models.unlimited_ocr:NGramPerReqLogitsProcessor \ + --no-enable-prefix-caching \ + --mm-processor-cache-gb 0 +``` + +Fedora ships Podman rather than Docker. For GPU access, install +`nvidia-container-toolkit` (needs sudo; ask the user) and generate the CDI +spec with `sudo nvidia-ctk cdi generate --output=/etc/cdi/nvidia.yaml`. Then +run: + +```bash +podman run --rm --device nvidia.com/gpu=all --network host --ipc host \ + -v ~/.cache/huggingface:/root/.cache/huggingface:Z \ + docker.io/vllm/vllm-openai:unlimited-ocr \ + baidu/Unlimited-OCR \ + --trust-remote-code \ + --logits_processors vllm.model_executor.models.unlimited_ocr:NGramPerReqLogitsProcessor \ + --no-enable-prefix-caching \ + --mm-processor-cache-gb 0 \ + --gpu-memory-utilization 0.90 --max-model-len 8192 +``` + +- The volume mount caches the ~6.7 GB of weights between runs. `:Z` is needed + because of SELinux. +- The recipe says 8 GB of VRAM is the minimum. If the GPU is that small, + `--max-model-len 8192` and `--gpu-memory-utilization` keep the KV cache from + running out of memory. Adjust them if it still fails. One dense page is + about 1,400 output tokens. +- vLLM reserves most of the GPU. Stop the server before running marker or the + transformers engine. +- The server listens on `http://localhost:8000/v1`. Check it with + `curl localhost:8000/v1/models`. + +Quick check with a single page. The request must follow the recipe exactly, +otherwise the output comes back empty or loops: + +```python +from openai import OpenAI +client = OpenAI(api_key="EMPTY", base_url="http://localhost:8000/v1", timeout=3600) +r = client.chat.completions.create( + model="baidu/Unlimited-OCR", + messages=[{"role": "user", "content": [ + {"type": "text", "text": "document parsing."}, # must start with + {"type": "image_url", "image_url": {"url": "data:image/png;base64,<...>"}}, + ]}], + max_tokens=8192, temperature=0.0, + extra_body={"skip_special_tokens": False, # keep the <|det|> tags + "vllm_xargs": {"ngram_size": 35, "window_size": 128}}, +) +``` + +### 5. Add a `vllm` engine to the program + +Add `--ocr-engine vllm` and `--vllm-url` (default `http://localhost:8000/v1`). + +- Put the client in a new module or next to the existing code. Share the page + rendering and `page_output_to_markdown()` with the transformers engine + instead of copying them. +- The vllm engine must not import or load the model locally. +- Send one request per page, with a few pages in flight at once (for example + `ThreadPoolExecutor`, about 4 to 8 workers). vLLM batches concurrent + requests, and most of the speedup comes from that. Keep the pages in order + in the output. +- Encode each page as a base64 PNG data URL, rendered at 200 DPI. +- Use the request parameters from section 4. For a single page, use + `window_size=128`. +- Use `requests` (already installed through transformers/marker) or add + `openai` to requirements. Whichever you pick, list it in the README. +- If the server is not running, fail with a clear message that shows the + `podman run` command. +- Put the same `.md`, `images/` and `_metadata.json` output next to the + PDF, with timings. Leave `mark2epub.py` alone. + +### 6. Benchmark and report + +Run marker, the transformers engine and the vllm engine on the same pages. +Report a table with seconds per page, total time and VRAM, plus a short +judgement of quality (compare the `.md` files). Update README with the vllm +engine, the Podman command and the measured numbers. Commit on this branch. +Do not push or merge unless the user asks. + +## Gotchas already found + +- `Some weights ... newly initialized: ['model.vision_model.embeddings.position_ids']` + is harmless. It is a buffer, not a trained weight. +- The "attention mask is not set" warnings come from Baidu's remote code. + Ignore them. +- In the transformers engine, `max_length=32768` can make a slow page look + like it hangs, because no output is streamed in `eval_mode`. To see + tokens/s, call `model.infer(..., tps_interval=5)` without `eval_mode`, and + use a smaller `max_length`. +- marker 1.x needs `--max-pages` whenever `--start-page` is used. +- The `docs/marker/marker_README.md` in this repo describes an older marker. + Its `OCR_ENGINE=ocrmypdf` option does not apply to marker 1.10. From f9b9f218da8421d0bff48ac0b43b7b3d580b5079 Mon Sep 17 00:00:00 2001 From: rfntnms Date: Fri, 25 Sep 2026 12:34:44 +0700 Subject: [PATCH 3/5] Add vLLM engine for Unlimited-OCR and benchmark it on Fedora Adds --ocr-engine vllm with --vllm-url and --vllm-workers. Pages are rendered locally and posted to a running vLLM server several at a time so vLLM can batch them; output order, images/ and _metadata.json match the transformers engine, which now shares the rendering and output helpers. If the server is unreachable the error prints the podman command. On an RTX 4060 (8 GB, also driving the desktop) the server only fits with FP8 weights and --skip-mm-profiling. It then reads a 12-page synthetic scan in ~12.5-15 s (~1-1.3 s/page) versus 71 s for marker and 178 s for the transformers engine, with 1-2 misread words in ~2,400. The transformers engine now sets expandable_segments, without which it runs out of memory on Linux, and the README lists torchvision for it. Co-Authored-By: Claude Opus 5.5 --- HANDOVER.md | 16 ++++ README.md | 118 ++++++++++++++++++++++++----- main.py | 27 ++++++- modules/unlimited_ocr.py | 75 ++++++++++++------ modules/vllm_ocr.py | 160 +++++++++++++++++++++++++++++++++++++++ 5 files changed, 351 insertions(+), 45 deletions(-) create mode 100644 modules/vllm_ocr.py diff --git a/HANDOVER.md b/HANDOVER.md index a004899..7665b79 100644 --- a/HANDOVER.md +++ b/HANDOVER.md @@ -5,6 +5,22 @@ Claude Code on Fedora should read this whole file first, then do the tasks in order. Ask the user before anything that needs `sudo`, a reboot, a push, or a merge into `main`. +## Status (2026-09-25, Fedora) + +All tasks below are done on branch `unlimited-ocr-backend` (not pushed). +`--ocr-engine vllm` works on the RTX 4060 at ~1.05–1.3 s per page, versus +~4–6 s for marker and ~14 s for the transformers engine. The README has the +working Podman command and the measured numbers. What differed from the plan: + +- The recipe's bf16 settings do not fit 8 GB next to the desktop. The server + needs `--quantization fp8 --skip-mm-profiling` (see README for why), and + rootless Podman needs `--security-opt label=disable` for SELinux. +- FP8 misreads roughly 1 word in 2,000; the bf16 transformers engine did not. +- The transformers engine needs `torchvision` and, on Linux, + `expandable_segments` (now set inside the module) to avoid running out of + memory. It is ~4x faster here than it was on Windows. +- Only a synthetic scan was tested. Re-check quality on a real scanned book. + ## Why this exists The user converts **scanned** books with this project. The slow step is diff --git a/README.md b/README.md index ad61795..cf38aed 100644 --- a/README.md +++ b/README.md @@ -133,7 +133,9 @@ python main.py [input_path] [output_path] [options] Options: --max-pages INT Maximum number of pages to process --start-page INT Page number to start from - --ocr-engine ENGINE marker (default) or unlimited + --ocr-engine ENGINE marker (default), unlimited or vllm + --vllm-url URL vLLM server for --ocr-engine vllm (default: http://localhost:8000/v1) + --vllm-workers INT Pages sent to the vLLM server at once (default: 8) --skip-epub Skip EPUB generation, only create markdown --skip-md Skip markdown generation, use existing markdown files ``` @@ -152,29 +154,111 @@ Convert to markdown only: python main.py thesis.pdf --skip-epub ``` -### Alternative OCR engine: Unlimited-OCR +### Alternative OCR engines: Unlimited-OCR -For scanned PDFs, `--ocr-engine unlimited` uses Baidu's +For scanned PDFs, two engines use Baidu's [Unlimited-OCR](https://github.com/baidu/Unlimited-OCR) vision-language model -instead of marker. Every page is rendered and parsed by the model, so it is not -worth using on digital PDFs, where marker reads the embedded text directly. - -- Requires an NVIDIA GPU with a CUDA build of PyTorch (no CPU or MPS support, - so it does not work in the CPU Docker image). Peak VRAM is about 7 GB, so an - 8 GB card is tight; close other GPU-heavy apps. -- It is slow through transformers: about 60 s per dense page on an RTX 4060, - roughly 4x slower than marker. The speed Baidu advertises comes from serving - the model with vLLM or SGLang on Linux, which is not wired in here. -- Needs extra packages: `pip install addict easydict matplotlib` -- The model (~6.7 GB) is downloaded from HuggingFace on first run. It is - loaded with `trust_remote_code=True`, which runs Python code from that - HuggingFace repository. -- Images are cropped from the page render at 200 DPI and saved to `images/`. +instead of marker. Every page is rendered at 200 DPI and parsed by the model, so +they are not worth using on digital PDFs, where marker reads the embedded text +directly. Blocks the model labels as images are cropped from the render and +saved to `images/`. + +- `--ocr-engine unlimited` loads the model in-process through transformers. +- `--ocr-engine vllm` sends the pages to a separate vLLM server, several at a + time, so vLLM can batch them. This is by far the fastest option for scans. + +Both need an NVIDIA GPU (no CPU or MPS support, so neither works in the CPU +Docker image). The model (~6.7 GB) is downloaded from HuggingFace on first use +and runs remote code from that repository (`trust_remote_code=True`). + +#### Measured speed + +Synthetic scanned book (A4 at 200 DPI, Indonesian text, one figure per page), +RTX 4060 8 GB that also drives the desktop, Fedora 44: + +| Engine | 3 pages | 12 pages | Per page (12 pages) | GPU memory | +|---|---|---|---|---| +| marker (Surya) | 27 s | 71 s | ~5.9 s (~4 s OCR only) | ~7.3 GB | +| unlimited (transformers, bf16) | 61 s | 178 s | ~13.9 s | ~7.4 GB | +| vllm (FP8, 8 workers) | 9.5 s | 15.4 s | ~1.3 s | ~7.4 GB, held while the server runs | +| vllm (FP8, 12 workers) | – | 12.6 s | ~1.05 s | same | + +Times for marker and unlimited include loading the models (~10 s). The vllm +times exclude starting the server, which takes about 2 minutes (the first start +also downloads the model), and the first batch after a start runs ~4 s slower. + +Quality on these pages: all three read the text correctly. Unlimited-OCR kept +every heading, while marker dropped the repeated "Bagian N" section headings as +page headers. Unlimited-OCR leaves headings as plain text rather than `##` and +keeps the printed page numbers. The FP8 server misread one word ("rumitnya" as +"mutinya") once or twice in ~2,400 words, depending on how pages were batched; +the bf16 transformers engine did not. + +#### `--ocr-engine unlimited` (transformers) + +- Needs extra packages: `pip install addict easydict matplotlib torchvision` + (torchvision must match your torch build). +- Peak VRAM is just under 8 GB. The engine sets + `PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True`, without which an 8 GB + card runs out of memory on Linux. Close other GPU-heavy apps. ```bash python main.py scanned_book.pdf --ocr-engine unlimited ``` +#### `--ocr-engine vllm` (vLLM server, Linux) + +The client only needs `requests`, which marker already installs. The model runs +in vLLM's dedicated image (the architecture is not in the regular vLLM wheel), +following the [vLLM recipe](https://recipes.vllm.ai/baidu/Unlimited-OCR). + +GPU access from Podman needs the NVIDIA Container Toolkit and a CDI spec (on +Fedora, add NVIDIA's repo from `https://nvidia.github.io/libnvidia-container/stable/rpm/nvidia-container-toolkit.repo` +to `/etc/yum.repos.d/`, then run `sudo dnf install nvidia-container-toolkit` and +`sudo nvidia-ctk cdi generate --output=/etc/cdi/nvidia.yaml`). Then start the +server and leave it running: + +```bash +podman run --rm --device nvidia.com/gpu=all --security-opt label=disable \ + --network host --ipc host \ + -e PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True \ + -v ~/.cache/huggingface:/root/.cache/huggingface \ + -v ~/.cache/vllm:/root/.cache/vllm \ + docker.io/vllm/vllm-openai:unlimited-ocr \ + baidu/Unlimited-OCR \ + --trust-remote-code \ + --logits_processors vllm.model_executor.models.unlimited_ocr:NGramPerReqLogitsProcessor \ + --no-enable-prefix-caching \ + --mm-processor-cache-gb 0 \ + --quantization fp8 --gpu-memory-utilization 0.85 \ + --max-model-len 8192 --skip-mm-profiling +``` + +It is ready once `curl localhost:8000/v1/models` answers. Then: + +```bash +python main.py scanned_book.pdf --ocr-engine vllm +``` + +Notes on the server flags: + +- `--security-opt label=disable` lets rootless Podman open the GPU under + SELinux; without it `nvidia-smi` in the container fails with "Insufficient + Permissions". With Docker, use `--gpus all` instead of `--device` and drop + `--security-opt`. +- The last three lines are for 8 GB cards. In bf16 the weights take 6.2 GB, and + each page needs another ~400 MB for the vision encoder, which does not fit + next to a KV cache. `--quantization fp8` (RTX 40xx or newer) halves the + weights and leaves room for ~33k tokens of KV cache, enough for ~12 pages in + flight. `--skip-mm-profiling` is needed because vLLM would otherwise profile a + 32-tile image, far larger than an A4 page (1 global view + 6 tiles). On a + GPU with 16 GB or more, drop these three lines to run the model in bf16. +- The two cache mounts keep the weights and vLLM's compiled kernels between + runs. On an 8 GB card, stop the server before running marker or the + transformers engine; it holds the GPU memory while it runs. +- If the server is not reachable, `--ocr-engine vllm` stops with an error that + prints this command. + ### Output Structure ``` diff --git a/main.py b/main.py index 627513f..4b6007c 100755 --- a/main.py +++ b/main.py @@ -45,10 +45,23 @@ def main(): ) parser.add_argument( '--ocr-engine', - choices=['marker', 'unlimited'], + choices=['marker', 'unlimited', 'vllm'], default='marker', - help='OCR backend: marker (default) or unlimited (Baidu Unlimited-OCR, ' - 'NVIDIA GPU only, meant for scanned PDFs)' + help='OCR backend: marker (default), unlimited (Baidu Unlimited-OCR ' + 'in-process, NVIDIA GPU only) or vllm (Unlimited-OCR served by a ' + 'running vLLM server); the last two are meant for scanned PDFs' + ) + parser.add_argument( + '--vllm-url', + default='http://localhost:8000/v1', + help='Base URL of the vLLM server for --ocr-engine vllm ' + '(default: http://localhost:8000/v1)' + ) + parser.add_argument( + '--vllm-workers', + type=int, + default=8, + help='Pages sent to the vLLM server at once (default: 8)' ) parser.add_argument( '--skip-epub', @@ -95,8 +108,15 @@ def main(): # Convert PDF to Markdown unless skipped if not args.skip_md: print("Converting PDF to Markdown...") + extra = {} if args.ocr_engine == 'unlimited': import modules.unlimited_ocr as converter + elif args.ocr_engine == 'vllm': + import modules.vllm_ocr as converter + extra = { + 'base_url': args.vllm_url, + 'workers': args.vllm_workers, + } else: converter = pdf2md converter.convert_pdf( @@ -104,6 +124,7 @@ def main(): markdown_dir, args.max_pages, args.start_page, + **extra, ) # Convert Markdown to EPUB unless skipped diff --git a/modules/unlimited_ocr.py b/modules/unlimited_ocr.py index dd82a76..7527036 100644 --- a/modules/unlimited_ocr.py +++ b/modules/unlimited_ocr.py @@ -6,12 +6,13 @@ aimed at scanned PDFs. It requires an NVIDIA GPU with CUDA; the model weights (~6.7 GB, bfloat16) are downloaded from HuggingFace on first use. -This runs the model through plain transformers, which is slow: about 60 s per -dense page on an 8 GB RTX 4060. The speed Baidu advertises comes from serving -the model with vLLM or SGLang on Linux, which this module does not do. +This runs the model in-process through plain transformers. The vllm engine +(modules/vllm_ocr.py) sends the same pages to a vLLM server instead and reuses +the rendering and post-processing helpers defined here. """ from pathlib import Path import json +import os import re import sys import tempfile @@ -41,6 +42,10 @@ def _load_model(): if _model is not None: return _model, _tokenizer + # Peak VRAM sits just under 8 GB; without this, fragmentation alone makes + # an 8 GB card run out of memory on Linux. Must be set before CUDA starts. + os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True") + import torch from transformers import AutoModel, AutoTokenizer @@ -62,12 +67,48 @@ def _load_model(): return _model, _tokenizer -def _page_indices(page_count: int, max_pages: int = None, start_page: int = None) -> list[int]: +def page_indices(page_count: int, max_pages: int = None, start_page: int = None) -> list[int]: start = start_page or 0 end = page_count if max_pages is None else min(page_count, start + max_pages) return list(range(start, end)) +def render_page(pdf, idx: int): + """Render page idx of an open pypdfium2 document to a PIL image.""" + page = pdf[idx] + image = page.render(scale=RENDER_DPI / 72).to_pil() + page.close() + return image + + +def write_output( + output_dir: Path, + stem: str, + page_texts: list[str], + pages: list[int], + elapsed: float, + engine: str, + **extra, +) -> None: + """Write .md and _metadata.json the way mark2epub expects.""" + md_output = output_dir / f"{stem}.md" + md_output.write_text("\n\n".join(t for t in page_texts if t), encoding="utf-8") + print(f"Markdown saved to: {md_output}") + + metadata = { + "ocr_engine": engine, + "pages": [i + 1 for i in pages], + "seconds_total": round(elapsed, 1), + "seconds_per_page": round(elapsed / max(len(pages), 1), 2), + **extra, + } + meta_output = output_dir / f"{stem}_metadata.json" + with open(meta_output, "w", encoding="utf-8") as f: + json.dump(metadata, f, indent=2) + print(f"Metadata saved to: {meta_output}") + print(f"OCR finished: {len(pages)} pages in {elapsed:.1f}s") + + def _parse_bbox(raw: str, width: int, height: int): nums = [float(n) for n in BBOX_RE.findall(raw or "")] if len(nums) < 4: @@ -145,7 +186,7 @@ def convert_pdf( stem = Path(input_path).stem pdf = pdfium.PdfDocument(input_path) - pages = _page_indices(len(pdf), max_pages, start_page) + pages = page_indices(len(pdf), max_pages, start_page) page_texts = [] started = time.time() @@ -153,9 +194,7 @@ def convert_pdf( tmp_dir = Path(tmp) for n, idx in enumerate(pages, 1): page_started = time.time() - page = pdf[idx] - page_image = page.render(scale=RENDER_DPI / 72).to_pil() - page.close() + page_image = render_page(pdf, idx) page_path = tmp_dir / f"page_{idx + 1:04d}.png" page_image.save(page_path) @@ -184,23 +223,9 @@ def convert_pdf( ) pdf.close() - - md_output = output_dir / f"{stem}.md" - md_output.write_text("\n\n".join(t for t in page_texts if t), encoding="utf-8") - print(f"Markdown saved to: {md_output}") - - elapsed = time.time() - started - metadata = { - "ocr_engine": MODEL_NAME, - "pages": [i + 1 for i in pages], - "seconds_total": round(elapsed, 1), - "seconds_per_page": round(elapsed / max(len(pages), 1), 2), - } - meta_output = output_dir / f"{stem}_metadata.json" - with open(meta_output, "w", encoding="utf-8") as f: - json.dump(metadata, f, indent=2) - print(f"Metadata saved to: {meta_output}") - print(f"OCR finished: {len(pages)} pages in {elapsed:.1f}s") + write_output( + output_dir, stem, page_texts, pages, time.time() - started, MODEL_NAME + ) except Exception as e: print(f"Error converting {input_path}: {str(e)}", file=sys.stderr) diff --git a/modules/vllm_ocr.py b/modules/vllm_ocr.py new file mode 100644 index 0000000..7fae393 --- /dev/null +++ b/modules/vllm_ocr.py @@ -0,0 +1,160 @@ +""" +OCR backend that sends pages to Baidu's Unlimited-OCR served by vLLM. + +The model is not loaded here; it runs in a separate vLLM server (Linux + +NVIDIA GPU, see README). Each page is rendered to a PNG and posted to the +server's OpenAI-compatible chat endpoint. Several pages are kept in flight so +vLLM can batch them, which is where the speedup over the transformers engine +comes from. Rendering and post-processing are shared with modules/unlimited_ocr. +""" +from collections import deque +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path +import base64 +import io +import sys +import time + +import requests + +from modules.unlimited_ocr import ( + MODEL_NAME, + page_indices, + page_output_to_markdown, + render_page, + write_output, +) + +DEFAULT_URL = "http://localhost:8000/v1" +DEFAULT_WORKERS = 8 + +# Settings that fit an 8 GB card that also drives the desktop; see README for +# what each memory flag is for and what to drop on a bigger GPU. +SERVER_COMMAND = """\ +podman run --rm --device nvidia.com/gpu=all --security-opt label=disable \\ + --network host --ipc host \\ + -e PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True \\ + -v ~/.cache/huggingface:/root/.cache/huggingface \\ + -v ~/.cache/vllm:/root/.cache/vllm \\ + docker.io/vllm/vllm-openai:unlimited-ocr \\ + baidu/Unlimited-OCR \\ + --trust-remote-code \\ + --logits_processors vllm.model_executor.models.unlimited_ocr:NGramPerReqLogitsProcessor \\ + --no-enable-prefix-caching \\ + --mm-processor-cache-gb 0 \\ + --quantization fp8 --gpu-memory-utilization 0.85 \\ + --max-model-len 8192 --skip-mm-profiling""" + + +def _check_server(base_url: str) -> str: + """Return the served model name, or fail with instructions to start vLLM.""" + try: + r = requests.get(f"{base_url}/models", timeout=5) + r.raise_for_status() + models = [m["id"] for m in r.json().get("data", [])] + except requests.RequestException as e: + raise RuntimeError( + f"No vLLM server reachable at {base_url} ({e.__class__.__name__}).\n" + f"Start one first (Linux + NVIDIA GPU), for example:\n\n{SERVER_COMMAND}\n" + ) from None + if MODEL_NAME in models: + return MODEL_NAME + if len(models) == 1: + return models[0] + raise RuntimeError(f"{MODEL_NAME} is not served at {base_url}; found {models}") + + +def _ocr_page(base_url: str, model: str, image) -> tuple[str, str, float]: + """Send one page image and return (raw output, finish reason, seconds).""" + started = time.time() + buf = io.BytesIO() + image.save(buf, format="PNG") + data_url = "data:image/png;base64," + base64.b64encode(buf.getvalue()).decode() + + # Follows the vLLM recipe exactly; other prompts or dropping the special + # tokens give empty or looping output. max_tokens is left to the server, + # which caps it at whatever --max-model-len leaves after the image tokens. + payload = { + "model": model, + "messages": [{"role": "user", "content": [ + {"type": "text", "text": "document parsing."}, + {"type": "image_url", "image_url": {"url": data_url}}, + ]}], + "temperature": 0.0, + "skip_special_tokens": False, + "vllm_xargs": {"ngram_size": 35, "window_size": 128}, + } + r = requests.post(f"{base_url}/chat/completions", json=payload, timeout=3600) + if not r.ok: + raise RuntimeError(f"vLLM returned {r.status_code}: {r.text[:500]}") + choice = r.json()["choices"][0] + return choice["message"]["content"] or "", choice.get("finish_reason"), time.time() - started + + +def convert_pdf( + input_path: str, + output_dir: Path, + max_pages: int = None, + start_page: int = None, + base_url: str = DEFAULT_URL, + workers: int = DEFAULT_WORKERS, +) -> None: + """Convert a PDF to markdown page by page through a vLLM server.""" + import pypdfium2 as pdfium + + base_url = base_url.rstrip("/") + try: + model = _check_server(base_url) + + output_dir.mkdir(parents=True, exist_ok=True) + image_dir = output_dir / "images" + stem = Path(input_path).stem + + pdf = pdfium.PdfDocument(input_path) + pages = page_indices(len(pdf), max_pages, start_page) + page_texts = [] + page_seconds = [] + started = time.time() + + # Pages are rendered here (pdfium is not thread-safe) and at most + # 2 * workers are held in memory, so long books do not pile up images. + pending = deque() + + def finish_oldest(): + idx, image, future = pending.popleft() + raw, reason, seconds = future.result() + if reason == "length": + print(f" warning: page {idx + 1} hit the token limit and is truncated") + page_texts.append(page_output_to_markdown(raw, image, image_dir, idx + 1)) + page_seconds.append(round(seconds, 1)) + image.close() + print( + f" page {idx + 1} ({len(page_texts)}/{len(pages)}) " + f"request {seconds:.1f}s, elapsed {time.time() - started:.1f}s" + ) + + with ThreadPoolExecutor(max_workers=workers) as pool: + for idx in pages: + image = render_page(pdf, idx) + pending.append((idx, image, pool.submit(_ocr_page, base_url, model, image))) + while len(pending) >= 2 * workers or pending and pending[0][2].done(): + finish_oldest() + while pending: + finish_oldest() + + pdf.close() + write_output( + output_dir, + stem, + page_texts, + pages, + time.time() - started, + f"{MODEL_NAME} (vLLM)", + vllm_url=base_url, + workers=workers, + request_seconds=page_seconds, + ) + + except Exception as e: + print(f"Error converting {input_path}: {str(e)}", file=sys.stderr) + raise From 2a3d529057b0081ca9ae98107331f7ba940883a7 Mon Sep 17 00:00:00 2001 From: rfntnms Date: Fri, 25 Sep 2026 12:58:23 +0700 Subject: [PATCH 4/5] Pad pages to 2:3 before sending them to vLLM vLLM picks the tile grid for Unlimited-OCR from the page's aspect ratio, with up to 32 tiles. On a real scanned book, half of the pages were just wide enough to get a 3x4 grid of upscaled tiles, and encoding them needed 704 MiB at once, which crashed the server on an 8 GB card. Padding each page with white to exactly 2:3 (3:2 for landscape) always gives 6 tiles, like the A4 pages that were benchmarked. 60 pages of that book now run at ~1.5 s per page. Also close the PDF when a request fails, which removes pdfium's "library is destroyed" warning. Co-Authored-By: Claude Opus 5.5 --- README.md | 7 +++++++ modules/vllm_ocr.py | 44 ++++++++++++++++++++++++++++++++++++-------- 2 files changed, 43 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index cf38aed..e33a889 100644 --- a/README.md +++ b/README.md @@ -194,6 +194,13 @@ keeps the printed page numbers. The FP8 server misread one word ("rumitnya" as "mutinya") once or twice in ~2,400 words, depending on how pages were batched; the bf16 transformers engine did not. +On a real scanned book (Indonesian, 841 small pages of ~9.6 x 13.7 cm) the +vllm engine read 60 pages in 92 s (~1.5 s/page) with only occasional +single-letter errors. Before sending a page, the engine pads it with white to +exactly 2:3. vLLM otherwise splits pages that are slightly wider than 2:3 into +12 or more upscaled 640 px tiles, which ran the 8 GB card out of memory on half +of that book's pages. + #### `--ocr-engine unlimited` (transformers) - Needs extra packages: `pip install addict easydict matplotlib torchvision` diff --git a/modules/vllm_ocr.py b/modules/vllm_ocr.py index 7fae393..9a48196 100644 --- a/modules/vllm_ocr.py +++ b/modules/vllm_ocr.py @@ -64,6 +64,30 @@ def _check_server(base_url: str) -> str: raise RuntimeError(f"{MODEL_NAME} is not served at {base_url}; found {models}") +def pad_to_tile_grid(image): + """ + Pad a page with white to exactly 2:3 (or 3:2 for landscape pages). + + vLLM tiles each image into 640 px crops on the grid (up to 32 crops) whose + aspect ratio is closest to the image's, so a book page only slightly wider + than 2:3 becomes a 3x4 or 4x5 grid of upscaled crops. Encoding those needs + 700 MB and more at once, which runs an 8 GB card out of memory. At exactly + 2:3 the page always gets 6 crops, like an A4 page. Detection boxes are + relative to the padded image, so image blocks are cropped from it. + """ + from PIL import Image + + w, h = image.size + tw, th = (2, 3) if w <= h else (3, 2) + new_w, new_h = max(w, -(-h * tw // th)), max(h, -(-w * th // tw)) + if (new_w, new_h) == (w, h): + return image + padded = Image.new(image.mode, (new_w, new_h), "white") + padded.paste(image, ((new_w - w) // 2, (new_h - h) // 2)) + image.close() + return padded + + def _ocr_page(base_url: str, model: str, image) -> tuple[str, str, float]: """Send one page image and return (raw output, finish reason, seconds).""" started = time.time() @@ -133,16 +157,20 @@ def finish_oldest(): f"request {seconds:.1f}s, elapsed {time.time() - started:.1f}s" ) - with ThreadPoolExecutor(max_workers=workers) as pool: - for idx in pages: - image = render_page(pdf, idx) - pending.append((idx, image, pool.submit(_ocr_page, base_url, model, image))) - while len(pending) >= 2 * workers or pending and pending[0][2].done(): + try: + with ThreadPoolExecutor(max_workers=workers) as pool: + for idx in pages: + image = pad_to_tile_grid(render_page(pdf, idx)) + pending.append((idx, image, pool.submit(_ocr_page, base_url, model, image))) + while len(pending) >= 2 * workers or pending and pending[0][2].done(): + finish_oldest() + while pending: finish_oldest() - while pending: - finish_oldest() + finally: + # Closing here rather than at interpreter exit avoids pdfium's + # "library is destroyed" warning when a request fails. + pdf.close() - pdf.close() write_output( output_dir, stem, From 69d967c4a2b652752fcd3a6a784af5d8250bc57c Mon Sep 17 00:00:00 2001 From: rfntnms Date: Fri, 25 Sep 2026 13:07:37 +0700 Subject: [PATCH 5/5] Start the vLLM server automatically in Podman With --ocr-engine vllm, main.py now checks --vllm-url before processing. If nothing answers and the URL is local, it starts the server in a Podman container (pdf2epub-vllm) with the settings that fit an 8 GB card, waits until it serves the model, and stops it again once all PDFs are done, including on errors and Ctrl+C, since it holds most of the GPU memory. --vllm-keep-server leaves it running for the next run. A server that was already running is used and left alone. If the container dies during startup, the error shows its last log lines. Co-Authored-By: Claude Opus 5.5 --- README.md | 35 +++++++++---- main.py | 114 +++++++++++++++++++++++++----------------- modules/vllm_ocr.py | 119 +++++++++++++++++++++++++++++++++++++++++--- 3 files changed, 203 insertions(+), 65 deletions(-) diff --git a/README.md b/README.md index e33a889..4cbb6fc 100644 --- a/README.md +++ b/README.md @@ -136,6 +136,7 @@ Options: --ocr-engine ENGINE marker (default), unlimited or vllm --vllm-url URL vLLM server for --ocr-engine vllm (default: http://localhost:8000/v1) --vllm-workers INT Pages sent to the vLLM server at once (default: 8) + --vllm-keep-server Leave an auto-started vLLM server running afterwards --skip-epub Skip EPUB generation, only create markdown --skip-md Skip markdown generation, use existing markdown files ``` @@ -222,8 +223,23 @@ following the [vLLM recipe](https://recipes.vllm.ai/baidu/Unlimited-OCR). GPU access from Podman needs the NVIDIA Container Toolkit and a CDI spec (on Fedora, add NVIDIA's repo from `https://nvidia.github.io/libnvidia-container/stable/rpm/nvidia-container-toolkit.repo` to `/etc/yum.repos.d/`, then run `sudo dnf install nvidia-container-toolkit` and -`sudo nvidia-ctk cdi generate --output=/etc/cdi/nvidia.yaml`). Then start the -server and leave it running: +`sudo nvidia-ctk cdi generate --output=/etc/cdi/nvidia.yaml`). Then: + +```bash +python main.py scanned_book.pdf --ocr-engine vllm +``` + +If no server answers at `--vllm-url` and the URL is local, the program starts +one itself in a Podman container named `pdf2epub-vllm`, waits until it is ready +(about 2 minutes; the first run also pulls the ~19 GB image and the model), and +stops it when all PDFs are done, on errors and on Ctrl+C too. The server holds +most of the GPU memory, so it is not left running. Pass `--vllm-keep-server` to +leave it running for the next run, which then starts right away; stop it with +`podman stop pdf2epub-vllm`. A server that was already running is used as is and +left alone. + +To run the server yourself instead (for example on another machine, with +`--vllm-url` pointing at it), this is the command the program uses: ```bash podman run --rm --device nvidia.com/gpu=all --security-opt label=disable \ @@ -241,11 +257,7 @@ podman run --rm --device nvidia.com/gpu=all --security-opt label=disable \ --max-model-len 8192 --skip-mm-profiling ``` -It is ready once `curl localhost:8000/v1/models` answers. Then: - -```bash -python main.py scanned_book.pdf --ocr-engine vllm -``` +It is ready once `curl localhost:8000/v1/models` answers. Notes on the server flags: @@ -259,12 +271,15 @@ Notes on the server flags: weights and leaves room for ~33k tokens of KV cache, enough for ~12 pages in flight. `--skip-mm-profiling` is needed because vLLM would otherwise profile a 32-tile image, far larger than an A4 page (1 global view + 6 tiles). On a - GPU with 16 GB or more, drop these three lines to run the model in bf16. + GPU with 16 GB or more, drop these three lines to run the model in bf16 + (for the auto-started server, edit `VLLM_ARGS` in `modules/vllm_ocr.py`). - The two cache mounts keep the weights and vLLM's compiled kernels between runs. On an 8 GB card, stop the server before running marker or the transformers engine; it holds the GPU memory while it runs. -- If the server is not reachable, `--ocr-engine vllm` stops with an error that - prints this command. +- If the server is not reachable and cannot be started automatically (a remote + `--vllm-url`, or no `podman`), `--ocr-engine vllm` stops with an error that + prints this command. If the container exits while starting, the error shows + its last log lines. ### Output Structure diff --git a/main.py b/main.py index 4b6007c..8a336d7 100755 --- a/main.py +++ b/main.py @@ -63,6 +63,11 @@ def main(): default=8, help='Pages sent to the vLLM server at once (default: 8)' ) + parser.add_argument( + '--vllm-keep-server', + action='store_true', + help='Leave the vLLM server running afterwards if this run started it' + ) parser.add_argument( '--skip-epub', action='store_true', @@ -83,59 +88,74 @@ def main(): queue = pdf2md.add_pdfs_to_queue(input_path) print(f"Found {len(queue)} PDF files to process") - # Process each PDF + # The vllm engine needs a running server; start one in podman if needed + # and stop it again at the end, since it holds most of the GPU memory. + server_started = False + if args.ocr_engine == 'vllm' and not args.skip_md: + import modules.vllm_ocr as vllm_ocr + try: + server_started = vllm_ocr.start_server(args.vllm_url) + except RuntimeError as e: + print(f"Error: {e}", file=sys.stderr) + sys.exit(1) + failed = [] - for pdf_path in queue: - print(f"\nProcessing: {pdf_path.name}") + try: + # Process each PDF + for pdf_path in queue: + print(f"\nProcessing: {pdf_path.name}") - # Get output directory for this PDF - if args.output_path: - output_path = Path(args.output_path) - markdown_dir = output_path / pdf_path.stem - else: - markdown_dir = pdf2md.get_default_output_dir(pdf_path) - output_path = markdown_dir.parent + # Get output directory for this PDF + if args.output_path: + output_path = Path(args.output_path) + markdown_dir = output_path / pdf_path.stem + else: + markdown_dir = pdf2md.get_default_output_dir(pdf_path) + output_path = markdown_dir.parent - try: - # Check if markdown directory exists when skipping MD generation - if args.skip_md: - if not markdown_dir.exists(): - print(f"Error: Markdown directory not found: {markdown_dir}", file=sys.stderr) - failed.append(pdf_path.name) - continue - print(f"Using existing markdown files from: {markdown_dir}") + try: + # Check if markdown directory exists when skipping MD generation + if args.skip_md: + if not markdown_dir.exists(): + print(f"Error: Markdown directory not found: {markdown_dir}", file=sys.stderr) + failed.append(pdf_path.name) + continue + print(f"Using existing markdown files from: {markdown_dir}") - # Convert PDF to Markdown unless skipped - if not args.skip_md: - print("Converting PDF to Markdown...") - extra = {} - if args.ocr_engine == 'unlimited': - import modules.unlimited_ocr as converter - elif args.ocr_engine == 'vllm': - import modules.vllm_ocr as converter - extra = { - 'base_url': args.vllm_url, - 'workers': args.vllm_workers, - } - else: - converter = pdf2md - converter.convert_pdf( - str(pdf_path), - markdown_dir, - args.max_pages, - args.start_page, - **extra, - ) + # Convert PDF to Markdown unless skipped + if not args.skip_md: + print("Converting PDF to Markdown...") + extra = {} + if args.ocr_engine == 'unlimited': + import modules.unlimited_ocr as converter + elif args.ocr_engine == 'vllm': + import modules.vllm_ocr as converter + extra = { + 'base_url': args.vllm_url, + 'workers': args.vllm_workers, + } + else: + converter = pdf2md + converter.convert_pdf( + str(pdf_path), + markdown_dir, + args.max_pages, + args.start_page, + **extra, + ) - # Convert Markdown to EPUB unless skipped - if not args.skip_epub: - print("Converting Markdown to EPUB...") - mark2epub.convert_to_epub(markdown_dir, output_path) + # Convert Markdown to EPUB unless skipped + if not args.skip_epub: + print("Converting Markdown to EPUB...") + mark2epub.convert_to_epub(markdown_dir, output_path) - except Exception as e: - print(f"Error processing {pdf_path.name}: {str(e)}", file=sys.stderr) - failed.append(pdf_path.name) - continue + except Exception as e: + print(f"Error processing {pdf_path.name}: {str(e)}", file=sys.stderr) + failed.append(pdf_path.name) + continue + finally: + if server_started and not args.vllm_keep_server: + vllm_ocr.stop_server() if failed: print( diff --git a/modules/vllm_ocr.py b/modules/vllm_ocr.py index 9a48196..356d58c 100644 --- a/modules/vllm_ocr.py +++ b/modules/vllm_ocr.py @@ -2,7 +2,8 @@ OCR backend that sends pages to Baidu's Unlimited-OCR served by vLLM. The model is not loaded here; it runs in a separate vLLM server (Linux + -NVIDIA GPU, see README). Each page is rendered to a PNG and posted to the +NVIDIA GPU, see README), which start_server() launches in Podman when none is +running. Each page is rendered to a PNG and posted to the server's OpenAI-compatible chat endpoint. Several pages are kept in flight so vLLM can batch them, which is where the speedup over the transformers engine comes from. Rendering and post-processing are shared with modules/unlimited_ocr. @@ -10,8 +11,11 @@ from collections import deque from concurrent.futures import ThreadPoolExecutor from pathlib import Path +from urllib.parse import urlparse import base64 import io +import shutil +import subprocess import sys import time @@ -28,8 +32,31 @@ DEFAULT_URL = "http://localhost:8000/v1" DEFAULT_WORKERS = 8 +CONTAINER_NAME = "pdf2epub-vllm" +IMAGE = "docker.io/vllm/vllm-openai:unlimited-ocr" +STARTUP_TIMEOUT = 20 * 60 # the first start pulls the image and the weights + # Settings that fit an 8 GB card that also drives the desktop; see README for # what each memory flag is for and what to drop on a bigger GPU. +PODMAN_ARGS = [ + "--device", "nvidia.com/gpu=all", + "--security-opt", "label=disable", + "--network", "host", + "--ipc", "host", + "-e", "PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True", +] +VLLM_ARGS = [ + "--trust-remote-code", + "--logits_processors", + "vllm.model_executor.models.unlimited_ocr:NGramPerReqLogitsProcessor", + "--no-enable-prefix-caching", + "--mm-processor-cache-gb", "0", + "--quantization", "fp8", + "--gpu-memory-utilization", "0.85", + "--max-model-len", "8192", + "--skip-mm-profiling", +] + SERVER_COMMAND = """\ podman run --rm --device nvidia.com/gpu=all --security-opt label=disable \\ --network host --ipc host \\ @@ -46,17 +73,93 @@ --max-model-len 8192 --skip-mm-profiling""" -def _check_server(base_url: str) -> str: - """Return the served model name, or fail with instructions to start vLLM.""" +def _served_models(base_url: str): + """Return the model ids served at base_url, or None if nothing answers.""" try: r = requests.get(f"{base_url}/models", timeout=5) r.raise_for_status() - models = [m["id"] for m in r.json().get("data", [])] - except requests.RequestException as e: + return [m["id"] for m in r.json().get("data", [])] + except requests.RequestException: + return None + + +def _podman(*args, check=False): + return subprocess.run(["podman", *args], capture_output=True, text=True, check=check) + + +def start_server(base_url: str = DEFAULT_URL) -> bool: + """ + Make sure a vLLM server answers at base_url, starting one in Podman if not. + + Returns True when this call started the container, so the caller knows to + stop it again with stop_server(). Only local URLs are started. + """ + base_url = base_url.rstrip("/") + if _served_models(base_url) is not None: + return False + + url = urlparse(base_url) + if url.hostname not in ("localhost", "127.0.0.1", "::1") or not shutil.which("podman"): + raise RuntimeError( + f"No vLLM server reachable at {base_url}, and it cannot be started " + f"automatically (needs a local URL and podman). Start one first, " + f"for example:\n\n{SERVER_COMMAND}\n" + ) + + mounts = [] + for name in ("huggingface", "vllm"): + cache = Path.home() / ".cache" / name + cache.mkdir(parents=True, exist_ok=True) + mounts += ["-v", f"{cache}:/root/.cache/{name}"] + + _podman("rm", "-f", CONTAINER_NAME) + cmd = [ + "run", "-d", "--name", CONTAINER_NAME, *PODMAN_ARGS, *mounts, + IMAGE, MODEL_NAME, *VLLM_ARGS, "--port", str(url.port or 8000), + ] + result = _podman(*cmd) + if result.returncode != 0: + raise RuntimeError(f"podman could not start the vLLM server:\n{result.stderr.strip()}") + + print( + f"Starting vLLM server in podman container {CONTAINER_NAME} " + "(about 2 minutes; the first run also downloads the image and model)..." + ) + started = time.time() + try: + while time.time() - started < STARTUP_TIMEOUT: + if _served_models(base_url) is not None: + print(f"vLLM server ready after {time.time() - started:.0f}s") + return True + state = _podman("inspect", "-f", "{{.State.Running}}", CONTAINER_NAME) + if state.stdout.strip() != "true": + logs = _podman("logs", "--tail", "15", CONTAINER_NAME) + raise RuntimeError( + "vLLM server exited during startup. Last log lines:\n" + + (logs.stdout + logs.stderr).strip() + ) + time.sleep(2) + raise RuntimeError(f"vLLM server not ready after {STARTUP_TIMEOUT // 60} minutes") + except BaseException: + stop_server() + raise + + +def stop_server() -> None: + """Stop and remove the container started by start_server().""" + print(f"Stopping vLLM server ({CONTAINER_NAME})...") + _podman("stop", "-t", "10", CONTAINER_NAME) + _podman("rm", "-f", CONTAINER_NAME) + + +def _check_server(base_url: str) -> str: + """Return the served model name, or fail with instructions to start vLLM.""" + models = _served_models(base_url) + if models is None: raise RuntimeError( - f"No vLLM server reachable at {base_url} ({e.__class__.__name__}).\n" - f"Start one first (Linux + NVIDIA GPU), for example:\n\n{SERVER_COMMAND}\n" - ) from None + f"No vLLM server reachable at {base_url}. Start one first, " + f"for example:\n\n{SERVER_COMMAND}\n" + ) if MODEL_NAME in models: return MODEL_NAME if len(models) == 1: