From ee807c47eed15a174751086a93c74481f5ecab90 Mon Sep 17 00:00:00 2001 From: Matanya Moses Date: Sun, 27 Sep 2026 14:06:57 +0300 Subject: [PATCH] Add RTL text direction and language support with CLI overrides - Add modules/bidi.py to detect text direction (RTL vs LTR) and language via Unicode Bidirectional character properties and script/word analysis - Update modules/mark2epub.py to inject dir="rtl"/"ltr", lang="XX", and page-progression-direction in package.opf, spine, cover, TOC, and chapters - Add --direction/--dir, --rtl, --ltr, and --lang/--language CLI arguments to main.py and mark2epub.py - Update pdf2md.py to store detected direction and language in metadata JSON - Add comprehensive test suite in tests/test_rtl.py and enable tests in CI - Update README.md with feature notes and usage examples --- .github/workflows/ci.yml | 5 + README.md | 16 ++ conftest.py | 7 + main.py | 57 ++++++-- modules/bidi.py | 303 ++++++++++++++++++++++++++++++++++++++ modules/mark2epub.py | 279 ++++++++++++++++++++++++++--------- modules/pdf2md.py | 9 ++ tests/test_rtl.py | 306 +++++++++++++++++++++++++++++++++++++++ 8 files changed, 902 insertions(+), 80 deletions(-) create mode 100644 conftest.py create mode 100644 modules/bidi.py create mode 100644 tests/test_rtl.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a9c1978..22a8f9b 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -31,3 +31,8 @@ jobs: # imports). Style rules are deliberately excluded. - name: Lint run: ruff check --select E9,F . + + - name: Run tests + run: | + pip install pytest markdown latex2mathml pillow regex + pytest diff --git a/README.md b/README.md index 16c6b4a..970599c 100644 --- a/README.md +++ b/README.md @@ -11,6 +11,7 @@ Convert PDF files to nicely structured Markdown and EPUB format with intelligent - 📝 Clean markdown output with preserved structure - 📱 EPUB generation with customizable styling - 🌍 Multi-language support +- 🔄 Automatic RTL/LTR text direction and language detection with page progression support - 🚀 GPU acceleration support (NVIDIA & AMD) - 🍎 Apple Silicon support @@ -135,6 +136,11 @@ Options: --start-page INT Page number to start from --skip-epub Skip EPUB generation, only create markdown --skip-md Skip markdown generation, use existing markdown files + --direction, --dir {rtl,ltr,auto} + Set text direction (default: auto) + --rtl Force Right-to-Left text direction + --ltr Force Left-to-Right text direction + --lang, --language CODE Override document language code (e.g., he, ar, en, fr) ``` If `input_path` is omitted, all PDFs in `./input/` are processed. @@ -151,6 +157,16 @@ Convert to markdown only: python main.py thesis.pdf --skip-epub ``` +Convert an RTL document with automatic direction and language detection: +```bash +python main.py hebrew_book.pdf +``` + +Force RTL direction and specify language code: +```bash +python main.py arabic_book.pdf --rtl --lang ar +``` + ### Output Structure ``` diff --git a/conftest.py b/conftest.py new file mode 100644 index 0000000..72f944a --- /dev/null +++ b/conftest.py @@ -0,0 +1,7 @@ +import sys +from pathlib import Path + +# Ensure repository root is on sys.path +root_dir = Path(__file__).resolve().parent +if str(root_dir) not in sys.path: + sys.path.insert(0, str(root_dir)) diff --git a/main.py b/main.py index 14a0f74..04c168c 100755 --- a/main.py +++ b/main.py @@ -4,17 +4,20 @@ from pathlib import Path import modules.pdf2md as pdf2md import modules.mark2epub as mark2epub -import torch - +try: + import torch +except ImportError: + torch = None def main(): - if torch.cuda.is_available(): - print("CUDA is available. Using GPU for processing.") - elif torch.backends.mps.is_available(): - print("MPS is available. Using Apple Silicon for processing.") - else: - print("CUDA is not available. Using CPU for processing.") + if torch is not None: + if torch.cuda.is_available(): + print("CUDA is available. Using GPU for processing.") + elif torch.backends.mps.is_available(): + print("MPS is available. Using Apple Silicon for processing.") + else: + print("CUDA is not available. Using CPU for processing.") parser = argparse.ArgumentParser( description='Convert PDF files to EPUB format via Markdown' @@ -53,9 +56,40 @@ def main(): action='store_true', help='Skip markdown generation, use existing markdown files' ) + parser.add_argument( + '--direction', '--dir', + dest='direction', + choices=['rtl', 'ltr', 'auto'], + type=str.lower, + default='auto', + help='Set text direction: rtl, ltr, or auto (default: auto)' + ) + parser.add_argument( + '--rtl', + action='store_true', + help='Convenience flag to force RTL text direction' + ) + parser.add_argument( + '--ltr', + action='store_true', + help='Convenience flag to force LTR text direction' + ) + parser.add_argument( + '--lang', '--language', + dest='language', + type=str, + default=None, + help='Override document language code (e.g., he, ar, en, fr)' + ) args = parser.parse_args() + direction = args.direction + if args.rtl: + direction = 'rtl' + elif args.ltr: + direction = 'ltr' + # Get input path input_path = Path(args.input_path) if args.input_path else pdf2md.get_default_input_dir() @@ -98,7 +132,12 @@ def main(): # Convert Markdown to EPUB unless skipped if not args.skip_epub: print("Converting Markdown to EPUB...") - mark2epub.convert_to_epub(markdown_dir, output_path) + mark2epub.convert_to_epub( + markdown_dir, + output_path, + direction=direction, + language=args.language, + ) except Exception as e: print(f"Error processing {pdf_path.name}: {str(e)}", file=sys.stderr) diff --git a/modules/bidi.py b/modules/bidi.py new file mode 100644 index 0000000..be4f049 --- /dev/null +++ b/modules/bidi.py @@ -0,0 +1,303 @@ +import re +import unicodedata +from collections import Counter +from pathlib import Path +from typing import Optional, Tuple, Union + + +def strip_markdown(text: str) -> str: + """ + Remove code blocks, inline code, images, links, LaTeX math formulas, + and HTML tags so direction and language detection operates purely + on natural language prose. + """ + # Remove fenced code blocks + text = re.sub(r"```[\s\S]*?```", " ", text) + # Remove inline code + text = re.sub(r"`[^`]+`", " ", text) + # Remove display math $$...$$ + text = re.sub(r"\$\$[\s\S]*?\$\$", " ", text) + # Remove inline math $...$ + text = re.sub(r"(?]+>", " ", text) + # Remove URLs + text = re.sub(r"https?://\S+", " ", text) + return text + + +def detect_direction(text: str) -> str: + """ + Detect whether the majority of directional characters in the text are RTL or LTR. + + Returns: + 'rtl' if majority of directional characters are Right-to-Left. + 'ltr' if majority are Left-to-Right or if text has no directional characters. + """ + cleaned = strip_markdown(text) + rtl_count = 0 + ltr_count = 0 + + for ch in cleaned: + bidi = unicodedata.bidirectional(ch) + if bidi in ("R", "AL"): + rtl_count += 1 + elif bidi == "L": + ltr_count += 1 + + if rtl_count > ltr_count: + return "rtl" + return "ltr" + + +# Stop words for Latin-script language detection +_LATIN_STOP_WORDS = { + "en": { + "the", "and", "to", "of", "in", "is", "that", "for", "with", "as", + "was", "on", "at", "by", "this", "from", "be", "are", "or", "an", + "have", "which", "one", "you", "were", "her", "all", "she", "there", + "would", "their", "we", "him", "been", "has", "when", "who", "will", + "more", "no", "if", "out", "so", "said", "what", "its", "about", + "into", "than", "them", "can", "only", "other", "new", "some", "could", + }, + "fr": { + "le", "la", "les", "de", "et", "en", "du", "un", "une", "pour", + "dans", "des", "est", "qui", "par", "sur", "au", "avec", "ce", "que", + "son", "sa", "ses", "sont", "pas", "plus", "aux", "ont", "mais", + "cette", "il", "elle", "ils", "elles", "comme", "nous", "vous", "leur", + }, + "de": { + "der", "die", "das", "und", "in", "zu", "den", "mit", "von", "ist", + "des", "nicht", "ein", "eine", "einer", "einem", "einen", "eines", + "auf", "für", "dem", "sich", "im", "auch", "es", "an", "werden", + "aus", "er", "hat", "dass", "sie", "nach", "wird", "bei", "um", "am", + }, + "es": { + "de", "la", "el", "en", "y", "a", "los", "que", "del", "las", "por", + "un", "para", "con", "una", "es", "al", "se", "su", "mas", "más", + "pero", "sus", "le", "ya", "o", "fue", "este", "ha", "si", "sí", + "porque", "esta", "son", "entre", "cuando", "muy", "sin", "sobre", + }, + "it": { + "di", "il", "la", "e", "ed", "in", "che", "per", "un", "una", "non", + "si", "da", "le", "del", "della", "dei", "delle", "nel", "nella", + "con", "al", "alla", "sono", "ha", "ma", "più", "se", "anche", + "come", "ci", "su", "dal", "dalla", "questo", "questa", "suo", "sua", + }, + "pt": { + "de", "a", "o", "que", "e", "do", "da", "em", "um", "para", "com", + "não", "uma", "os", "no", "se", "na", "por", "mais", "as", "dos", + "como", "mas", "ao", "ele", "das", "à", "seu", "sua", "ou", "quando", + "muito", "nos", "já", "eu", "também", "só", "pelo", "pela", "até", + }, + "nl": { + "de", "het", "en", "van", "ik", "te", "dat", "die", "in", "een", + "hij", "op", "voor", "met", "maar", "zijn", "was", "niet", "is", + "er", "om", "aan", "zo", "door", "over", "ze", "bij", "wij", "al", + }, +} + +_PERSIAN_LETTERS = set("پچژگیک") +_URDU_LETTERS = set("ٹڈڑںےھہ") + + +def detect_language(text: str, direction: Optional[str] = None) -> str: + """ + Detect language code from text. If direction is not specified, it is detected first. + + Returns ISO 639-1 / 639-2 language code (e.g. 'he', 'ar', 'fa', 'ur', 'en', 'fr', etc.). + """ + if direction is None: + direction = detect_direction(text) + + cleaned = strip_markdown(text) + + if direction == "rtl": + hebrew_count = 0 + arabic_count = 0 + persian_specific = 0 + urdu_specific = 0 + syriac_count = 0 + thaana_count = 0 + + for ch in cleaned: + code = ord(ch) + if (0x0590 <= code <= 0x05FF) or (0xFB1D <= code <= 0xFB4F): + hebrew_count += 1 + elif ( + (0x0600 <= code <= 0x06FF) + or (0x0750 <= code <= 0x077F) + or (0x08A0 <= code <= 0x08FF) + or (0xFB50 <= code <= 0xFDFF) + or (0xFE70 <= code <= 0xFEFF) + ): + arabic_count += 1 + if ch in _PERSIAN_LETTERS: + persian_specific += 1 + if ch in _URDU_LETTERS: + urdu_specific += 1 + elif 0x0700 <= code <= 0x074F: + syriac_count += 1 + elif 0x0780 <= code <= 0x07BF: + thaana_count += 1 + + total_rtl_chars = hebrew_count + arabic_count + syriac_count + thaana_count + if hebrew_count > arabic_count and hebrew_count > 0: + return "he" + if arabic_count > 0: + if urdu_specific > 0: + return "ur" + if persian_specific > 0: + return "fa" + return "ar" + if syriac_count > 0: + return "syr" + if thaana_count > 0: + return "dv" + if total_rtl_chars == 0: + return _detect_ltr_language(cleaned) + return "he" if hebrew_count > 0 else "ar" + + return _detect_ltr_language(cleaned) + + +def _detect_ltr_language(cleaned: str) -> str: + """Detect language code for LTR or non-RTL text.""" + greek_count = 0 + cyrillic_count = 0 + cjk_count = 0 + kana_count = 0 + hangul_count = 0 + devanagari_count = 0 + bengali_count = 0 + thai_count = 0 + ukrainian_specific = 0 + belarusian_specific = 0 + + ukr_letters = set("іїєґІЇЄҐ") + bel_letters = set("ўЎ") + + for ch in cleaned: + code = ord(ch) + if 0x0370 <= code <= 0x03FF: + greek_count += 1 + elif 0x0400 <= code <= 0x04FF: + cyrillic_count += 1 + if ch in ukr_letters: + ukrainian_specific += 1 + elif ch in bel_letters: + belarusian_specific += 1 + elif 0x3040 <= code <= 0x30FF: + kana_count += 1 + elif (0xAC00 <= code <= 0xD7AF) or (0x1100 <= code <= 0x11FF): + hangul_count += 1 + elif 0x4E00 <= code <= 0x9FFF: + cjk_count += 1 + elif 0x0900 <= code <= 0x097F: + devanagari_count += 1 + elif 0x0980 <= code <= 0x09FF: + bengali_count += 1 + elif 0x0E00 <= code <= 0x0E7F: + thai_count += 1 + + words = re.findall(r"\b[a-zà-ÿ]+\b", cleaned.lower()) + latin_word_count = len(words) + + if kana_count > 0: + return "ja" + if hangul_count > 0: + return "ko" + if cjk_count > 0 and cjk_count >= latin_word_count: + return "zh" + if greek_count > 0 and greek_count >= latin_word_count: + return "el" + if cyrillic_count > 0 and cyrillic_count >= latin_word_count: + if ukrainian_specific > 0: + return "uk" + if belarusian_specific > 0: + return "be" + return "ru" + if devanagari_count > 0 and devanagari_count >= latin_word_count: + return "hi" + if bengali_count > 0 and bengali_count >= latin_word_count: + return "bn" + if thai_count > 0 and thai_count >= latin_word_count: + return "th" + + # Latin-script language detection using stop words + if not words: + return "en" + + word_counts = Counter(words) + scores = {} + for lang, stop_words in _LATIN_STOP_WORDS.items(): + score = sum(word_counts[w] for w in stop_words if w in word_counts) + scores[lang] = score + + best_lang, best_score = max(scores.items(), key=lambda item: item[1]) + if best_score > 0: + return best_lang + + return "en" + + +def detect_direction_and_language( + text: str, + override_direction: Optional[str] = None, + override_language: Optional[str] = None, +) -> Tuple[str, str]: + """ + Determine (direction, language) with optional overrides. + + Direction: 'rtl' or 'ltr' + Language: ISO code, e.g. 'he', 'ar', 'en' + """ + norm_dir = ( + override_direction.strip().lower() + if override_direction and override_direction.strip().lower() != "auto" + else None + ) + + direction = norm_dir if norm_dir in ("rtl", "ltr") else detect_direction(text) + + if override_language and override_language.strip(): + language = override_language.strip().lower() + else: + language = detect_language(text, direction=direction) + + return direction, language + + +def detect_from_markdown_files( + markdown_path: Union[str, Path], + override_direction: Optional[str] = None, + override_language: Optional[str] = None, +) -> Tuple[str, str]: + """ + Detect direction and language by reading markdown files from a directory or single file. + """ + path = Path(markdown_path) + combined_text = [] + + if path.is_file() and path.suffix.lower() == ".md": + try: + combined_text.append(path.read_text(encoding="utf-8")) + except Exception: + pass + elif path.is_dir(): + for md_file in sorted(path.glob("*.md")): + try: + combined_text.append(md_file.read_text(encoding="utf-8")) + except Exception: + pass + + all_text = "\n".join(combined_text) + return detect_direction_and_language( + all_text, + override_direction=override_direction, + override_language=override_language, + ) diff --git a/modules/mark2epub.py b/modules/mark2epub.py index 7a289aa..fdea6da 100644 --- a/modules/mark2epub.py +++ b/modules/mark2epub.py @@ -4,22 +4,32 @@ import zipfile import sys import json +import io +import argparse from PIL import Image import regex as re from pathlib import Path from datetime import datetime, timezone import subprocess -from typing import Dict, Optional +from typing import Dict, Optional, Union from urllib.parse import quote from xml.sax.saxutils import escape as xml_escape import latex2mathml.converter +try: + from modules.bidi import detect_direction_and_language +except ImportError: + from bidi import detect_direction_and_language + def get_user_input(prompt: str, default: str = "") -> str: """Get user input with a default value.""" - user_input = input(f"{prompt} [{default}]: ").strip() - return user_input if user_input else default + try: + user_input = input(f"{prompt} [{default}]: ").strip() + return user_input if user_input else default + except (EOFError, io.UnsupportedOperation, OSError): + return default -def get_metadata_from_user(existing_metadata: Optional[Dict] = None) -> Dict: +def get_metadata_from_user(existing_metadata: Optional[Dict] = None, default_lang: str = "en") -> Dict: """Interactively collect metadata from user with defaults from existing metadata.""" if existing_metadata is None: existing_metadata = {} @@ -32,7 +42,7 @@ def get_metadata_from_user(existing_metadata: Optional[Dict] = None) -> Dict: "dc:title": ("Title", metadata.get("dc:title", "Untitled Document")), "dc:creator": ("Author(s)", metadata.get("dc:creator", "Unknown Author")), "dc:identifier": ("Unique Identifier", metadata.get("dc:identifier", f"id-{datetime.now().strftime('%Y%m%d%H%M%S')}")), - "dc:language": ("Language (e.g., en, de, fr)", metadata.get("dc:language", "en")), + "dc:language": ("Language (e.g., en, de, fr)", metadata.get("dc:language", default_lang)), "dc:rights": ("Rights", metadata.get("dc:rights", "All rights reserved")), "dc:publisher": ("Publisher", metadata.get("dc:publisher", "PDF2EPUB")), "dc:date": ("Publication Date (YYYY-MM-DD)", metadata.get("dc:date", datetime.now().strftime("%Y-%m-%d"))) @@ -47,7 +57,8 @@ def get_metadata_from_user(existing_metadata: Optional[Dict] = None) -> Dict: "metadata": updated_metadata, "default_css": existing_metadata.get("default_css", ["style.css"]), "chapters": existing_metadata.get("chapters", []), - "cover_image": existing_metadata.get("cover_image", None) + "cover_image": existing_metadata.get("cover_image", None), + "direction": existing_metadata.get("direction", "ltr") } def review_markdown(markdown_path: Path) -> tuple[bool, str]: @@ -55,7 +66,10 @@ def review_markdown(markdown_path: Path) -> tuple[bool, str]: content = markdown_path.read_text(encoding='utf-8') while True: - response = input("\nWould you like to review the markdown file before conversion? (y/n): ").lower() + try: + response = input("\nWould you like to review the markdown file before conversion? (y/n): ").lower() + except (EOFError, io.UnsupportedOperation, OSError): + return True, content if response in ['y', 'yes']: try: if sys.platform == 'darwin': @@ -170,13 +184,17 @@ def get_all_filenames(the_dir, extensions=[]): all_files = [x for x in all_files if x.split(".")[-1] in extensions] return all_files -def get_packageOPF_XML(md_filenames=[], image_filenames=[], css_filenames=[], description_data=None): +def get_packageOPF_XML(md_filenames=[], image_filenames=[], css_filenames=[], description_data=None, direction="ltr", lang="en"): doc = minidom.Document() + dir_val = (description_data.get("direction") if description_data else None) or direction + lang_val = (description_data.get("metadata", {}).get("dc:language") if description_data else None) or lang + package = doc.createElement('package') package.setAttribute('xmlns',"http://www.idpf.org/2007/opf") package.setAttribute('version',"3.0") - package.setAttribute('xml:lang',"en") + package.setAttribute('xml:lang', lang_val) + package.setAttribute('dir', dir_val) package.setAttribute("unique-identifier","pub-id") ## Now building the metadata @@ -184,14 +202,23 @@ def get_packageOPF_XML(md_filenames=[], image_filenames=[], css_filenames=[], de metadata = doc.createElement('metadata') metadata.setAttribute('xmlns:dc', 'http://purl.org/dc/elements/1.1/') - for k,v in description_data["metadata"].items(): - if len(v): - x = doc.createElement(k) - for metadata_type,id_label in [("dc:title","title"),("dc:creator","creator"),("dc:identifier","pub-id")]: - if k==metadata_type: - x.setAttribute('id',id_label) - x.appendChild(doc.createTextNode(v)) - metadata.appendChild(x) + has_language = False + if description_data and "metadata" in description_data: + for k, v in description_data["metadata"].items(): + if len(v): + if k == "dc:language": + has_language = True + x = doc.createElement(k) + for metadata_type, id_label in [("dc:title","title"),("dc:creator","creator"),("dc:identifier","pub-id")]: + if k == metadata_type: + x.setAttribute('id', id_label) + x.appendChild(doc.createTextNode(v)) + metadata.appendChild(x) + + if not has_language and lang_val: + x = doc.createElement("dc:language") + x.appendChild(doc.createTextNode(lang_val)) + metadata.appendChild(x) # Required by EPUB 3: dcterms:modified timestamp modified_meta = doc.createElement('meta') @@ -264,6 +291,7 @@ def get_packageOPF_XML(md_filenames=[], image_filenames=[], css_filenames=[], de spine = doc.createElement('spine') spine.setAttribute('toc', "ncx") + spine.setAttribute('page-progression-direction', dir_val) x = doc.createElement('itemref') x.setAttribute('idref',"titlepage") @@ -300,10 +328,10 @@ def get_container_XML(): container_data += """\n""" return container_data -def get_coverpage_XML(title, authors): +def get_coverpage_XML(title, authors, direction: str = "ltr", lang: str = "en"): """Generate a simple cover page with title and optional author input.""" return f""" - + Cover Page