Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -31,3 +31,8 @@ jobs:
# imports). Style rules are deliberately excluded.
- name: Lint
run: ruff check --select E9,F .

- name: Run tests
run: |
pip install pytest markdown latex2mathml pillow regex
pytest
16 changes: 16 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@ Convert PDF files to nicely structured Markdown and EPUB format with intelligent
- 📝 Clean markdown output with preserved structure
- 📱 EPUB generation with customizable styling
- 🌍 Multi-language support
- 🔄 Automatic RTL/LTR text direction and language detection with page progression support
- 🚀 GPU acceleration support (NVIDIA & AMD)
- 🍎 Apple Silicon support

Expand Down Expand Up @@ -135,6 +136,11 @@ Options:
--start-page INT Page number to start from
--skip-epub Skip EPUB generation, only create markdown
--skip-md Skip markdown generation, use existing markdown files
--direction, --dir {rtl,ltr,auto}
Set text direction (default: auto)
--rtl Force Right-to-Left text direction
--ltr Force Left-to-Right text direction
--lang, --language CODE Override document language code (e.g., he, ar, en, fr)
```

If `input_path` is omitted, all PDFs in `./input/` are processed.
Expand All @@ -151,6 +157,16 @@ Convert to markdown only:
python main.py thesis.pdf --skip-epub
```

Convert an RTL document with automatic direction and language detection:
```bash
python main.py hebrew_book.pdf
```

Force RTL direction and specify language code:
```bash
python main.py arabic_book.pdf --rtl --lang ar
```

### Output Structure

```
Expand Down
7 changes: 7 additions & 0 deletions conftest.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
import sys
from pathlib import Path

# Ensure repository root is on sys.path
root_dir = Path(__file__).resolve().parent
if str(root_dir) not in sys.path:
sys.path.insert(0, str(root_dir))
57 changes: 48 additions & 9 deletions main.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,17 +4,20 @@
from pathlib import Path
import modules.pdf2md as pdf2md
import modules.mark2epub as mark2epub
import torch

try:
import torch
except ImportError:
torch = None


def main():
if torch.cuda.is_available():
print("CUDA is available. Using GPU for processing.")
elif torch.backends.mps.is_available():
print("MPS is available. Using Apple Silicon for processing.")
else:
print("CUDA is not available. Using CPU for processing.")
if torch is not None:
if torch.cuda.is_available():
print("CUDA is available. Using GPU for processing.")
elif torch.backends.mps.is_available():
print("MPS is available. Using Apple Silicon for processing.")
else:
print("CUDA is not available. Using CPU for processing.")

parser = argparse.ArgumentParser(
description='Convert PDF files to EPUB format via Markdown'
Expand Down Expand Up @@ -53,9 +56,40 @@ def main():
action='store_true',
help='Skip markdown generation, use existing markdown files'
)
parser.add_argument(
'--direction', '--dir',
dest='direction',
choices=['rtl', 'ltr', 'auto'],
type=str.lower,
default='auto',
help='Set text direction: rtl, ltr, or auto (default: auto)'
)
parser.add_argument(
'--rtl',
action='store_true',
help='Convenience flag to force RTL text direction'
)
parser.add_argument(
'--ltr',
action='store_true',
help='Convenience flag to force LTR text direction'
)
parser.add_argument(
'--lang', '--language',
dest='language',
type=str,
default=None,
help='Override document language code (e.g., he, ar, en, fr)'
)

args = parser.parse_args()

direction = args.direction
if args.rtl:
direction = 'rtl'
elif args.ltr:
direction = 'ltr'

# Get input path
input_path = Path(args.input_path) if args.input_path else pdf2md.get_default_input_dir()

Expand Down Expand Up @@ -98,7 +132,12 @@ def main():
# Convert Markdown to EPUB unless skipped
if not args.skip_epub:
print("Converting Markdown to EPUB...")
mark2epub.convert_to_epub(markdown_dir, output_path)
mark2epub.convert_to_epub(
markdown_dir,
output_path,
direction=direction,
language=args.language,
)

except Exception as e:
print(f"Error processing {pdf_path.name}: {str(e)}", file=sys.stderr)
Expand Down
Loading