From ca8a2e8eb8475cd8f0be554f4657d105e35fb2de Mon Sep 17 00:00:00 2001 From: Palash Debnath Date: Sun, 14 Jun 2026 17:14:28 +0530 Subject: [PATCH] feat(audiobook): PDF ingest for /audiobook/import (ebook-in core value) (#459) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The audiobook importer accepted .txt/.md/.epub but not PDF — the single most common "ebook in" format. Add a pure `pdf_to_chapter_script(data)` that extracts the text layer page-by-page and runs it through the existing chapterizer, so PDFs land in the same `# Heading` + body grammar EPUB and plaintext already produce (one front door onto the unchanged render pipeline). - Dep: `pypdf>=4.0` — pure-Python, MIT, zero native deps, so PDF import behaves identically on macOS/Windows/Linux (default-feature cross-platform rule). EPUB + plaintext stay stdlib-only; only PDF needs a real parser. - Robustness, surfaced as actionable 400s rather than silent empty imports: corrupt file, password-protected (empty-password decrypt attempted first), scanned/image-only (no text layer → clear "scanned PDF" message), and a page-count ceiling. A single unparseable page is skipped, not fatal. - Route: `.pdf` branch in audiobook_import; frontend accept filter + api-client doc updated to `.txt,.md,.epub,.pdf`. tests/test_longform_import.py: 5 PDF cases (extract+chapterize, no-marker single chapter, corrupt, image-only, page-cap) using a hand-built in-memory PDF — no PDF-authoring test dep, mirroring the in-memory-EPUB approach. 16 passed; frontend suite 401; CJK guard green. --- backend/api/routers/audiobook.py | 20 +++++-- backend/services/longform_import.py | 59 ++++++++++++++++++++ frontend/src/api/audiobook.ts | 2 +- frontend/src/pages/AudiobookTab.jsx | 2 +- pyproject.toml | 5 ++ tests/test_longform_import.py | 84 ++++++++++++++++++++++++++++- uv.lock | 11 ++++ 7 files changed, 175 insertions(+), 8 deletions(-) diff --git a/backend/api/routers/audiobook.py b/backend/api/routers/audiobook.py index 757bdd05..52cda953 100644 --- a/backend/api/routers/audiobook.py +++ b/backend/api/routers/audiobook.py @@ -97,12 +97,17 @@ _MAX_CHAPTERS = 10_000 @router.post("/audiobook/import") async def audiobook_import(file: UploadFile = File(...)) -> dict: - """Import a ``.txt``/``.md``/``.epub`` into a chapter-delimited script. + """Import a ``.txt``/``.md``/``.epub``/``.pdf`` into a chapter-delimited script. - EPUB is parsed in spine order (stdlib only, local); plain text gets ``# `` - headings inserted ahead of obvious chapter-title lines. Returns the script - text (for the editor) + the resulting chapter count.""" - from services.longform_import import chapterize_plaintext, epub_to_chapter_script + EPUB is parsed in spine order (stdlib only, local); PDF text is extracted + with pypdf (pure-Python) then chapterized; plain text gets ``# `` headings + inserted ahead of obvious chapter-title lines. Returns the script text (for + the editor) + the resulting chapter count.""" + from services.longform_import import ( + chapterize_plaintext, + epub_to_chapter_script, + pdf_to_chapter_script, + ) name = (file.filename or "").lower() data = await file.read() @@ -115,6 +120,11 @@ async def audiobook_import(file: UploadFile = File(...)) -> dict: script = epub_to_chapter_script(data) except ValueError as e: raise HTTPException(status_code=400, detail=f"couldn't parse EPUB: {e}") + elif name.endswith(".pdf"): + try: + script = pdf_to_chapter_script(data) + except ValueError as e: + raise HTTPException(status_code=400, detail=f"couldn't parse PDF: {e}") else: script = chapterize_plaintext(data.decode("utf-8", "ignore")) if not script.strip(): diff --git a/backend/services/longform_import.py b/backend/services/longform_import.py index 2745f889..2f2d5a41 100644 --- a/backend/services/longform_import.py +++ b/backend/services/longform_import.py @@ -190,3 +190,62 @@ def epub_to_chapter_script( if not blocks: raise ValueError("no readable chapters found in the EPUB") return "\n\n".join(blocks) + + +# Page-count ceiling for PDF ingestion — a defence against a pathological +# document tying up the worker. 5000 pages comfortably covers any real book. +_PDF_MAX_PAGES = 5000 + + +def pdf_to_chapter_script(data: bytes, *, max_pages: int = _PDF_MAX_PAGES) -> str: + """Convert PDF bytes into a ``# Chapter`` / body script. + + Extracts the embedded text layer page-by-page (in page order), joins it, + and runs it through :func:`chapterize_plaintext` so ``Chapter N`` / + ``Prologue`` lines become headings — same grammar EPUB and plaintext emit. + Unlike EPUB this needs a real parser (``pypdf``, pure-Python, no native + deps → identical on every platform). + + Limitations surfaced as ``ValueError`` (the route maps these to a 400 with + the message, so the user gets actionable feedback rather than a silent + empty import): + + * **Scanned / image-only PDFs** have no text layer — there's nothing to + extract without OCR, so we raise rather than return an empty script. + * **Password-protected PDFs** that don't open with an empty password can't + be read. + """ + from pypdf import PdfReader + from pypdf.errors import PdfReadError + + try: + reader = PdfReader(io.BytesIO(data)) + except (PdfReadError, OSError, ValueError) as e: + raise ValueError(f"not a valid PDF file: {e}") from e + + if reader.is_encrypted: + # Many PDFs are encrypted with an empty user password (owner-locked but + # freely readable). Try that; a real password we can't supply. + try: + if reader.decrypt("") == 0: # 0 == wrong password + raise ValueError("PDF is password-protected") + except (NotImplementedError, PdfReadError) as e: + raise ValueError(f"can't read this encrypted PDF: {e}") from e + + pages = reader.pages + if len(pages) > max_pages: + raise ValueError(f"PDF has too many pages (max {max_pages})") + + parts: list[str] = [] + for page in pages: + try: + text = page.extract_text() or "" + except Exception: # noqa: BLE001 — one bad page shouldn't kill the import + continue + if text.strip(): + parts.append(text) + + if not parts: + raise ValueError( + "no extractable text — this looks like a scanned or image-only PDF") + return chapterize_plaintext("\n\n".join(parts)) diff --git a/frontend/src/api/audiobook.ts b/frontend/src/api/audiobook.ts index cd89f942..0d00b058 100644 --- a/frontend/src/api/audiobook.ts +++ b/frontend/src/api/audiobook.ts @@ -89,7 +89,7 @@ export async function audiobookUploadCover(file: File): Promise<{ path: string } return res.json(); } -/** Import a .txt/.md/.epub into a chapter-delimited script. */ +/** Import a .txt/.md/.epub/.pdf into a chapter-delimited script. */ export async function audiobookImport(file: File): Promise<{ text: string; chapters: number }> { const form = new FormData(); form.append('file', file); diff --git a/frontend/src/pages/AudiobookTab.jsx b/frontend/src/pages/AudiobookTab.jsx index d1d6a9cc..3f093e15 100644 --- a/frontend/src/pages/AudiobookTab.jsx +++ b/frontend/src/pages/AudiobookTab.jsx @@ -209,7 +209,7 @@ export default function AudiobookTab({ profiles = [] }) {