feat(audiobook): PDF ingest for /audiobook/import (ebook-in core value) (#459)

The audiobook importer accepted .txt/.md/.epub but not PDF — the single most
common "ebook in" format. Add a pure `pdf_to_chapter_script(data)` that
extracts the text layer page-by-page and runs it through the existing
chapterizer, so PDFs land in the same `# Heading` + body grammar EPUB and
plaintext already produce (one front door onto the unchanged render pipeline).

- Dep: `pypdf>=4.0` — pure-Python, MIT, zero native deps, so PDF import behaves
  identically on macOS/Windows/Linux (default-feature cross-platform rule).
  EPUB + plaintext stay stdlib-only; only PDF needs a real parser.
- Robustness, surfaced as actionable 400s rather than silent empty imports:
  corrupt file, password-protected (empty-password decrypt attempted first),
  scanned/image-only (no text layer → clear "scanned PDF" message), and a
  page-count ceiling. A single unparseable page is skipped, not fatal.
- Route: `.pdf` branch in audiobook_import; frontend accept filter +
  api-client doc updated to `.txt,.md,.epub,.pdf`.

tests/test_longform_import.py: 5 PDF cases (extract+chapterize, no-marker
single chapter, corrupt, image-only, page-cap) using a hand-built in-memory
PDF — no PDF-authoring test dep, mirroring the in-memory-EPUB approach.
16 passed; frontend suite 401; CJK guard green.
This commit is contained in:
Palash Debnath
2026-06-14 17:14:28 +05:30
committed by GitHub
parent 142b4bc25a
commit ca8a2e8eb8
7 changed files with 175 additions and 8 deletions
+15 -5
View File
@@ -97,12 +97,17 @@ _MAX_CHAPTERS = 10_000
@router.post("/audiobook/import")
async def audiobook_import(file: UploadFile = File(...)) -> dict:
"""Import a ``.txt``/``.md``/``.epub`` into a chapter-delimited script.
"""Import a ``.txt``/``.md``/``.epub``/``.pdf`` into a chapter-delimited script.
EPUB is parsed in spine order (stdlib only, local); plain text gets ``# ``
headings inserted ahead of obvious chapter-title lines. Returns the script
text (for the editor) + the resulting chapter count."""
from services.longform_import import chapterize_plaintext, epub_to_chapter_script
EPUB is parsed in spine order (stdlib only, local); PDF text is extracted
with pypdf (pure-Python) then chapterized; plain text gets ``# `` headings
inserted ahead of obvious chapter-title lines. Returns the script text (for
the editor) + the resulting chapter count."""
from services.longform_import import (
chapterize_plaintext,
epub_to_chapter_script,
pdf_to_chapter_script,
)
name = (file.filename or "").lower()
data = await file.read()
@@ -115,6 +120,11 @@ async def audiobook_import(file: UploadFile = File(...)) -> dict:
script = epub_to_chapter_script(data)
except ValueError as e:
raise HTTPException(status_code=400, detail=f"couldn't parse EPUB: {e}")
elif name.endswith(".pdf"):
try:
script = pdf_to_chapter_script(data)
except ValueError as e:
raise HTTPException(status_code=400, detail=f"couldn't parse PDF: {e}")
else:
script = chapterize_plaintext(data.decode("utf-8", "ignore"))
if not script.strip():
+59
View File
@@ -190,3 +190,62 @@ def epub_to_chapter_script(
if not blocks:
raise ValueError("no readable chapters found in the EPUB")
return "\n\n".join(blocks)
# Page-count ceiling for PDF ingestion — a defence against a pathological
# document tying up the worker. 5000 pages comfortably covers any real book.
_PDF_MAX_PAGES = 5000
def pdf_to_chapter_script(data: bytes, *, max_pages: int = _PDF_MAX_PAGES) -> str:
"""Convert PDF bytes into a ``# Chapter`` / body script.
Extracts the embedded text layer page-by-page (in page order), joins it,
and runs it through :func:`chapterize_plaintext` so ``Chapter N`` /
``Prologue`` lines become headings — same grammar EPUB and plaintext emit.
Unlike EPUB this needs a real parser (``pypdf``, pure-Python, no native
deps → identical on every platform).
Limitations surfaced as ``ValueError`` (the route maps these to a 400 with
the message, so the user gets actionable feedback rather than a silent
empty import):
* **Scanned / image-only PDFs** have no text layer — there's nothing to
extract without OCR, so we raise rather than return an empty script.
* **Password-protected PDFs** that don't open with an empty password can't
be read.
"""
from pypdf import PdfReader
from pypdf.errors import PdfReadError
try:
reader = PdfReader(io.BytesIO(data))
except (PdfReadError, OSError, ValueError) as e:
raise ValueError(f"not a valid PDF file: {e}") from e
if reader.is_encrypted:
# Many PDFs are encrypted with an empty user password (owner-locked but
# freely readable). Try that; a real password we can't supply.
try:
if reader.decrypt("") == 0: # 0 == wrong password
raise ValueError("PDF is password-protected")
except (NotImplementedError, PdfReadError) as e:
raise ValueError(f"can't read this encrypted PDF: {e}") from e
pages = reader.pages
if len(pages) > max_pages:
raise ValueError(f"PDF has too many pages (max {max_pages})")
parts: list[str] = []
for page in pages:
try:
text = page.extract_text() or ""
except Exception: # noqa: BLE001 — one bad page shouldn't kill the import
continue
if text.strip():
parts.append(text)
if not parts:
raise ValueError(
"no extractable text — this looks like a scanned or image-only PDF")
return chapterize_plaintext("\n\n".join(parts))
+1 -1
View File
@@ -89,7 +89,7 @@ export async function audiobookUploadCover(file: File): Promise<{ path: string }
return res.json();
}
/** Import a .txt/.md/.epub into a chapter-delimited script. */
/** Import a .txt/.md/.epub/.pdf into a chapter-delimited script. */
export async function audiobookImport(file: File): Promise<{ text: string; chapters: number }> {
const form = new FormData();
form.append('file', file);
+1 -1
View File
@@ -209,7 +209,7 @@ export default function AudiobookTab({ profiles = [] }) {
<div className="audiobook-tab__actions">
<label className="ui-btn ui-btn--subtle" style={{ cursor: busy ? 'default' : 'pointer', display: 'inline-flex', alignItems: 'center', gap: 6 }}>
{importing ? <Loader size={14} className="spin" /> : <Upload size={14} />} {t('audiobook.import')}
<input type="file" accept=".txt,.md,.epub" onChange={onImport} disabled={busy} style={{ display: 'none' }} />
<input type="file" accept=".txt,.md,.epub,.pdf" onChange={onImport} disabled={busy} style={{ display: 'none' }} />
</label>
<button className="ui-btn ui-btn--subtle" onClick={onPreview} disabled={!canRun}>
{planLoading ? <Loader size={14} className="spin" /> : null} {t('audiobook.preview_plan')}
+5
View File
@@ -120,6 +120,11 @@ dependencies = [
# callbacks (the accurate-progress work in this same plan depends on tqdm).
"huggingface_hub>=1.7",
"hf-xet>=1.1",
# Audiobook PDF ingest (/audiobook/import). Pure-Python, MIT, zero native
# deps → identical behaviour on macOS/Windows/Linux (default-parity rule).
# EPUB + plaintext stay stdlib-only; only PDF needs a real parser, and
# pypdf is the lightest one that ships no C extensions.
"pypdf>=4.0",
]
[project.optional-dependencies]
+83 -1
View File
@@ -10,10 +10,53 @@ import zipfile
import pytest
from services.longform_import import chapterize_plaintext, epub_to_chapter_script
from services.longform_import import (
chapterize_plaintext,
epub_to_chapter_script,
pdf_to_chapter_script,
)
from services.audiobook import parse_audiobook_script
# ── PDF fixture builder ───────────────────────────────────────────────────
# A minimal hand-built single-page PDF with a Helvetica text layer, so the PDF
# tests need no PDF-authoring dependency (mirrors the in-memory-EPUB approach).
def _make_pdf(lines: list[str], *, content_override: bytes | None = None) -> bytes:
if content_override is not None:
content = content_override
else:
show = "BT /F1 12 Tf 72 720 Td 16 TL\n"
for ln in lines:
esc = ln.replace("\\", "\\\\").replace("(", r"\(").replace(")", r"\)")
show += f"({esc}) Tj T*\n"
show += "ET"
content = show.encode("latin-1")
objs = [
b"<< /Type /Catalog /Pages 2 0 R >>",
b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] "
b"/Resources << /Font << /F1 4 0 R >> >> /Contents 5 0 R >>",
b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
b"<< /Length " + str(len(content)).encode() + b" >>\nstream\n" + content + b"\nendstream",
]
out = io.BytesIO()
out.write(b"%PDF-1.4\n")
offsets = []
for i, body in enumerate(objs, 1):
offsets.append(out.tell())
out.write(f"{i} 0 obj\n".encode() + body + b"\nendobj\n")
xref_pos = out.tell()
n = len(objs) + 1
out.write(f"xref\n0 {n}\n".encode())
out.write(b"0000000000 65535 f \n")
for off in offsets:
out.write(f"{off:010d} 00000 n \n".encode())
out.write(f"trailer\n<< /Size {n} /Root 1 0 R >>\nstartxref\n{xref_pos}\n%%EOF".encode())
return out.getvalue()
# ── plain text ──────────────────────────────────────────────────────────────
def test_plaintext_leaves_existing_h1_untouched():
@@ -137,3 +180,42 @@ def test_epub_oversize_entry_skipped():
data = _make_epub([("Big", "x" * 500)])
with pytest.raises(ValueError): # the one entry exceeds the cap → all skipped
epub_to_chapter_script(data, max_entry_bytes=50)
# ── PDF ──────────────────────────────────────────────────────────────────
def test_pdf_extracts_and_chapterizes():
data = _make_pdf(["Chapter 1", "Once upon a time.", "Chapter 2", "The end."])
script = pdf_to_chapter_script(data)
# Chapter-keyword lines from the extracted text become headings …
assert "# Chapter 1" in script
assert "# Chapter 2" in script
# … and the body survives.
assert "Once upon a time." in script
# And it parses into two chapters via the shared grammar.
assert len(parse_audiobook_script(script).chapters) == 2
def test_pdf_without_chapter_markers_is_single_chapter():
data = _make_pdf(["Just some flowing prose.", "With no chapter headings at all."])
script = pdf_to_chapter_script(data)
assert "With no chapter headings" in script
assert len(parse_audiobook_script(script).chapters) == 1
def test_pdf_corrupt_raises_valueerror():
with pytest.raises(ValueError):
pdf_to_chapter_script(b"this is definitely not a pdf")
def test_pdf_image_only_raises_actionable_error():
# A valid PDF whose page has no text-showing operators → nothing to extract.
data = _make_pdf([], content_override=b"q Q") # graphics-only, no BT/Tj
with pytest.raises(ValueError, match="scanned or image-only"):
pdf_to_chapter_script(data)
def test_pdf_too_many_pages_guard():
data = _make_pdf(["Chapter 1", "Hi."])
with pytest.raises(ValueError, match="too many pages"):
pdf_to_chapter_script(data, max_pages=0)
Generated
+11
View File
@@ -3117,6 +3117,7 @@ dependencies = [
{ name = "pyannote-audio" },
{ name = "pydub" },
{ name = "pyinstaller" },
{ name = "pypdf" },
{ name = "python-multipart" },
{ name = "scalar-fastapi" },
{ name = "setuptools" },
@@ -3190,6 +3191,7 @@ requires-dist = [
{ name = "pyannote-audio", specifier = ">=3.3.2,<4.0" },
{ name = "pydub" },
{ name = "pyinstaller", specifier = ">=6.19.0" },
{ name = "pypdf", specifier = ">=4.0" },
{ name = "python-multipart" },
{ name = "requests", marker = "extra == 'ui'" },
{ name = "s3prl", marker = "extra == 'eval'" },
@@ -4209,6 +4211,15 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/10/bd/c038d7cc38edc1aa5bf91ab8068b63d4308c66c4c8bb3cbba7dfbc049f9c/pyparsing-3.3.2-py3-none-any.whl", hash = "sha256:850ba148bd908d7e2411587e247a1e4f0327839c40e2e5e6d05a007ecc69911d", size = 122781, upload-time = "2026-01-21T03:57:55.912Z" },
]
[[package]]
name = "pypdf"
version = "6.13.2"
source = { registry = "https://pypi.org/simple" }
sdist = { url = "https://files.pythonhosted.org/packages/99/0a/48fe05c6bb3aa4bb4d2a4079a383d33c0dfec1edf613a642f07d8b8b5c2e/pypdf-6.13.2.tar.gz", hash = "sha256:5a96a17dbdfbf9c2ab24c0a13fa0aba182be22ba6f283098712c16fc242f509f", size = 6479250, upload-time = "2026-06-10T16:42:34.5Z" }
wheels = [
{ url = "https://files.pythonhosted.org/packages/cb/17/378943705992f74e451a06de3401ce68e3213763c81e44d0614559c45599/pypdf-6.13.2-py3-none-any.whl", hash = "sha256:6eeb9e57693f29d41bd01255d02660cbbb41fd7fc818a982677389a35e4f2083", size = 346555, upload-time = "2026-06-10T16:42:32.37Z" },
]
[[package]]
name = "pytest"
version = "9.0.3"