diff --git a/CHANGELOG.md b/CHANGELOG.md index 534142f5a..b9e80355f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,8 @@ metadata and the backend fallback mirror it. ### Fixed +- EPUB imports preserve word and paragraph boundaries around block elements (#2630) — thanks @rudycelekli! + - MCP speech tools wait through model loading and progress-extended CPU renders instead of timing out before the backend (#2609) ## [0.5.7] — 2026-10-05 diff --git a/backend/services/longform_import.py b/backend/services/longform_import.py index 9ab091677..860ab74e0 100644 --- a/backend/services/longform_import.py +++ b/backend/services/longform_import.py @@ -129,7 +129,12 @@ class _TextExtractor(HTMLParser): whitespace. First

/

/ seen is kept as the chapter title.""" _SKIP = {"script", "style", "head"} - _BREAK = {"p", "br", "div", "h1", "h2", "h3", "li", "tr"} + _BREAK = { + "p", "br", "hr", "div", "h1", "h2", "h3", "h4", "h5", "h6", + "li", "tr", "td", "th", "table", "ul", "ol", "dl", "dt", "dd", + "section", "article", "main", "header", "footer", "aside", + "blockquote", "pre", "address", "figure", "figcaption", + } #: A print page number carried into the EPUB (EPUB 3 ``epub:type="pagebreak"``, #: ARIA ``role="doc-pagebreak"``, or a publisher class such as #: ``pagebreak-rw``). Inline, it glues onto prose ("happily as 2Zoe threw"); @@ -223,6 +228,11 @@ def handle_endtag(self, tag): if self._in_title and tag == self._title_tag: self._in_title = False self.title = " ".join("".join(self._title_parts).split()) + if tag in self._BREAK and tag not in self._VOID and not self._skip_depth: + if self._in_title: + self._title_parts.append(" ") + else: + self._parts.append("\n") def handle_data(self, data): if self._skip_depth or self._pagebreak_stack: diff --git a/backend/tests/test_longform_import_parse_failure.py b/backend/tests/test_longform_import_parse_failure.py index f4c5be4e5..c78982542 100644 --- a/backend/tests/test_longform_import_parse_failure.py +++ b/backend/tests/test_longform_import_parse_failure.py @@ -15,6 +15,23 @@ from services import longform_import as li +@pytest.mark.parametrize("tag", ["p", "div", "li", "tr", "section", "blockquote", "dd", "h4"]) +def test_block_edges_separate_adjacent_prose(tag): + _, body = li._html_to_title_body(f"<body>Before<{tag}>Inside</{tag}>After</body>") + assert body.splitlines() == ["Before", "Inside", "After"] + + +def test_inline_markup_keeps_word_fragments_joined(): + _, body = li._html_to_title_body("<body><p>un<em>break</em>able</p></body>") + assert body == "unbreakable" + + +def test_heading_and_pagebreak_do_not_become_body_text(): + title, body = li._html_to_title_body('<body><h1>Chapter <em>One</em></h1><p>First<span role="doc-pagebreak">20</span> word.</p>After.</body>') + assert title == "Chapter One" + assert body.splitlines() == ["First word.", "After."] + + def _make_epub(n_chapters: int = 3) -> bytes: buf = io.BytesIO() with zipfile.ZipFile(buf, "w") as z: @@ -50,6 +67,7 @@ def _make_epub(n_chapters: int = 3) -> bytes: def test_clean_epub_yields_all_chapters(): script = li.epub_to_chapter_script(_make_epub()) assert script.count("# Chapter") == 3 + assert "Opening paragraph of chapter 1.\n\nClosing paragraph of chapter 1." in script def test_mid_chapter_parse_failure_keeps_partial_text(monkeypatch, caplog): @@ -92,3 +110,31 @@ def dead_feed(self, data): script = li.epub_to_chapter_script(_make_epub()) assert "# Chapter 1" in script and "# Chapter 3" in script assert "chapter 2" not in script.lower() + + +@pytest.mark.parametrize("line_break", ["<br>", "<br/>", "<br />", "<BR/>"]) +def test_void_line_break_is_not_a_paragraph_break(line_break): + from services import longform_import + + _, body = longform_import._html_to_title_body(f"<body><p>Line one.{line_break}Line two.</p>After.</body>") + + assert body == "Line one.\nLine two.\nAfter." + + +@pytest.mark.parametrize("line_break", ["<br>", "<br/>"]) +def test_epub_preserves_line_break_inside_paragraph(line_break): + from services import longform_import + + # Repack the actual EPUB fixture with one multiline opening paragraph. + packed = io.BytesIO() + with zipfile.ZipFile(io.BytesIO(_make_epub(1))) as source, zipfile.ZipFile(packed, "w") as dest: + for info in source.infolist(): + content = source.read(info.filename) + if info.filename == "OEBPS/ch1.xhtml": + content = content.replace(b"Opening paragraph of chapter 1.", + f"Line one.{line_break}Line two.".encode()) + dest.writestr(info, content) + + script = longform_import.epub_to_chapter_script(packed.getvalue()) + + assert "Line one.\nLine two.\n\nClosing paragraph of chapter 1." in script diff --git a/docs/electron-longform.md b/docs/electron-longform.md index 7040b2373..db6fca3ef 100644 --- a/docs/electron-longform.md +++ b/docs/electron-longform.md @@ -71,3 +71,7 @@ The finished-render library filters for Stories and Audiobooks before applying i Resuming starts a fresh job while preserving the original checkpoint until all chapters render successfully. Checkpoint writes and syncs are best effort, so an interrupted resumed job can leave both the original and new checkpoints available; either can reuse the shared chapter cache. Closing an unstarted response or interrupting the render leaves the original plan available, including when a replacement checkpoint could not be saved. A render with no failed chapters retires the original even if the replacement checkpoint could not be saved. A partial output keeps the original plan available to retry its failed chapters. Cache pruning evicts the oldest chapter and segment audio, including interrupted partial files, while retaining JSON bookkeeping needed to locate older cache entries after moving the data directory. Bookkeeping contributes to the reported cache size; if it alone exceeds the budget, pruning remains best effort. + +EPUB block elements separate adjacent prose at both their opening and closing edges. Inline emphasis keeps word fragments joined, and heading text stays in chapter metadata. + +EPUB line breaks such as `<br/>` remain single line breaks inside a paragraph; separate paragraph elements retain their paragraph boundaries.