From 4811cab33c4f22b0e5d1c1033ab3e9fc78bbc547 Mon Sep 17 00:00:00 2001 From: Rudy Celekli <47457359+rudycelekli@users.noreply.github.com> Date: Tue, 6 Oct 2026 07:25:01 -0400 Subject: [PATCH 1/3] fix(longform): preserve EPUB block element boundaries Signed-off-by: Rudy Celekli <47457359+rudycelekli@users.noreply.github.com> --- CHANGELOG.md | 2 ++ backend/services/longform_import.py | 12 +++++++++++- .../test_longform_import_parse_failure.py | 18 ++++++++++++++++++ docs/electron-longform.md | 2 ++ 4 files changed, 33 insertions(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 534142f5a..a93c4b139 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,8 @@ metadata and the backend fallback mirror it. ### Fixed +- EPUB imports preserve word and paragraph boundaries around block elements (#0) — thanks @rudycelekli! + - MCP speech tools wait through model loading and progress-extended CPU renders instead of timing out before the backend (#2609) ## [0.5.7] — 2026-10-05 diff --git a/backend/services/longform_import.py b/backend/services/longform_import.py index 9ab091677..4056cf775 100644 --- a/backend/services/longform_import.py +++ b/backend/services/longform_import.py @@ -129,7 +129,12 @@ class _TextExtractor(HTMLParser): whitespace. First
unbreakable
") + assert body == "unbreakable" + + +def test_heading_and_pagebreak_do_not_become_body_text(): + title, body = li._html_to_title_body('First20 word.
After.') + assert title == "Chapter One" + assert body.splitlines() == ["First word.", "After."] + + def _make_epub(n_chapters: int = 3) -> bytes: buf = io.BytesIO() with zipfile.ZipFile(buf, "w") as z: @@ -50,6 +67,7 @@ def _make_epub(n_chapters: int = 3) -> bytes: def test_clean_epub_yields_all_chapters(): script = li.epub_to_chapter_script(_make_epub()) assert script.count("# Chapter") == 3 + assert "Opening paragraph of chapter 1.\n\nClosing paragraph of chapter 1." in script def test_mid_chapter_parse_failure_keeps_partial_text(monkeypatch, caplog): diff --git a/docs/electron-longform.md b/docs/electron-longform.md index 7040b2373..b8ca0007a 100644 --- a/docs/electron-longform.md +++ b/docs/electron-longform.md @@ -71,3 +71,5 @@ The finished-render library filters for Stories and Audiobooks before applying i Resuming starts a fresh job while preserving the original checkpoint until all chapters render successfully. Checkpoint writes and syncs are best effort, so an interrupted resumed job can leave both the original and new checkpoints available; either can reuse the shared chapter cache. Closing an unstarted response or interrupting the render leaves the original plan available, including when a replacement checkpoint could not be saved. A render with no failed chapters retires the original even if the replacement checkpoint could not be saved. A partial output keeps the original plan available to retry its failed chapters. Cache pruning evicts the oldest chapter and segment audio, including interrupted partial files, while retaining JSON bookkeeping needed to locate older cache entries after moving the data directory. Bookkeeping contributes to the reported cache size; if it alone exceeds the budget, pruning remains best effort. + +EPUB block elements separate adjacent prose at both their opening and closing edges. Inline emphasis keeps word fragments joined, and heading text stays in chapter metadata. From 32471d3c2565590dd25a8fc6bded1a8f5fce45a0 Mon Sep 17 00:00:00 2001 From: Rudy Celekli <47457359+rudycelekli@users.noreply.github.com> Date: Tue, 6 Oct 2026 07:49:19 -0400 Subject: [PATCH 2/3] docs: link epub-block regression report (#2630) Signed-off-by: Rudy Celekli <47457359+rudycelekli@users.noreply.github.com> --- CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a93c4b139..b9e80355f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,7 +14,7 @@ metadata and the backend fallback mirror it. ### Fixed -- EPUB imports preserve word and paragraph boundaries around block elements (#0) — thanks @rudycelekli! +- EPUB imports preserve word and paragraph boundaries around block elements (#2630) — thanks @rudycelekli! - MCP speech tools wait through model loading and progress-extended CPU renders instead of timing out before the backend (#2609) From 15242771049c06463cbdc242f7a65a433c382b86 Mon Sep 17 00:00:00 2001 From: Rudy Celekli <47457359+rudycelekli@users.noreply.github.com> Date: Tue, 6 Oct 2026 08:01:27 -0400 Subject: [PATCH 3/3] fix(longform): retain single breaks for EPUB void tags Signed-off-by: Rudy Celekli <47457359+rudycelekli@users.noreply.github.com> --- backend/services/longform_import.py | 2 +- .../test_longform_import_parse_failure.py | 28 +++++++++++++++++++ docs/electron-longform.md | 2 ++ 3 files changed, 31 insertions(+), 1 deletion(-) diff --git a/backend/services/longform_import.py b/backend/services/longform_import.py index 4056cf775..860ab74e0 100644 --- a/backend/services/longform_import.py +++ b/backend/services/longform_import.py @@ -228,7 +228,7 @@ def handle_endtag(self, tag): if self._in_title and tag == self._title_tag: self._in_title = False self.title = " ".join("".join(self._title_parts).split()) - if tag in self._BREAK and not self._skip_depth: + if tag in self._BREAK and tag not in self._VOID and not self._skip_depth: if self._in_title: self._title_parts.append(" ") else: diff --git a/backend/tests/test_longform_import_parse_failure.py b/backend/tests/test_longform_import_parse_failure.py index 095989ce3..c78982542 100644 --- a/backend/tests/test_longform_import_parse_failure.py +++ b/backend/tests/test_longform_import_parse_failure.py @@ -110,3 +110,31 @@ def dead_feed(self, data): script = li.epub_to_chapter_script(_make_epub()) assert "# Chapter 1" in script and "# Chapter 3" in script assert "chapter 2" not in script.lower() + + +@pytest.mark.parametrize("line_break", ["Line one.{line_break}Line two.
After.") + + assert body == "Line one.\nLine two.\nAfter." + + +@pytest.mark.parametrize("line_break", ["