From ad1befcc7eec3c57c2b7eaae22beb489e5727c53 Mon Sep 17 00:00:00 2001 From: kisjelica Date: Fri, 25 Sep 2026 11:53:28 +0200 Subject: [PATCH] fix: strip script/style elements when converting HTML to XHTML XPath text extraction (e.g. normalize-space() over a content container) returns a node's full string-value, which includes descendant text nodes inside nested + +

Real content after.

+ + +""" + xhtml_output = converter.convert(html_input) + + assert " None: def _remove_problem_nodes(self, doc: Any) -> None: if not hasattr(doc, "xpath"): return - for query in ("//comment()", "//processing-instruction()"): + for query in ( + "//comment()", + "//processing-instruction()", + "//script", + "//style", + ): for node in doc.xpath(query): parent = node.getparent() if parent is not None: