Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions src/note_mcp/utils/html_to_markdown.py
Original file line number Diff line number Diff line change
Expand Up @@ -313,6 +313,9 @@ def _convert_list(html_content: str, ordered: bool = False, indent_level: int =
text = _OL_PATTERN.sub("", text)
text = text.strip()

# Preserve supported inline elements before removing residual HTML tags.
text = _convert_inline_elements(text)

# Clean up any remaining HTML tags from text
text = re.sub(r"<[^>]+>", "", text).strip()

Expand Down
43 changes: 43 additions & 0 deletions tests/unit/test_html_to_markdown.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,8 @@
Tests for html_to_markdown function that converts note.com HTML format back to Markdown.
"""

from bs4 import BeautifulSoup

from note_mcp.utils.html_to_markdown import html_to_markdown
from note_mcp.utils.markdown_to_html import markdown_to_html

Expand Down Expand Up @@ -102,6 +104,47 @@ def test_nested_list_conversion(self) -> None:
for line in sub_items:
assert line.startswith(" ") or line.startswith(" ")

def test_ordered_and_unordered_list_links_are_preserved(self) -> None:
"""Test that links survive conversion from paragraph-wrapped list items."""
for tag, marker in (("ul", "-"), ("ol", "1.")):
html = f"""<{tag}>
<li><p>Read the <a href="https://github.com/drillan/note-mcp">official docs</a>.</p></li>
</{tag}>"""

result = html_to_markdown(html)

assert f"{marker} Read the [official docs](https://github.com/drillan/note-mcp)." in result

def test_nested_list_links_preserve_formatted_labels(self) -> None:
"""Test links and inline formatting in nested list items."""
reference_url = "https://note.com/drillan/n/n7379c02632c9"
html = f"""<ul>
<li><p>Start with <a href="https://github.com/drillan/note-mcp"><strong>API docs</strong></a>.</p>
<ol>
<li><p>Then read the <a href="{reference_url}"><em>reference guide</em></a>.</p></li>
</ol>
</li>
</ul>"""

result = html_to_markdown(html)

assert "- Start with [**API docs**](https://github.com/drillan/note-mcp)." in result
assert f" 1. Then read the [*reference guide*]({reference_url})." in result

def test_list_link_roundtrip_preserves_label_and_href(self) -> None:
"""Test that a list link keeps its label and href through HTML and Markdown."""
source_html = """<ol>
<li><p>Consult <a href="https://github.com/drillan/note-mcp"><strong>the guide</strong></a>.</p></li>
</ol>"""

markdown = html_to_markdown(source_html)
roundtrip_html = markdown_to_html(markdown)
links = BeautifulSoup(roundtrip_html, "html.parser").select('li a[href="https://github.com/drillan/note-mcp"]')

assert "1. Consult [**the guide**](https://github.com/drillan/note-mcp)." in markdown
assert len(links) == 1
assert links[0].get_text() == "the guide"

# === Blockquote Tests ===

def test_blockquote_conversion(self) -> None:
Expand Down