diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index 32a6720..b886dcc 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -37,7 +37,7 @@ jobs:
steps:
- name: Checkout
- uses: actions/checkout@v4
+ uses: actions/checkout@v6
- name: Set up Python ${{ matrix.python-version }}
uses: actions/setup-python@v6
@@ -86,7 +86,7 @@ jobs:
PIP_DISABLE_PIP_VERSION_CHECK: "1"
steps:
- name: Checkout
- uses: actions/checkout@v4
+ uses: actions/checkout@v6
- name: Set up Python 3.12
uses: actions/setup-python@v6
@@ -134,7 +134,7 @@ jobs:
SKIP_CF_TEST: "0"
steps:
- name: Checkout
- uses: actions/checkout@v4
+ uses: actions/checkout@v6
- name: Set up Python 3.12
uses: actions/setup-python@v6
diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml
index 8033321..3c8e06b 100644
--- a/.github/workflows/publish.yml
+++ b/.github/workflows/publish.yml
@@ -26,7 +26,7 @@ jobs:
PIP_DISABLE_PIP_VERSION_CHECK: "1"
steps:
- name: Checkout
- uses: actions/checkout@v4
+ uses: actions/checkout@v6
- name: Set up Python
uses: actions/setup-python@v6
diff --git a/run_server.py b/run_server.py
index 859f22e..4881cdc 100644
--- a/run_server.py
+++ b/run_server.py
@@ -1,11 +1,11 @@
-# ./run_server.py
-"""
-Shim for local development.
-Please use 'web-scraper-server' (or 'python -m web_scraper_toolkit.server.mcp_server')
-when installed.
-"""
-
-from web_scraper_toolkit.server.mcp_server import main
-
-if __name__ == "__main__":
- main()
+# ./run_server.py
+"""
+Shim for local development.
+Please use 'web-scraper-server' (or 'python -m web_scraper_toolkit.server.mcp_server')
+when installed.
+"""
+
+from web_scraper_toolkit.server.mcp_server import main
+
+if __name__ == "__main__":
+ main()
diff --git a/src/web_scraper_toolkit/core/diagnostics.py b/src/web_scraper_toolkit/core/diagnostics.py
index fcbc27a..f96c0a1 100644
--- a/src/web_scraper_toolkit/core/diagnostics.py
+++ b/src/web_scraper_toolkit/core/diagnostics.py
@@ -1,112 +1,112 @@
-# ./src/web_scraper_toolkit/core/diagnostics.py
-"""
-Diagnostics Module
-==================
-
-Provides system self-checks to ensure the environment is correctly configured.
-Checks Playwright installation, browser binaries, and network connectivity.
-
-Usage:
- print_diagnostics()
-
-Outputs:
- - Printed report of system status to stdout.
-"""
-
-import asyncio
-import logging
-import sys
-from typing import Dict, Any
-
-from playwright.async_api import async_playwright
-
-logger = logging.getLogger(__name__)
-
-
-async def verify_environment() -> Dict[str, Any]:
- """
- Checks the scraping environment for necessary dependencies and functionality.
-
- Returns:
- Dict containing status of:
- - python_version
- - playwright_installed
- - browser_launch_successful
- - browsers_found
- """
- report: Dict[str, Any] = {
- "python_version": sys.version,
- "playwright_installed": False,
- "browser_launch_successful": False,
- "browsers_found": [],
- "errors": [],
- }
-
- # 1. Check Playwright Import
- try:
- from importlib.util import find_spec
-
- if find_spec("playwright"):
- report["playwright_installed"] = True
- else:
- report["errors"].append("Playwright package not found.")
- except ImportError:
- report["errors"].append("Playwright package not installed.")
- return report
-
- # 2. Check Browsers
- try:
- async with async_playwright() as p:
- # Try launching Chromium
- try:
- browser = await p.chromium.launch(headless=True)
- report["browsers_found"].append("chromium")
- report["browser_launch_successful"] = True
- await browser.close()
- except Exception as e:
- report["errors"].append(f"Chromium launch failed: {e}")
-
- # Try launching Firefox (optional but good to know)
- try:
- browser = await p.firefox.launch(headless=True)
- report["browsers_found"].append("firefox")
- await browser.close()
- except Exception:
- pass # Firefox might not be installed, that's okay
-
- except Exception as e:
- report["errors"].append(f"Playwright runtime error: {e}")
-
- return report
-
-
-def print_diagnostics() -> None:
- """Runs the verification and prints a human-readable report."""
- print("Running WebScraperToolkit Diagnostics...")
- try:
- results = asyncio.run(verify_environment())
-
- print(f"\nPython Version: {results['python_version'].split()[0]}")
- print(
- f"Playwright Installed: {'✅' if results['playwright_installed'] else '❌'}"
- )
-
- if results["playwright_installed"]:
- if results["browser_launch_successful"]:
- print("Browser Launch: ✅ (Chromium)")
- else:
- print("Browser Launch: ❌")
-
- print(f"Browsers Detected: {', '.join(results['browsers_found'])}")
-
- if results["errors"]:
- print("\n⚠️ Issues Found:")
- for err in results["errors"]:
- print(f" - {err}")
-
- if "Executable doesn't exist" in str(results["errors"]):
- print(
- "\nSuggestion: Run `playwright install` to download necessary browsers."
- )
- except Exception as e:
- print(f"Diagnostics failed to run: {e}")
+# ./src/web_scraper_toolkit/core/diagnostics.py
+"""
+Diagnostics Module
+==================
+
+Provides system self-checks to ensure the environment is correctly configured.
+Checks Playwright installation, browser binaries, and network connectivity.
+
+Usage:
+ print_diagnostics()
+
+Outputs:
+ - Printed report of system status to stdout.
+"""
+
+import asyncio
+import logging
+import sys
+from typing import Dict, Any
+
+from playwright.async_api import async_playwright
+
+logger = logging.getLogger(__name__)
+
+
+async def verify_environment() -> Dict[str, Any]:
+ """
+ Checks the scraping environment for necessary dependencies and functionality.
+
+ Returns:
+ Dict containing status of:
+ - python_version
+ - playwright_installed
+ - browser_launch_successful
+ - browsers_found
+ """
+ report: Dict[str, Any] = {
+ "python_version": sys.version,
+ "playwright_installed": False,
+ "browser_launch_successful": False,
+ "browsers_found": [],
+ "errors": [],
+ }
+
+ # 1. Check Playwright Import
+ try:
+ from importlib.util import find_spec
+
+ if find_spec("playwright"):
+ report["playwright_installed"] = True
+ else:
+ report["errors"].append("Playwright package not found.")
+ except ImportError:
+ report["errors"].append("Playwright package not installed.")
+ return report
+
+ # 2. Check Browsers
+ try:
+ async with async_playwright() as p:
+ # Try launching Chromium
+ try:
+ browser = await p.chromium.launch(headless=True)
+ report["browsers_found"].append("chromium")
+ report["browser_launch_successful"] = True
+ await browser.close()
+ except Exception as e:
+ report["errors"].append(f"Chromium launch failed: {e}")
+
+ # Try launching Firefox (optional but good to know)
+ try:
+ browser = await p.firefox.launch(headless=True)
+ report["browsers_found"].append("firefox")
+ await browser.close()
+ except Exception:
+ pass # Firefox might not be installed, that's okay
+
+ except Exception as e:
+ report["errors"].append(f"Playwright runtime error: {e}")
+
+ return report
+
+
+def print_diagnostics() -> None:
+ """Runs the verification and prints a human-readable report."""
+ print("Running WebScraperToolkit Diagnostics...")
+ try:
+ results = asyncio.run(verify_environment())
+
+ print(f"\nPython Version: {results['python_version'].split()[0]}")
+ print(
+ f"Playwright Installed: {'✅' if results['playwright_installed'] else '❌'}"
+ )
+
+ if results["playwright_installed"]:
+ if results["browser_launch_successful"]:
+ print("Browser Launch: ✅ (Chromium)")
+ else:
+ print("Browser Launch: ❌")
+
+ print(f"Browsers Detected: {', '.join(results['browsers_found'])}")
+
+ if results["errors"]:
+ print("\n⚠️ Issues Found:")
+ for err in results["errors"]:
+ print(f" - {err}")
+
+ if "Executable doesn't exist" in str(results["errors"]):
+ print(
+ "\nSuggestion: Run `playwright install` to download necessary browsers."
+ )
+ except Exception as e:
+ print(f"Diagnostics failed to run: {e}")
diff --git a/src/web_scraper_toolkit/core/utils.py b/src/web_scraper_toolkit/core/utils.py
index c8c9d42..64ad6eb 100644
--- a/src/web_scraper_toolkit/core/utils.py
+++ b/src/web_scraper_toolkit/core/utils.py
@@ -1,60 +1,60 @@
-# ./src/web_scraper_toolkit/core/utils.py
-# WebScraperToolkit/src/web_scraper_toolkit/utils.py
-from typing import Optional
-from urllib.parse import urlparse, urljoin
-import logging
-
-logger = logging.getLogger(__name__)
-
-
-def normalize_url(url: str, base_url: Optional[str] = None) -> Optional[str]:
- """Normalizes a URL, making it absolute if a base_url is provided."""
- try:
- if base_url:
- url = urljoin(base_url, url.strip())
-
- parsed = urlparse(url.strip())
- if not parsed.scheme or parsed.scheme not in ["http", "https"]:
- return None # Scheme required and must be http/https
- if not parsed.netloc:
- return None # Domain name required
-
- # Remove 'www.' prefix for consistency in domain comparison
- netloc = parsed.netloc.lower()
- if netloc.startswith("www."):
- netloc = netloc[4:]
-
- # Reconstruct with lowercase scheme and netloc, and stripped path/query/fragment
- path = (
- parsed.path.rstrip("/") if parsed.path else ""
- ) # Remove trailing slash from path for consistency
-
- normalized = f"{parsed.scheme.lower()}://{netloc}{path}"
- if parsed.query:
- normalized += f"?{parsed.query}"
-
- return normalized
- except Exception as e:
- logger.debug(f"Failed to normalize URL '{url}': {e}")
- return None
-
-
-def get_domain_from_url(url: str) -> Optional[str]:
- """Extracts the normalized domain (e.g., example.com) from a URL."""
- try:
- parsed_url = urlparse(url.lower())
- domain = parsed_url.netloc
- if domain.startswith("www."):
- domain = domain[4:]
- return domain
- except Exception:
- return None
-
-
-def truncate_text(text: str, max_length: int = 100) -> str:
- """Truncates text to a maximum length, adding ellipsis if truncated."""
- if not text:
- return ""
- if len(text) <= max_length:
- return text
- return text[:max_length].rstrip() + "..."
+# ./src/web_scraper_toolkit/core/utils.py
+# WebScraperToolkit/src/web_scraper_toolkit/utils.py
+from typing import Optional
+from urllib.parse import urlparse, urljoin
+import logging
+
+logger = logging.getLogger(__name__)
+
+
+def normalize_url(url: str, base_url: Optional[str] = None) -> Optional[str]:
+ """Normalizes a URL, making it absolute if a base_url is provided."""
+ try:
+ if base_url:
+ url = urljoin(base_url, url.strip())
+
+ parsed = urlparse(url.strip())
+ if not parsed.scheme or parsed.scheme not in ["http", "https"]:
+ return None # Scheme required and must be http/https
+ if not parsed.netloc:
+ return None # Domain name required
+
+ # Remove 'www.' prefix for consistency in domain comparison
+ netloc = parsed.netloc.lower()
+ if netloc.startswith("www."):
+ netloc = netloc[4:]
+
+ # Reconstruct with lowercase scheme and netloc, and stripped path/query/fragment
+ path = (
+ parsed.path.rstrip("/") if parsed.path else ""
+ ) # Remove trailing slash from path for consistency
+
+ normalized = f"{parsed.scheme.lower()}://{netloc}{path}"
+ if parsed.query:
+ normalized += f"?{parsed.query}"
+
+ return normalized
+ except Exception as e:
+ logger.debug(f"Failed to normalize URL '{url}': {e}")
+ return None
+
+
+def get_domain_from_url(url: str) -> Optional[str]:
+ """Extracts the normalized domain (e.g., example.com) from a URL."""
+ try:
+ parsed_url = urlparse(url.lower())
+ domain = parsed_url.netloc
+ if domain.startswith("www."):
+ domain = domain[4:]
+ return domain
+ except Exception:
+ return None
+
+
+def truncate_text(text: str, max_length: int = 100) -> str:
+ """Truncates text to a maximum length, adding ellipsis if truncated."""
+ if not text:
+ return ""
+ if len(text) <= max_length:
+ return text
+ return text[:max_length].rstrip() + "..."
diff --git a/src/web_scraper_toolkit/core/verify_deps.py b/src/web_scraper_toolkit/core/verify_deps.py
index 1b012ec..8119c45 100644
--- a/src/web_scraper_toolkit/core/verify_deps.py
+++ b/src/web_scraper_toolkit/core/verify_deps.py
@@ -1,60 +1,60 @@
-# ./src/web_scraper_toolkit/core/verify_deps.py
-"""
-Dependency Verification Module
-==============================
-
-Checks that critical runtime dependencies meet the required versions.
-Implements the "Fail Gently" philosophy by guiding users to upgrade instead of crashing hard.
-
-Usage:
- if not verify_dependencies(): sys.exit(1)
-
-Key Checks:
- - playwright-stealth >= 2.0.0
- - playwright >= 1.56.0 (or base compatible)
-
-Key Outputs:
- - Boolean (True if all good, False if missing/old).
- - Printed instructions for the user.
-"""
-
-import importlib.metadata
-from packaging import version
-from .logger import setup_logger
-
-logger = setup_logger()
-
-
-def verify_dependencies():
- """
- Verifies that critical dependencies meet the 'Expert' standards.
- If not, it prints a helpful message and returns False.
- """
- required = {
- "playwright-stealth": "2.0.0",
- "playwright": "1.40.0", # User mentioned 1.56, but let's be safe. 1.40 is stable base.
- # Actually user said "playwright>=1.56.0". Let's use that.
- }
- required["playwright"] = "1.56.0"
-
- all_good = True
-
- for package, min_ver in required.items():
- try:
- installed_ver = importlib.metadata.version(package)
- if version.parse(installed_ver) < version.parse(min_ver):
- logger.error(
- f"❌ Dependency Error: {package} {installed_ver} is too old. Required: >={min_ver}"
- )
- print(f"\n[!] Critical Dependency Update Required for '{package}':")
- print(f" Current: {installed_ver}")
- print(f" Required: >={min_ver}")
- print(f" Selected: pip install --upgrade {package}\n")
- all_good = False
- except importlib.metadata.PackageNotFoundError:
- logger.error(f"❌ Missing Dependency: {package}")
- print(f"\n[!] Missing Critical Dependency: '{package}'")
- print(f" Selected: pip install {package}\n")
- all_good = False
-
- return all_good
+# ./src/web_scraper_toolkit/core/verify_deps.py
+"""
+Dependency Verification Module
+==============================
+
+Checks that critical runtime dependencies meet the required versions.
+Implements the "Fail Gently" philosophy by guiding users to upgrade instead of crashing hard.
+
+Usage:
+ if not verify_dependencies(): sys.exit(1)
+
+Key Checks:
+ - playwright-stealth >= 2.0.0
+ - playwright >= 1.56.0 (or base compatible)
+
+Key Outputs:
+ - Boolean (True if all good, False if missing/old).
+ - Printed instructions for the user.
+"""
+
+import importlib.metadata
+from packaging import version
+from .logger import setup_logger
+
+logger = setup_logger()
+
+
+def verify_dependencies():
+ """
+ Verifies that critical dependencies meet the 'Expert' standards.
+ If not, it prints a helpful message and returns False.
+ """
+ required = {
+ "playwright-stealth": "2.0.0",
+ "playwright": "1.40.0", # User mentioned 1.56, but let's be safe. 1.40 is stable base.
+ # Actually user said "playwright>=1.56.0". Let's use that.
+ }
+ required["playwright"] = "1.56.0"
+
+ all_good = True
+
+ for package, min_ver in required.items():
+ try:
+ installed_ver = importlib.metadata.version(package)
+ if version.parse(installed_ver) < version.parse(min_ver):
+ logger.error(
+ f"❌ Dependency Error: {package} {installed_ver} is too old. Required: >={min_ver}"
+ )
+ print(f"\n[!] Critical Dependency Update Required for '{package}':")
+ print(f" Current: {installed_ver}")
+ print(f" Required: >={min_ver}")
+ print(f" Selected: pip install --upgrade {package}\n")
+ all_good = False
+ except importlib.metadata.PackageNotFoundError:
+ logger.error(f"❌ Missing Dependency: {package}")
+ print(f"\n[!] Missing Critical Dependency: '{package}'")
+ print(f" Selected: pip install {package}\n")
+ all_good = False
+
+ return all_good
diff --git a/src/web_scraper_toolkit/parsers/html_to_markdown.py b/src/web_scraper_toolkit/parsers/html_to_markdown.py
index 5ab5e58..ae56891 100644
--- a/src/web_scraper_toolkit/parsers/html_to_markdown.py
+++ b/src/web_scraper_toolkit/parsers/html_to_markdown.py
@@ -1,396 +1,396 @@
-# ./src/web_scraper_toolkit/parsers/html_to_markdown.py
-"""
-HTML to Markdown Converter
-==========================
-
-Utility for converting HTML content into clean, readable Markdown.
-Specializes in handling tables, lists, and links properly.
-
-Usage:
- md = MarkdownConverter.to_markdown(html_content, base_url="...")
-
-Key Features:
- - Table conversion (ASCII/GD style).
- - Link resolution (absolute paths).
- - Noise removal (scripts/styles).
-"""
-
-from bs4 import BeautifulSoup, NavigableString, Tag
-import re
-from typing import Any, Optional
-
-
-class MarkdownConverter:
- """
- Converts HTML content to clean Markdown, preserving structure for LLM consumption.
- """
-
- @staticmethod
- def to_markdown(html_content: str, base_url: str = "") -> str:
- """
- Main entry point. Converts valid HTML string to Markdown.
- """
- if not html_content:
- return ""
-
- soup = BeautifulSoup(html_content, "lxml")
-
- # 1. Cleanup Noise
- # Removing semantic noise is critical for clean output
- for tag in soup(
- [
- "script",
- "style",
- "noscript",
- "iframe",
- "svg",
- "meta",
- "link",
- "head",
- "nav",
- "footer",
- "header",
- "aside",
- ]
- ):
- tag.decompose()
-
- # 2. Process Content
- # We process the body if available, else the whole soup
- root = soup.body if soup.body else soup
-
- markdown = MarkdownConverter._process_element(root, base_url).strip()
-
- # 3. Post-Processing: Collapse excessive newlines
- # Nested blocks (div > div > p) often generate \n\n\n\n. We want max 2.
- markdown = re.sub(r"\n{3,}", "\n\n", markdown)
-
- return markdown
-
- @staticmethod
- def _attr_to_str(value: Any) -> Optional[str]:
- """Normalize BeautifulSoup attribute values into strings."""
- if value is None:
- return None
- if isinstance(value, str):
- return value
- if isinstance(value, (list, tuple)):
- for item in value:
- if isinstance(item, str) and item:
- return item
- return None
- return str(value)
-
- @staticmethod
- def _is_block_element(tag: Any) -> bool:
- """Checks if a tag is a block element."""
- if not isinstance(tag, Tag):
- return False
- tag_name = (tag.name or "").lower()
- return tag_name in [
- "div",
- "section",
- "article",
- "main",
- "header",
- "footer",
- "ul",
- "ol",
- "li",
- "h1",
- "h2",
- "h3",
- "h4",
- "h5",
- "h6",
- "p",
- "table",
- "blockquote",
- "pre",
- ]
-
- @staticmethod
- def _has_block_children(element: Tag) -> bool:
- """Checks if an element contains any direct block-level children."""
- for child in element.children:
- if MarkdownConverter._is_block_element(child):
- return True
- # Also check recursively? No, usually direct children are enough if structure is good.
- # But ..
is possible.
- # Let's do a simple non-recursive check first, or shallow scan.
- # Deep scan:
- if isinstance(child, Tag) and child.find(
- [
- "div",
- "p",
- "h1",
- "h2",
- "h3",
- "h4",
- "h5",
- "h6",
- "ul",
- "ol",
- "li",
- "table",
- ]
- ):
- return True
- return False
-
- @staticmethod
- def _process_element(
- element: Any, base_url: str = "", link_context: Optional[str] = None
- ) -> str:
- if element is None:
- return ""
-
- # TEXT
- if isinstance(element, NavigableString):
- raw_text = str(element)
- # Normalize whitespace but KEEP leading/trailing spaces for separation
- normalized_text = re.sub(r"\s+", " ", raw_text)
- if not normalized_text:
- return ""
-
- # If we are in a link context, we link the text directly
- if link_context and normalized_text.strip():
- # Avoid linking whitespace only
- return f"[{normalized_text}]({link_context})"
- return normalized_text
-
- if not isinstance(element, Tag):
- return ""
-
- # TAGS mapping
- tag_name = (element.name or "").lower()
-
- # --- BLOCK ELEMENTS --- #
-
- if tag_name in ["h1", "h2", "h3", "h4", "h5", "h6"]:
- level = int(tag_name[1])
- # Pass link_context down to inner text
- content = MarkdownConverter._get_inner_text(element, base_url, link_context)
- # Use fewer newlines to avoid stacking up to 4+
- return f"\n{'#' * level} {content}\n"
-
- if tag_name == "p":
- content = MarkdownConverter._get_inner_text(element, base_url, link_context)
- return f"\n{content}\n" if content else ""
-
- if tag_name == "br":
- return "\n"
-
- if tag_name == "hr":
- return "\n\n---\n\n"
-
- if tag_name in ["ul", "ol"]:
- return f"\n{MarkdownConverter._process_list(element, base_url, link_context)}\n"
-
- if tag_name == "li":
- # Handled by parent UL/OL usually, but if standalone:
- content = MarkdownConverter._get_inner_text(element, base_url, link_context)
- return f"- {content}\n"
-
- if tag_name == "blockquote":
- content = MarkdownConverter._get_inner_text(element, base_url, link_context)
- lines = content.split("\n")
- quoted = "\n".join(f"> {line}" for line in lines if line.strip())
- return f"\n{quoted}\n"
-
- if tag_name == "pre":
- # Code blocks inside links are rare and likely break markdown parsers anyway.
- # We ignore link_context here for code blocks usually.
- code_tag = element.find("code")
- content = ""
- if code_tag:
- content = code_tag.get_text() # Raw text for code
- else:
- content = element.get_text()
- return f"\n\n```\n{content}\n```\n\n"
-
- # --- TABLES --- #
- if tag_name == "table":
- return f"\n\n{MarkdownConverter._process_table(element, base_url)}\n\n"
-
- # --- INLINE ELEMENTS --- #
-
- if tag_name == "a":
- href = MarkdownConverter._attr_to_str(element.get("href")) or ""
- if not href or href.startswith("#") or href.startswith("javascript:"):
- return MarkdownConverter._get_inner_text(
- element, base_url, link_context
- )
-
- # Check if this link wraps BLOCK elements
- if MarkdownConverter._has_block_children(element):
- # DISTRIBUTE THE LINK DOWN
- # We do NOT wrap this tag. We recurse.
- return MarkdownConverter._get_inner_text(
- element, base_url, link_context=href
- )
-
- # STANDARD INLINE LINK
- text = MarkdownConverter._get_inner_text(element, base_url, link_context)
- # If we are already inside a link context, we shouldn't nest links technically.
- # Markdown doesn't support nested links.
- # We defer to the outer link? Or inner?
- # Usually inner link takes precedence in HTML, but markdown breaks.
- # Let's assume inner takes precedence.
- return f"[{text}]({href})"
-
- if tag_name == "img":
- alt = MarkdownConverter._attr_to_str(element.get("alt")) or "Image"
- src = MarkdownConverter._attr_to_str(element.get("src")) or ""
- if not src:
- return ""
- img_md = f""
- if link_context:
- return f"[{img_md}]({link_context})"
- return img_md
-
- if tag_name in ["strong", "b"]:
- content = MarkdownConverter._get_inner_text(element, base_url, link_context)
- return f"**{content}**" if content else ""
-
- if tag_name in ["em", "i"]:
- content = MarkdownConverter._get_inner_text(element, base_url, link_context)
- return f"_{content}_" if content else ""
-
- if tag_name == "code":
- # If inside pre, handled by pre. If inline:
- if element.parent and element.parent.name == "pre":
- return "" # Let pre handle it
- text = element.get_text()
- return f"`{text}`"
-
- if tag_name in ["span", "html"]:
- # Just traverse children transparently
- return MarkdownConverter._get_inner_text(element, base_url, link_context)
-
- if tag_name in [
- "div",
- "section",
- "article",
- "main",
- "header",
- "footer",
- "body",
- ]:
- # Treat as block structural elements.
- # We add newlines to ensure separation of content blocks.
- content = MarkdownConverter._get_inner_text(element, base_url, link_context)
- if not content.strip():
- return ""
- return f"\n{content}\n"
-
- # Default fallback
- return MarkdownConverter._get_inner_text(element, base_url, link_context)
-
- @staticmethod
- def _get_inner_text(
- element: Tag, base_url: str, link_context: Optional[str] = None
- ) -> str:
- """Helper to recursively process children and join them."""
- results = []
- for child in element.children:
- res = MarkdownConverter._process_element(child, base_url, link_context)
- if res:
- results.append(res)
-
- # Intelligent joining
- # To fix "Paragraph**Bold**", we can insert spaces if the elements are inline.
- # But this might break "word," -> "word ,".
- # Simple heuristic: if we are in a block element context, spacing is handled by block logic.
- # If we are in an inline context, we might need a space.
- # A simpler fix for this specific tool: join with " " then remove excessive spaces.
-
- joined = "".join(results)
-
- # Quick fix for Paragraph**Bold** -> Paragraph **Bold**
- # We can just leave it as is if the user didn't put a space in HTML "Paragraph Bold".
- # But BS4 often strips distinct text nodes.
- # Let's try joining with nothing, because usually HTML handles space via text nodes.
- # If the failure 'Paragraph**Bold**' happened, it means the input HTML was "
".
- # "Paragraph " should be a NavigableString.
- # Wait, the test input was "".
- # "Paragraph " is a text node. "Bold" is a tag.
- # If _process_element("Paragraph ") returns "Paragraph" (stripped), then we lose the space.
-
- return joined
-
- @staticmethod
- def _process_list(
- list_element: Tag, base_url: str, link_context: Optional[str] = None
- ) -> str:
- items = []
- is_ordered = (list_element.name or "").lower() == "ol"
-
- for i, child in enumerate(list_element.find_all("li", recursive=False)):
- content = MarkdownConverter._get_inner_text(
- child, base_url, link_context
- ).strip()
- # Handle nested lists
- nested_list = child.find(["ul", "ol"])
- if nested_list:
- # remove the text content that belongs to the nested list so we don't duplicate
- # actually _get_inner_text handles recursion, so 'content' already includes the nested list markdown!
- # This is tricky using simple recursion.
- # Let's trust _get_inner_text returns formatting.
- pass
-
- # Simple handling: replace internal newlines for sub-items alignment
- content = content.replace("\n", "\n ")
-
- prefix = f"{i + 1}." if is_ordered else "-"
- items.append(f"{prefix} {content}")
-
- return "\n".join(items)
-
- @staticmethod
- def _process_table(table_element: Tag, base_url: str) -> str:
- """
- Simple Markdown table converter.
- Only handles standard thead/tbody structures nicely.
- """
- rows = []
-
- # 1. Headers
- headers = []
- thead = table_element.find("thead")
- if thead:
- header_row = thead.find("tr")
- if header_row:
- headers = [
- th.get_text(strip=True) for th in header_row.find_all(["th", "td"])
- ]
-
- # Fallback: if no thead, check first tr
- if not headers:
- first_tr = table_element.find("tr")
- if first_tr and first_tr.find("th"):
- headers = [th.get_text(strip=True) for th in first_tr.find_all(["th"])]
-
- if headers:
- rows.append(f"| {' | '.join(headers)} |")
- rows.append(f"| {' | '.join(['---'] * len(headers))} |")
-
- # 2. Body
- tbody = table_element.find("tbody")
- row_container = tbody if tbody else table_element
-
- for tr in row_container.find_all("tr"):
- # specific check to skip the header row if we already processed it from 'not headers' logic
- if not headers and tr == table_element.find("tr") and tr.find("th"):
- continue
-
- cols = [td.get_text(strip=True) for td in tr.find_all(["td", "th"])]
- if cols:
- # Match header length if possible
- if headers and len(cols) != len(headers):
- # Pad or truncate? Markdown tables are forgiving of mismatches usually,
- # but let's just print what we have
- pass
- rows.append(f"| {' | '.join(cols)} |")
-
- return "\n".join(rows)
+# ./src/web_scraper_toolkit/parsers/html_to_markdown.py
+"""
+HTML to Markdown Converter
+==========================
+
+Utility for converting HTML content into clean, readable Markdown.
+Specializes in handling tables, lists, and links properly.
+
+Usage:
+ md = MarkdownConverter.to_markdown(html_content, base_url="...")
+
+Key Features:
+ - Table conversion (ASCII/GD style).
+ - Link resolution (absolute paths).
+ - Noise removal (scripts/styles).
+"""
+
+from bs4 import BeautifulSoup, NavigableString, Tag
+import re
+from typing import Any, Optional
+
+
+class MarkdownConverter:
+ """
+ Converts HTML content to clean Markdown, preserving structure for LLM consumption.
+ """
+
+ @staticmethod
+ def to_markdown(html_content: str, base_url: str = "") -> str:
+ """
+ Main entry point. Converts valid HTML string to Markdown.
+ """
+ if not html_content:
+ return ""
+
+ soup = BeautifulSoup(html_content, "lxml")
+
+ # 1. Cleanup Noise
+ # Removing semantic noise is critical for clean output
+ for tag in soup(
+ [
+ "script",
+ "style",
+ "noscript",
+ "iframe",
+ "svg",
+ "meta",
+ "link",
+ "head",
+ "nav",
+ "footer",
+ "header",
+ "aside",
+ ]
+ ):
+ tag.decompose()
+
+ # 2. Process Content
+ # We process the body if available, else the whole soup
+ root = soup.body if soup.body else soup
+
+ markdown = MarkdownConverter._process_element(root, base_url).strip()
+
+ # 3. Post-Processing: Collapse excessive newlines
+ # Nested blocks (div > div > p) often generate \n\n\n\n. We want max 2.
+ markdown = re.sub(r"\n{3,}", "\n\n", markdown)
+
+ return markdown
+
+ @staticmethod
+ def _attr_to_str(value: Any) -> Optional[str]:
+ """Normalize BeautifulSoup attribute values into strings."""
+ if value is None:
+ return None
+ if isinstance(value, str):
+ return value
+ if isinstance(value, (list, tuple)):
+ for item in value:
+ if isinstance(item, str) and item:
+ return item
+ return None
+ return str(value)
+
+ @staticmethod
+ def _is_block_element(tag: Any) -> bool:
+ """Checks if a tag is a block element."""
+ if not isinstance(tag, Tag):
+ return False
+ tag_name = (tag.name or "").lower()
+ return tag_name in [
+ "div",
+ "section",
+ "article",
+ "main",
+ "header",
+ "footer",
+ "ul",
+ "ol",
+ "li",
+ "h1",
+ "h2",
+ "h3",
+ "h4",
+ "h5",
+ "h6",
+ "p",
+ "table",
+ "blockquote",
+ "pre",
+ ]
+
+ @staticmethod
+ def _has_block_children(element: Tag) -> bool:
+ """Checks if an element contains any direct block-level children."""
+ for child in element.children:
+ if MarkdownConverter._is_block_element(child):
+ return True
+ # Also check recursively? No, usually direct children are enough if structure is good.
+ # But ..
is possible.
+ # Let's do a simple non-recursive check first, or shallow scan.
+ # Deep scan:
+ if isinstance(child, Tag) and child.find(
+ [
+ "div",
+ "p",
+ "h1",
+ "h2",
+ "h3",
+ "h4",
+ "h5",
+ "h6",
+ "ul",
+ "ol",
+ "li",
+ "table",
+ ]
+ ):
+ return True
+ return False
+
+ @staticmethod
+ def _process_element(
+ element: Any, base_url: str = "", link_context: Optional[str] = None
+ ) -> str:
+ if element is None:
+ return ""
+
+ # TEXT
+ if isinstance(element, NavigableString):
+ raw_text = str(element)
+ # Normalize whitespace but KEEP leading/trailing spaces for separation
+ normalized_text = re.sub(r"\s+", " ", raw_text)
+ if not normalized_text:
+ return ""
+
+ # If we are in a link context, we link the text directly
+ if link_context and normalized_text.strip():
+ # Avoid linking whitespace only
+ return f"[{normalized_text}]({link_context})"
+ return normalized_text
+
+ if not isinstance(element, Tag):
+ return ""
+
+ # TAGS mapping
+ tag_name = (element.name or "").lower()
+
+ # --- BLOCK ELEMENTS --- #
+
+ if tag_name in ["h1", "h2", "h3", "h4", "h5", "h6"]:
+ level = int(tag_name[1])
+ # Pass link_context down to inner text
+ content = MarkdownConverter._get_inner_text(element, base_url, link_context)
+ # Use fewer newlines to avoid stacking up to 4+
+ return f"\n{'#' * level} {content}\n"
+
+ if tag_name == "p":
+ content = MarkdownConverter._get_inner_text(element, base_url, link_context)
+ return f"\n{content}\n" if content else ""
+
+ if tag_name == "br":
+ return "\n"
+
+ if tag_name == "hr":
+ return "\n\n---\n\n"
+
+ if tag_name in ["ul", "ol"]:
+ return f"\n{MarkdownConverter._process_list(element, base_url, link_context)}\n"
+
+ if tag_name == "li":
+ # Handled by parent UL/OL usually, but if standalone:
+ content = MarkdownConverter._get_inner_text(element, base_url, link_context)
+ return f"- {content}\n"
+
+ if tag_name == "blockquote":
+ content = MarkdownConverter._get_inner_text(element, base_url, link_context)
+ lines = content.split("\n")
+ quoted = "\n".join(f"> {line}" for line in lines if line.strip())
+ return f"\n{quoted}\n"
+
+ if tag_name == "pre":
+ # Code blocks inside links are rare and likely break markdown parsers anyway.
+ # We ignore link_context here for code blocks usually.
+ code_tag = element.find("code")
+ content = ""
+ if code_tag:
+ content = code_tag.get_text() # Raw text for code
+ else:
+ content = element.get_text()
+ return f"\n\n```\n{content}\n```\n\n"
+
+ # --- TABLES --- #
+ if tag_name == "table":
+ return f"\n\n{MarkdownConverter._process_table(element, base_url)}\n\n"
+
+ # --- INLINE ELEMENTS --- #
+
+ if tag_name == "a":
+ href = MarkdownConverter._attr_to_str(element.get("href")) or ""
+ if not href or href.startswith("#") or href.startswith("javascript:"):
+ return MarkdownConverter._get_inner_text(
+ element, base_url, link_context
+ )
+
+ # Check if this link wraps BLOCK elements
+ if MarkdownConverter._has_block_children(element):
+ # DISTRIBUTE THE LINK DOWN
+ # We do NOT wrap this tag. We recurse.
+ return MarkdownConverter._get_inner_text(
+ element, base_url, link_context=href
+ )
+
+ # STANDARD INLINE LINK
+ text = MarkdownConverter._get_inner_text(element, base_url, link_context)
+ # If we are already inside a link context, we shouldn't nest links technically.
+ # Markdown doesn't support nested links.
+ # We defer to the outer link? Or inner?
+ # Usually inner link takes precedence in HTML, but markdown breaks.
+ # Let's assume inner takes precedence.
+ return f"[{text}]({href})"
+
+ if tag_name == "img":
+ alt = MarkdownConverter._attr_to_str(element.get("alt")) or "Image"
+ src = MarkdownConverter._attr_to_str(element.get("src")) or ""
+ if not src:
+ return ""
+ img_md = f""
+ if link_context:
+ return f"[{img_md}]({link_context})"
+ return img_md
+
+ if tag_name in ["strong", "b"]:
+ content = MarkdownConverter._get_inner_text(element, base_url, link_context)
+ return f"**{content}**" if content else ""
+
+ if tag_name in ["em", "i"]:
+ content = MarkdownConverter._get_inner_text(element, base_url, link_context)
+ return f"_{content}_" if content else ""
+
+ if tag_name == "code":
+ # If inside pre, handled by pre. If inline:
+ if element.parent and element.parent.name == "pre":
+ return "" # Let pre handle it
+ text = element.get_text()
+ return f"`{text}`"
+
+ if tag_name in ["span", "html"]:
+ # Just traverse children transparently
+ return MarkdownConverter._get_inner_text(element, base_url, link_context)
+
+ if tag_name in [
+ "div",
+ "section",
+ "article",
+ "main",
+ "header",
+ "footer",
+ "body",
+ ]:
+ # Treat as block structural elements.
+ # We add newlines to ensure separation of content blocks.
+ content = MarkdownConverter._get_inner_text(element, base_url, link_context)
+ if not content.strip():
+ return ""
+ return f"\n{content}\n"
+
+ # Default fallback
+ return MarkdownConverter._get_inner_text(element, base_url, link_context)
+
+ @staticmethod
+ def _get_inner_text(
+ element: Tag, base_url: str, link_context: Optional[str] = None
+ ) -> str:
+ """Helper to recursively process children and join them."""
+ results = []
+ for child in element.children:
+ res = MarkdownConverter._process_element(child, base_url, link_context)
+ if res:
+ results.append(res)
+
+ # Intelligent joining
+ # To fix "Paragraph**Bold**", we can insert spaces if the elements are inline.
+ # But this might break "word," -> "word ,".
+ # Simple heuristic: if we are in a block element context, spacing is handled by block logic.
+ # If we are in an inline context, we might need a space.
+ # A simpler fix for this specific tool: join with " " then remove excessive spaces.
+
+ joined = "".join(results)
+
+ # Quick fix for Paragraph**Bold** -> Paragraph **Bold**
+ # We can just leave it as is if the user didn't put a space in HTML "Paragraph Bold".
+ # But BS4 often strips distinct text nodes.
+ # Let's try joining with nothing, because usually HTML handles space via text nodes.
+ # If the failure 'Paragraph**Bold**' happened, it means the input HTML was "".
+ # "Paragraph " should be a NavigableString.
+ # Wait, the test input was "".
+ # "Paragraph " is a text node. "Bold" is a tag.
+ # If _process_element("Paragraph ") returns "Paragraph" (stripped), then we lose the space.
+
+ return joined
+
+ @staticmethod
+ def _process_list(
+ list_element: Tag, base_url: str, link_context: Optional[str] = None
+ ) -> str:
+ items = []
+ is_ordered = (list_element.name or "").lower() == "ol"
+
+ for i, child in enumerate(list_element.find_all("li", recursive=False)):
+ content = MarkdownConverter._get_inner_text(
+ child, base_url, link_context
+ ).strip()
+ # Handle nested lists
+ nested_list = child.find(["ul", "ol"])
+ if nested_list:
+ # remove the text content that belongs to the nested list so we don't duplicate
+ # actually _get_inner_text handles recursion, so 'content' already includes the nested list markdown!
+ # This is tricky using simple recursion.
+ # Let's trust _get_inner_text returns formatting.
+ pass
+
+ # Simple handling: replace internal newlines for sub-items alignment
+ content = content.replace("\n", "\n ")
+
+ prefix = f"{i + 1}." if is_ordered else "-"
+ items.append(f"{prefix} {content}")
+
+ return "\n".join(items)
+
+ @staticmethod
+ def _process_table(table_element: Tag, base_url: str) -> str:
+ """
+ Simple Markdown table converter.
+ Only handles standard thead/tbody structures nicely.
+ """
+ rows = []
+
+ # 1. Headers
+ headers = []
+ thead = table_element.find("thead")
+ if thead:
+ header_row = thead.find("tr")
+ if header_row:
+ headers = [
+ th.get_text(strip=True) for th in header_row.find_all(["th", "td"])
+ ]
+
+ # Fallback: if no thead, check first tr
+ if not headers:
+ first_tr = table_element.find("tr")
+ if first_tr and first_tr.find("th"):
+ headers = [th.get_text(strip=True) for th in first_tr.find_all(["th"])]
+
+ if headers:
+ rows.append(f"| {' | '.join(headers)} |")
+ rows.append(f"| {' | '.join(['---'] * len(headers))} |")
+
+ # 2. Body
+ tbody = table_element.find("tbody")
+ row_container = tbody if tbody else table_element
+
+ for tr in row_container.find_all("tr"):
+ # specific check to skip the header row if we already processed it from 'not headers' logic
+ if not headers and tr == table_element.find("tr") and tr.find("th"):
+ continue
+
+ cols = [td.get_text(strip=True) for td in tr.find_all(["td", "th"])]
+ if cols:
+ # Match header length if possible
+ if headers and len(cols) != len(headers):
+ # Pad or truncate? Markdown tables are forgiving of mismatches usually,
+ # but let's just print what we have
+ pass
+ rows.append(f"| {' | '.join(cols)} |")
+
+ return "\n".join(rows)
diff --git a/src/web_scraper_toolkit/parsers/utils.py b/src/web_scraper_toolkit/parsers/utils.py
index 8a693de..9a8bb87 100644
--- a/src/web_scraper_toolkit/parsers/utils.py
+++ b/src/web_scraper_toolkit/parsers/utils.py
@@ -1,67 +1,67 @@
-# ./src/web_scraper_toolkit/parsers/utils.py
-"""
-Parser Utilities
-================
-
-Helper definitions for text extraction, URL normalization, and data cleanup.
-Used by serp_parser and other parsing modules.
-
-Usage:
- norm_url = normalize_url("/foo", "https://base.com")
-
-Key Functions:
- - normalize_url: Resolves relative links.
- - truncate_text: Safe string shortening.
-"""
-
-from urllib.parse import urljoin, urlparse
-from typing import Optional
-
-
-def normalize_url(url: str, base_url: str = "") -> Optional[str]:
- """
- Resolves relative URLs against a base URL.
- Returns None if URL is invalid or empty.
- Strips trailing slashes to canonicalize.
- """
- if not url:
- return None
-
- try:
- # Strip whitespace
- url = url.strip()
-
- # Handle javascript: links
- if url.startswith("javascript:") or url.startswith("#"):
- return None
-
- # Join
- full_url = urljoin(base_url, url)
-
- # Validate scheme
- parsed = urlparse(full_url)
- if parsed.scheme not in ["http", "https"]:
- return None
-
- # Canonicalize: Remove trailing slash if path > 1 (keep root /)
- if parsed.path.endswith("/") and len(parsed.path) > 1:
- full_url = full_url.rstrip("/")
-
- return full_url
-
- except Exception:
- return None
-
-
-def truncate_text(text: str, max_length: int = 100) -> str:
- """
- Truncates text to max_length, appending '...' if truncated.
- """
- if not text:
- return ""
-
- clean_text = text.strip()
- if len(clean_text) <= max_length:
- return clean_text
-
- return clean_text[:max_length].rstrip() + "..."
+# ./src/web_scraper_toolkit/parsers/utils.py
+"""
+Parser Utilities
+================
+
+Helper definitions for text extraction, URL normalization, and data cleanup.
+Used by serp_parser and other parsing modules.
+
+Usage:
+ norm_url = normalize_url("/foo", "https://base.com")
+
+Key Functions:
+ - normalize_url: Resolves relative links.
+ - truncate_text: Safe string shortening.
+"""
+
+from urllib.parse import urljoin, urlparse
+from typing import Optional
+
+
+def normalize_url(url: str, base_url: str = "") -> Optional[str]:
+ """
+ Resolves relative URLs against a base URL.
+ Returns None if URL is invalid or empty.
+ Strips trailing slashes to canonicalize.
+ """
+ if not url:
+ return None
+
+ try:
+ # Strip whitespace
+ url = url.strip()
+
+ # Handle javascript: links
+ if url.startswith("javascript:") or url.startswith("#"):
+ return None
+
+ # Join
+ full_url = urljoin(base_url, url)
+
+ # Validate scheme
+ parsed = urlparse(full_url)
+ if parsed.scheme not in ["http", "https"]:
+ return None
+
+ # Canonicalize: Remove trailing slash if path > 1 (keep root /)
+ if parsed.path.endswith("/") and len(parsed.path) > 1:
+ full_url = full_url.rstrip("/")
+
+ return full_url
+
+ except Exception:
+ return None
+
+
+def truncate_text(text: str, max_length: int = 100) -> str:
+ """
+ Truncates text to max_length, appending '...' if truncated.
+ """
+ if not text:
+ return ""
+
+ clean_text = text.strip()
+ if len(clean_text) <= max_length:
+ return clean_text
+
+ return clean_text[:max_length].rstrip() + "..."
diff --git a/src/web_scraper_toolkit/server/__init__.py b/src/web_scraper_toolkit/server/__init__.py
index 71cf36c..4a2950a 100644
--- a/src/web_scraper_toolkit/server/__init__.py
+++ b/src/web_scraper_toolkit/server/__init__.py
@@ -1,19 +1,19 @@
-# ./src/web_scraper_toolkit/server/__init__.py
-"""
-Server Module
-=============
-
-This module exposes the FastMCP server instance for the Web Scraper Toolkit.
-It allows the server to be imported and run programmatically or via specific runners.
-
-Usage:
- from src.web_scraper_toolkit.server import mcp
- mcp.run()
-
-Components:
- - mcp: The FastMCP instance configured with scraping tools.
-"""
-
-from .mcp_server import mcp
-
-__all__ = ["mcp"]
+# ./src/web_scraper_toolkit/server/__init__.py
+"""
+Server Module
+=============
+
+This module exposes the FastMCP server instance for the Web Scraper Toolkit.
+It allows the server to be imported and run programmatically or via specific runners.
+
+Usage:
+ from src.web_scraper_toolkit.server import mcp
+ mcp.run()
+
+Components:
+ - mcp: The FastMCP instance configured with scraping tools.
+"""
+
+from .mcp_server import mcp
+
+__all__ = ["mcp"]
diff --git a/tests/test_basics.py b/tests/test_basics.py
index 4d1232e..2742028 100644
--- a/tests/test_basics.py
+++ b/tests/test_basics.py
@@ -1,30 +1,30 @@
-import unittest
-
-# Ensure src is in path for testing without installing
-# sys.path handled by run_tests.py
-
-from web_scraper_toolkit.parsers.utils import normalize_url, truncate_text
-from web_scraper_toolkit.parsers import SerpParser
-
-
-class TestUtils(unittest.TestCase):
- def test_normalize_url(self):
- self.assertEqual(
- normalize_url("https://example.com/foo/"), "https://example.com/foo"
- )
- self.assertEqual(normalize_url("example.com"), None) # Needs scheme
-
- def test_truncate_text(self):
- text = "Hello World"
- self.assertEqual(truncate_text(text, 5), "Hello...")
- self.assertEqual(truncate_text(text, 50), "Hello World")
-
-
-class TestSerpParser(unittest.TestCase):
- def test_parse_empty(self):
- results = SerpParser.parse_ddg_html("", "https://example.com")
- self.assertEqual(results, [])
-
-
-if __name__ == "__main__":
- unittest.main()
+import unittest
+
+# Ensure src is in path for testing without installing
+# sys.path handled by run_tests.py
+
+from web_scraper_toolkit.parsers.utils import normalize_url, truncate_text
+from web_scraper_toolkit.parsers import SerpParser
+
+
+class TestUtils(unittest.TestCase):
+ def test_normalize_url(self):
+ self.assertEqual(
+ normalize_url("https://example.com/foo/"), "https://example.com/foo"
+ )
+ self.assertEqual(normalize_url("example.com"), None) # Needs scheme
+
+ def test_truncate_text(self):
+ text = "Hello World"
+ self.assertEqual(truncate_text(text, 5), "Hello...")
+ self.assertEqual(truncate_text(text, 50), "Hello World")
+
+
+class TestSerpParser(unittest.TestCase):
+ def test_parse_empty(self):
+ results = SerpParser.parse_ddg_html("", "https://example.com")
+ self.assertEqual(results, [])
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_cli.py b/tests/test_cli.py
index f49b5e5..b9e71e6 100644
--- a/tests/test_cli.py
+++ b/tests/test_cli.py
@@ -1,116 +1,116 @@
-import unittest
-from unittest.mock import patch, AsyncMock, Mock
-import sys
-import os
-
-# Ensure src is in path
-# sys.path handled by run_tests.py
-
-from web_scraper_toolkit import cli
-
-
-class TestCLI(unittest.TestCase):
- def setUp(self):
- # Silence the rich console
- patcher = patch("web_scraper_toolkit.cli.console", Mock())
- self.mock_console = patcher.start()
- self.addCleanup(patcher.stop)
-
- def test_argument_parser(self):
- # Test default args
- parser = cli.parse_arguments(["--url", "https://example.com"])
- self.assertEqual(parser.url, "https://example.com")
- self.assertEqual(parser.format, "markdown") # Default
- self.assertFalse(parser.headless)
-
- # Test complex args
- args = [
- "--input",
- "file.txt",
- "--format",
- "pdf",
- "--headless",
- "--workers",
- "max",
- ]
- parser = cli.parse_arguments(args)
- self.assertEqual(parser.input, "file.txt")
- self.assertEqual(parser.format, "pdf")
- self.assertTrue(parser.headless)
- self.assertEqual(parser.workers, "max")
-
- @patch("web_scraper_toolkit.cli.WebCrawler")
- def test_main_execution_single_url(self, MockCrawler):
- # Mocking the Crawler
- instance = MockCrawler.return_value
- instance.run = AsyncMock(return_value="Success")
-
- # Mock sys.argv
- test_args = ["web-scraper", "--url", "http://example.com", "--format", "json"]
- with patch.object(sys, "argv", test_args):
- cli.main()
-
- # Verify Config was created
- MockCrawler.assert_called_once()
- # Verify run was called
- instance.run.assert_called_once()
- _, kwargs = instance.run.call_args
- self.assertEqual(kwargs["urls"], ["http://example.com"])
- self.assertEqual(kwargs["output_format"], "json")
-
- @patch("web_scraper_toolkit.cli.load_urls_from_source")
- @patch("web_scraper_toolkit.cli.WebCrawler")
- def test_main_execution_input_file(self, MockCrawler, mock_load):
- # Mock file loader
- mock_load.return_value = [
- "https://site-a.example.com",
- "https://site-b.example.com",
- ]
-
- instance = MockCrawler.return_value
- instance.run = AsyncMock(return_value="Success")
-
- test_args = ["web-scraper", "--input", "list.txt"]
- with patch.object(sys, "argv", test_args):
- cli.main()
-
- # Verify loader called
- mock_load.assert_called_with("list.txt")
- # Verify run called with list
- _, kwargs = instance.run.call_args
- self.assertEqual(
- kwargs["urls"], ["https://site-a.example.com", "https://site-b.example.com"]
- )
-
- @patch("web_scraper_toolkit.extract_sitemap_tree", new_callable=AsyncMock)
- def test_site_tree_mode(self, mock_tree):
- mock_tree.return_value = ["https://example.com/item1"]
-
- # Redirect output to tests_output for cleanliness
- output_dir = os.path.abspath(
- os.path.join(os.path.dirname(__file__), "../tests_output")
- )
- os.makedirs(output_dir, exist_ok=True)
- output_file = os.path.join(output_dir, "sitemap_tree.csv")
-
- # Override argv to include output_name
- test_args = [
- "web-scraper",
- "--input",
- "https://example.com/sitemap.xml",
- "--site-tree",
- "--output-name",
- output_file,
- ]
-
- # We also need to mock print or file writing, as main() prints result or writes file
- # But we just want to ensure it calls extract_sitemap_tree
- with patch.object(sys, "argv", test_args):
- # main() might exit? No, it just finishes.
- cli.main()
-
- mock_tree.assert_called_with("https://example.com/sitemap.xml")
-
-
-if __name__ == "__main__":
- unittest.main()
+import unittest
+from unittest.mock import patch, AsyncMock, Mock
+import sys
+import os
+
+# Ensure src is in path
+# sys.path handled by run_tests.py
+
+from web_scraper_toolkit import cli
+
+
+class TestCLI(unittest.TestCase):
+ def setUp(self):
+ # Silence the rich console
+ patcher = patch("web_scraper_toolkit.cli.console", Mock())
+ self.mock_console = patcher.start()
+ self.addCleanup(patcher.stop)
+
+ def test_argument_parser(self):
+ # Test default args
+ parser = cli.parse_arguments(["--url", "https://example.com"])
+ self.assertEqual(parser.url, "https://example.com")
+ self.assertEqual(parser.format, "markdown") # Default
+ self.assertFalse(parser.headless)
+
+ # Test complex args
+ args = [
+ "--input",
+ "file.txt",
+ "--format",
+ "pdf",
+ "--headless",
+ "--workers",
+ "max",
+ ]
+ parser = cli.parse_arguments(args)
+ self.assertEqual(parser.input, "file.txt")
+ self.assertEqual(parser.format, "pdf")
+ self.assertTrue(parser.headless)
+ self.assertEqual(parser.workers, "max")
+
+ @patch("web_scraper_toolkit.cli.WebCrawler")
+ def test_main_execution_single_url(self, MockCrawler):
+ # Mocking the Crawler
+ instance = MockCrawler.return_value
+ instance.run = AsyncMock(return_value="Success")
+
+ # Mock sys.argv
+ test_args = ["web-scraper", "--url", "http://example.com", "--format", "json"]
+ with patch.object(sys, "argv", test_args):
+ cli.main()
+
+ # Verify Config was created
+ MockCrawler.assert_called_once()
+ # Verify run was called
+ instance.run.assert_called_once()
+ _, kwargs = instance.run.call_args
+ self.assertEqual(kwargs["urls"], ["http://example.com"])
+ self.assertEqual(kwargs["output_format"], "json")
+
+ @patch("web_scraper_toolkit.cli.load_urls_from_source")
+ @patch("web_scraper_toolkit.cli.WebCrawler")
+ def test_main_execution_input_file(self, MockCrawler, mock_load):
+ # Mock file loader
+ mock_load.return_value = [
+ "https://site-a.example.com",
+ "https://site-b.example.com",
+ ]
+
+ instance = MockCrawler.return_value
+ instance.run = AsyncMock(return_value="Success")
+
+ test_args = ["web-scraper", "--input", "list.txt"]
+ with patch.object(sys, "argv", test_args):
+ cli.main()
+
+ # Verify loader called
+ mock_load.assert_called_with("list.txt")
+ # Verify run called with list
+ _, kwargs = instance.run.call_args
+ self.assertEqual(
+ kwargs["urls"], ["https://site-a.example.com", "https://site-b.example.com"]
+ )
+
+ @patch("web_scraper_toolkit.extract_sitemap_tree", new_callable=AsyncMock)
+ def test_site_tree_mode(self, mock_tree):
+ mock_tree.return_value = ["https://example.com/item1"]
+
+ # Redirect output to tests_output for cleanliness
+ output_dir = os.path.abspath(
+ os.path.join(os.path.dirname(__file__), "../tests_output")
+ )
+ os.makedirs(output_dir, exist_ok=True)
+ output_file = os.path.join(output_dir, "sitemap_tree.csv")
+
+ # Override argv to include output_name
+ test_args = [
+ "web-scraper",
+ "--input",
+ "https://example.com/sitemap.xml",
+ "--site-tree",
+ "--output-name",
+ output_file,
+ ]
+
+ # We also need to mock print or file writing, as main() prints result or writes file
+ # But we just want to ensure it calls extract_sitemap_tree
+ with patch.object(sys, "argv", test_args):
+ # main() might exit? No, it just finishes.
+ cli.main()
+
+ mock_tree.assert_called_with("https://example.com/sitemap.xml")
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_html_to_markdown.py b/tests/test_html_to_markdown.py
index 3d60a19..db04587 100644
--- a/tests/test_html_to_markdown.py
+++ b/tests/test_html_to_markdown.py
@@ -1,99 +1,99 @@
-import unittest
-import os
-import re
-
-# sys.path handled by run_tests.py
-from web_scraper_toolkit.parsers.html_to_markdown import MarkdownConverter
-
-
-class TestMarkdownConverter(unittest.TestCase):
- def setUp(self):
- # Cache Scrub (Roy-Standard)
- import shutil
-
- cache_path = os.path.join(os.path.dirname(__file__), "__pycache__")
- if os.path.exists(cache_path):
- try:
- shutil.rmtree(cache_path)
- except Exception:
- pass
-
- def test_basic_conversion(self):
- html = "Title
Body text.
"
- # We strip to avoid whitespace nitpicking on exact newlines
- self.assertEqual(
- MarkdownConverter.to_markdown(html).strip(), "# Title\n\nBody text."
- )
-
- def test_links(self):
- html = 'Example'
- expected = "[Example](https://example.com)"
- self.assertEqual(MarkdownConverter.to_markdown(html).strip(), expected)
-
- def test_lists(self):
- html = """
-
- """
- # Expected:
- # - Item 1
- # - Item 2
- md = MarkdownConverter.to_markdown(html).strip()
- self.assertIn("- Item 1", md)
- self.assertIn("- Item 2", md)
-
- def test_nested_structure(self):
- html = ""
- md = MarkdownConverter.to_markdown(html).strip()
- self.assertEqual(md, "Paragraph **Bold**")
-
- def test_tables(self):
- html = """
-
-
- | Header 1 | Header 2 |
-
-
- | Row 1 Col 1 | Row 1 Col 2 |
-
-
- """
- md = MarkdownConverter.to_markdown(html).strip()
- self.assertIn("| Header 1 | Header 2 |", md)
- self.assertIn("| --- | --- |", md)
- self.assertIn("| Row 1 Col 1 | Row 1 Col 2 |", md)
-
- def test_noise_removal(self):
- html = "Text
"
- md = MarkdownConverter.to_markdown(html).strip()
- self.assertEqual(md, "Text")
-
- def test_div_separation(self):
- html = "Line 1
Line 2
"
- md = MarkdownConverter.to_markdown(html).strip()
- cleaned = re.sub(r"\n+", "\n", md)
- self.assertEqual(cleaned, "Line 1\nLine 2")
-
- def test_block_link_distribution(self):
- # Case A: Complex link with Header and Text
- html = 'Header
Content
'
- md = MarkdownConverter.to_markdown(html).strip()
-
- # We expect distribution:
- # ### [Header](/test)
- # [Content](/test)
-
- self.assertIn("### [Header](/test)", md)
- self.assertIn("[Content](/test)", md)
- self.assertFalse("[ ### Header" in md) # Should NOT wrap the block markers
-
- # Case B: Image inside link
- html = '
'
- md = MarkdownConverter.to_markdown(html).strip()
- self.assertEqual(md, "[](/img)")
-
-
-if __name__ == "__main__":
- unittest.main()
+import unittest
+import os
+import re
+
+# sys.path handled by run_tests.py
+from web_scraper_toolkit.parsers.html_to_markdown import MarkdownConverter
+
+
+class TestMarkdownConverter(unittest.TestCase):
+ def setUp(self):
+ # Cache Scrub (Roy-Standard)
+ import shutil
+
+ cache_path = os.path.join(os.path.dirname(__file__), "__pycache__")
+ if os.path.exists(cache_path):
+ try:
+ shutil.rmtree(cache_path)
+ except Exception:
+ pass
+
+ def test_basic_conversion(self):
+ html = "Title
Body text.
"
+ # We strip to avoid whitespace nitpicking on exact newlines
+ self.assertEqual(
+ MarkdownConverter.to_markdown(html).strip(), "# Title\n\nBody text."
+ )
+
+ def test_links(self):
+ html = 'Example'
+ expected = "[Example](https://example.com)"
+ self.assertEqual(MarkdownConverter.to_markdown(html).strip(), expected)
+
+ def test_lists(self):
+ html = """
+
+ """
+ # Expected:
+ # - Item 1
+ # - Item 2
+ md = MarkdownConverter.to_markdown(html).strip()
+ self.assertIn("- Item 1", md)
+ self.assertIn("- Item 2", md)
+
+ def test_nested_structure(self):
+ html = ""
+ md = MarkdownConverter.to_markdown(html).strip()
+ self.assertEqual(md, "Paragraph **Bold**")
+
+ def test_tables(self):
+ html = """
+
+
+ | Header 1 | Header 2 |
+
+
+ | Row 1 Col 1 | Row 1 Col 2 |
+
+
+ """
+ md = MarkdownConverter.to_markdown(html).strip()
+ self.assertIn("| Header 1 | Header 2 |", md)
+ self.assertIn("| --- | --- |", md)
+ self.assertIn("| Row 1 Col 1 | Row 1 Col 2 |", md)
+
+ def test_noise_removal(self):
+ html = "Text
"
+ md = MarkdownConverter.to_markdown(html).strip()
+ self.assertEqual(md, "Text")
+
+ def test_div_separation(self):
+ html = "Line 1
Line 2
"
+ md = MarkdownConverter.to_markdown(html).strip()
+ cleaned = re.sub(r"\n+", "\n", md)
+ self.assertEqual(cleaned, "Line 1\nLine 2")
+
+ def test_block_link_distribution(self):
+ # Case A: Complex link with Header and Text
+ html = 'Header
Content
'
+ md = MarkdownConverter.to_markdown(html).strip()
+
+ # We expect distribution:
+ # ### [Header](/test)
+ # [Content](/test)
+
+ self.assertIn("### [Header](/test)", md)
+ self.assertIn("[Content](/test)", md)
+ self.assertFalse("[ ### Header" in md) # Should NOT wrap the block markers
+
+ # Case B: Image inside link
+ html = '
'
+ md = MarkdownConverter.to_markdown(html).strip()
+ self.assertEqual(md, "[](/img)")
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_multimodal.py b/tests/test_multimodal.py
index 5c7aff8..b3a2c25 100644
--- a/tests/test_multimodal.py
+++ b/tests/test_multimodal.py
@@ -1,115 +1,115 @@
-import unittest
-import os
-import shutil
-from unittest.mock import patch, AsyncMock
-
-# Adjust path to import from src
-
-# sys.path handled by run_tests.py
-
-from web_scraper_toolkit.parsers.scraping_tools import (
- extract_metadata,
- capture_screenshot,
- save_as_pdf,
-)
-
-
-class TestMultimodal(unittest.TestCase):
- def setUp(self):
- # Cache Scrub (Roy-Standard)
- import shutil
-
- cache_path = os.path.join(os.path.dirname(__file__), "__pycache__")
- if os.path.exists(cache_path):
- try:
- shutil.rmtree(cache_path)
- except Exception:
- pass
-
- self.output_dir = "test_outputs"
- os.makedirs(self.output_dir, exist_ok=True)
-
- def tearDown(self):
- if os.path.exists(self.output_dir):
- shutil.rmtree(self.output_dir)
-
- def test_extract_metadata(self):
- # We'll mock the internal _arun_extract_metadata to avoid network calls
- # But we want to test the parsing logic.
- # So we mock PlaywrightManager.smart_fetch to return a known HTML string.
-
- html_content = """
-
-
- Test Page
-
-
-
-
-
-
- """
-
- with patch(
- "web_scraper_toolkit.browser.playwright_handler.PlaywrightManager"
- ) as MockManager:
- instance = MockManager.return_value
- instance.start = AsyncMock()
- instance.stop = AsyncMock()
- # return content, url, status
- instance.smart_fetch = AsyncMock(
- return_value=(html_content, "https://example.com", 200)
- )
-
- output = extract_metadata("https://example.com")
-
- self.assertIn("METADATA REPORT", output)
- self.assertIn("Test Corp", output)
- self.assertIn("OpenGraph Title", output)
- self.assertIn("Meta Description", output)
-
- @patch("web_scraper_toolkit.browser.playwright_handler.PlaywrightManager")
- def test_capture_screenshot(self, MockManager):
- instance = MockManager.return_value
- instance.start = AsyncMock()
- instance.stop = AsyncMock()
- instance.capture_screenshot = AsyncMock(return_value=True)
-
- path = os.path.join(self.output_dir, "test.png")
- result = capture_screenshot("https://example.com", path)
-
- self.assertIn(f"Screenshot saved to {path}", result)
- instance.capture_screenshot.assert_called_once()
- args, kwargs = instance.capture_screenshot.call_args
- self.assertEqual(args[0], "https://example.com")
- self.assertEqual(args[1], path)
- self.assertTrue(kwargs.get("full_page", True))
-
- @patch("web_scraper_toolkit.browser.playwright_handler.PlaywrightManager")
- def test_save_as_pdf(self, MockManager):
- instance = MockManager.return_value
- instance.start = AsyncMock()
- instance.stop = AsyncMock()
- instance.save_pdf = AsyncMock(return_value=True)
-
- path = os.path.join(self.output_dir, "test.pdf")
- # PDF forces headless in the code
- result = save_as_pdf("https://example.com", path)
-
- self.assertIn(f"PDF saved to {path}", result)
- instance.save_pdf.assert_called_once()
-
- # Verify config update was passed to Manager constructor
- # We can't easily check constructor args on the mocked class instance creation
- # unless we check the CALL context of MockManager.
- # But for unit test, just verifying save_pdf is called is sufficient.
-
-
-if __name__ == "__main__":
- unittest.main()
+import unittest
+import os
+import shutil
+from unittest.mock import patch, AsyncMock
+
+# Adjust path to import from src
+
+# sys.path handled by run_tests.py
+
+from web_scraper_toolkit.parsers.scraping_tools import (
+ extract_metadata,
+ capture_screenshot,
+ save_as_pdf,
+)
+
+
+class TestMultimodal(unittest.TestCase):
+ def setUp(self):
+ # Cache Scrub (Roy-Standard)
+ import shutil
+
+ cache_path = os.path.join(os.path.dirname(__file__), "__pycache__")
+ if os.path.exists(cache_path):
+ try:
+ shutil.rmtree(cache_path)
+ except Exception:
+ pass
+
+ self.output_dir = "test_outputs"
+ os.makedirs(self.output_dir, exist_ok=True)
+
+ def tearDown(self):
+ if os.path.exists(self.output_dir):
+ shutil.rmtree(self.output_dir)
+
+ def test_extract_metadata(self):
+ # We'll mock the internal _arun_extract_metadata to avoid network calls
+ # But we want to test the parsing logic.
+ # So we mock PlaywrightManager.smart_fetch to return a known HTML string.
+
+ html_content = """
+
+
+ Test Page
+
+
+
+
+
+
+ """
+
+ with patch(
+ "web_scraper_toolkit.browser.playwright_handler.PlaywrightManager"
+ ) as MockManager:
+ instance = MockManager.return_value
+ instance.start = AsyncMock()
+ instance.stop = AsyncMock()
+ # return content, url, status
+ instance.smart_fetch = AsyncMock(
+ return_value=(html_content, "https://example.com", 200)
+ )
+
+ output = extract_metadata("https://example.com")
+
+ self.assertIn("METADATA REPORT", output)
+ self.assertIn("Test Corp", output)
+ self.assertIn("OpenGraph Title", output)
+ self.assertIn("Meta Description", output)
+
+ @patch("web_scraper_toolkit.browser.playwright_handler.PlaywrightManager")
+ def test_capture_screenshot(self, MockManager):
+ instance = MockManager.return_value
+ instance.start = AsyncMock()
+ instance.stop = AsyncMock()
+ instance.capture_screenshot = AsyncMock(return_value=True)
+
+ path = os.path.join(self.output_dir, "test.png")
+ result = capture_screenshot("https://example.com", path)
+
+ self.assertIn(f"Screenshot saved to {path}", result)
+ instance.capture_screenshot.assert_called_once()
+ args, kwargs = instance.capture_screenshot.call_args
+ self.assertEqual(args[0], "https://example.com")
+ self.assertEqual(args[1], path)
+ self.assertTrue(kwargs.get("full_page", True))
+
+ @patch("web_scraper_toolkit.browser.playwright_handler.PlaywrightManager")
+ def test_save_as_pdf(self, MockManager):
+ instance = MockManager.return_value
+ instance.start = AsyncMock()
+ instance.stop = AsyncMock()
+ instance.save_pdf = AsyncMock(return_value=True)
+
+ path = os.path.join(self.output_dir, "test.pdf")
+ # PDF forces headless in the code
+ result = save_as_pdf("https://example.com", path)
+
+ self.assertIn(f"PDF saved to {path}", result)
+ instance.save_pdf.assert_called_once()
+
+ # Verify config update was passed to Manager constructor
+ # We can't easily check constructor args on the mocked class instance creation
+ # unless we check the CALL context of MockManager.
+ # But for unit test, just verifying save_pdf is called is sufficient.
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_scraping_tools.py b/tests/test_scraping_tools.py
index 18fb751..7d95916 100644
--- a/tests/test_scraping_tools.py
+++ b/tests/test_scraping_tools.py
@@ -1,116 +1,116 @@
-import unittest
-from unittest.mock import patch, AsyncMock
-import os
-import asyncio
-
-# Ensure src is in path
-# sys.path handled by run_tests.py
-
-# We need to test specific logic inside content module.
-# Since the logic is inside _arun_scrape, we can mock PlaywrightManager
-# to return specific HTML content and verify the extraction output.
-
-from web_scraper_toolkit.parsers.content import _arun_scrape
-from web_scraper_toolkit.parsers.scraping_tools import get_sitemap_urls
-
-
-class TestScrapingTools(unittest.TestCase):
- def setUp(self):
- # Cache Scrub (Roy-Standard)
- import shutil
-
- cache_path = os.path.join(os.path.dirname(__file__), "__pycache__")
- if os.path.exists(cache_path):
- try:
- shutil.rmtree(cache_path)
- except Exception:
- pass
-
- self.loop = asyncio.new_event_loop()
- asyncio.set_event_loop(self.loop)
-
- def tearDown(self):
- self.loop.close()
-
- @patch("web_scraper_toolkit.browser.playwright_handler.PlaywrightManager")
- def test_arun_scrape_extraction(self, MockPlaywrightManager):
- # Setup Mock
- mock_manager = MockPlaywrightManager.return_value
- mock_manager.start = AsyncMock()
- mock_manager.stop = AsyncMock()
- mock_manager.get_new_page = AsyncMock()
- mock_manager.fetch_page_content = AsyncMock()
-
- # Mock Page/Context
- mock_page = AsyncMock()
- mock_context = AsyncMock()
- mock_manager.get_new_page.return_value = (mock_page, mock_context)
-
- # FIX: We now use smart_fetch in the tools, so we must mock it instead of/in addition to fetch_page_content
- mock_manager.smart_fetch = AsyncMock()
-
- # Mock Content
- html_content = """
-
- Test Company
-
- Welcome to Test Company.
- Our CEO is John Doe.
- Contact us at contact@example.com
-
-
- """
- # mock_manager.smart_fetch.return_value = (html_content, "https://example.com", 200)
- mock_manager.smart_fetch.return_value = (
- html_content,
- "https://example.com",
- 200,
- )
-
- # Run extraction
- result = self.loop.run_until_complete(_arun_scrape("https://example.com"))
-
- # Verify assertions
- self.assertIn("TITLE: Test Company", result)
- self.assertIn("John Doe", result) # Leadership extraction
- self.assertIn("contact@example.com", result) # Email extraction
-
- @patch(
- "web_scraper_toolkit.parsers.sitemap.tools.extract_sitemap_tree",
- new_callable=AsyncMock,
- )
- @patch(
- "web_scraper_toolkit.parsers.sitemap.tools.peek_sitemap_index",
- new_callable=AsyncMock,
- )
- @patch(
- "web_scraper_toolkit.parsers.sitemap.tools.find_sitemap_urls",
- new_callable=AsyncMock,
- )
- def test_sitemap_extraction(self, mock_find, mock_peek, mock_extract):
- # Mock finding a sitemap
- mock_find.return_value = ["https://example.com/sitemap.xml"]
-
- # Mock extracting URLs from it
- mock_extract.return_value = [
- "https://example.com/about",
- "https://example.com/contact",
- "https://example.com/product/123",
- ]
-
- # Mock peeking - return what correct extract returns
- mock_peek.return_value = {"type": "urlset", "urls": mock_extract.return_value}
-
- result = get_sitemap_urls("https://example.com")
-
- # The function sums up found URLs.
- # Total unique = 3.
- self.assertIn("Contains 3 relevant URLs", result)
- self.assertIn("https://example.com/about", result)
- self.assertIn("https://example.com/contact", result)
- # Products are typically included unless they are assets
- self.assertIn("https://example.com/product/123", result)
-
-
-if __name__ == "__main__":
- unittest.main()
+import unittest
+from unittest.mock import patch, AsyncMock
+import os
+import asyncio
+
+# Ensure src is in path
+# sys.path handled by run_tests.py
+
+# We need to test specific logic inside content module.
+# Since the logic is inside _arun_scrape, we can mock PlaywrightManager
+# to return specific HTML content and verify the extraction output.
+
+from web_scraper_toolkit.parsers.content import _arun_scrape
+from web_scraper_toolkit.parsers.scraping_tools import get_sitemap_urls
+
+
+class TestScrapingTools(unittest.TestCase):
+ def setUp(self):
+ # Cache Scrub (Roy-Standard)
+ import shutil
+
+ cache_path = os.path.join(os.path.dirname(__file__), "__pycache__")
+ if os.path.exists(cache_path):
+ try:
+ shutil.rmtree(cache_path)
+ except Exception:
+ pass
+
+ self.loop = asyncio.new_event_loop()
+ asyncio.set_event_loop(self.loop)
+
+ def tearDown(self):
+ self.loop.close()
+
+ @patch("web_scraper_toolkit.browser.playwright_handler.PlaywrightManager")
+ def test_arun_scrape_extraction(self, MockPlaywrightManager):
+ # Setup Mock
+ mock_manager = MockPlaywrightManager.return_value
+ mock_manager.start = AsyncMock()
+ mock_manager.stop = AsyncMock()
+ mock_manager.get_new_page = AsyncMock()
+ mock_manager.fetch_page_content = AsyncMock()
+
+ # Mock Page/Context
+ mock_page = AsyncMock()
+ mock_context = AsyncMock()
+ mock_manager.get_new_page.return_value = (mock_page, mock_context)
+
+ # FIX: We now use smart_fetch in the tools, so we must mock it instead of/in addition to fetch_page_content
+ mock_manager.smart_fetch = AsyncMock()
+
+ # Mock Content
+ html_content = """
+
+ Test Company
+
+ Welcome to Test Company.
+ Our CEO is John Doe.
+ Contact us at contact@example.com
+
+
+ """
+ # mock_manager.smart_fetch.return_value = (html_content, "https://example.com", 200)
+ mock_manager.smart_fetch.return_value = (
+ html_content,
+ "https://example.com",
+ 200,
+ )
+
+ # Run extraction
+ result = self.loop.run_until_complete(_arun_scrape("https://example.com"))
+
+ # Verify assertions
+ self.assertIn("TITLE: Test Company", result)
+ self.assertIn("John Doe", result) # Leadership extraction
+ self.assertIn("contact@example.com", result) # Email extraction
+
+ @patch(
+ "web_scraper_toolkit.parsers.sitemap.tools.extract_sitemap_tree",
+ new_callable=AsyncMock,
+ )
+ @patch(
+ "web_scraper_toolkit.parsers.sitemap.tools.peek_sitemap_index",
+ new_callable=AsyncMock,
+ )
+ @patch(
+ "web_scraper_toolkit.parsers.sitemap.tools.find_sitemap_urls",
+ new_callable=AsyncMock,
+ )
+ def test_sitemap_extraction(self, mock_find, mock_peek, mock_extract):
+ # Mock finding a sitemap
+ mock_find.return_value = ["https://example.com/sitemap.xml"]
+
+ # Mock extracting URLs from it
+ mock_extract.return_value = [
+ "https://example.com/about",
+ "https://example.com/contact",
+ "https://example.com/product/123",
+ ]
+
+ # Mock peeking - return what correct extract returns
+ mock_peek.return_value = {"type": "urlset", "urls": mock_extract.return_value}
+
+ result = get_sitemap_urls("https://example.com")
+
+ # The function sums up found URLs.
+ # Total unique = 3.
+ self.assertIn("Contains 3 relevant URLs", result)
+ self.assertIn("https://example.com/about", result)
+ self.assertIn("https://example.com/contact", result)
+ # Products are typically included unless they are assets
+ self.assertIn("https://example.com/product/123", result)
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/test_serp_parser.py b/tests/test_serp_parser.py
index 41a8fd2..d4ef620 100644
--- a/tests/test_serp_parser.py
+++ b/tests/test_serp_parser.py
@@ -1,64 +1,64 @@
-import unittest
-
-# Ensure src is in path
-# sys.path handled by run_tests.py
-
-from web_scraper_toolkit.parsers import SerpParser
-
-
-class TestSerpParser(unittest.TestCase):
- def test_parse_ddg_html_sample(self):
- # Simulated DuckDuckGo HTML structure
- html = """
-
-
-
-
-
This is a sample snippet for the example domain.
-
-
-
-
Another snippet here.
-
-
-
- """
- results = SerpParser.parse_ddg_html(html, "https://duckduckgo.com")
- self.assertEqual(len(results), 2)
- self.assertEqual(results[0]["url"], "https://example.com/foo")
- self.assertEqual(results[0]["title"], "Example Domain")
- self.assertIn("sample snippet", results[0]["snippet"])
-
- def test_parse_google_style_sample(self):
- # Simulated Google HTML structure (div.g)
- html = """
-
-
-
-
-
Google snippet text...
-
-
-
- """
- results = SerpParser.parse_google_direct_links_style(html, "https://google.com")
- self.assertEqual(len(results), 1)
- self.assertEqual(results[0]["url"], "https://example.com/result")
- self.assertEqual(results[0]["title"], "Google Result Title")
- self.assertEqual(results[0]["snippet"], "Google snippet text...")
-
- def test_empty_content(self):
- results = SerpParser.parse_ddg_html("", "https://duckduckgo.com")
- self.assertEqual(results, [])
-
-
-if __name__ == "__main__":
- unittest.main()
+import unittest
+
+# Ensure src is in path
+# sys.path handled by run_tests.py
+
+from web_scraper_toolkit.parsers import SerpParser
+
+
+class TestSerpParser(unittest.TestCase):
+ def test_parse_ddg_html_sample(self):
+ # Simulated DuckDuckGo HTML structure
+ html = """
+
+
+
+
+
This is a sample snippet for the example domain.
+
+
+
+
Another snippet here.
+
+
+
+ """
+ results = SerpParser.parse_ddg_html(html, "https://duckduckgo.com")
+ self.assertEqual(len(results), 2)
+ self.assertEqual(results[0]["url"], "https://example.com/foo")
+ self.assertEqual(results[0]["title"], "Example Domain")
+ self.assertIn("sample snippet", results[0]["snippet"])
+
+ def test_parse_google_style_sample(self):
+ # Simulated Google HTML structure (div.g)
+ html = """
+
+
+
+
+
Google snippet text...
+
+
+
+ """
+ results = SerpParser.parse_google_direct_links_style(html, "https://google.com")
+ self.assertEqual(len(results), 1)
+ self.assertEqual(results[0]["url"], "https://example.com/result")
+ self.assertEqual(results[0]["title"], "Google Result Title")
+ self.assertEqual(results[0]["snippet"], "Google snippet text...")
+
+ def test_empty_content(self):
+ results = SerpParser.parse_ddg_html("", "https://duckduckgo.com")
+ self.assertEqual(results, [])
+
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/tests/verify_imports.py b/tests/verify_imports.py
index 5727aa8..eceae68 100644
--- a/tests/verify_imports.py
+++ b/tests/verify_imports.py
@@ -1,42 +1,42 @@
-import sys
-import os
-import pkgutil
-import importlib
-import traceback
-
-# Add src to path so we can import as if installed
-sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../src")))
-
-
-def verify_package(package_name):
- print(f"🔍 Verifying package: {package_name}")
- try:
- root_pkg = importlib.import_module(package_name)
- print(f"✅ Root import successful: {package_name}")
- except Exception as e:
- print(f"❌ Root import failed: {e}")
- traceback.print_exc()
- return
-
- # Walk through all submodules
- path = root_pkg.__path__
- prefix = package_name + "."
-
- for _, name, ispkg in pkgutil.walk_packages(path, prefix):
- print(f" Checking {name}...", end=" ")
- try:
- importlib.import_module(name)
- print("✅ OK")
- except Exception as e:
- print("❌ FAILED")
- print(f" Error: {e}")
- # traceback.print_exc()
-
-
-if __name__ == "__main__":
- # verify_package("web_scraper_toolkit")
- try:
- importlib.import_module("web_scraper_toolkit.parsers.scraping_tools")
- print("✅ scraping_tools OK")
- except Exception:
- traceback.print_exc()
+import sys
+import os
+import pkgutil
+import importlib
+import traceback
+
+# Add src to path so we can import as if installed
+sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../src")))
+
+
+def verify_package(package_name):
+ print(f"🔍 Verifying package: {package_name}")
+ try:
+ root_pkg = importlib.import_module(package_name)
+ print(f"✅ Root import successful: {package_name}")
+ except Exception as e:
+ print(f"❌ Root import failed: {e}")
+ traceback.print_exc()
+ return
+
+ # Walk through all submodules
+ path = root_pkg.__path__
+ prefix = package_name + "."
+
+ for _, name, ispkg in pkgutil.walk_packages(path, prefix):
+ print(f" Checking {name}...", end=" ")
+ try:
+ importlib.import_module(name)
+ print("✅ OK")
+ except Exception as e:
+ print("❌ FAILED")
+ print(f" Error: {e}")
+ # traceback.print_exc()
+
+
+if __name__ == "__main__":
+ # verify_package("web_scraper_toolkit")
+ try:
+ importlib.import_module("web_scraper_toolkit.parsers.scraping_tools")
+ print("✅ scraping_tools OK")
+ except Exception:
+ traceback.print_exc()