diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 32a6720..b886dcc 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -37,7 +37,7 @@ jobs: steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@v6 - name: Set up Python ${{ matrix.python-version }} uses: actions/setup-python@v6 @@ -86,7 +86,7 @@ jobs: PIP_DISABLE_PIP_VERSION_CHECK: "1" steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@v6 - name: Set up Python 3.12 uses: actions/setup-python@v6 @@ -134,7 +134,7 @@ jobs: SKIP_CF_TEST: "0" steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@v6 - name: Set up Python 3.12 uses: actions/setup-python@v6 diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index 8033321..3c8e06b 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -26,7 +26,7 @@ jobs: PIP_DISABLE_PIP_VERSION_CHECK: "1" steps: - name: Checkout - uses: actions/checkout@v4 + uses: actions/checkout@v6 - name: Set up Python uses: actions/setup-python@v6 diff --git a/run_server.py b/run_server.py index 859f22e..4881cdc 100644 --- a/run_server.py +++ b/run_server.py @@ -1,11 +1,11 @@ -# ./run_server.py -""" -Shim for local development. -Please use 'web-scraper-server' (or 'python -m web_scraper_toolkit.server.mcp_server') -when installed. -""" - -from web_scraper_toolkit.server.mcp_server import main - -if __name__ == "__main__": - main() +# ./run_server.py +""" +Shim for local development. +Please use 'web-scraper-server' (or 'python -m web_scraper_toolkit.server.mcp_server') +when installed. +""" + +from web_scraper_toolkit.server.mcp_server import main + +if __name__ == "__main__": + main() diff --git a/src/web_scraper_toolkit/core/diagnostics.py b/src/web_scraper_toolkit/core/diagnostics.py index fcbc27a..f96c0a1 100644 --- a/src/web_scraper_toolkit/core/diagnostics.py +++ b/src/web_scraper_toolkit/core/diagnostics.py @@ -1,112 +1,112 @@ -# ./src/web_scraper_toolkit/core/diagnostics.py -""" -Diagnostics Module -================== - -Provides system self-checks to ensure the environment is correctly configured. -Checks Playwright installation, browser binaries, and network connectivity. - -Usage: - print_diagnostics() - -Outputs: - - Printed report of system status to stdout. -""" - -import asyncio -import logging -import sys -from typing import Dict, Any - -from playwright.async_api import async_playwright - -logger = logging.getLogger(__name__) - - -async def verify_environment() -> Dict[str, Any]: - """ - Checks the scraping environment for necessary dependencies and functionality. - - Returns: - Dict containing status of: - - python_version - - playwright_installed - - browser_launch_successful - - browsers_found - """ - report: Dict[str, Any] = { - "python_version": sys.version, - "playwright_installed": False, - "browser_launch_successful": False, - "browsers_found": [], - "errors": [], - } - - # 1. Check Playwright Import - try: - from importlib.util import find_spec - - if find_spec("playwright"): - report["playwright_installed"] = True - else: - report["errors"].append("Playwright package not found.") - except ImportError: - report["errors"].append("Playwright package not installed.") - return report - - # 2. Check Browsers - try: - async with async_playwright() as p: - # Try launching Chromium - try: - browser = await p.chromium.launch(headless=True) - report["browsers_found"].append("chromium") - report["browser_launch_successful"] = True - await browser.close() - except Exception as e: - report["errors"].append(f"Chromium launch failed: {e}") - - # Try launching Firefox (optional but good to know) - try: - browser = await p.firefox.launch(headless=True) - report["browsers_found"].append("firefox") - await browser.close() - except Exception: - pass # Firefox might not be installed, that's okay - - except Exception as e: - report["errors"].append(f"Playwright runtime error: {e}") - - return report - - -def print_diagnostics() -> None: - """Runs the verification and prints a human-readable report.""" - print("Running WebScraperToolkit Diagnostics...") - try: - results = asyncio.run(verify_environment()) - - print(f"\nPython Version: {results['python_version'].split()[0]}") - print( - f"Playwright Installed: {'✅' if results['playwright_installed'] else '❌'}" - ) - - if results["playwright_installed"]: - if results["browser_launch_successful"]: - print("Browser Launch: ✅ (Chromium)") - else: - print("Browser Launch: ❌") - - print(f"Browsers Detected: {', '.join(results['browsers_found'])}") - - if results["errors"]: - print("\n⚠️ Issues Found:") - for err in results["errors"]: - print(f" - {err}") - - if "Executable doesn't exist" in str(results["errors"]): - print( - "\nSuggestion: Run `playwright install` to download necessary browsers." - ) - except Exception as e: - print(f"Diagnostics failed to run: {e}") +# ./src/web_scraper_toolkit/core/diagnostics.py +""" +Diagnostics Module +================== + +Provides system self-checks to ensure the environment is correctly configured. +Checks Playwright installation, browser binaries, and network connectivity. + +Usage: + print_diagnostics() + +Outputs: + - Printed report of system status to stdout. +""" + +import asyncio +import logging +import sys +from typing import Dict, Any + +from playwright.async_api import async_playwright + +logger = logging.getLogger(__name__) + + +async def verify_environment() -> Dict[str, Any]: + """ + Checks the scraping environment for necessary dependencies and functionality. + + Returns: + Dict containing status of: + - python_version + - playwright_installed + - browser_launch_successful + - browsers_found + """ + report: Dict[str, Any] = { + "python_version": sys.version, + "playwright_installed": False, + "browser_launch_successful": False, + "browsers_found": [], + "errors": [], + } + + # 1. Check Playwright Import + try: + from importlib.util import find_spec + + if find_spec("playwright"): + report["playwright_installed"] = True + else: + report["errors"].append("Playwright package not found.") + except ImportError: + report["errors"].append("Playwright package not installed.") + return report + + # 2. Check Browsers + try: + async with async_playwright() as p: + # Try launching Chromium + try: + browser = await p.chromium.launch(headless=True) + report["browsers_found"].append("chromium") + report["browser_launch_successful"] = True + await browser.close() + except Exception as e: + report["errors"].append(f"Chromium launch failed: {e}") + + # Try launching Firefox (optional but good to know) + try: + browser = await p.firefox.launch(headless=True) + report["browsers_found"].append("firefox") + await browser.close() + except Exception: + pass # Firefox might not be installed, that's okay + + except Exception as e: + report["errors"].append(f"Playwright runtime error: {e}") + + return report + + +def print_diagnostics() -> None: + """Runs the verification and prints a human-readable report.""" + print("Running WebScraperToolkit Diagnostics...") + try: + results = asyncio.run(verify_environment()) + + print(f"\nPython Version: {results['python_version'].split()[0]}") + print( + f"Playwright Installed: {'✅' if results['playwright_installed'] else '❌'}" + ) + + if results["playwright_installed"]: + if results["browser_launch_successful"]: + print("Browser Launch: ✅ (Chromium)") + else: + print("Browser Launch: ❌") + + print(f"Browsers Detected: {', '.join(results['browsers_found'])}") + + if results["errors"]: + print("\n⚠️ Issues Found:") + for err in results["errors"]: + print(f" - {err}") + + if "Executable doesn't exist" in str(results["errors"]): + print( + "\nSuggestion: Run `playwright install` to download necessary browsers." + ) + except Exception as e: + print(f"Diagnostics failed to run: {e}") diff --git a/src/web_scraper_toolkit/core/utils.py b/src/web_scraper_toolkit/core/utils.py index c8c9d42..64ad6eb 100644 --- a/src/web_scraper_toolkit/core/utils.py +++ b/src/web_scraper_toolkit/core/utils.py @@ -1,60 +1,60 @@ -# ./src/web_scraper_toolkit/core/utils.py -# WebScraperToolkit/src/web_scraper_toolkit/utils.py -from typing import Optional -from urllib.parse import urlparse, urljoin -import logging - -logger = logging.getLogger(__name__) - - -def normalize_url(url: str, base_url: Optional[str] = None) -> Optional[str]: - """Normalizes a URL, making it absolute if a base_url is provided.""" - try: - if base_url: - url = urljoin(base_url, url.strip()) - - parsed = urlparse(url.strip()) - if not parsed.scheme or parsed.scheme not in ["http", "https"]: - return None # Scheme required and must be http/https - if not parsed.netloc: - return None # Domain name required - - # Remove 'www.' prefix for consistency in domain comparison - netloc = parsed.netloc.lower() - if netloc.startswith("www."): - netloc = netloc[4:] - - # Reconstruct with lowercase scheme and netloc, and stripped path/query/fragment - path = ( - parsed.path.rstrip("/") if parsed.path else "" - ) # Remove trailing slash from path for consistency - - normalized = f"{parsed.scheme.lower()}://{netloc}{path}" - if parsed.query: - normalized += f"?{parsed.query}" - - return normalized - except Exception as e: - logger.debug(f"Failed to normalize URL '{url}': {e}") - return None - - -def get_domain_from_url(url: str) -> Optional[str]: - """Extracts the normalized domain (e.g., example.com) from a URL.""" - try: - parsed_url = urlparse(url.lower()) - domain = parsed_url.netloc - if domain.startswith("www."): - domain = domain[4:] - return domain - except Exception: - return None - - -def truncate_text(text: str, max_length: int = 100) -> str: - """Truncates text to a maximum length, adding ellipsis if truncated.""" - if not text: - return "" - if len(text) <= max_length: - return text - return text[:max_length].rstrip() + "..." +# ./src/web_scraper_toolkit/core/utils.py +# WebScraperToolkit/src/web_scraper_toolkit/utils.py +from typing import Optional +from urllib.parse import urlparse, urljoin +import logging + +logger = logging.getLogger(__name__) + + +def normalize_url(url: str, base_url: Optional[str] = None) -> Optional[str]: + """Normalizes a URL, making it absolute if a base_url is provided.""" + try: + if base_url: + url = urljoin(base_url, url.strip()) + + parsed = urlparse(url.strip()) + if not parsed.scheme or parsed.scheme not in ["http", "https"]: + return None # Scheme required and must be http/https + if not parsed.netloc: + return None # Domain name required + + # Remove 'www.' prefix for consistency in domain comparison + netloc = parsed.netloc.lower() + if netloc.startswith("www."): + netloc = netloc[4:] + + # Reconstruct with lowercase scheme and netloc, and stripped path/query/fragment + path = ( + parsed.path.rstrip("/") if parsed.path else "" + ) # Remove trailing slash from path for consistency + + normalized = f"{parsed.scheme.lower()}://{netloc}{path}" + if parsed.query: + normalized += f"?{parsed.query}" + + return normalized + except Exception as e: + logger.debug(f"Failed to normalize URL '{url}': {e}") + return None + + +def get_domain_from_url(url: str) -> Optional[str]: + """Extracts the normalized domain (e.g., example.com) from a URL.""" + try: + parsed_url = urlparse(url.lower()) + domain = parsed_url.netloc + if domain.startswith("www."): + domain = domain[4:] + return domain + except Exception: + return None + + +def truncate_text(text: str, max_length: int = 100) -> str: + """Truncates text to a maximum length, adding ellipsis if truncated.""" + if not text: + return "" + if len(text) <= max_length: + return text + return text[:max_length].rstrip() + "..." diff --git a/src/web_scraper_toolkit/core/verify_deps.py b/src/web_scraper_toolkit/core/verify_deps.py index 1b012ec..8119c45 100644 --- a/src/web_scraper_toolkit/core/verify_deps.py +++ b/src/web_scraper_toolkit/core/verify_deps.py @@ -1,60 +1,60 @@ -# ./src/web_scraper_toolkit/core/verify_deps.py -""" -Dependency Verification Module -============================== - -Checks that critical runtime dependencies meet the required versions. -Implements the "Fail Gently" philosophy by guiding users to upgrade instead of crashing hard. - -Usage: - if not verify_dependencies(): sys.exit(1) - -Key Checks: - - playwright-stealth >= 2.0.0 - - playwright >= 1.56.0 (or base compatible) - -Key Outputs: - - Boolean (True if all good, False if missing/old). - - Printed instructions for the user. -""" - -import importlib.metadata -from packaging import version -from .logger import setup_logger - -logger = setup_logger() - - -def verify_dependencies(): - """ - Verifies that critical dependencies meet the 'Expert' standards. - If not, it prints a helpful message and returns False. - """ - required = { - "playwright-stealth": "2.0.0", - "playwright": "1.40.0", # User mentioned 1.56, but let's be safe. 1.40 is stable base. - # Actually user said "playwright>=1.56.0". Let's use that. - } - required["playwright"] = "1.56.0" - - all_good = True - - for package, min_ver in required.items(): - try: - installed_ver = importlib.metadata.version(package) - if version.parse(installed_ver) < version.parse(min_ver): - logger.error( - f"❌ Dependency Error: {package} {installed_ver} is too old. Required: >={min_ver}" - ) - print(f"\n[!] Critical Dependency Update Required for '{package}':") - print(f" Current: {installed_ver}") - print(f" Required: >={min_ver}") - print(f" Selected: pip install --upgrade {package}\n") - all_good = False - except importlib.metadata.PackageNotFoundError: - logger.error(f"❌ Missing Dependency: {package}") - print(f"\n[!] Missing Critical Dependency: '{package}'") - print(f" Selected: pip install {package}\n") - all_good = False - - return all_good +# ./src/web_scraper_toolkit/core/verify_deps.py +""" +Dependency Verification Module +============================== + +Checks that critical runtime dependencies meet the required versions. +Implements the "Fail Gently" philosophy by guiding users to upgrade instead of crashing hard. + +Usage: + if not verify_dependencies(): sys.exit(1) + +Key Checks: + - playwright-stealth >= 2.0.0 + - playwright >= 1.56.0 (or base compatible) + +Key Outputs: + - Boolean (True if all good, False if missing/old). + - Printed instructions for the user. +""" + +import importlib.metadata +from packaging import version +from .logger import setup_logger + +logger = setup_logger() + + +def verify_dependencies(): + """ + Verifies that critical dependencies meet the 'Expert' standards. + If not, it prints a helpful message and returns False. + """ + required = { + "playwright-stealth": "2.0.0", + "playwright": "1.40.0", # User mentioned 1.56, but let's be safe. 1.40 is stable base. + # Actually user said "playwright>=1.56.0". Let's use that. + } + required["playwright"] = "1.56.0" + + all_good = True + + for package, min_ver in required.items(): + try: + installed_ver = importlib.metadata.version(package) + if version.parse(installed_ver) < version.parse(min_ver): + logger.error( + f"❌ Dependency Error: {package} {installed_ver} is too old. Required: >={min_ver}" + ) + print(f"\n[!] Critical Dependency Update Required for '{package}':") + print(f" Current: {installed_ver}") + print(f" Required: >={min_ver}") + print(f" Selected: pip install --upgrade {package}\n") + all_good = False + except importlib.metadata.PackageNotFoundError: + logger.error(f"❌ Missing Dependency: {package}") + print(f"\n[!] Missing Critical Dependency: '{package}'") + print(f" Selected: pip install {package}\n") + all_good = False + + return all_good diff --git a/src/web_scraper_toolkit/parsers/html_to_markdown.py b/src/web_scraper_toolkit/parsers/html_to_markdown.py index 5ab5e58..ae56891 100644 --- a/src/web_scraper_toolkit/parsers/html_to_markdown.py +++ b/src/web_scraper_toolkit/parsers/html_to_markdown.py @@ -1,396 +1,396 @@ -# ./src/web_scraper_toolkit/parsers/html_to_markdown.py -""" -HTML to Markdown Converter -========================== - -Utility for converting HTML content into clean, readable Markdown. -Specializes in handling tables, lists, and links properly. - -Usage: - md = MarkdownConverter.to_markdown(html_content, base_url="...") - -Key Features: - - Table conversion (ASCII/GD style). - - Link resolution (absolute paths). - - Noise removal (scripts/styles). -""" - -from bs4 import BeautifulSoup, NavigableString, Tag -import re -from typing import Any, Optional - - -class MarkdownConverter: - """ - Converts HTML content to clean Markdown, preserving structure for LLM consumption. - """ - - @staticmethod - def to_markdown(html_content: str, base_url: str = "") -> str: - """ - Main entry point. Converts valid HTML string to Markdown. - """ - if not html_content: - return "" - - soup = BeautifulSoup(html_content, "lxml") - - # 1. Cleanup Noise - # Removing semantic noise is critical for clean output - for tag in soup( - [ - "script", - "style", - "noscript", - "iframe", - "svg", - "meta", - "link", - "head", - "nav", - "footer", - "header", - "aside", - ] - ): - tag.decompose() - - # 2. Process Content - # We process the body if available, else the whole soup - root = soup.body if soup.body else soup - - markdown = MarkdownConverter._process_element(root, base_url).strip() - - # 3. Post-Processing: Collapse excessive newlines - # Nested blocks (div > div > p) often generate \n\n\n\n. We want max 2. - markdown = re.sub(r"\n{3,}", "\n\n", markdown) - - return markdown - - @staticmethod - def _attr_to_str(value: Any) -> Optional[str]: - """Normalize BeautifulSoup attribute values into strings.""" - if value is None: - return None - if isinstance(value, str): - return value - if isinstance(value, (list, tuple)): - for item in value: - if isinstance(item, str) and item: - return item - return None - return str(value) - - @staticmethod - def _is_block_element(tag: Any) -> bool: - """Checks if a tag is a block element.""" - if not isinstance(tag, Tag): - return False - tag_name = (tag.name or "").lower() - return tag_name in [ - "div", - "section", - "article", - "main", - "header", - "footer", - "ul", - "ol", - "li", - "h1", - "h2", - "h3", - "h4", - "h5", - "h6", - "p", - "table", - "blockquote", - "pre", - ] - - @staticmethod - def _has_block_children(element: Tag) -> bool: - """Checks if an element contains any direct block-level children.""" - for child in element.children: - if MarkdownConverter._is_block_element(child): - return True - # Also check recursively? No, usually direct children are enough if structure is good. - # But
..
is possible. - # Let's do a simple non-recursive check first, or shallow scan. - # Deep scan: - if isinstance(child, Tag) and child.find( - [ - "div", - "p", - "h1", - "h2", - "h3", - "h4", - "h5", - "h6", - "ul", - "ol", - "li", - "table", - ] - ): - return True - return False - - @staticmethod - def _process_element( - element: Any, base_url: str = "", link_context: Optional[str] = None - ) -> str: - if element is None: - return "" - - # TEXT - if isinstance(element, NavigableString): - raw_text = str(element) - # Normalize whitespace but KEEP leading/trailing spaces for separation - normalized_text = re.sub(r"\s+", " ", raw_text) - if not normalized_text: - return "" - - # If we are in a link context, we link the text directly - if link_context and normalized_text.strip(): - # Avoid linking whitespace only - return f"[{normalized_text}]({link_context})" - return normalized_text - - if not isinstance(element, Tag): - return "" - - # TAGS mapping - tag_name = (element.name or "").lower() - - # --- BLOCK ELEMENTS --- # - - if tag_name in ["h1", "h2", "h3", "h4", "h5", "h6"]: - level = int(tag_name[1]) - # Pass link_context down to inner text - content = MarkdownConverter._get_inner_text(element, base_url, link_context) - # Use fewer newlines to avoid stacking up to 4+ - return f"\n{'#' * level} {content}\n" - - if tag_name == "p": - content = MarkdownConverter._get_inner_text(element, base_url, link_context) - return f"\n{content}\n" if content else "" - - if tag_name == "br": - return "\n" - - if tag_name == "hr": - return "\n\n---\n\n" - - if tag_name in ["ul", "ol"]: - return f"\n{MarkdownConverter._process_list(element, base_url, link_context)}\n" - - if tag_name == "li": - # Handled by parent UL/OL usually, but if standalone: - content = MarkdownConverter._get_inner_text(element, base_url, link_context) - return f"- {content}\n" - - if tag_name == "blockquote": - content = MarkdownConverter._get_inner_text(element, base_url, link_context) - lines = content.split("\n") - quoted = "\n".join(f"> {line}" for line in lines if line.strip()) - return f"\n{quoted}\n" - - if tag_name == "pre": - # Code blocks inside links are rare and likely break markdown parsers anyway. - # We ignore link_context here for code blocks usually. - code_tag = element.find("code") - content = "" - if code_tag: - content = code_tag.get_text() # Raw text for code - else: - content = element.get_text() - return f"\n\n```\n{content}\n```\n\n" - - # --- TABLES --- # - if tag_name == "table": - return f"\n\n{MarkdownConverter._process_table(element, base_url)}\n\n" - - # --- INLINE ELEMENTS --- # - - if tag_name == "a": - href = MarkdownConverter._attr_to_str(element.get("href")) or "" - if not href or href.startswith("#") or href.startswith("javascript:"): - return MarkdownConverter._get_inner_text( - element, base_url, link_context - ) - - # Check if this link wraps BLOCK elements - if MarkdownConverter._has_block_children(element): - # DISTRIBUTE THE LINK DOWN - # We do NOT wrap this tag. We recurse. - return MarkdownConverter._get_inner_text( - element, base_url, link_context=href - ) - - # STANDARD INLINE LINK - text = MarkdownConverter._get_inner_text(element, base_url, link_context) - # If we are already inside a link context, we shouldn't nest links technically. - # Markdown doesn't support nested links. - # We defer to the outer link? Or inner? - # Usually inner link takes precedence in HTML, but markdown breaks. - # Let's assume inner takes precedence. - return f"[{text}]({href})" - - if tag_name == "img": - alt = MarkdownConverter._attr_to_str(element.get("alt")) or "Image" - src = MarkdownConverter._attr_to_str(element.get("src")) or "" - if not src: - return "" - img_md = f"![{alt}]({src})" - if link_context: - return f"[{img_md}]({link_context})" - return img_md - - if tag_name in ["strong", "b"]: - content = MarkdownConverter._get_inner_text(element, base_url, link_context) - return f"**{content}**" if content else "" - - if tag_name in ["em", "i"]: - content = MarkdownConverter._get_inner_text(element, base_url, link_context) - return f"_{content}_" if content else "" - - if tag_name == "code": - # If inside pre, handled by pre. If inline: - if element.parent and element.parent.name == "pre": - return "" # Let pre handle it - text = element.get_text() - return f"`{text}`" - - if tag_name in ["span", "html"]: - # Just traverse children transparently - return MarkdownConverter._get_inner_text(element, base_url, link_context) - - if tag_name in [ - "div", - "section", - "article", - "main", - "header", - "footer", - "body", - ]: - # Treat as block structural elements. - # We add newlines to ensure separation of content blocks. - content = MarkdownConverter._get_inner_text(element, base_url, link_context) - if not content.strip(): - return "" - return f"\n{content}\n" - - # Default fallback - return MarkdownConverter._get_inner_text(element, base_url, link_context) - - @staticmethod - def _get_inner_text( - element: Tag, base_url: str, link_context: Optional[str] = None - ) -> str: - """Helper to recursively process children and join them.""" - results = [] - for child in element.children: - res = MarkdownConverter._process_element(child, base_url, link_context) - if res: - results.append(res) - - # Intelligent joining - # To fix "Paragraph**Bold**", we can insert spaces if the elements are inline. - # But this might break "word," -> "word ,". - # Simple heuristic: if we are in a block element context, spacing is handled by block logic. - # If we are in an inline context, we might need a space. - # A simpler fix for this specific tool: join with " " then remove excessive spaces. - - joined = "".join(results) - - # Quick fix for Paragraph**Bold** -> Paragraph **Bold** - # We can just leave it as is if the user didn't put a space in HTML "Paragraph Bold". - # But BS4 often strips distinct text nodes. - # Let's try joining with nothing, because usually HTML handles space via text nodes. - # If the failure 'Paragraph**Bold**' happened, it means the input HTML was "

Paragraph Bold

". - # "Paragraph " should be a NavigableString. - # Wait, the test input was "

Paragraph Bold

". - # "Paragraph " is a text node. "Bold" is a tag. - # If _process_element("Paragraph ") returns "Paragraph" (stripped), then we lose the space. - - return joined - - @staticmethod - def _process_list( - list_element: Tag, base_url: str, link_context: Optional[str] = None - ) -> str: - items = [] - is_ordered = (list_element.name or "").lower() == "ol" - - for i, child in enumerate(list_element.find_all("li", recursive=False)): - content = MarkdownConverter._get_inner_text( - child, base_url, link_context - ).strip() - # Handle nested lists - nested_list = child.find(["ul", "ol"]) - if nested_list: - # remove the text content that belongs to the nested list so we don't duplicate - # actually _get_inner_text handles recursion, so 'content' already includes the nested list markdown! - # This is tricky using simple recursion. - # Let's trust _get_inner_text returns formatting. - pass - - # Simple handling: replace internal newlines for sub-items alignment - content = content.replace("\n", "\n ") - - prefix = f"{i + 1}." if is_ordered else "-" - items.append(f"{prefix} {content}") - - return "\n".join(items) - - @staticmethod - def _process_table(table_element: Tag, base_url: str) -> str: - """ - Simple Markdown table converter. - Only handles standard thead/tbody structures nicely. - """ - rows = [] - - # 1. Headers - headers = [] - thead = table_element.find("thead") - if thead: - header_row = thead.find("tr") - if header_row: - headers = [ - th.get_text(strip=True) for th in header_row.find_all(["th", "td"]) - ] - - # Fallback: if no thead, check first tr - if not headers: - first_tr = table_element.find("tr") - if first_tr and first_tr.find("th"): - headers = [th.get_text(strip=True) for th in first_tr.find_all(["th"])] - - if headers: - rows.append(f"| {' | '.join(headers)} |") - rows.append(f"| {' | '.join(['---'] * len(headers))} |") - - # 2. Body - tbody = table_element.find("tbody") - row_container = tbody if tbody else table_element - - for tr in row_container.find_all("tr"): - # specific check to skip the header row if we already processed it from 'not headers' logic - if not headers and tr == table_element.find("tr") and tr.find("th"): - continue - - cols = [td.get_text(strip=True) for td in tr.find_all(["td", "th"])] - if cols: - # Match header length if possible - if headers and len(cols) != len(headers): - # Pad or truncate? Markdown tables are forgiving of mismatches usually, - # but let's just print what we have - pass - rows.append(f"| {' | '.join(cols)} |") - - return "\n".join(rows) +# ./src/web_scraper_toolkit/parsers/html_to_markdown.py +""" +HTML to Markdown Converter +========================== + +Utility for converting HTML content into clean, readable Markdown. +Specializes in handling tables, lists, and links properly. + +Usage: + md = MarkdownConverter.to_markdown(html_content, base_url="...") + +Key Features: + - Table conversion (ASCII/GD style). + - Link resolution (absolute paths). + - Noise removal (scripts/styles). +""" + +from bs4 import BeautifulSoup, NavigableString, Tag +import re +from typing import Any, Optional + + +class MarkdownConverter: + """ + Converts HTML content to clean Markdown, preserving structure for LLM consumption. + """ + + @staticmethod + def to_markdown(html_content: str, base_url: str = "") -> str: + """ + Main entry point. Converts valid HTML string to Markdown. + """ + if not html_content: + return "" + + soup = BeautifulSoup(html_content, "lxml") + + # 1. Cleanup Noise + # Removing semantic noise is critical for clean output + for tag in soup( + [ + "script", + "style", + "noscript", + "iframe", + "svg", + "meta", + "link", + "head", + "nav", + "footer", + "header", + "aside", + ] + ): + tag.decompose() + + # 2. Process Content + # We process the body if available, else the whole soup + root = soup.body if soup.body else soup + + markdown = MarkdownConverter._process_element(root, base_url).strip() + + # 3. Post-Processing: Collapse excessive newlines + # Nested blocks (div > div > p) often generate \n\n\n\n. We want max 2. + markdown = re.sub(r"\n{3,}", "\n\n", markdown) + + return markdown + + @staticmethod + def _attr_to_str(value: Any) -> Optional[str]: + """Normalize BeautifulSoup attribute values into strings.""" + if value is None: + return None + if isinstance(value, str): + return value + if isinstance(value, (list, tuple)): + for item in value: + if isinstance(item, str) and item: + return item + return None + return str(value) + + @staticmethod + def _is_block_element(tag: Any) -> bool: + """Checks if a tag is a block element.""" + if not isinstance(tag, Tag): + return False + tag_name = (tag.name or "").lower() + return tag_name in [ + "div", + "section", + "article", + "main", + "header", + "footer", + "ul", + "ol", + "li", + "h1", + "h2", + "h3", + "h4", + "h5", + "h6", + "p", + "table", + "blockquote", + "pre", + ] + + @staticmethod + def _has_block_children(element: Tag) -> bool: + """Checks if an element contains any direct block-level children.""" + for child in element.children: + if MarkdownConverter._is_block_element(child): + return True + # Also check recursively? No, usually direct children are enough if structure is good. + # But
..
is possible. + # Let's do a simple non-recursive check first, or shallow scan. + # Deep scan: + if isinstance(child, Tag) and child.find( + [ + "div", + "p", + "h1", + "h2", + "h3", + "h4", + "h5", + "h6", + "ul", + "ol", + "li", + "table", + ] + ): + return True + return False + + @staticmethod + def _process_element( + element: Any, base_url: str = "", link_context: Optional[str] = None + ) -> str: + if element is None: + return "" + + # TEXT + if isinstance(element, NavigableString): + raw_text = str(element) + # Normalize whitespace but KEEP leading/trailing spaces for separation + normalized_text = re.sub(r"\s+", " ", raw_text) + if not normalized_text: + return "" + + # If we are in a link context, we link the text directly + if link_context and normalized_text.strip(): + # Avoid linking whitespace only + return f"[{normalized_text}]({link_context})" + return normalized_text + + if not isinstance(element, Tag): + return "" + + # TAGS mapping + tag_name = (element.name or "").lower() + + # --- BLOCK ELEMENTS --- # + + if tag_name in ["h1", "h2", "h3", "h4", "h5", "h6"]: + level = int(tag_name[1]) + # Pass link_context down to inner text + content = MarkdownConverter._get_inner_text(element, base_url, link_context) + # Use fewer newlines to avoid stacking up to 4+ + return f"\n{'#' * level} {content}\n" + + if tag_name == "p": + content = MarkdownConverter._get_inner_text(element, base_url, link_context) + return f"\n{content}\n" if content else "" + + if tag_name == "br": + return "\n" + + if tag_name == "hr": + return "\n\n---\n\n" + + if tag_name in ["ul", "ol"]: + return f"\n{MarkdownConverter._process_list(element, base_url, link_context)}\n" + + if tag_name == "li": + # Handled by parent UL/OL usually, but if standalone: + content = MarkdownConverter._get_inner_text(element, base_url, link_context) + return f"- {content}\n" + + if tag_name == "blockquote": + content = MarkdownConverter._get_inner_text(element, base_url, link_context) + lines = content.split("\n") + quoted = "\n".join(f"> {line}" for line in lines if line.strip()) + return f"\n{quoted}\n" + + if tag_name == "pre": + # Code blocks inside links are rare and likely break markdown parsers anyway. + # We ignore link_context here for code blocks usually. + code_tag = element.find("code") + content = "" + if code_tag: + content = code_tag.get_text() # Raw text for code + else: + content = element.get_text() + return f"\n\n```\n{content}\n```\n\n" + + # --- TABLES --- # + if tag_name == "table": + return f"\n\n{MarkdownConverter._process_table(element, base_url)}\n\n" + + # --- INLINE ELEMENTS --- # + + if tag_name == "a": + href = MarkdownConverter._attr_to_str(element.get("href")) or "" + if not href or href.startswith("#") or href.startswith("javascript:"): + return MarkdownConverter._get_inner_text( + element, base_url, link_context + ) + + # Check if this link wraps BLOCK elements + if MarkdownConverter._has_block_children(element): + # DISTRIBUTE THE LINK DOWN + # We do NOT wrap this tag. We recurse. + return MarkdownConverter._get_inner_text( + element, base_url, link_context=href + ) + + # STANDARD INLINE LINK + text = MarkdownConverter._get_inner_text(element, base_url, link_context) + # If we are already inside a link context, we shouldn't nest links technically. + # Markdown doesn't support nested links. + # We defer to the outer link? Or inner? + # Usually inner link takes precedence in HTML, but markdown breaks. + # Let's assume inner takes precedence. + return f"[{text}]({href})" + + if tag_name == "img": + alt = MarkdownConverter._attr_to_str(element.get("alt")) or "Image" + src = MarkdownConverter._attr_to_str(element.get("src")) or "" + if not src: + return "" + img_md = f"![{alt}]({src})" + if link_context: + return f"[{img_md}]({link_context})" + return img_md + + if tag_name in ["strong", "b"]: + content = MarkdownConverter._get_inner_text(element, base_url, link_context) + return f"**{content}**" if content else "" + + if tag_name in ["em", "i"]: + content = MarkdownConverter._get_inner_text(element, base_url, link_context) + return f"_{content}_" if content else "" + + if tag_name == "code": + # If inside pre, handled by pre. If inline: + if element.parent and element.parent.name == "pre": + return "" # Let pre handle it + text = element.get_text() + return f"`{text}`" + + if tag_name in ["span", "html"]: + # Just traverse children transparently + return MarkdownConverter._get_inner_text(element, base_url, link_context) + + if tag_name in [ + "div", + "section", + "article", + "main", + "header", + "footer", + "body", + ]: + # Treat as block structural elements. + # We add newlines to ensure separation of content blocks. + content = MarkdownConverter._get_inner_text(element, base_url, link_context) + if not content.strip(): + return "" + return f"\n{content}\n" + + # Default fallback + return MarkdownConverter._get_inner_text(element, base_url, link_context) + + @staticmethod + def _get_inner_text( + element: Tag, base_url: str, link_context: Optional[str] = None + ) -> str: + """Helper to recursively process children and join them.""" + results = [] + for child in element.children: + res = MarkdownConverter._process_element(child, base_url, link_context) + if res: + results.append(res) + + # Intelligent joining + # To fix "Paragraph**Bold**", we can insert spaces if the elements are inline. + # But this might break "word," -> "word ,". + # Simple heuristic: if we are in a block element context, spacing is handled by block logic. + # If we are in an inline context, we might need a space. + # A simpler fix for this specific tool: join with " " then remove excessive spaces. + + joined = "".join(results) + + # Quick fix for Paragraph**Bold** -> Paragraph **Bold** + # We can just leave it as is if the user didn't put a space in HTML "Paragraph Bold". + # But BS4 often strips distinct text nodes. + # Let's try joining with nothing, because usually HTML handles space via text nodes. + # If the failure 'Paragraph**Bold**' happened, it means the input HTML was "

Paragraph Bold

". + # "Paragraph " should be a NavigableString. + # Wait, the test input was "

Paragraph Bold

". + # "Paragraph " is a text node. "Bold" is a tag. + # If _process_element("Paragraph ") returns "Paragraph" (stripped), then we lose the space. + + return joined + + @staticmethod + def _process_list( + list_element: Tag, base_url: str, link_context: Optional[str] = None + ) -> str: + items = [] + is_ordered = (list_element.name or "").lower() == "ol" + + for i, child in enumerate(list_element.find_all("li", recursive=False)): + content = MarkdownConverter._get_inner_text( + child, base_url, link_context + ).strip() + # Handle nested lists + nested_list = child.find(["ul", "ol"]) + if nested_list: + # remove the text content that belongs to the nested list so we don't duplicate + # actually _get_inner_text handles recursion, so 'content' already includes the nested list markdown! + # This is tricky using simple recursion. + # Let's trust _get_inner_text returns formatting. + pass + + # Simple handling: replace internal newlines for sub-items alignment + content = content.replace("\n", "\n ") + + prefix = f"{i + 1}." if is_ordered else "-" + items.append(f"{prefix} {content}") + + return "\n".join(items) + + @staticmethod + def _process_table(table_element: Tag, base_url: str) -> str: + """ + Simple Markdown table converter. + Only handles standard thead/tbody structures nicely. + """ + rows = [] + + # 1. Headers + headers = [] + thead = table_element.find("thead") + if thead: + header_row = thead.find("tr") + if header_row: + headers = [ + th.get_text(strip=True) for th in header_row.find_all(["th", "td"]) + ] + + # Fallback: if no thead, check first tr + if not headers: + first_tr = table_element.find("tr") + if first_tr and first_tr.find("th"): + headers = [th.get_text(strip=True) for th in first_tr.find_all(["th"])] + + if headers: + rows.append(f"| {' | '.join(headers)} |") + rows.append(f"| {' | '.join(['---'] * len(headers))} |") + + # 2. Body + tbody = table_element.find("tbody") + row_container = tbody if tbody else table_element + + for tr in row_container.find_all("tr"): + # specific check to skip the header row if we already processed it from 'not headers' logic + if not headers and tr == table_element.find("tr") and tr.find("th"): + continue + + cols = [td.get_text(strip=True) for td in tr.find_all(["td", "th"])] + if cols: + # Match header length if possible + if headers and len(cols) != len(headers): + # Pad or truncate? Markdown tables are forgiving of mismatches usually, + # but let's just print what we have + pass + rows.append(f"| {' | '.join(cols)} |") + + return "\n".join(rows) diff --git a/src/web_scraper_toolkit/parsers/utils.py b/src/web_scraper_toolkit/parsers/utils.py index 8a693de..9a8bb87 100644 --- a/src/web_scraper_toolkit/parsers/utils.py +++ b/src/web_scraper_toolkit/parsers/utils.py @@ -1,67 +1,67 @@ -# ./src/web_scraper_toolkit/parsers/utils.py -""" -Parser Utilities -================ - -Helper definitions for text extraction, URL normalization, and data cleanup. -Used by serp_parser and other parsing modules. - -Usage: - norm_url = normalize_url("/foo", "https://base.com") - -Key Functions: - - normalize_url: Resolves relative links. - - truncate_text: Safe string shortening. -""" - -from urllib.parse import urljoin, urlparse -from typing import Optional - - -def normalize_url(url: str, base_url: str = "") -> Optional[str]: - """ - Resolves relative URLs against a base URL. - Returns None if URL is invalid or empty. - Strips trailing slashes to canonicalize. - """ - if not url: - return None - - try: - # Strip whitespace - url = url.strip() - - # Handle javascript: links - if url.startswith("javascript:") or url.startswith("#"): - return None - - # Join - full_url = urljoin(base_url, url) - - # Validate scheme - parsed = urlparse(full_url) - if parsed.scheme not in ["http", "https"]: - return None - - # Canonicalize: Remove trailing slash if path > 1 (keep root /) - if parsed.path.endswith("/") and len(parsed.path) > 1: - full_url = full_url.rstrip("/") - - return full_url - - except Exception: - return None - - -def truncate_text(text: str, max_length: int = 100) -> str: - """ - Truncates text to max_length, appending '...' if truncated. - """ - if not text: - return "" - - clean_text = text.strip() - if len(clean_text) <= max_length: - return clean_text - - return clean_text[:max_length].rstrip() + "..." +# ./src/web_scraper_toolkit/parsers/utils.py +""" +Parser Utilities +================ + +Helper definitions for text extraction, URL normalization, and data cleanup. +Used by serp_parser and other parsing modules. + +Usage: + norm_url = normalize_url("/foo", "https://base.com") + +Key Functions: + - normalize_url: Resolves relative links. + - truncate_text: Safe string shortening. +""" + +from urllib.parse import urljoin, urlparse +from typing import Optional + + +def normalize_url(url: str, base_url: str = "") -> Optional[str]: + """ + Resolves relative URLs against a base URL. + Returns None if URL is invalid or empty. + Strips trailing slashes to canonicalize. + """ + if not url: + return None + + try: + # Strip whitespace + url = url.strip() + + # Handle javascript: links + if url.startswith("javascript:") or url.startswith("#"): + return None + + # Join + full_url = urljoin(base_url, url) + + # Validate scheme + parsed = urlparse(full_url) + if parsed.scheme not in ["http", "https"]: + return None + + # Canonicalize: Remove trailing slash if path > 1 (keep root /) + if parsed.path.endswith("/") and len(parsed.path) > 1: + full_url = full_url.rstrip("/") + + return full_url + + except Exception: + return None + + +def truncate_text(text: str, max_length: int = 100) -> str: + """ + Truncates text to max_length, appending '...' if truncated. + """ + if not text: + return "" + + clean_text = text.strip() + if len(clean_text) <= max_length: + return clean_text + + return clean_text[:max_length].rstrip() + "..." diff --git a/src/web_scraper_toolkit/server/__init__.py b/src/web_scraper_toolkit/server/__init__.py index 71cf36c..4a2950a 100644 --- a/src/web_scraper_toolkit/server/__init__.py +++ b/src/web_scraper_toolkit/server/__init__.py @@ -1,19 +1,19 @@ -# ./src/web_scraper_toolkit/server/__init__.py -""" -Server Module -============= - -This module exposes the FastMCP server instance for the Web Scraper Toolkit. -It allows the server to be imported and run programmatically or via specific runners. - -Usage: - from src.web_scraper_toolkit.server import mcp - mcp.run() - -Components: - - mcp: The FastMCP instance configured with scraping tools. -""" - -from .mcp_server import mcp - -__all__ = ["mcp"] +# ./src/web_scraper_toolkit/server/__init__.py +""" +Server Module +============= + +This module exposes the FastMCP server instance for the Web Scraper Toolkit. +It allows the server to be imported and run programmatically or via specific runners. + +Usage: + from src.web_scraper_toolkit.server import mcp + mcp.run() + +Components: + - mcp: The FastMCP instance configured with scraping tools. +""" + +from .mcp_server import mcp + +__all__ = ["mcp"] diff --git a/tests/test_basics.py b/tests/test_basics.py index 4d1232e..2742028 100644 --- a/tests/test_basics.py +++ b/tests/test_basics.py @@ -1,30 +1,30 @@ -import unittest - -# Ensure src is in path for testing without installing -# sys.path handled by run_tests.py - -from web_scraper_toolkit.parsers.utils import normalize_url, truncate_text -from web_scraper_toolkit.parsers import SerpParser - - -class TestUtils(unittest.TestCase): - def test_normalize_url(self): - self.assertEqual( - normalize_url("https://example.com/foo/"), "https://example.com/foo" - ) - self.assertEqual(normalize_url("example.com"), None) # Needs scheme - - def test_truncate_text(self): - text = "Hello World" - self.assertEqual(truncate_text(text, 5), "Hello...") - self.assertEqual(truncate_text(text, 50), "Hello World") - - -class TestSerpParser(unittest.TestCase): - def test_parse_empty(self): - results = SerpParser.parse_ddg_html("", "https://example.com") - self.assertEqual(results, []) - - -if __name__ == "__main__": - unittest.main() +import unittest + +# Ensure src is in path for testing without installing +# sys.path handled by run_tests.py + +from web_scraper_toolkit.parsers.utils import normalize_url, truncate_text +from web_scraper_toolkit.parsers import SerpParser + + +class TestUtils(unittest.TestCase): + def test_normalize_url(self): + self.assertEqual( + normalize_url("https://example.com/foo/"), "https://example.com/foo" + ) + self.assertEqual(normalize_url("example.com"), None) # Needs scheme + + def test_truncate_text(self): + text = "Hello World" + self.assertEqual(truncate_text(text, 5), "Hello...") + self.assertEqual(truncate_text(text, 50), "Hello World") + + +class TestSerpParser(unittest.TestCase): + def test_parse_empty(self): + results = SerpParser.parse_ddg_html("", "https://example.com") + self.assertEqual(results, []) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_cli.py b/tests/test_cli.py index f49b5e5..b9e71e6 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -1,116 +1,116 @@ -import unittest -from unittest.mock import patch, AsyncMock, Mock -import sys -import os - -# Ensure src is in path -# sys.path handled by run_tests.py - -from web_scraper_toolkit import cli - - -class TestCLI(unittest.TestCase): - def setUp(self): - # Silence the rich console - patcher = patch("web_scraper_toolkit.cli.console", Mock()) - self.mock_console = patcher.start() - self.addCleanup(patcher.stop) - - def test_argument_parser(self): - # Test default args - parser = cli.parse_arguments(["--url", "https://example.com"]) - self.assertEqual(parser.url, "https://example.com") - self.assertEqual(parser.format, "markdown") # Default - self.assertFalse(parser.headless) - - # Test complex args - args = [ - "--input", - "file.txt", - "--format", - "pdf", - "--headless", - "--workers", - "max", - ] - parser = cli.parse_arguments(args) - self.assertEqual(parser.input, "file.txt") - self.assertEqual(parser.format, "pdf") - self.assertTrue(parser.headless) - self.assertEqual(parser.workers, "max") - - @patch("web_scraper_toolkit.cli.WebCrawler") - def test_main_execution_single_url(self, MockCrawler): - # Mocking the Crawler - instance = MockCrawler.return_value - instance.run = AsyncMock(return_value="Success") - - # Mock sys.argv - test_args = ["web-scraper", "--url", "http://example.com", "--format", "json"] - with patch.object(sys, "argv", test_args): - cli.main() - - # Verify Config was created - MockCrawler.assert_called_once() - # Verify run was called - instance.run.assert_called_once() - _, kwargs = instance.run.call_args - self.assertEqual(kwargs["urls"], ["http://example.com"]) - self.assertEqual(kwargs["output_format"], "json") - - @patch("web_scraper_toolkit.cli.load_urls_from_source") - @patch("web_scraper_toolkit.cli.WebCrawler") - def test_main_execution_input_file(self, MockCrawler, mock_load): - # Mock file loader - mock_load.return_value = [ - "https://site-a.example.com", - "https://site-b.example.com", - ] - - instance = MockCrawler.return_value - instance.run = AsyncMock(return_value="Success") - - test_args = ["web-scraper", "--input", "list.txt"] - with patch.object(sys, "argv", test_args): - cli.main() - - # Verify loader called - mock_load.assert_called_with("list.txt") - # Verify run called with list - _, kwargs = instance.run.call_args - self.assertEqual( - kwargs["urls"], ["https://site-a.example.com", "https://site-b.example.com"] - ) - - @patch("web_scraper_toolkit.extract_sitemap_tree", new_callable=AsyncMock) - def test_site_tree_mode(self, mock_tree): - mock_tree.return_value = ["https://example.com/item1"] - - # Redirect output to tests_output for cleanliness - output_dir = os.path.abspath( - os.path.join(os.path.dirname(__file__), "../tests_output") - ) - os.makedirs(output_dir, exist_ok=True) - output_file = os.path.join(output_dir, "sitemap_tree.csv") - - # Override argv to include output_name - test_args = [ - "web-scraper", - "--input", - "https://example.com/sitemap.xml", - "--site-tree", - "--output-name", - output_file, - ] - - # We also need to mock print or file writing, as main() prints result or writes file - # But we just want to ensure it calls extract_sitemap_tree - with patch.object(sys, "argv", test_args): - # main() might exit? No, it just finishes. - cli.main() - - mock_tree.assert_called_with("https://example.com/sitemap.xml") - - -if __name__ == "__main__": - unittest.main() +import unittest +from unittest.mock import patch, AsyncMock, Mock +import sys +import os + +# Ensure src is in path +# sys.path handled by run_tests.py + +from web_scraper_toolkit import cli + + +class TestCLI(unittest.TestCase): + def setUp(self): + # Silence the rich console + patcher = patch("web_scraper_toolkit.cli.console", Mock()) + self.mock_console = patcher.start() + self.addCleanup(patcher.stop) + + def test_argument_parser(self): + # Test default args + parser = cli.parse_arguments(["--url", "https://example.com"]) + self.assertEqual(parser.url, "https://example.com") + self.assertEqual(parser.format, "markdown") # Default + self.assertFalse(parser.headless) + + # Test complex args + args = [ + "--input", + "file.txt", + "--format", + "pdf", + "--headless", + "--workers", + "max", + ] + parser = cli.parse_arguments(args) + self.assertEqual(parser.input, "file.txt") + self.assertEqual(parser.format, "pdf") + self.assertTrue(parser.headless) + self.assertEqual(parser.workers, "max") + + @patch("web_scraper_toolkit.cli.WebCrawler") + def test_main_execution_single_url(self, MockCrawler): + # Mocking the Crawler + instance = MockCrawler.return_value + instance.run = AsyncMock(return_value="Success") + + # Mock sys.argv + test_args = ["web-scraper", "--url", "http://example.com", "--format", "json"] + with patch.object(sys, "argv", test_args): + cli.main() + + # Verify Config was created + MockCrawler.assert_called_once() + # Verify run was called + instance.run.assert_called_once() + _, kwargs = instance.run.call_args + self.assertEqual(kwargs["urls"], ["http://example.com"]) + self.assertEqual(kwargs["output_format"], "json") + + @patch("web_scraper_toolkit.cli.load_urls_from_source") + @patch("web_scraper_toolkit.cli.WebCrawler") + def test_main_execution_input_file(self, MockCrawler, mock_load): + # Mock file loader + mock_load.return_value = [ + "https://site-a.example.com", + "https://site-b.example.com", + ] + + instance = MockCrawler.return_value + instance.run = AsyncMock(return_value="Success") + + test_args = ["web-scraper", "--input", "list.txt"] + with patch.object(sys, "argv", test_args): + cli.main() + + # Verify loader called + mock_load.assert_called_with("list.txt") + # Verify run called with list + _, kwargs = instance.run.call_args + self.assertEqual( + kwargs["urls"], ["https://site-a.example.com", "https://site-b.example.com"] + ) + + @patch("web_scraper_toolkit.extract_sitemap_tree", new_callable=AsyncMock) + def test_site_tree_mode(self, mock_tree): + mock_tree.return_value = ["https://example.com/item1"] + + # Redirect output to tests_output for cleanliness + output_dir = os.path.abspath( + os.path.join(os.path.dirname(__file__), "../tests_output") + ) + os.makedirs(output_dir, exist_ok=True) + output_file = os.path.join(output_dir, "sitemap_tree.csv") + + # Override argv to include output_name + test_args = [ + "web-scraper", + "--input", + "https://example.com/sitemap.xml", + "--site-tree", + "--output-name", + output_file, + ] + + # We also need to mock print or file writing, as main() prints result or writes file + # But we just want to ensure it calls extract_sitemap_tree + with patch.object(sys, "argv", test_args): + # main() might exit? No, it just finishes. + cli.main() + + mock_tree.assert_called_with("https://example.com/sitemap.xml") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_html_to_markdown.py b/tests/test_html_to_markdown.py index 3d60a19..db04587 100644 --- a/tests/test_html_to_markdown.py +++ b/tests/test_html_to_markdown.py @@ -1,99 +1,99 @@ -import unittest -import os -import re - -# sys.path handled by run_tests.py -from web_scraper_toolkit.parsers.html_to_markdown import MarkdownConverter - - -class TestMarkdownConverter(unittest.TestCase): - def setUp(self): - # Cache Scrub (Roy-Standard) - import shutil - - cache_path = os.path.join(os.path.dirname(__file__), "__pycache__") - if os.path.exists(cache_path): - try: - shutil.rmtree(cache_path) - except Exception: - pass - - def test_basic_conversion(self): - html = "

Title

Body text.

" - # We strip to avoid whitespace nitpicking on exact newlines - self.assertEqual( - MarkdownConverter.to_markdown(html).strip(), "# Title\n\nBody text." - ) - - def test_links(self): - html = 'Example' - expected = "[Example](https://example.com)" - self.assertEqual(MarkdownConverter.to_markdown(html).strip(), expected) - - def test_lists(self): - html = """ - - """ - # Expected: - # - Item 1 - # - Item 2 - md = MarkdownConverter.to_markdown(html).strip() - self.assertIn("- Item 1", md) - self.assertIn("- Item 2", md) - - def test_nested_structure(self): - html = "

Paragraph Bold

" - md = MarkdownConverter.to_markdown(html).strip() - self.assertEqual(md, "Paragraph **Bold**") - - def test_tables(self): - html = """ - - - - - - - -
Header 1Header 2
Row 1 Col 1Row 1 Col 2
- """ - md = MarkdownConverter.to_markdown(html).strip() - self.assertIn("| Header 1 | Header 2 |", md) - self.assertIn("| --- | --- |", md) - self.assertIn("| Row 1 Col 1 | Row 1 Col 2 |", md) - - def test_noise_removal(self): - html = "

Text

" - md = MarkdownConverter.to_markdown(html).strip() - self.assertEqual(md, "Text") - - def test_div_separation(self): - html = "
Line 1
Line 2
" - md = MarkdownConverter.to_markdown(html).strip() - cleaned = re.sub(r"\n+", "\n", md) - self.assertEqual(cleaned, "Line 1\nLine 2") - - def test_block_link_distribution(self): - # Case A: Complex link with Header and Text - html = '

Header

Content

' - md = MarkdownConverter.to_markdown(html).strip() - - # We expect distribution: - # ### [Header](/test) - # [Content](/test) - - self.assertIn("### [Header](/test)", md) - self.assertIn("[Content](/test)", md) - self.assertFalse("[ ### Header" in md) # Should NOT wrap the block markers - - # Case B: Image inside link - html = 'Alt' - md = MarkdownConverter.to_markdown(html).strip() - self.assertEqual(md, "[![Alt](pic.jpg)](/img)") - - -if __name__ == "__main__": - unittest.main() +import unittest +import os +import re + +# sys.path handled by run_tests.py +from web_scraper_toolkit.parsers.html_to_markdown import MarkdownConverter + + +class TestMarkdownConverter(unittest.TestCase): + def setUp(self): + # Cache Scrub (Roy-Standard) + import shutil + + cache_path = os.path.join(os.path.dirname(__file__), "__pycache__") + if os.path.exists(cache_path): + try: + shutil.rmtree(cache_path) + except Exception: + pass + + def test_basic_conversion(self): + html = "

Title

Body text.

" + # We strip to avoid whitespace nitpicking on exact newlines + self.assertEqual( + MarkdownConverter.to_markdown(html).strip(), "# Title\n\nBody text." + ) + + def test_links(self): + html = 'Example' + expected = "[Example](https://example.com)" + self.assertEqual(MarkdownConverter.to_markdown(html).strip(), expected) + + def test_lists(self): + html = """ + + """ + # Expected: + # - Item 1 + # - Item 2 + md = MarkdownConverter.to_markdown(html).strip() + self.assertIn("- Item 1", md) + self.assertIn("- Item 2", md) + + def test_nested_structure(self): + html = "

Paragraph Bold

" + md = MarkdownConverter.to_markdown(html).strip() + self.assertEqual(md, "Paragraph **Bold**") + + def test_tables(self): + html = """ + + + + + + + +
Header 1Header 2
Row 1 Col 1Row 1 Col 2
+ """ + md = MarkdownConverter.to_markdown(html).strip() + self.assertIn("| Header 1 | Header 2 |", md) + self.assertIn("| --- | --- |", md) + self.assertIn("| Row 1 Col 1 | Row 1 Col 2 |", md) + + def test_noise_removal(self): + html = "

Text

" + md = MarkdownConverter.to_markdown(html).strip() + self.assertEqual(md, "Text") + + def test_div_separation(self): + html = "
Line 1
Line 2
" + md = MarkdownConverter.to_markdown(html).strip() + cleaned = re.sub(r"\n+", "\n", md) + self.assertEqual(cleaned, "Line 1\nLine 2") + + def test_block_link_distribution(self): + # Case A: Complex link with Header and Text + html = '

Header

Content

' + md = MarkdownConverter.to_markdown(html).strip() + + # We expect distribution: + # ### [Header](/test) + # [Content](/test) + + self.assertIn("### [Header](/test)", md) + self.assertIn("[Content](/test)", md) + self.assertFalse("[ ### Header" in md) # Should NOT wrap the block markers + + # Case B: Image inside link + html = 'Alt' + md = MarkdownConverter.to_markdown(html).strip() + self.assertEqual(md, "[![Alt](pic.jpg)](/img)") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_multimodal.py b/tests/test_multimodal.py index 5c7aff8..b3a2c25 100644 --- a/tests/test_multimodal.py +++ b/tests/test_multimodal.py @@ -1,115 +1,115 @@ -import unittest -import os -import shutil -from unittest.mock import patch, AsyncMock - -# Adjust path to import from src - -# sys.path handled by run_tests.py - -from web_scraper_toolkit.parsers.scraping_tools import ( - extract_metadata, - capture_screenshot, - save_as_pdf, -) - - -class TestMultimodal(unittest.TestCase): - def setUp(self): - # Cache Scrub (Roy-Standard) - import shutil - - cache_path = os.path.join(os.path.dirname(__file__), "__pycache__") - if os.path.exists(cache_path): - try: - shutil.rmtree(cache_path) - except Exception: - pass - - self.output_dir = "test_outputs" - os.makedirs(self.output_dir, exist_ok=True) - - def tearDown(self): - if os.path.exists(self.output_dir): - shutil.rmtree(self.output_dir) - - def test_extract_metadata(self): - # We'll mock the internal _arun_extract_metadata to avoid network calls - # But we want to test the parsing logic. - # So we mock PlaywrightManager.smart_fetch to return a known HTML string. - - html_content = """ - - - Test Page - - - - - - - """ - - with patch( - "web_scraper_toolkit.browser.playwright_handler.PlaywrightManager" - ) as MockManager: - instance = MockManager.return_value - instance.start = AsyncMock() - instance.stop = AsyncMock() - # return content, url, status - instance.smart_fetch = AsyncMock( - return_value=(html_content, "https://example.com", 200) - ) - - output = extract_metadata("https://example.com") - - self.assertIn("METADATA REPORT", output) - self.assertIn("Test Corp", output) - self.assertIn("OpenGraph Title", output) - self.assertIn("Meta Description", output) - - @patch("web_scraper_toolkit.browser.playwright_handler.PlaywrightManager") - def test_capture_screenshot(self, MockManager): - instance = MockManager.return_value - instance.start = AsyncMock() - instance.stop = AsyncMock() - instance.capture_screenshot = AsyncMock(return_value=True) - - path = os.path.join(self.output_dir, "test.png") - result = capture_screenshot("https://example.com", path) - - self.assertIn(f"Screenshot saved to {path}", result) - instance.capture_screenshot.assert_called_once() - args, kwargs = instance.capture_screenshot.call_args - self.assertEqual(args[0], "https://example.com") - self.assertEqual(args[1], path) - self.assertTrue(kwargs.get("full_page", True)) - - @patch("web_scraper_toolkit.browser.playwright_handler.PlaywrightManager") - def test_save_as_pdf(self, MockManager): - instance = MockManager.return_value - instance.start = AsyncMock() - instance.stop = AsyncMock() - instance.save_pdf = AsyncMock(return_value=True) - - path = os.path.join(self.output_dir, "test.pdf") - # PDF forces headless in the code - result = save_as_pdf("https://example.com", path) - - self.assertIn(f"PDF saved to {path}", result) - instance.save_pdf.assert_called_once() - - # Verify config update was passed to Manager constructor - # We can't easily check constructor args on the mocked class instance creation - # unless we check the CALL context of MockManager. - # But for unit test, just verifying save_pdf is called is sufficient. - - -if __name__ == "__main__": - unittest.main() +import unittest +import os +import shutil +from unittest.mock import patch, AsyncMock + +# Adjust path to import from src + +# sys.path handled by run_tests.py + +from web_scraper_toolkit.parsers.scraping_tools import ( + extract_metadata, + capture_screenshot, + save_as_pdf, +) + + +class TestMultimodal(unittest.TestCase): + def setUp(self): + # Cache Scrub (Roy-Standard) + import shutil + + cache_path = os.path.join(os.path.dirname(__file__), "__pycache__") + if os.path.exists(cache_path): + try: + shutil.rmtree(cache_path) + except Exception: + pass + + self.output_dir = "test_outputs" + os.makedirs(self.output_dir, exist_ok=True) + + def tearDown(self): + if os.path.exists(self.output_dir): + shutil.rmtree(self.output_dir) + + def test_extract_metadata(self): + # We'll mock the internal _arun_extract_metadata to avoid network calls + # But we want to test the parsing logic. + # So we mock PlaywrightManager.smart_fetch to return a known HTML string. + + html_content = """ + + + Test Page + + + + + + + """ + + with patch( + "web_scraper_toolkit.browser.playwright_handler.PlaywrightManager" + ) as MockManager: + instance = MockManager.return_value + instance.start = AsyncMock() + instance.stop = AsyncMock() + # return content, url, status + instance.smart_fetch = AsyncMock( + return_value=(html_content, "https://example.com", 200) + ) + + output = extract_metadata("https://example.com") + + self.assertIn("METADATA REPORT", output) + self.assertIn("Test Corp", output) + self.assertIn("OpenGraph Title", output) + self.assertIn("Meta Description", output) + + @patch("web_scraper_toolkit.browser.playwright_handler.PlaywrightManager") + def test_capture_screenshot(self, MockManager): + instance = MockManager.return_value + instance.start = AsyncMock() + instance.stop = AsyncMock() + instance.capture_screenshot = AsyncMock(return_value=True) + + path = os.path.join(self.output_dir, "test.png") + result = capture_screenshot("https://example.com", path) + + self.assertIn(f"Screenshot saved to {path}", result) + instance.capture_screenshot.assert_called_once() + args, kwargs = instance.capture_screenshot.call_args + self.assertEqual(args[0], "https://example.com") + self.assertEqual(args[1], path) + self.assertTrue(kwargs.get("full_page", True)) + + @patch("web_scraper_toolkit.browser.playwright_handler.PlaywrightManager") + def test_save_as_pdf(self, MockManager): + instance = MockManager.return_value + instance.start = AsyncMock() + instance.stop = AsyncMock() + instance.save_pdf = AsyncMock(return_value=True) + + path = os.path.join(self.output_dir, "test.pdf") + # PDF forces headless in the code + result = save_as_pdf("https://example.com", path) + + self.assertIn(f"PDF saved to {path}", result) + instance.save_pdf.assert_called_once() + + # Verify config update was passed to Manager constructor + # We can't easily check constructor args on the mocked class instance creation + # unless we check the CALL context of MockManager. + # But for unit test, just verifying save_pdf is called is sufficient. + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_scraping_tools.py b/tests/test_scraping_tools.py index 18fb751..7d95916 100644 --- a/tests/test_scraping_tools.py +++ b/tests/test_scraping_tools.py @@ -1,116 +1,116 @@ -import unittest -from unittest.mock import patch, AsyncMock -import os -import asyncio - -# Ensure src is in path -# sys.path handled by run_tests.py - -# We need to test specific logic inside content module. -# Since the logic is inside _arun_scrape, we can mock PlaywrightManager -# to return specific HTML content and verify the extraction output. - -from web_scraper_toolkit.parsers.content import _arun_scrape -from web_scraper_toolkit.parsers.scraping_tools import get_sitemap_urls - - -class TestScrapingTools(unittest.TestCase): - def setUp(self): - # Cache Scrub (Roy-Standard) - import shutil - - cache_path = os.path.join(os.path.dirname(__file__), "__pycache__") - if os.path.exists(cache_path): - try: - shutil.rmtree(cache_path) - except Exception: - pass - - self.loop = asyncio.new_event_loop() - asyncio.set_event_loop(self.loop) - - def tearDown(self): - self.loop.close() - - @patch("web_scraper_toolkit.browser.playwright_handler.PlaywrightManager") - def test_arun_scrape_extraction(self, MockPlaywrightManager): - # Setup Mock - mock_manager = MockPlaywrightManager.return_value - mock_manager.start = AsyncMock() - mock_manager.stop = AsyncMock() - mock_manager.get_new_page = AsyncMock() - mock_manager.fetch_page_content = AsyncMock() - - # Mock Page/Context - mock_page = AsyncMock() - mock_context = AsyncMock() - mock_manager.get_new_page.return_value = (mock_page, mock_context) - - # FIX: We now use smart_fetch in the tools, so we must mock it instead of/in addition to fetch_page_content - mock_manager.smart_fetch = AsyncMock() - - # Mock Content - html_content = """ - - Test Company - -

Welcome to Test Company.

-

Our CEO is John Doe.

-

Contact us at contact@example.com

- - - """ - # mock_manager.smart_fetch.return_value = (html_content, "https://example.com", 200) - mock_manager.smart_fetch.return_value = ( - html_content, - "https://example.com", - 200, - ) - - # Run extraction - result = self.loop.run_until_complete(_arun_scrape("https://example.com")) - - # Verify assertions - self.assertIn("TITLE: Test Company", result) - self.assertIn("John Doe", result) # Leadership extraction - self.assertIn("contact@example.com", result) # Email extraction - - @patch( - "web_scraper_toolkit.parsers.sitemap.tools.extract_sitemap_tree", - new_callable=AsyncMock, - ) - @patch( - "web_scraper_toolkit.parsers.sitemap.tools.peek_sitemap_index", - new_callable=AsyncMock, - ) - @patch( - "web_scraper_toolkit.parsers.sitemap.tools.find_sitemap_urls", - new_callable=AsyncMock, - ) - def test_sitemap_extraction(self, mock_find, mock_peek, mock_extract): - # Mock finding a sitemap - mock_find.return_value = ["https://example.com/sitemap.xml"] - - # Mock extracting URLs from it - mock_extract.return_value = [ - "https://example.com/about", - "https://example.com/contact", - "https://example.com/product/123", - ] - - # Mock peeking - return what correct extract returns - mock_peek.return_value = {"type": "urlset", "urls": mock_extract.return_value} - - result = get_sitemap_urls("https://example.com") - - # The function sums up found URLs. - # Total unique = 3. - self.assertIn("Contains 3 relevant URLs", result) - self.assertIn("https://example.com/about", result) - self.assertIn("https://example.com/contact", result) - # Products are typically included unless they are assets - self.assertIn("https://example.com/product/123", result) - - -if __name__ == "__main__": - unittest.main() +import unittest +from unittest.mock import patch, AsyncMock +import os +import asyncio + +# Ensure src is in path +# sys.path handled by run_tests.py + +# We need to test specific logic inside content module. +# Since the logic is inside _arun_scrape, we can mock PlaywrightManager +# to return specific HTML content and verify the extraction output. + +from web_scraper_toolkit.parsers.content import _arun_scrape +from web_scraper_toolkit.parsers.scraping_tools import get_sitemap_urls + + +class TestScrapingTools(unittest.TestCase): + def setUp(self): + # Cache Scrub (Roy-Standard) + import shutil + + cache_path = os.path.join(os.path.dirname(__file__), "__pycache__") + if os.path.exists(cache_path): + try: + shutil.rmtree(cache_path) + except Exception: + pass + + self.loop = asyncio.new_event_loop() + asyncio.set_event_loop(self.loop) + + def tearDown(self): + self.loop.close() + + @patch("web_scraper_toolkit.browser.playwright_handler.PlaywrightManager") + def test_arun_scrape_extraction(self, MockPlaywrightManager): + # Setup Mock + mock_manager = MockPlaywrightManager.return_value + mock_manager.start = AsyncMock() + mock_manager.stop = AsyncMock() + mock_manager.get_new_page = AsyncMock() + mock_manager.fetch_page_content = AsyncMock() + + # Mock Page/Context + mock_page = AsyncMock() + mock_context = AsyncMock() + mock_manager.get_new_page.return_value = (mock_page, mock_context) + + # FIX: We now use smart_fetch in the tools, so we must mock it instead of/in addition to fetch_page_content + mock_manager.smart_fetch = AsyncMock() + + # Mock Content + html_content = """ + + Test Company + +

Welcome to Test Company.

+

Our CEO is John Doe.

+

Contact us at contact@example.com

+ + + """ + # mock_manager.smart_fetch.return_value = (html_content, "https://example.com", 200) + mock_manager.smart_fetch.return_value = ( + html_content, + "https://example.com", + 200, + ) + + # Run extraction + result = self.loop.run_until_complete(_arun_scrape("https://example.com")) + + # Verify assertions + self.assertIn("TITLE: Test Company", result) + self.assertIn("John Doe", result) # Leadership extraction + self.assertIn("contact@example.com", result) # Email extraction + + @patch( + "web_scraper_toolkit.parsers.sitemap.tools.extract_sitemap_tree", + new_callable=AsyncMock, + ) + @patch( + "web_scraper_toolkit.parsers.sitemap.tools.peek_sitemap_index", + new_callable=AsyncMock, + ) + @patch( + "web_scraper_toolkit.parsers.sitemap.tools.find_sitemap_urls", + new_callable=AsyncMock, + ) + def test_sitemap_extraction(self, mock_find, mock_peek, mock_extract): + # Mock finding a sitemap + mock_find.return_value = ["https://example.com/sitemap.xml"] + + # Mock extracting URLs from it + mock_extract.return_value = [ + "https://example.com/about", + "https://example.com/contact", + "https://example.com/product/123", + ] + + # Mock peeking - return what correct extract returns + mock_peek.return_value = {"type": "urlset", "urls": mock_extract.return_value} + + result = get_sitemap_urls("https://example.com") + + # The function sums up found URLs. + # Total unique = 3. + self.assertIn("Contains 3 relevant URLs", result) + self.assertIn("https://example.com/about", result) + self.assertIn("https://example.com/contact", result) + # Products are typically included unless they are assets + self.assertIn("https://example.com/product/123", result) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_serp_parser.py b/tests/test_serp_parser.py index 41a8fd2..d4ef620 100644 --- a/tests/test_serp_parser.py +++ b/tests/test_serp_parser.py @@ -1,64 +1,64 @@ -import unittest - -# Ensure src is in path -# sys.path handled by run_tests.py - -from web_scraper_toolkit.parsers import SerpParser - - -class TestSerpParser(unittest.TestCase): - def test_parse_ddg_html_sample(self): - # Simulated DuckDuckGo HTML structure - html = """ - - -
-

- Example Domain -

-
This is a sample snippet for the example domain.
-
-
-

- Google Result Title -

-
Another snippet here.
-
- - - """ - results = SerpParser.parse_ddg_html(html, "https://duckduckgo.com") - self.assertEqual(len(results), 2) - self.assertEqual(results[0]["url"], "https://example.com/foo") - self.assertEqual(results[0]["title"], "Example Domain") - self.assertIn("sample snippet", results[0]["snippet"]) - - def test_parse_google_style_sample(self): - # Simulated Google HTML structure (div.g) - html = """ - - -
-
- -

Google Result Title

-
-
-
Google snippet text...
-
- - - """ - results = SerpParser.parse_google_direct_links_style(html, "https://google.com") - self.assertEqual(len(results), 1) - self.assertEqual(results[0]["url"], "https://example.com/result") - self.assertEqual(results[0]["title"], "Google Result Title") - self.assertEqual(results[0]["snippet"], "Google snippet text...") - - def test_empty_content(self): - results = SerpParser.parse_ddg_html("", "https://duckduckgo.com") - self.assertEqual(results, []) - - -if __name__ == "__main__": - unittest.main() +import unittest + +# Ensure src is in path +# sys.path handled by run_tests.py + +from web_scraper_toolkit.parsers import SerpParser + + +class TestSerpParser(unittest.TestCase): + def test_parse_ddg_html_sample(self): + # Simulated DuckDuckGo HTML structure + html = """ + + +
+

+ Example Domain +

+
This is a sample snippet for the example domain.
+
+
+

+ Google Result Title +

+
Another snippet here.
+
+ + + """ + results = SerpParser.parse_ddg_html(html, "https://duckduckgo.com") + self.assertEqual(len(results), 2) + self.assertEqual(results[0]["url"], "https://example.com/foo") + self.assertEqual(results[0]["title"], "Example Domain") + self.assertIn("sample snippet", results[0]["snippet"]) + + def test_parse_google_style_sample(self): + # Simulated Google HTML structure (div.g) + html = """ + + +
+
+ +

Google Result Title

+
+
+
Google snippet text...
+
+ + + """ + results = SerpParser.parse_google_direct_links_style(html, "https://google.com") + self.assertEqual(len(results), 1) + self.assertEqual(results[0]["url"], "https://example.com/result") + self.assertEqual(results[0]["title"], "Google Result Title") + self.assertEqual(results[0]["snippet"], "Google snippet text...") + + def test_empty_content(self): + results = SerpParser.parse_ddg_html("", "https://duckduckgo.com") + self.assertEqual(results, []) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/verify_imports.py b/tests/verify_imports.py index 5727aa8..eceae68 100644 --- a/tests/verify_imports.py +++ b/tests/verify_imports.py @@ -1,42 +1,42 @@ -import sys -import os -import pkgutil -import importlib -import traceback - -# Add src to path so we can import as if installed -sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../src"))) - - -def verify_package(package_name): - print(f"🔍 Verifying package: {package_name}") - try: - root_pkg = importlib.import_module(package_name) - print(f"✅ Root import successful: {package_name}") - except Exception as e: - print(f"❌ Root import failed: {e}") - traceback.print_exc() - return - - # Walk through all submodules - path = root_pkg.__path__ - prefix = package_name + "." - - for _, name, ispkg in pkgutil.walk_packages(path, prefix): - print(f" Checking {name}...", end=" ") - try: - importlib.import_module(name) - print("✅ OK") - except Exception as e: - print("❌ FAILED") - print(f" Error: {e}") - # traceback.print_exc() - - -if __name__ == "__main__": - # verify_package("web_scraper_toolkit") - try: - importlib.import_module("web_scraper_toolkit.parsers.scraping_tools") - print("✅ scraping_tools OK") - except Exception: - traceback.print_exc() +import sys +import os +import pkgutil +import importlib +import traceback + +# Add src to path so we can import as if installed +sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../src"))) + + +def verify_package(package_name): + print(f"🔍 Verifying package: {package_name}") + try: + root_pkg = importlib.import_module(package_name) + print(f"✅ Root import successful: {package_name}") + except Exception as e: + print(f"❌ Root import failed: {e}") + traceback.print_exc() + return + + # Walk through all submodules + path = root_pkg.__path__ + prefix = package_name + "." + + for _, name, ispkg in pkgutil.walk_packages(path, prefix): + print(f" Checking {name}...", end=" ") + try: + importlib.import_module(name) + print("✅ OK") + except Exception as e: + print("❌ FAILED") + print(f" Error: {e}") + # traceback.print_exc() + + +if __name__ == "__main__": + # verify_package("web_scraper_toolkit") + try: + importlib.import_module("web_scraper_toolkit.parsers.scraping_tools") + print("✅ scraping_tools OK") + except Exception: + traceback.print_exc()