diff --git a/common/chunkers/structured.py b/common/chunkers/structured.py index aa3ec26..0980140 100644 --- a/common/chunkers/structured.py +++ b/common/chunkers/structured.py @@ -140,15 +140,28 @@ class Element: # pymupdf4llm artifacts: # • "==> picture [WxH] intentionally omitted <==" — image dropped (skip line) # • "----- Start of picture text -----" / "----- End of picture text -----" -# bracket OCR'd content inside an image; we fold the body into the figure -# so chart-internal labels stay with the image chunk. +# (markdown) or ```` (HTML-comment form) +# bracket pymupdf4llm picture-text inside a figure; we fold the body into +# the figure so chart-internal labels stay with the image chunk. _MD_PICTURE_OMITTED = re.compile(r"^\s*\*+\s*==>\s*picture\b.*intentionally omitted\s*<==\s*\*+.*$", re.IGNORECASE) -_MD_PICTURE_TEXT_START = re.compile(r"^\s*\*+\s*-+\s*Start of picture text\s*-+\s*\*+\s*()?\s*$", re.IGNORECASE) -_MD_PICTURE_TEXT_END = re.compile(r"^\s*\*+\s*-+\s*End of picture text\s*-+\s*\*+\s*()?\s*$", re.IGNORECASE) +_MD_PICTURE_TEXT_START = re.compile( + r"^\s*(?:\*+\s*-+\s*Start of picture text\s*-+\s*\*+|" + r")\s*(?:)?\s*$", + re.IGNORECASE, +) +_MD_PICTURE_TEXT_END = re.compile( + r"^\s*(?:\*+\s*-+\s*End of picture text\s*-+\s*\*+|" + r")\s*(?:)?\s*$", + re.IGNORECASE, +) # Inline variant of the End marker: the picture-text body can arrive as a # single
-joined line with the marker on its tail, so it is not always # line-anchored. Searched anywhere in a line to terminate the block. -_MD_PICTURE_TEXT_END_INLINE = re.compile(r"\*+\s*-+\s*End of picture text\s*-+\s*\*+\s*(?:)?", re.IGNORECASE) +_MD_PICTURE_TEXT_END_INLINE = re.compile( + r"(?:\*+\s*-+\s*End of picture text\s*-+\s*\*+|" + r")\s*(?:)?", + re.IGNORECASE, +) def _flush_prose(buf: List[str], heading: Optional[str], page: Optional[int], out: List[Element]) -> None: @@ -312,18 +325,14 @@ def markdown_to_elements(md: str, page: Optional[int] = None) -> List[Element]: i += 1 continue - # 5b. Other HTML comments (chunk markers etc.) — skip. - if _MD_HTML_COMMENT.match(line): - i += 1 - continue - # 5c. pymupdf4llm "==> picture ... intentionally omitted <==" — drop. if _MD_PICTURE_OMITTED.match(line): i += 1 continue - # 5d. pymupdf4llm picture-text block: ----- Start ... End of picture - # text ----- wraps OCR'd content (chart axis labels, legends). + # 5d. pymupdf4llm picture-text block (markdown dash markers OR HTML + # comments). Must run before the generic HTML-comment skip so + # ```` is not dropped. # Fold the body into the immediately preceding figure when # present so chart-internal text travels with the image. if _MD_PICTURE_TEXT_START.match(line): @@ -361,6 +370,12 @@ def markdown_to_elements(md: str, page: Optional[int] = None) -> List[Element]: out.append(Element(kind="figure", text=body, heading=heading, page=page)) continue + # 5b. Other HTML comments (chunk markers etc.) — skip. + # Picture-text comments are handled above. + if _MD_HTML_COMMENT.match(line): + i += 1 + continue + # 6. Blank line — flush current prose paragraph. if not stripped: _flush_prose(prose_buf, heading, page, out) diff --git a/common/embeddings/tigergraph_embedding_store.py b/common/embeddings/tigergraph_embedding_store.py index bfd1978..47dbed8 100644 --- a/common/embeddings/tigergraph_embedding_store.py +++ b/common/embeddings/tigergraph_embedding_store.py @@ -566,13 +566,14 @@ def retrieve_similar_with_score(self, query_embedding, top_k=10, similarity_thre logger.info(f"Fetch {top_k} similar entries from {vertex_types} with filter {filter_expr}") start_time = time() + # GSQL declares STRING expr=""; pyTigerGraph rejects null. verts = self.conn.runInstalledQuery( "get_topk_similar", params={ "vertex_types": vertex_types, "query_vector": query_embedding, "top_k": top_k*2, - "expr": filter_expr, + "expr": filter_expr or "", } ) end_time = time() diff --git a/common/utils/image_data_extractor.py b/common/utils/image_data_extractor.py index 6be929c..51721f7 100644 --- a/common/utils/image_data_extractor.py +++ b/common/utils/image_data_extractor.py @@ -119,7 +119,14 @@ def describe_image_with_llm(file_path): "stacked bar with a time-period axis), TRANSCRIBE every " "(period, value) pair you can read in the format " "`period: value; period: value; …` — do not summarize " - "the trend in place of the values; (3) the entities, " + "the trend in place of the values. For categorical " + "bar/rank charts (categories such as regions, " + "companies, or age groups on one axis and a numeric " + "unit on the other), TRANSCRIBE every visible " + "category with its value as " + "`category: value; category: value; …` including the " + "unit — never list categories without their numbers; " + "(3) the entities, " "relationships, or process steps in any diagram or " "flowchart; (4) any logo or branding mark, identified by " "name. Do NOT describe layout, background color, " diff --git a/common/utils/text_extractors.py b/common/utils/text_extractors.py index 09399a9..41a441e 100644 --- a/common/utils/text_extractors.py +++ b/common/utils/text_extractors.py @@ -41,8 +41,11 @@ # which bloat tokens 3-5x and confuse retrieval embeddings. The CJK # Unicode ranges below cover CJK Unified Ideographs (U+4E00-U+9FFF), # Hiragana / Katakana / CJK Symbols (U+3000-U+30FF), and full-width -# / half-width forms (U+FF00-U+FFEF). -_CJK_CHAR_CLASS = r"[ -鿿＀-￯]" +# / half-width forms (U+FF00-U+FFEF) excluding fullwidth digits +# (U+FF10-U+FF19). Collapsing digit runs would glue distinct chart +# values such as 767 and 808 into ``767808``. +# One CJK char excluding fullwidth digits (U+FF10-U+FF19). +_CJK_CHAR_CLASS = r"(?:[ -鿿]|[＀-/]|[:-￯])" _VERTICAL_BOLD_CJK = re.compile( rf"(?:\*\*{_CJK_CHAR_CLASS}\*\*(?:)){{2,}}\*\*{_CJK_CHAR_CLASS}\*\*" ) @@ -60,6 +63,20 @@ _TABLE_LINE_RE = re.compile(r"^\s*\|") _BR_TAG_RE = re.compile(r"", re.IGNORECASE) +# pymupdf4llm picture-text blocks (HTML-comment or markdown-dash markers). +# These are figure-associated text from pymupdf4llm, not a separate OCR engine. +_PICTURE_TEXT_BLOCK_RE = re.compile( + r"(?:|\*{0,3}\s*-+\s*Start of picture text\s*-+\s*\*{0,3})" + r"(.*?)" + r"(?:|\*{0,3}\s*-+\s*End of picture text\s*-+\s*\*{0,3})", + re.IGNORECASE | re.DOTALL, +) + +# Adjacent comma-grouped numbers glued with no separator, e.g. ``1,5461,518``. +_GLUED_COMMA_NUMBERS_RE = re.compile( + r"(? list[dict]: def _strip_br_in_table_rows(text: str) -> str: - """Remove ``
`` tags inside markdown table rows. + """Replace ``
`` tags inside markdown table rows with spaces. - Rationale documented at _TABLE_LINE_RE. + Using a space (not empty string) keeps stacked chart values distinct + — ``|767
808|`` becomes ``|767 808|``, never ``|767808|``. """ out: list[str] = [] for line in text.split("\n"): if _TABLE_LINE_RE.match(line): line = _BR_TAG_RE.sub(" ", line) + # Collapse runs of whitespace left by consecutive
tags. + line = re.sub(r"[ \t]{2,}", " ", line) out.append(line) return "\n".join(out) +def _split_glued_comma_numbers(text: str) -> str: + """Insert a space between adjacent comma-grouped numbers. + + pymupdf4llm / chart extraction sometimes emits ``1,5461,518`` instead of + ``1,546 1,518``. Repeat until stable for longer glued runs. + """ + prev = None + while prev != text: + prev = text + text = _GLUED_COMMA_NUMBERS_RE.sub(r"\1 \2", text) + return text + + +def _normalize_picture_text_blocks(text: str) -> str: + """Normalize pymupdf4llm picture-text blocks for chunking + retrieval. + + - Rewrite HTML-comment markers to the markdown form StructuredChunker + already recognizes. + - Turn in-block ``
`` into newlines (not empty joins) so values like + ``767`` and ``808`` stay separable. + - Split glued comma-numbers inside the block. + """ + + def _rewrite(match: re.Match) -> str: + body = match.group(1) or "" + body = _BR_TAG_RE.sub("\n", body) + body = _split_glued_comma_numbers(body) + # Trim excess blank lines inside the block. + body = re.sub(r"\n{3,}", "\n\n", body).strip("\n") + return ( + "***----- Start of picture text -----***\n" + f"{body}\n" + "***----- End of picture text -----***" + ) + + return _PICTURE_TEXT_BLOCK_RE.sub(_rewrite, text) + + +_PAGE_MARKER_RE = re.compile(r"") + + +def _recover_mojibake_pages( + file_path, + markdown: str, + graphname=None, + max_pages: int = 3, +) -> str: + """Recover table/chart text from PDF pages with broken ToUnicode CMaps. + + When glyph mapping fails, embedded text often keeps numbers but corrupts + labels. Drawn (non-embedded) figures also skip the normal image-describe + pass. For page sections that look both corrupted and table-like, render + the page and multimodal-transcribe it, then append the result next to the + original page body. Capped by ``max_pages`` to bound cost. + """ + if not markdown or max_pages <= 0: + return markdown + + # Collect page numbers whose section text looks corrupted. + parts = _PAGE_MARKER_RE.split(markdown) + # parts: [pre, pageNo, body, pageNo, body, ...] + bad_pages: list[int] = [] + if len(parts) >= 3: + for i in range(1, len(parts), 2): + try: + page_no = int(parts[i]) + except (TypeError, ValueError): + continue + body = parts[i + 1] if i + 1 < len(parts) else "" + findings = _detect_mojibake(body, source_hint=f"{file_path}:p{page_no}") + # Prefer pages that look like broken tables (pipe rows + mojibake). + pipe_rows = sum(1 for ln in body.splitlines() if ln.strip().startswith("|")) + if len(findings) >= 3 and pipe_rows >= 3: + bad_pages.append(page_no) + if not bad_pages: + return markdown + + try: + import pymupdf + from common.utils.image_data_extractor import ( + describe_image_with_llm, + should_extract_images, + ) + except Exception as e: # noqa: BLE001 + logger.warning("mojibake page recovery unavailable: %s", e) + return markdown + + if not should_extract_images(graphname): + return markdown + + recovered: dict[int, str] = {} + try: + doc = pymupdf.open(str(file_path)) + except Exception as e: # noqa: BLE001 + logger.warning("mojibake recovery: cannot open %s: %s", file_path, e) + return markdown + + try: + for page_no in bad_pages[:max_pages]: + idx = page_no - 1 + if idx < 0 or idx >= doc.page_count: + continue + try: + page = doc[idx] + # ~150 dpi — enough to read table cells without huge payloads. + pix = page.get_pixmap(matrix=pymupdf.Matrix(2.0, 2.0), alpha=False) + tmp = Path(tempfile.mkdtemp(prefix="mojibake_page_")) / f"p{page_no}.png" + pix.save(str(tmp)) + desc = describe_image_with_llm(str(tmp)) + try: + shutil.rmtree(tmp.parent, ignore_errors=True) + except Exception: + pass + if not desc or "decorative image" in desc.lower(): + continue + # Skip if the transcription itself looks glyph-broken. + if len(_detect_mojibake(desc)) >= 3: + continue + recovered[page_no] = desc.strip() + logger.info( + "mojibake page recovery: %s page %s recovered %s chars", + file_path, + page_no, + len(desc), + ) + except Exception as e: # noqa: BLE001 + logger.warning( + "mojibake page recovery failed for %s p%s: %s", + file_path, + page_no, + e, + ) + finally: + doc.close() + + if not recovered: + return markdown + + # Append recovered transcription under each page marker body. + out_parts: list[str] = [parts[0]] + for i in range(1, len(parts), 2): + page_no_s = parts[i] + body = parts[i + 1] if i + 1 < len(parts) else "" + out_parts.append(f"") + out_parts.append(body) + try: + page_no = int(page_no_s) + except (TypeError, ValueError): + continue + if page_no in recovered: + out_parts.append( + "\n\n\n" + + recovered[page_no] + + "\n" + ) + return "\n".join(out_parts) + + def _collapse_vertical_cjk(text: str) -> str: """Collapse pymupdf4llm's per-character vertical-CJK runs back into a single token. Bold runs ``**X**
**Y**
**Z**`` become ``**XYZ**``; @@ -158,11 +336,21 @@ def _clean_pdf_markdown(markdown: str, source_hint: str = "") -> str: with ``
`` separators and per-character bold markers. The run is collapsed back into a single token so embedding and retrieval see the intended word (e.g. ``**個別信用購入あっせん**``) rather than ten - fragments. + fragments. Fullwidth digits are excluded so chart values are not glued. + + 4. **Picture-text blocks** — figure text wrapped in + ```` (or the dash-marker form) by + pymupdf4llm is rewritten so StructuredChunker keeps the block atomic, + ``
`` becomes newlines, and glued comma-numbers like ``1,5461,518`` + are split. """ # --- Pass 1: remove ColN placeholders --- markdown = _coln_pattern.sub('', markdown) + # --- Pass 1b: normalize pymupdf4llm picture-text blocks before CJK/table + # passes so chart
stacks become newlines rather than empty joins. + markdown = _normalize_picture_text_blocks(markdown) + # --- Pass 2: collapse vertical-CJK runs (do this BEFORE row dedup so # rows that differ only by the collapsed form aren't treated as # distinct rows). @@ -171,7 +359,10 @@ def _clean_pdf_markdown(markdown: str, source_hint: str = "") -> str: # --- Pass 2b: strip
inside markdown table rows --- markdown = _strip_br_in_table_rows(markdown) - # --- Pass 2c: log lines that look like mojibake (failed glyph decode). + # --- Pass 2c: split glued comma-grouped numbers globally --- + markdown = _split_glued_comma_numbers(markdown) + + # --- Pass 2d: log lines that look like mojibake (failed glyph decode). # We don't repair these — the underlying glyphs aren't recoverable # from the markdown — but logging gives operators a grep target. findings = _detect_mojibake(markdown, source_hint) @@ -697,6 +888,12 @@ def _to_markdown_paged(strategy: str | None = None): # Clean up artefacts common in form PDFs (duplicate rows, ColN headers) markdown_content = _clean_pdf_markdown(markdown_content, source_hint=str(file_path)) + # Pages with broken CMaps (mojibake row labels, intact numbers) need a + # page-screenshot multimodal pass — embedded images are often absent. + markdown_content = _recover_mojibake_pages( + file_path, markdown_content, graphname=graphname + ) + # Rename image files that contain spaces to avoid path-parsing issues markdown_content = _sanitize_image_filenames(image_output_folder, markdown_content)