From a59e002cc71fc761873d6ce0957038967efb065a Mon Sep 17 00:00:00 2001 From: Lia Date: Thu, 24 Sep 2026 00:20:19 +0000 Subject: [PATCH 1/5] =?UTF-8?q?=F0=9F=93=84=20feat:=20Add=20Opt-In=20DOCX?= =?UTF-8?q?=20Extraction=20Profile?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .github/workflows/ci.yml | 1 + EXTRACTION.md | 55 +++++ app/models.py | 18 +- app/routes/extraction_routes.py | 124 +++++++++++ app/services/extraction.py | 100 +++++++++ app/services/extraction_worker.py | 136 +++++++++++ main.py | 3 +- requirements.extraction.txt | 3 + tests/fixtures/structured.docx | Bin 0 -> 1803 bytes tests/test_extraction_api.py | 359 ++++++++++++++++++++++++++++++ 10 files changed, 797 insertions(+), 2 deletions(-) create mode 100644 EXTRACTION.md create mode 100644 app/routes/extraction_routes.py create mode 100644 app/services/extraction.py create mode 100644 app/services/extraction_worker.py create mode 100644 requirements.extraction.txt create mode 100644 tests/fixtures/structured.docx create mode 100644 tests/test_extraction_api.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e6f4bfbf..cec318a0 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -31,6 +31,7 @@ jobs: python -m pip install --upgrade pip pip install -r requirements.txt pip install -r test_requirements.txt + pip install -r requirements.extraction.txt - name: Run unit tests env: diff --git a/EXTRACTION.md b/EXTRACTION.md new file mode 100644 index 00000000..f600d9fe --- /dev/null +++ b/EXTRACTION.md @@ -0,0 +1,55 @@ +# Opt-in document extraction contract + +`POST /v1/extract` implements a **document-v1** profile for DOCX only. It does not +replace `/text`, alter ingestion/chunking, or invoke embeddings, OCR or the vector +store. The route is disabled by default and requires a verified `JWT_SECRET` +token with an `id` claim when enabled. The main RAG application **still initializes +its vector store and embeddings at startup**. A parsing-only deployment remains a +separate migration. + +Install the pinned optional engine in a custom image or Python environment, +then opt in explicitly: + +```sh +pip install -r requirements.extraction.txt +export RAG_EXTRACTION_API_ENABLED=true +``` + +Send multipart `file` and `profile=document-v1`. Use DOCX MIME or a `.docx` +filename with a generic MIME. Success includes `text` (Markdown), `format`, +`profile`, `completeness` (`complete`/`partial`), `may_omit_content`, +`pages_needing_ocr` (empty until PDF support), `truncated` (always false), and +`parser: {name, version}`. An embedded image marks the DOCX as *partial*. Never +use partial text as proof of full content inspection. `complete` means the +supported conversion finished without *known* omitted image entries, not that +all information in the source is provably inspectable. No hosted OCR or +second-parser fallback runs inside this endpoint. + +Failures use `detail.code`, not the native exception message: + +| Status | Codes | Action | +|---|---|---| +| 400/415 | `UNSUPPORTED_PROFILE`, `UNSUPPORTED_DOCUMENT_TYPE` | Select a supported profile/type | +| 401/404 | `EXTRACTION_AUTH_REQUIRED`, `EXTRACTION_DISABLED` | Authenticate/opt in | +| 413 | `PARSER_INPUT_LIMIT`, `PARSER_OUTPUT_LIMIT`, `ZIP_BOMB` | Hard refusal; never send the same bytes to another parser | +| 422 | `ARCHIVE_INVALID`, `NO_DOCUMENT_TEXT`, `PARSE_FAILED` | Unusable archive or empty/unconvertible document | +| 429 | `CONCURRENCY_LIMIT` | Retry later; not a reason to invoke paid OCR | +| 503/504 | `PARSER_UNAVAILABLE`, `PARSER_CRASH`, `PARSER_TIMEOUT` | Retry or fix the service | + +The route checks the input limit of 15 MiB while staging the upload; +serialized output is capped at 15 MiB before IPC. Starlette may have already +spooled a multipart upload before the route runs: configure an upstream HTTP +body-size limit as well for internet-facing deployments. The child checks +*actual decompressed* ZIP entry bytes: at most +25 MiB per entry, 100 MiB in total and 4,096 entries. Defaults are two active +parses and six queued per API process. Set `RAG_EXTRACTION_CONCURRENT`, +`RAG_EXTRACTION_QUEUED` and `RAG_EXTRACTION_TIMEOUT_SECONDS` to tune admission +and the overall 30-second default deadline (queue wait, upload staging, parse). +On cancellation/timeout the child is killed and reaped before its temp file is +removed and its slot is reused. + +This is the first **service-side** slice. Existing LibreChat local parsing and +RAG `/text` behavior remain in place until cross-service tests establish policy, +authorization, preview, failure and compatibility behavior for each consumer. +The real DOCX test fixture is copied from Marco's LibreChat AnyDoc PR #14701 at +`fb7bbcd9cf75f4f78ecbd5a8780685c481600be2`. diff --git a/app/models.py b/app/models.py index 57c01130..684b859c 100644 --- a/app/models.py +++ b/app/models.py @@ -2,7 +2,23 @@ import hashlib from enum import Enum from pydantic import BaseModel -from typing import Optional, List +from typing import Optional, List, Literal + + +class ParserProvenance(BaseModel): + name: Literal["anydoc"] + version: str + + +class ExtractionResult(BaseModel): + profile: Literal["document-v1"] + text: str + format: Literal["markdown"] + completeness: Literal["complete", "partial"] + may_omit_content: bool + pages_needing_ocr: List[int] + truncated: bool + parser: ParserProvenance class DocumentResponse(BaseModel): diff --git a/app/routes/extraction_routes.py b/app/routes/extraction_routes.py new file mode 100644 index 00000000..1082a6ad --- /dev/null +++ b/app/routes/extraction_routes.py @@ -0,0 +1,124 @@ +"""Versioned document extraction. No embeddings, vector writes, or OCR calls.""" + +import asyncio +import math +import os +import tempfile +from pathlib import Path + +import aiofiles +from fastapi import APIRouter, File, Form, HTTPException, Request, UploadFile + +from app.config import RAG_UPLOAD_DIR, logger +from app.models import ExtractionResult +from app.services.extraction import ( + ExtractionAdmission, + ExtractionBusy, + ExtractionFailure, + run_worker, +) + +router = APIRouter(prefix="/v1") +DOCX_TYPE = "application/vnd.openxmlformats-officedocument.wordprocessingml.document" +MAX_INPUT_BYTES = 15 * 1024 * 1024 +_DEFAULT_TIMEOUT = 30.0 +_admission: ExtractionAdmission | None = None + + +def _error(code: str, status_code: int) -> HTTPException: + return HTTPException(status_code=status_code, detail={"code": code}) + + +def _get_admission() -> ExtractionAdmission: + global _admission + if _admission is None: + _admission = ExtractionAdmission( + concurrent=int(os.getenv("RAG_EXTRACTION_CONCURRENT", "2")), + queued=int(os.getenv("RAG_EXTRACTION_QUEUED", "6")), + ) + return _admission + + +async def _save_bounded(file: UploadFile, path: Path) -> None: + size = 0 + async with aiofiles.open(path, "wb") as output: + while chunk := await file.read(64 * 1024): + size += len(chunk) + if size > MAX_INPUT_BYTES: + raise _error("PARSER_INPUT_LIMIT", 413) + await output.write(chunk) + + +@router.post("/extract", response_model=ExtractionResult) +async def extract_document( + request: Request, + file: UploadFile = File(...), + profile: str = Form(...), +) -> ExtractionResult: + # Existing /text remains unchanged. An operator must explicitly enable + # and install this separate profile before moving any LibreChat caller. + if os.getenv("RAG_EXTRACTION_API_ENABLED", "false").lower() not in { + "1", + "true", + "yes", + "on", + }: + raise _error("EXTRACTION_DISABLED", 404) + # Legacy RAG deployments may run without auth; expensive extraction is + # never allowed anonymously, even when those older routes are public. + if not os.getenv("JWT_SECRET") or not getattr(request.state, "user", {}).get("id"): + raise _error("EXTRACTION_AUTH_REQUIRED", 401) + if profile != "document-v1": + raise _error("UNSUPPORTED_PROFILE", 400) + content_type = (file.content_type or "").split(";")[0].strip().lower() + extension = Path(file.filename or "").suffix.lower() + if content_type == "application/pdf" or not ( + content_type == DOCX_TYPE + or ( + content_type in {"application/octet-stream", "binary/octet-stream", ""} + and extension == ".docx" + ) + ): + raise _error("UNSUPPORTED_DOCUMENT_TYPE", 415) + if file.size is not None and file.size > MAX_INPUT_BYTES: + raise _error("PARSER_INPUT_LIMIT", 413) + + try: + admission = _get_admission() + timeout = float( + os.getenv("RAG_EXTRACTION_TIMEOUT_SECONDS", str(_DEFAULT_TIMEOUT)) + ) + if not math.isfinite(timeout) or timeout <= 0: + raise ValueError("Invalid extraction timeout") + async with asyncio.timeout(timeout): + async with admission.slot(): + fd, filename = tempfile.mkstemp( + prefix="rag-extract-", suffix=".docx", dir=RAG_UPLOAD_DIR + ) + os.close(fd) + path = Path(filename) + try: + await _save_bounded(file, path) + return await run_worker(path) + finally: + path.unlink(missing_ok=True) + except ExtractionBusy: + raise _error("CONCURRENCY_LIMIT", 429) + except ExtractionFailure as exc: + status_code = { + "ZIP_BOMB": 413, + "ARCHIVE_INVALID": 422, + "PARSER_OUTPUT_LIMIT": 413, + "NO_DOCUMENT_TEXT": 422, + "PARSE_FAILED": 422, + "PARSER_UNAVAILABLE": 503, + "PARSER_CRASH": 503, + }[exc.code] + raise _error(exc.code, status_code) + except TimeoutError: + raise _error("PARSER_TIMEOUT", 504) + except (OSError, ValueError) as exc: + logger.error( + "Extraction infrastructure unavailable | error=%s", type(exc).__name__ + ) + raise _error("PARSER_UNAVAILABLE", 503) diff --git a/app/services/extraction.py b/app/services/extraction.py new file mode 100644 index 00000000..06148362 --- /dev/null +++ b/app/services/extraction.py @@ -0,0 +1,100 @@ +"""Bounded, cancellable process boundary for opt-in document extraction.""" + +import asyncio +import json +import sys +from contextlib import asynccontextmanager +from pathlib import Path +from typing import AsyncIterator + +from app.models import ExtractionResult + +MAX_IPC_BYTES = 15 * 1024 * 1024 + + +class ExtractionBusy(Exception): + pass + + +class ExtractionFailure(Exception): + def __init__(self, code: str): + self.code = code + + +class ExtractionAdmission: + """One per serving process: bound uploads waiting and native children running.""" + + def __init__(self, concurrent: int = 2, queued: int = 6): + if concurrent < 1 or queued < 0: + raise ValueError("Invalid extraction admission limits") + self._capacity = concurrent + queued + self._pending = 0 + self._slots = asyncio.Semaphore(concurrent) + + @asynccontextmanager + async def slot(self) -> AsyncIterator[None]: + # The serving process has one event loop. No await separates the check + # and increment, so concurrent requests cannot exceed the queue limit. + if self._pending >= self._capacity: + raise ExtractionBusy() + self._pending += 1 + acquired = False + try: + await self._slots.acquire() + acquired = True + yield + finally: + self._pending -= 1 + if acquired: + self._slots.release() + + +def _command(path: Path) -> tuple[str, ...]: + return (sys.executable, "-m", "app.services.extraction_worker", str(path)) + + +async def run_worker(path: Path) -> ExtractionResult: + """Read a bounded child response; always reap a child before releasing its slot.""" + process = await asyncio.create_subprocess_exec( + *_command(path), + stdout=asyncio.subprocess.PIPE, + stderr=asyncio.subprocess.DEVNULL, + cwd=str(Path(__file__).resolve().parents[2]), + ) + try: + output = bytearray() + while chunk := await process.stdout.read(64 * 1024): + output.extend(chunk) + if len(output) > MAX_IPC_BYTES: + raise ExtractionFailure("PARSER_OUTPUT_LIMIT") + await process.wait() + except BaseException: + if process.returncode is None: + try: + process.kill() + except ProcessLookupError: + pass # The child exited while cancellation was being delivered. + # Draining and reaping are necessary before the temporary file can be + # removed and admission can be granted to the next upload. + await process.communicate() + raise + if process.returncode != 0: + raise ExtractionFailure("PARSER_CRASH") + try: + message = json.loads(output) + if message.get("ok") is False: + code = message["code"] + if code in { + "ZIP_BOMB", + "ARCHIVE_INVALID", + "PARSER_UNAVAILABLE", + "PARSER_OUTPUT_LIMIT", + "NO_DOCUMENT_TEXT", + "PARSE_FAILED", + }: + raise ExtractionFailure(code) + return ExtractionResult.model_validate(message["result"]) + except ExtractionFailure: + raise + except (AttributeError, KeyError, TypeError, ValueError) as exc: + raise ExtractionFailure("PARSER_CRASH") from exc diff --git a/app/services/extraction_worker.py b/app/services/extraction_worker.py new file mode 100644 index 00000000..8bd827be --- /dev/null +++ b/app/services/extraction_worker.py @@ -0,0 +1,136 @@ +"""Isolated document parsing. Run only as ``python -m app.services.extraction_worker``. + +The web process does not import native parsing bindings. Child exit, crash, and +SIGKILL cannot terminate the API process or leave its event loop blocked. +""" + +import json +import sys +import zipfile +from importlib.metadata import version +from pathlib import Path + +MAX_ARCHIVE_ENTRIES = 4096 +MAX_ENTRY_BYTES = 25 * 1024 * 1024 +MAX_TOTAL_BYTES = 100 * 1024 * 1024 +MAX_OUTPUT_BYTES = 15 * 1024 * 1024 +IMAGE_EXTENSIONS = ( + ".jpg", + ".jpeg", + ".png", + ".gif", + ".tif", + ".tiff", + ".bmp", + ".webp", + ".jp2", + ".jpx", + ".avif", + ".heic", + ".heif", + ".emf", + ".wmf", + ".svg", +) + + +class ExtractionRefusal(Exception): + def __init__(self, code: str): + self.code = code + + +def inspect_docx(path: Path) -> bool: + """Validate *actually inflated* archive bytes before handing any to AnyDoc. + + Reading in 64-KiB chunks catches false central-directory sizes without + keeping decompressed entries in memory. This is deliberately separate from + the text-output limit, since even an empty parse can inflate a zip bomb. + """ + try: + with zipfile.ZipFile(path) as archive: + entries = archive.infolist() + if len(entries) > MAX_ARCHIVE_ENTRIES: + raise ExtractionRefusal("ZIP_BOMB") + names = {entry.filename for entry in entries} + if not {"[Content_Types].xml", "word/document.xml"} <= names: + raise ExtractionRefusal("ARCHIVE_INVALID") + total = 0 + may_omit_content = False + for entry in entries: + if entry.is_dir(): + continue + if ( + entry.file_size > MAX_ENTRY_BYTES + or total + entry.file_size > MAX_TOTAL_BYTES + ): + raise ExtractionRefusal("ZIP_BOMB") + name = entry.filename.lower() + if not name.startswith(("docprops/", "thumbnails/")) and name.endswith( + IMAGE_EXTENSIONS + ): + may_omit_content = True + entry_bytes = 0 + with archive.open(entry) as stream: + while chunk := stream.read(64 * 1024): + entry_bytes += len(chunk) + total += len(chunk) + if entry_bytes > MAX_ENTRY_BYTES or total > MAX_TOTAL_BYTES: + raise ExtractionRefusal("ZIP_BOMB") + return may_omit_content + except ( + zipfile.BadZipFile, + zipfile.LargeZipFile, + RuntimeError, + EOFError, + NotImplementedError, + OSError, + ) as exc: + raise ExtractionRefusal("ARCHIVE_INVALID") from exc + + +def extract(path: Path) -> dict: + may_omit_content = inspect_docx(path) + try: + import anydoc + + # The MIME/extension are never passed to the binding. DOCX identity was + # checked in the archive above; explicit format avoids filename fallback. + text = anydoc.to_markdown_bytes(path.read_bytes(), "docx") + except ImportError as exc: + raise ExtractionRefusal("PARSER_UNAVAILABLE") from exc + except Exception as exc: + # Native error messages can echo source content. Never send them to clients. + raise ExtractionRefusal("PARSE_FAILED") from exc + if not isinstance(text, str) or not text.strip(): + raise ExtractionRefusal("NO_DOCUMENT_TEXT") + if len(text.encode("utf-8")) > MAX_OUTPUT_BYTES: + raise ExtractionRefusal("PARSER_OUTPUT_LIMIT") + return { + "profile": "document-v1", + "text": text, + "format": "markdown", + "completeness": "partial" if may_omit_content else "complete", + "may_omit_content": may_omit_content, + "pages_needing_ocr": [], + "truncated": False, + "parser": {"name": "anydoc", "version": version("firecrawl-anydoc")}, + } + + +def main() -> None: + if len(sys.argv) != 2: + raise SystemExit(2) + try: + payload = {"ok": True, "result": extract(Path(sys.argv[1]))} + except ExtractionRefusal as exc: + payload = {"ok": False, "code": exc.code} + except Exception: + payload = {"ok": False, "code": "PARSE_FAILED"} + serialized = json.dumps(payload, ensure_ascii=False) + if len(serialized.encode("utf-8")) > MAX_OUTPUT_BYTES: + serialized = json.dumps({"ok": False, "code": "PARSER_OUTPUT_LIMIT"}) + sys.stdout.write(serialized) + + +if __name__ == "__main__": + main() diff --git a/main.py b/main.py index e300951a..f653980f 100644 --- a/main.py +++ b/main.py @@ -23,7 +23,7 @@ vector_store, ) from app.middleware import security_middleware -from app.routes import document_routes, pgvector_routes +from app.routes import document_routes, extraction_routes, pgvector_routes from app.services.database import PSQLDatabase, ensure_vector_indexes from app.services.vector_store.factory import close_vector_store_connections @@ -90,6 +90,7 @@ async def lifespan(app: FastAPI): # Include routers app.include_router(document_routes.router) +app.include_router(extraction_routes.router) if debug_mode: app.include_router(router=pgvector_routes.router) diff --git a/requirements.extraction.txt b/requirements.extraction.txt new file mode 100644 index 00000000..e8276a3a --- /dev/null +++ b/requirements.extraction.txt @@ -0,0 +1,3 @@ +# Optional document-v1 extraction engine; install only when enabling /v1/extract. +# Match LibreChat AnyDoc PR #14701 (Node @firecrawl/anydoc 0.1.3) for corpus parity. +firecrawl-anydoc==0.1.3 diff --git a/tests/fixtures/structured.docx b/tests/fixtures/structured.docx new file mode 100644 index 0000000000000000000000000000000000000000..eff663f8f0bb831b03770efe2bed9f109182fa81 GIT binary patch literal 1803 zcmWIWW@Zs#U|`^2__l#1wkxPQ?I)0D#KgcL45Xu-^Ycnl^Gf1FDhpDJWA!R>bJkAW z?RVHfq^*43p-)k&d^{Wrm&xwvIS?TC;Dnv0PR->0=vmvJFKOgiao|wK|LO0i@4Y9t zSD5`wN^-{2)Vm&yj_%3M_fK8Fn_Q~QD$2nY!*k=5pl6=B`0KkfCa|L>ZJr%ThP>%&P6wE%Qlp|+Q|H@i<@+DqDcAX z;-;-aK}BKCiLtUxKigK?37V?=%Sq)d5|v(J)jwm&e9riT>2-5*c)P?;Jms{xeWm)0 z+sqZ;=PmyF-Zp*Ov2~lJrvJGzt>}u!k@fFl_iD)HMEtdEIrU<~zxrc$c7M3~^0c{p z07^*Nul@IHJurlH7#SG2fpmOPYEH4f9*FEcZ_RhefQRA2++de`nNe&l0nXwrS45{W zyL8SDGm$^~@bc}+jf`4sZvW@K-Bo$3^RD5g7jqj{X*4bjj_W_iv7==9S3U{Vms{rS zab`3YGHDO3J?6R0#`&;BN9vJ@4T?)QJfCx=AuQ-2^RAeewOfpy8c4}4+2z$-6FeRy!0~+4zm-5Q7>Udm`4SOrO_RQ1vWtKwP9qNq1=L*kTuH>A%M9byNQ=_&gDSC#I za!bJ;y&gU99`6sQ$n~sC;>-`UsXnr=kT@Rs&~U|U=lS&|1*f`onQtgeKB)6x>gq=f z`hWidk)caDg*G@0|vElON{I;*!do)M9WV@7w5i#DE8w$dx{wTl%&kp!8)f zhgie|MxW$cUY){6FHYTO+xdc%^I*e~wAY{i*W1mzw>LQ5YZHsv1v#mJBgq@M{n_H? zy`Mag9pP*Dqa?fC`I2afN zVID`0*|TBZ#mx!=|FnJd=184c?KyYGoLetfUVdh1D}2{2T|nZ1#P8SL{jXee^;hoI zS+M*5_TQXqUcWU>d%@)8-c(&-RA#{Ru3Of4O$gmW4J8dz@g5)9=kt!3 zJX$RCNu^0gJ~c5e)bNe(F@~k@T8;9*@R*#>F;bdsy0VrlSmjZcXYK5zAqyO18S?K5 zt>V#`wouk%p^sczS3;=Mm6lqKc7sbR)Q2r-In!j$_C&i+C$@B!7C(}WX;8o5>zj2kU+ZtTE!#r@CZ9Lc zq?^k375(b?@9}M~N87}|iH04oHrFrwrF=EF{NLN}ziP}nUUjFwS6m&UUYef45PGse zi*t2oxV?@Q-y*JnLzzM@Y*Es3Kl5HK`I&g5hB3gKkx7IZcd-qOS1@P*Ml{w!9Nhr) z(hQ=VfuVs>3upvV*@mtWy|_SVWCB(p_zDknlhAV{!laillMwk6T{C*BMreM-40Q%_ V8V~ShWdkW;1wsd)2e-0 bytes: + result = io.BytesIO() + with zipfile.ZipFile(io.BytesIO(original)) as source, zipfile.ZipFile( + result, "w", zipfile.ZIP_DEFLATED + ) as target: + for entry in source.infolist(): + if entry.filename != name: + target.writestr(entry, source.read(entry)) + target.writestr(name, value) + return result.getvalue() + + +async def test_real_docx_from_marcos_pr_returns_markdown_and_provenance( + headers, configured +): + response = await post( + FIXTURE.read_bytes(), headers, name="renamed.csv", mime=DOCX_TYPE + ) + assert response.status_code == 200, response.text + payload = response.json() + assert payload == { + "profile": "document-v1", + "format": "markdown", + "text": ( + "# Quarterly Report\n\nThis document summarizes the results for the period.\n\n" + "## Regional Totals\n\n| | | |\n| --- | --- | --- |\n" + "| Region | Units | Revenue |\n| North | 1200 | 48000 |\n" + "| South | 950 | 38000 |\n| East | 1430 | 57200 |\n\n" + "**Totals are unaudited.**\n" + ), + "completeness": "complete", + "may_omit_content": False, + "pages_needing_ocr": [], + "truncated": False, + "parser": {"name": "anydoc", "version": "0.1.3"}, + } + assert not list(configured[0].iterdir()) + + +async def test_embedded_image_cannot_claim_complete_text(headers, configured): + image_docx = with_entry(FIXTURE.read_bytes(), "word/media/scan.png", b"\x89PNG\r\n") + response = await post( + image_docx, headers, name="report.docx", mime="application/octet-stream" + ) + assert response.status_code == 200, response.text + assert response.json()["completeness"] == "partial" + assert response.json()["may_omit_content"] is True + assert response.json()["pages_needing_ocr"] == [] + assert not list(configured[0].iterdir()) + + +async def test_disabled_requires_explicit_opt_in(headers, configured, monkeypatch): + monkeypatch.delenv("RAG_EXTRACTION_API_ENABLED") + response = await post(FIXTURE.read_bytes(), headers) + assert response.status_code == 404 + assert response.json()["detail"]["code"] == "EXTRACTION_DISABLED" + assert not list(configured[0].iterdir()) + + +async def test_requires_verified_identity_and_signing_secret( + headers, configured, monkeypatch +): + missing = await post(FIXTURE.read_bytes(), {}) + assert missing.status_code == 401 + monkeypatch.delenv("JWT_SECRET") + unsigned = await post(FIXTURE.read_bytes(), {}) + assert unsigned.status_code == 401 + assert unsigned.json()["detail"]["code"] == "EXTRACTION_AUTH_REQUIRED" + assert not list(configured[0].iterdir()) + + +@pytest.mark.parametrize( + "name,mime,code", + [ + ("report.md", "text/markdown", "UNSUPPORTED_DOCUMENT_TYPE"), + ("report.docx", "application/pdf", "UNSUPPORTED_DOCUMENT_TYPE"), + ("report.pdf", "application/octet-stream", "UNSUPPORTED_DOCUMENT_TYPE"), + ], +) +async def test_unrelated_formats_never_reach_the_parser( + headers, configured, name, mime, code +): + response = await post(FIXTURE.read_bytes(), headers, name=name, mime=mime) + assert response.status_code == 415 + assert response.json()["detail"]["code"] == code + assert not list(configured[0].iterdir()) + + +async def test_nonfinite_deadline_cannot_disable_worker_timeout( + headers, configured, monkeypatch +): + monkeypatch.setenv("RAG_EXTRACTION_TIMEOUT_SECONDS", "inf") + response = await post(FIXTURE.read_bytes(), headers) + assert response.status_code == 503 + assert response.json()["detail"]["code"] == "PARSER_UNAVAILABLE" + assert not list(configured[0].iterdir()) + + +async def test_unknown_profile_and_invalid_archive_fail_closed(headers, configured): + bad_profile = await post(FIXTURE.read_bytes(), headers, profile="raw-v1") + assert bad_profile.status_code == 400 + assert bad_profile.json()["detail"]["code"] == "UNSUPPORTED_PROFILE" + bad_archive = await post(b"private secret from a forged DOCX", headers) + assert bad_archive.status_code == 422 + assert bad_archive.json() == {"detail": {"code": "ARCHIVE_INVALID"}} + assert not list(configured[0].iterdir()) + + +async def test_refuses_zip_bomb_before_native_parsing(headers, configured): + bomb = with_entry( + FIXTURE.read_bytes(), "word/bomb.xml", b"x" * (25 * 1024 * 1024 + 1) + ) + response = await post(bomb, headers) + assert response.status_code == 413 + assert response.json()["detail"]["code"] == "ZIP_BOMB" + assert not list(configured[0].iterdir()) + + +async def test_empty_docx_reports_no_text_instead_of_success(headers, configured): + result = io.BytesIO() + with zipfile.ZipFile(FIXTURE) as source, zipfile.ZipFile( + result, "w", zipfile.ZIP_DEFLATED + ) as target: + for entry in source.infolist(): + data = source.read(entry) + if entry.filename == "word/document.xml": + data = ( + b'' + b"" + ) + target.writestr(entry, data) + response = await post(result.getvalue(), headers) + assert response.status_code == 422 + assert response.json()["detail"]["code"] == "NO_DOCUMENT_TEXT" + assert not list(configured[0].iterdir()) + + +async def test_archive_entry_count_has_an_independent_limit(headers, configured): + result = io.BytesIO() + with zipfile.ZipFile(FIXTURE) as source, zipfile.ZipFile( + result, "w", zipfile.ZIP_DEFLATED + ) as target: + for entry in source.infolist(): + target.writestr(entry, source.read(entry)) + for index in range(4096): + target.writestr(f"word/noise/{index}", b"") + response = await post(result.getvalue(), headers) + assert response.status_code == 413 + assert response.json()["detail"]["code"] == "ZIP_BOMB" + assert not list(configured[0].iterdir()) + + +async def test_input_limit_is_checked_while_streaming_even_without_size_hint( + configured, monkeypatch +): + monkeypatch.setattr(extraction_routes, "MAX_INPUT_BYTES", 8) + file = UploadFile(file=io.BytesIO(b"0123456789"), filename="report.docx", size=None) + with pytest.raises(HTTPException) as caught: + await extraction_routes._save_bounded(file, configured[0] / "bounded.docx") + assert caught.value.status_code == 413 + assert caught.value.detail == {"code": "PARSER_INPUT_LIMIT"} + + +async def test_worker_output_limit_is_a_refusal(configured, monkeypatch): + monkeypatch.setattr(extraction_worker, "MAX_OUTPUT_BYTES", 10) + with pytest.raises(extraction_worker.ExtractionRefusal) as caught: + extraction_worker.extract(FIXTURE) + assert caught.value.code == "PARSER_OUTPUT_LIMIT" + + +async def test_child_crash_and_malformed_output_are_sanitized( + headers, configured, monkeypatch +): + monkeypatch.setattr( + extraction, + "_command", + lambda path: (sys.executable, "-c", "import sys; sys.exit(11)"), + ) + crash = await post(FIXTURE.read_bytes(), headers) + assert crash.status_code == 503 + assert crash.json() == {"detail": {"code": "PARSER_CRASH"}} + assert not list(configured[0].iterdir()) + monkeypatch.setattr( + extraction, + "_command", + lambda path: (sys.executable, "-c", "print('private secret')"), + ) + invalid = await post(FIXTURE.read_bytes(), headers) + assert invalid.status_code == 503 + assert invalid.json() == {"detail": {"code": "PARSER_CRASH"}} + assert "private secret" not in invalid.text + assert not list(configured[0].iterdir()) + + +async def test_api_kills_worker_that_overproduces_ipc(headers, configured, monkeypatch): + monkeypatch.setattr(extraction, "MAX_IPC_BYTES", 64) + monkeypatch.setattr( + extraction, + "_command", + lambda path: (sys.executable, "-c", "import os; os.write(1, b'x' * 1024)"), + ) + response = await post(FIXTURE.read_bytes(), headers) + assert response.status_code == 413 + assert response.json()["detail"]["code"] == "PARSER_OUTPUT_LIMIT" + assert not list(configured[0].iterdir()) + assert configured[1]._pending == 0 + + +def _sleeping_command(path: Path) -> tuple[str, ...]: + return ( + sys.executable, + "-c", + "import os,sys,time; open(sys.argv[1]+'.pid','w').write(str(os.getpid())); time.sleep(60)", + str(path), + ) + + +async def _await_child(tmp_path: Path) -> Path: + for _ in range(200): + pids = list(tmp_path.glob("*.pid")) + if pids: + return pids[0] + await asyncio.sleep(0.01) + raise AssertionError("parser child did not start") + + +async def _assert_child_reaped( + pid_file: Path, admission: extraction.ExtractionAdmission +) -> None: + pid = int(pid_file.read_text()) + for _ in range(200): + try: + os.kill(pid, 0) + except ProcessLookupError: + if admission._pending == 0 and not list(pid_file.parent.glob("*.docx")): + pid_file.unlink() + return + await asyncio.sleep(0.01) + raise AssertionError("parser child or staging file survived cancellation") + + +async def test_timeout_kills_and_reaps_child(headers, configured, monkeypatch): + monkeypatch.setattr(extraction, "_command", _sleeping_command) + monkeypatch.setenv("RAG_EXTRACTION_TIMEOUT_SECONDS", "0.3") + task = asyncio.create_task(post(FIXTURE.read_bytes(), headers)) + pid_file = await _await_child(configured[0]) + response = await asyncio.wait_for(task, 2) + assert response.status_code == 504 + assert response.json()["detail"]["code"] == "PARSER_TIMEOUT" + await _assert_child_reaped(pid_file, configured[1]) + assert not list(configured[0].iterdir()) + assert configured[1]._pending == 0 + + +async def test_cancel_reaps_child_and_frees_slot_for_next_upload( + headers, configured, monkeypatch +): + original_command = extraction._command + monkeypatch.setattr(extraction, "_command", _sleeping_command) + task = asyncio.create_task(post(FIXTURE.read_bytes(), headers)) + pid_file = await _await_child(configured[0]) + task.cancel() + with pytest.raises(asyncio.CancelledError): + await asyncio.wait_for(task, 2) + await _assert_child_reaped(pid_file, configured[1]) + assert not list(configured[0].iterdir()) + assert configured[1]._pending == 0 + monkeypatch.setattr(extraction, "_command", original_command) + response = await post(FIXTURE.read_bytes(), headers) + assert response.status_code == 200 + + +async def test_busy_parser_refuses_before_staging_anything( + headers, configured, monkeypatch +): + monkeypatch.setattr(extraction, "_command", _sleeping_command) + monkeypatch.setattr( + extraction_routes, "_admission", extraction.ExtractionAdmission(1, 0) + ) + task = asyncio.create_task(post(FIXTURE.read_bytes(), headers)) + pid_file = await _await_child(configured[0]) + busy = await post(FIXTURE.read_bytes(), headers) + assert busy.status_code == 429 + assert busy.json()["detail"]["code"] == "CONCURRENCY_LIMIT" + assert len(list(configured[0].glob("*.docx"))) == 1 + task.cancel() + with pytest.raises(asyncio.CancelledError): + await asyncio.wait_for(task, 2) + await _assert_child_reaped(pid_file, extraction_routes._admission) + assert not list(configured[0].iterdir()) + + +async def test_existing_text_endpoint_still_preserves_raw_markdown(headers, configured): + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=app), base_url="http://test" + ) as client: + response = await client.post( + "/text", + headers=headers, + data={"file_id": "md-test"}, + files={"file": ("readme.md", b"# Raw **Markdown**\n", "text/markdown")}, + ) + assert response.status_code == 200, response.text + assert response.json()["text"] == "# Raw **Markdown**" From 15d84b85443f8cf4e23d083c4af6b135ce3f5344 Mon Sep 17 00:00:00 2001 From: Lia Date: Thu, 24 Sep 2026 00:25:50 +0000 Subject: [PATCH 2/5] =?UTF-8?q?=F0=9F=94=92=20fix:=20Register=20Extraction?= =?UTF-8?q?=20Only=20When=20Enabled=20at=20Startup?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- EXTRACTION.md | 3 ++- main.py | 8 ++++++- tests/test_extraction_api.py | 42 ++++++++++++++++++++++++++++++++---- 3 files changed, 47 insertions(+), 6 deletions(-) diff --git a/EXTRACTION.md b/EXTRACTION.md index f600d9fe..5678aa37 100644 --- a/EXTRACTION.md +++ b/EXTRACTION.md @@ -8,7 +8,8 @@ its vector store and embeddings at startup**. A parsing-only deployment remains separate migration. Install the pinned optional engine in a custom image or Python environment, -then opt in explicitly: +then opt in explicitly before starting (or restarting) the API. With the flag +off, the route is not registered and does not parse multipart bodies: ```sh pip install -r requirements.extraction.txt diff --git a/main.py b/main.py index f653980f..714f5752 100644 --- a/main.py +++ b/main.py @@ -90,7 +90,13 @@ async def lifespan(app: FastAPI): # Include routers app.include_router(document_routes.router) -app.include_router(extraction_routes.router) +if os.getenv("RAG_EXTRACTION_API_ENABLED", "false").lower() in { + "1", + "true", + "yes", + "on", +}: + app.include_router(extraction_routes.router) if debug_mode: app.include_router(router=pgvector_routes.router) diff --git a/tests/test_extraction_api.py b/tests/test_extraction_api.py index 9973948e..59dc1592 100644 --- a/tests/test_extraction_api.py +++ b/tests/test_extraction_api.py @@ -3,6 +3,7 @@ import asyncio import io import os +import subprocess import sys import zipfile from concurrent.futures import ThreadPoolExecutor @@ -11,11 +12,16 @@ import httpx import jwt import pytest -from fastapi import HTTPException, UploadFile +from fastapi import FastAPI, HTTPException, UploadFile +from app.middleware import security_middleware from app.routes import extraction_routes from app.services import extraction, extraction_worker -from main import app +from main import app as main_app + +app = FastAPI() +app.middleware("http")(security_middleware) +app.include_router(extraction_routes.router) FIXTURE = Path(__file__).parent / "fixtures" / "structured.docx" DOCX_TYPE = extraction_routes.DOCX_TYPE @@ -29,7 +35,7 @@ def configured(monkeypatch, tmp_path): admission = extraction.ExtractionAdmission() monkeypatch.setattr(extraction_routes, "_admission", admission) with ThreadPoolExecutor(max_workers=2) as pool: - monkeypatch.setattr(app.state, "thread_pool", pool, raising=False) + monkeypatch.setattr(main_app.state, "thread_pool", pool, raising=False) yield tmp_path, admission @@ -104,6 +110,34 @@ async def test_embedded_image_cannot_claim_complete_text(headers, configured): assert not list(configured[0].iterdir()) +def test_real_app_registers_route_only_when_enabled(): + # A fresh process verifies main.py registration, not just the test router. + script = ( + "from langchain_community.vectorstores.pgvector import PGVector\n" + "from app.services.vector_store.async_pg_vector import AsyncPgVector\n" + "PGVector.__post_init__ = lambda self: None\n" + "AsyncPgVector.__post_init__ = lambda self: None\n" + "from main import app\n" + "import sys\n" + "print(int(any(getattr(r, 'path', None) == '/v1/extract' for r in app.routes)), " + "int('anydoc' in sys.modules))\n" + ) + for enabled, expected in (("false", "0 0"), ("true", "1 0")): + env = { + **os.environ, + "RAG_EXTRACTION_API_ENABLED": enabled, + "OPENAI_API_KEY": "test_key", + } + result = subprocess.run( + [sys.executable, "-c", script], + env=env, + capture_output=True, + text=True, + check=True, + ) + assert result.stdout.strip().splitlines()[-1] == expected, result.stderr + + async def test_disabled_requires_explicit_opt_in(headers, configured, monkeypatch): monkeypatch.delenv("RAG_EXTRACTION_API_ENABLED") response = await post(FIXTURE.read_bytes(), headers) @@ -347,7 +381,7 @@ async def test_busy_parser_refuses_before_staging_anything( async def test_existing_text_endpoint_still_preserves_raw_markdown(headers, configured): async with httpx.AsyncClient( - transport=httpx.ASGITransport(app=app), base_url="http://test" + transport=httpx.ASGITransport(app=main_app), base_url="http://test" ) as client: response = await client.post( "/text", From 611c5c94a1b4e992cdf7bcf3706b3651980f8926 Mon Sep 17 00:00:00 2001 From: Lia Date: Wed, 30 Sep 2026 09:52:28 +0000 Subject: [PATCH 3/5] =?UTF-8?q?=F0=9F=8D=9E=20feat:=20Move=20Document=20Ex?= =?UTF-8?q?traction=20to=20Standalone=20Bun=20and=20Hono=20Service?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .dockerignore | 5 +- .github/workflows/bun-service.yml | 32 +++ .github/workflows/ci.yml | 1 - .gitignore | 2 + EXTRACTION.md | 170 +++++++++---- app/models.py | 18 +- app/routes/extraction_routes.py | 124 ---------- app/services/extraction.py | 100 -------- app/services/extraction_worker.py | 136 ----------- main.py | 9 +- requirements.extraction.txt | 3 - service/Dockerfile | 13 + service/bun.lock | 74 ++++++ service/package.json | 30 +++ service/src/admission.ts | 56 +++++ service/src/app.ts | 128 ++++++++++ service/src/config.ts | 64 +++++ service/src/contract.ts | 55 +++++ service/src/process.ts | 67 +++++ service/src/server.ts | 29 +++ service/src/upload.ts | 126 ++++++++++ service/src/worker.ts | 159 ++++++++++++ service/test/extraction.test.ts | 387 +++++++++++++++++++++++++++++ service/test/image.sh | 28 +++ service/test/lifecycle.test.ts | 125 ++++++++++ service/test/sleep.fixture.ts | 6 + service/tsconfig.json | 15 ++ tests/test_extraction_api.py | 393 ------------------------------ 28 files changed, 1525 insertions(+), 830 deletions(-) create mode 100644 .github/workflows/bun-service.yml delete mode 100644 app/routes/extraction_routes.py delete mode 100644 app/services/extraction.py delete mode 100644 app/services/extraction_worker.py delete mode 100644 requirements.extraction.txt create mode 100644 service/Dockerfile create mode 100644 service/bun.lock create mode 100644 service/package.json create mode 100644 service/src/admission.ts create mode 100644 service/src/app.ts create mode 100644 service/src/config.ts create mode 100644 service/src/contract.ts create mode 100644 service/src/process.ts create mode 100644 service/src/server.ts create mode 100644 service/src/upload.ts create mode 100644 service/src/worker.ts create mode 100644 service/test/extraction.test.ts create mode 100644 service/test/image.sh create mode 100644 service/test/lifecycle.test.ts create mode 100644 service/test/sleep.fixture.ts create mode 100644 service/tsconfig.json delete mode 100644 tests/test_extraction_api.py diff --git a/.dockerignore b/.dockerignore index 4e7bb6db..288e2245 100644 --- a/.dockerignore +++ b/.dockerignore @@ -2,4 +2,7 @@ __pycache__ uploads/ myenv/ -venv/ \ No newline at end of file +venv/ +.git +.worktrees +node_modules diff --git a/.github/workflows/bun-service.yml b/.github/workflows/bun-service.yml new file mode 100644 index 00000000..6c1d4871 --- /dev/null +++ b/.github/workflows/bun-service.yml @@ -0,0 +1,32 @@ +name: Bun service + +on: + pull_request: + push: + branches: [main] + +jobs: + extraction: + runs-on: ubuntu-latest + defaults: + run: + working-directory: service + steps: + - uses: actions/checkout@v4 + - uses: oven-sh/setup-bun@v2 + with: + bun-version: '1.4.2' + - name: Install locked dependencies + run: bun install --frozen-lockfile + - name: Typecheck + run: bun run typecheck + - name: Formatting + run: bun run format:check + - name: Native extraction and service contract tests + run: bun test + - name: Build standalone image + run: docker build -f service/Dockerfile -t rag-bun-test . + working-directory: . + - name: Verify image starts without Python, database or provider credentials + run: bash service/test/image.sh rag-bun-test + working-directory: . diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index cec318a0..e6f4bfbf 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -31,7 +31,6 @@ jobs: python -m pip install --upgrade pip pip install -r requirements.txt pip install -r test_requirements.txt - pip install -r requirements.extraction.txt - name: Run unit tests env: diff --git a/.gitignore b/.gitignore index 38921790..57141d6e 100644 --- a/.gitignore +++ b/.gitignore @@ -9,3 +9,5 @@ venv/ *.pyc dev.yml SHOPIFY.md + +node_modules/ diff --git a/EXTRACTION.md b/EXTRACTION.md index 5678aa37..06e46083 100644 --- a/EXTRACTION.md +++ b/EXTRACTION.md @@ -1,56 +1,132 @@ -# Opt-in document extraction contract +# Bun/Hono RAG service: first extraction slice -`POST /v1/extract` implements a **document-v1** profile for DOCX only. It does not -replace `/text`, alter ingestion/chunking, or invoke embeddings, OCR or the vector -store. The route is disabled by default and requires a verified `JWT_SECRET` -token with an `id` claim when enabled. The main RAG application **still initializes -its vector store and embeddings at startup**. A parsing-only deployment remains a -separate migration. +PR #330 starts the **new Bun/Hono service**, under `service/`. It is not an +extension of FastAPI and does not proxy to Python. The existing Python API, +requirements, ingestion, `/text`, collection schema and deployment remain +unchanged. Keep that service available throughout compatibility evaluation. -Install the pinned optional engine in a custom image or Python environment, -then opt in explicitly before starting (or restarting) the API. With the flag -off, the route is not registered and does not parse multipart bodies: +This first slice implements only `POST /v1/extract` with the `document-v1` DOCX +profile. Retrieval, embeddings, reranking, PDF accounting, OCR and the LibreChat +client migration remain follow-up slices. **Do not point LibreChat's existing +`RAG_API_URL` at this service yet:** it does not implement the legacy endpoints. + +## Run independently + +Requires Bun 1.4.2. The Node AnyDoc binding is pinned to 0.1.3, matching Marco's +LibreChat PR #14701. The binding runs in a separate, killable Bun process, never +in the serving process. No Node server, Python runtime, database or embedding +provider is required. + +```sh +cd service +bun install --frozen-lockfile +bun run typecheck +bun test +bun run start +``` + +The default port is **8001**, so the existing Python service may stay on 8000. +`GET /health` works without external dependencies. Extraction is disabled by +default, and `/v1/extract` is not registered when off. + +Build a separate image from the repository root: ```sh -pip install -r requirements.extraction.txt -export RAG_EXTRACTION_API_ENABLED=true +docker build -f service/Dockerfile -t rag-bun . +``` + +Opt in at startup with `RAG_EXTRACTION_API_ENABLED=true` and `RAG_JWT_SECRET` +(minimum 32 characters). The secret must be the dedicated RAG signing key, **not +LibreChat's session key**. Each extraction request needs an HS256 JWT with +`sub`, `exp`, issuer `librechat`, audience `rag-api`, and a `scopes` array +containing `rag:documents`. Issuer and audience are configurable. Legacy `{id}` +tokens and inference-only tokens do not authorize this new route. That is a +new-service contract, not a change to the old Python API. Coordinate the +LibreChat token supplier with the strict-auth migration before sending traffic. + +## Request and response + +Send multipart `profile=document-v1` and one `file`. DOCX MIME is authoritative; +a generic MIME requires a `.docx` filename. No caller-provided path or URL is +read, and no hosted parser, OCR service, or inference provider is called. + +Successful response: + +```json +{ + "profile": "document-v1", + "text": "# Extracted Markdown\n", + "format": "markdown", + "completeness": "complete", + "may_omit_content": false, + "pages_needing_ocr": [], + "truncated": false, + "parser": { "name": "anydoc", "version": "0.1.3" } +} ``` -Send multipart `file` and `profile=document-v1`. Use DOCX MIME or a `.docx` -filename with a generic MIME. Success includes `text` (Markdown), `format`, -`profile`, `completeness` (`complete`/`partial`), `may_omit_content`, -`pages_needing_ocr` (empty until PDF support), `truncated` (always false), and -`parser: {name, version}`. An embedded image marks the DOCX as *partial*. Never -use partial text as proof of full content inspection. `complete` means the -supported conversion finished without *known* omitted image entries, not that -all information in the source is provably inspectable. No hosted OCR or -second-parser fallback runs inside this endpoint. +An archive with non-thumbnail artwork or embedded objects is marked `partial` +and `may_omit_content=true`. This is conservative omission detection, **not a +proof that every source element is inspectable**. Never use partial text or a +preview as complete content-inspection input. No automatic fallback occurs +inside the service, so hard refusals cannot accidentally become paid OCR calls. -Failures use `detail.code`, not the native exception message: +`detail.code` classifies failures without source text, tokens, or native error +messages: -| Status | Codes | Action | +| Status | Codes | Meaning | |---|---|---| -| 400/415 | `UNSUPPORTED_PROFILE`, `UNSUPPORTED_DOCUMENT_TYPE` | Select a supported profile/type | -| 401/404 | `EXTRACTION_AUTH_REQUIRED`, `EXTRACTION_DISABLED` | Authenticate/opt in | -| 413 | `PARSER_INPUT_LIMIT`, `PARSER_OUTPUT_LIMIT`, `ZIP_BOMB` | Hard refusal; never send the same bytes to another parser | -| 422 | `ARCHIVE_INVALID`, `NO_DOCUMENT_TEXT`, `PARSE_FAILED` | Unusable archive or empty/unconvertible document | -| 429 | `CONCURRENCY_LIMIT` | Retry later; not a reason to invoke paid OCR | -| 503/504 | `PARSER_UNAVAILABLE`, `PARSER_CRASH`, `PARSER_TIMEOUT` | Retry or fix the service | - -The route checks the input limit of 15 MiB while staging the upload; -serialized output is capped at 15 MiB before IPC. Starlette may have already -spooled a multipart upload before the route runs: configure an upstream HTTP -body-size limit as well for internet-facing deployments. The child checks -*actual decompressed* ZIP entry bytes: at most -25 MiB per entry, 100 MiB in total and 4,096 entries. Defaults are two active -parses and six queued per API process. Set `RAG_EXTRACTION_CONCURRENT`, -`RAG_EXTRACTION_QUEUED` and `RAG_EXTRACTION_TIMEOUT_SECONDS` to tune admission -and the overall 30-second default deadline (queue wait, upload staging, parse). -On cancellation/timeout the child is killed and reaped before its temp file is -removed and its slot is reused. - -This is the first **service-side** slice. Existing LibreChat local parsing and -RAG `/text` behavior remain in place until cross-service tests establish policy, -authorization, preview, failure and compatibility behavior for each consumer. -The real DOCX test fixture is copied from Marco's LibreChat AnyDoc PR #14701 at -`fb7bbcd9cf75f4f78ecbd5a8780685c481600be2`. +| 400/415 | `INVALID_MULTIPART`, `UNSUPPORTED_PROFILE`, `UNSUPPORTED_DOCUMENT_TYPE` | Malformed or unsupported request | +| 401/403 | `EXTRACTION_AUTH_REQUIRED`, `EXTRACTION_FORBIDDEN` | Missing/invalid service token or missing document scope | +| 404 | `EXTRACTION_DISABLED` | Extraction not registered | +| 413 | `PARSER_INPUT_LIMIT`, `PARSER_OUTPUT_LIMIT`, `ZIP_BOMB` | Hard refusal; do not send those bytes to another parser | +| 422 | `ARCHIVE_INVALID`, `NO_DOCUMENT_TEXT`, `PARSE_FAILED` | Unusable archive or no usable conversion | +| 429 | `CONCURRENCY_LIMIT` | Busy; retry later rather than escalating to OCR | +| 503/504 | `PARSER_UNAVAILABLE`, `PARSER_CRASH`, `PARSER_TIMEOUT` | Infrastructure failure or overall deadline | +| 408 | `REQUEST_CANCELLED` | Cancelled operation; no result retained | + +## Limits and lifecycle + +Authentication and parser admission happen **before** multipart consumption. +Multipart uploads stream to a private random directory. Both declared length +and actual streamed bytes are bounded; there is no `request.formData()` buffer +or health-check round trip. The Bun listener also enforces the body ceiling. + +| Environment variable | Default | +|---|---:| +| `RAG_PORT` / `RAG_HOST` | 8001 / 0.0.0.0 | +| `RAG_EXTRACTION_CONCURRENT` | 2 | +| `RAG_EXTRACTION_QUEUED` | 6 | +| `RAG_EXTRACTION_TIMEOUT_MS` | 30,000 | +| `RAG_EXTRACTION_MAX_FILE_BYTES` | 15 MiB | +| `RAG_EXTRACTION_MAX_BODY_BYTES` | 16 MiB | +| `RAG_EXTRACTION_MAX_OUTPUT_BYTES` | 15 MiB | +| `RAG_EXTRACTION_MAX_ENTRY_BYTES` | 25 MiB | +| `RAG_EXTRACTION_MAX_ARCHIVE_BYTES` | 100 MiB | +| `RAG_EXTRACTION_MAX_ENTRIES` | 4,096 | +| `RAG_EXTRACTION_TEMP_DIR` | operating-system temp directory | +| `RAG_JWT_ISSUER` / `RAG_JWT_AUDIENCE` | librechat / rag-api | + +Limits are per service process, not cluster-wide quotas. Queue wait, upload +staging and parsing share one deadline. Cancelled queued work is removed; an +active child is killed and reaped before cleanup and slot reuse. The child +receives no JWT secret or provider credentials. Both sides cap serialized IPC +output. Actual decompressed entry bytes are checked before native parsing; +metadata-only size claims are not trusted. Graceful shutdown stops accepting +requests and allows active operations to finish within their deadlines. + +## Verification and adoption + +The Bun corpus uses Marco's structured DOCX at +`fb7bbcd9cf75f4f78ecbd5a8780685c481600be2` and the exact output expected by the +Python prototype. Tests use real native parsing, Hono requests and child +processes, with injected crash/hang/overproduction programs for failure cases. +Typecheck and native tests run in a separate Bun CI job alongside the unchanged +Python jobs. + +No traffic cutover or database migration is part of this PR. Next, add a flagged +LibreChat adapter using this result contract, then prove content inspection, +preview/sharing, token scopes, outages and mixed-version behavior end to end. +Keep raw Markdown, rich HTML preview and complete inspection as distinct +contracts. Expand formats or enable reranking only behind their own fidelity +and quality gates. No measured speedup or full service parity is claimed here. diff --git a/app/models.py b/app/models.py index 684b859c..57c01130 100644 --- a/app/models.py +++ b/app/models.py @@ -2,23 +2,7 @@ import hashlib from enum import Enum from pydantic import BaseModel -from typing import Optional, List, Literal - - -class ParserProvenance(BaseModel): - name: Literal["anydoc"] - version: str - - -class ExtractionResult(BaseModel): - profile: Literal["document-v1"] - text: str - format: Literal["markdown"] - completeness: Literal["complete", "partial"] - may_omit_content: bool - pages_needing_ocr: List[int] - truncated: bool - parser: ParserProvenance +from typing import Optional, List class DocumentResponse(BaseModel): diff --git a/app/routes/extraction_routes.py b/app/routes/extraction_routes.py deleted file mode 100644 index 1082a6ad..00000000 --- a/app/routes/extraction_routes.py +++ /dev/null @@ -1,124 +0,0 @@ -"""Versioned document extraction. No embeddings, vector writes, or OCR calls.""" - -import asyncio -import math -import os -import tempfile -from pathlib import Path - -import aiofiles -from fastapi import APIRouter, File, Form, HTTPException, Request, UploadFile - -from app.config import RAG_UPLOAD_DIR, logger -from app.models import ExtractionResult -from app.services.extraction import ( - ExtractionAdmission, - ExtractionBusy, - ExtractionFailure, - run_worker, -) - -router = APIRouter(prefix="/v1") -DOCX_TYPE = "application/vnd.openxmlformats-officedocument.wordprocessingml.document" -MAX_INPUT_BYTES = 15 * 1024 * 1024 -_DEFAULT_TIMEOUT = 30.0 -_admission: ExtractionAdmission | None = None - - -def _error(code: str, status_code: int) -> HTTPException: - return HTTPException(status_code=status_code, detail={"code": code}) - - -def _get_admission() -> ExtractionAdmission: - global _admission - if _admission is None: - _admission = ExtractionAdmission( - concurrent=int(os.getenv("RAG_EXTRACTION_CONCURRENT", "2")), - queued=int(os.getenv("RAG_EXTRACTION_QUEUED", "6")), - ) - return _admission - - -async def _save_bounded(file: UploadFile, path: Path) -> None: - size = 0 - async with aiofiles.open(path, "wb") as output: - while chunk := await file.read(64 * 1024): - size += len(chunk) - if size > MAX_INPUT_BYTES: - raise _error("PARSER_INPUT_LIMIT", 413) - await output.write(chunk) - - -@router.post("/extract", response_model=ExtractionResult) -async def extract_document( - request: Request, - file: UploadFile = File(...), - profile: str = Form(...), -) -> ExtractionResult: - # Existing /text remains unchanged. An operator must explicitly enable - # and install this separate profile before moving any LibreChat caller. - if os.getenv("RAG_EXTRACTION_API_ENABLED", "false").lower() not in { - "1", - "true", - "yes", - "on", - }: - raise _error("EXTRACTION_DISABLED", 404) - # Legacy RAG deployments may run without auth; expensive extraction is - # never allowed anonymously, even when those older routes are public. - if not os.getenv("JWT_SECRET") or not getattr(request.state, "user", {}).get("id"): - raise _error("EXTRACTION_AUTH_REQUIRED", 401) - if profile != "document-v1": - raise _error("UNSUPPORTED_PROFILE", 400) - content_type = (file.content_type or "").split(";")[0].strip().lower() - extension = Path(file.filename or "").suffix.lower() - if content_type == "application/pdf" or not ( - content_type == DOCX_TYPE - or ( - content_type in {"application/octet-stream", "binary/octet-stream", ""} - and extension == ".docx" - ) - ): - raise _error("UNSUPPORTED_DOCUMENT_TYPE", 415) - if file.size is not None and file.size > MAX_INPUT_BYTES: - raise _error("PARSER_INPUT_LIMIT", 413) - - try: - admission = _get_admission() - timeout = float( - os.getenv("RAG_EXTRACTION_TIMEOUT_SECONDS", str(_DEFAULT_TIMEOUT)) - ) - if not math.isfinite(timeout) or timeout <= 0: - raise ValueError("Invalid extraction timeout") - async with asyncio.timeout(timeout): - async with admission.slot(): - fd, filename = tempfile.mkstemp( - prefix="rag-extract-", suffix=".docx", dir=RAG_UPLOAD_DIR - ) - os.close(fd) - path = Path(filename) - try: - await _save_bounded(file, path) - return await run_worker(path) - finally: - path.unlink(missing_ok=True) - except ExtractionBusy: - raise _error("CONCURRENCY_LIMIT", 429) - except ExtractionFailure as exc: - status_code = { - "ZIP_BOMB": 413, - "ARCHIVE_INVALID": 422, - "PARSER_OUTPUT_LIMIT": 413, - "NO_DOCUMENT_TEXT": 422, - "PARSE_FAILED": 422, - "PARSER_UNAVAILABLE": 503, - "PARSER_CRASH": 503, - }[exc.code] - raise _error(exc.code, status_code) - except TimeoutError: - raise _error("PARSER_TIMEOUT", 504) - except (OSError, ValueError) as exc: - logger.error( - "Extraction infrastructure unavailable | error=%s", type(exc).__name__ - ) - raise _error("PARSER_UNAVAILABLE", 503) diff --git a/app/services/extraction.py b/app/services/extraction.py deleted file mode 100644 index 06148362..00000000 --- a/app/services/extraction.py +++ /dev/null @@ -1,100 +0,0 @@ -"""Bounded, cancellable process boundary for opt-in document extraction.""" - -import asyncio -import json -import sys -from contextlib import asynccontextmanager -from pathlib import Path -from typing import AsyncIterator - -from app.models import ExtractionResult - -MAX_IPC_BYTES = 15 * 1024 * 1024 - - -class ExtractionBusy(Exception): - pass - - -class ExtractionFailure(Exception): - def __init__(self, code: str): - self.code = code - - -class ExtractionAdmission: - """One per serving process: bound uploads waiting and native children running.""" - - def __init__(self, concurrent: int = 2, queued: int = 6): - if concurrent < 1 or queued < 0: - raise ValueError("Invalid extraction admission limits") - self._capacity = concurrent + queued - self._pending = 0 - self._slots = asyncio.Semaphore(concurrent) - - @asynccontextmanager - async def slot(self) -> AsyncIterator[None]: - # The serving process has one event loop. No await separates the check - # and increment, so concurrent requests cannot exceed the queue limit. - if self._pending >= self._capacity: - raise ExtractionBusy() - self._pending += 1 - acquired = False - try: - await self._slots.acquire() - acquired = True - yield - finally: - self._pending -= 1 - if acquired: - self._slots.release() - - -def _command(path: Path) -> tuple[str, ...]: - return (sys.executable, "-m", "app.services.extraction_worker", str(path)) - - -async def run_worker(path: Path) -> ExtractionResult: - """Read a bounded child response; always reap a child before releasing its slot.""" - process = await asyncio.create_subprocess_exec( - *_command(path), - stdout=asyncio.subprocess.PIPE, - stderr=asyncio.subprocess.DEVNULL, - cwd=str(Path(__file__).resolve().parents[2]), - ) - try: - output = bytearray() - while chunk := await process.stdout.read(64 * 1024): - output.extend(chunk) - if len(output) > MAX_IPC_BYTES: - raise ExtractionFailure("PARSER_OUTPUT_LIMIT") - await process.wait() - except BaseException: - if process.returncode is None: - try: - process.kill() - except ProcessLookupError: - pass # The child exited while cancellation was being delivered. - # Draining and reaping are necessary before the temporary file can be - # removed and admission can be granted to the next upload. - await process.communicate() - raise - if process.returncode != 0: - raise ExtractionFailure("PARSER_CRASH") - try: - message = json.loads(output) - if message.get("ok") is False: - code = message["code"] - if code in { - "ZIP_BOMB", - "ARCHIVE_INVALID", - "PARSER_UNAVAILABLE", - "PARSER_OUTPUT_LIMIT", - "NO_DOCUMENT_TEXT", - "PARSE_FAILED", - }: - raise ExtractionFailure(code) - return ExtractionResult.model_validate(message["result"]) - except ExtractionFailure: - raise - except (AttributeError, KeyError, TypeError, ValueError) as exc: - raise ExtractionFailure("PARSER_CRASH") from exc diff --git a/app/services/extraction_worker.py b/app/services/extraction_worker.py deleted file mode 100644 index 8bd827be..00000000 --- a/app/services/extraction_worker.py +++ /dev/null @@ -1,136 +0,0 @@ -"""Isolated document parsing. Run only as ``python -m app.services.extraction_worker``. - -The web process does not import native parsing bindings. Child exit, crash, and -SIGKILL cannot terminate the API process or leave its event loop blocked. -""" - -import json -import sys -import zipfile -from importlib.metadata import version -from pathlib import Path - -MAX_ARCHIVE_ENTRIES = 4096 -MAX_ENTRY_BYTES = 25 * 1024 * 1024 -MAX_TOTAL_BYTES = 100 * 1024 * 1024 -MAX_OUTPUT_BYTES = 15 * 1024 * 1024 -IMAGE_EXTENSIONS = ( - ".jpg", - ".jpeg", - ".png", - ".gif", - ".tif", - ".tiff", - ".bmp", - ".webp", - ".jp2", - ".jpx", - ".avif", - ".heic", - ".heif", - ".emf", - ".wmf", - ".svg", -) - - -class ExtractionRefusal(Exception): - def __init__(self, code: str): - self.code = code - - -def inspect_docx(path: Path) -> bool: - """Validate *actually inflated* archive bytes before handing any to AnyDoc. - - Reading in 64-KiB chunks catches false central-directory sizes without - keeping decompressed entries in memory. This is deliberately separate from - the text-output limit, since even an empty parse can inflate a zip bomb. - """ - try: - with zipfile.ZipFile(path) as archive: - entries = archive.infolist() - if len(entries) > MAX_ARCHIVE_ENTRIES: - raise ExtractionRefusal("ZIP_BOMB") - names = {entry.filename for entry in entries} - if not {"[Content_Types].xml", "word/document.xml"} <= names: - raise ExtractionRefusal("ARCHIVE_INVALID") - total = 0 - may_omit_content = False - for entry in entries: - if entry.is_dir(): - continue - if ( - entry.file_size > MAX_ENTRY_BYTES - or total + entry.file_size > MAX_TOTAL_BYTES - ): - raise ExtractionRefusal("ZIP_BOMB") - name = entry.filename.lower() - if not name.startswith(("docprops/", "thumbnails/")) and name.endswith( - IMAGE_EXTENSIONS - ): - may_omit_content = True - entry_bytes = 0 - with archive.open(entry) as stream: - while chunk := stream.read(64 * 1024): - entry_bytes += len(chunk) - total += len(chunk) - if entry_bytes > MAX_ENTRY_BYTES or total > MAX_TOTAL_BYTES: - raise ExtractionRefusal("ZIP_BOMB") - return may_omit_content - except ( - zipfile.BadZipFile, - zipfile.LargeZipFile, - RuntimeError, - EOFError, - NotImplementedError, - OSError, - ) as exc: - raise ExtractionRefusal("ARCHIVE_INVALID") from exc - - -def extract(path: Path) -> dict: - may_omit_content = inspect_docx(path) - try: - import anydoc - - # The MIME/extension are never passed to the binding. DOCX identity was - # checked in the archive above; explicit format avoids filename fallback. - text = anydoc.to_markdown_bytes(path.read_bytes(), "docx") - except ImportError as exc: - raise ExtractionRefusal("PARSER_UNAVAILABLE") from exc - except Exception as exc: - # Native error messages can echo source content. Never send them to clients. - raise ExtractionRefusal("PARSE_FAILED") from exc - if not isinstance(text, str) or not text.strip(): - raise ExtractionRefusal("NO_DOCUMENT_TEXT") - if len(text.encode("utf-8")) > MAX_OUTPUT_BYTES: - raise ExtractionRefusal("PARSER_OUTPUT_LIMIT") - return { - "profile": "document-v1", - "text": text, - "format": "markdown", - "completeness": "partial" if may_omit_content else "complete", - "may_omit_content": may_omit_content, - "pages_needing_ocr": [], - "truncated": False, - "parser": {"name": "anydoc", "version": version("firecrawl-anydoc")}, - } - - -def main() -> None: - if len(sys.argv) != 2: - raise SystemExit(2) - try: - payload = {"ok": True, "result": extract(Path(sys.argv[1]))} - except ExtractionRefusal as exc: - payload = {"ok": False, "code": exc.code} - except Exception: - payload = {"ok": False, "code": "PARSE_FAILED"} - serialized = json.dumps(payload, ensure_ascii=False) - if len(serialized.encode("utf-8")) > MAX_OUTPUT_BYTES: - serialized = json.dumps({"ok": False, "code": "PARSER_OUTPUT_LIMIT"}) - sys.stdout.write(serialized) - - -if __name__ == "__main__": - main() diff --git a/main.py b/main.py index 714f5752..e300951a 100644 --- a/main.py +++ b/main.py @@ -23,7 +23,7 @@ vector_store, ) from app.middleware import security_middleware -from app.routes import document_routes, extraction_routes, pgvector_routes +from app.routes import document_routes, pgvector_routes from app.services.database import PSQLDatabase, ensure_vector_indexes from app.services.vector_store.factory import close_vector_store_connections @@ -90,13 +90,6 @@ async def lifespan(app: FastAPI): # Include routers app.include_router(document_routes.router) -if os.getenv("RAG_EXTRACTION_API_ENABLED", "false").lower() in { - "1", - "true", - "yes", - "on", -}: - app.include_router(extraction_routes.router) if debug_mode: app.include_router(router=pgvector_routes.router) diff --git a/requirements.extraction.txt b/requirements.extraction.txt deleted file mode 100644 index e8276a3a..00000000 --- a/requirements.extraction.txt +++ /dev/null @@ -1,3 +0,0 @@ -# Optional document-v1 extraction engine; install only when enabling /v1/extract. -# Match LibreChat AnyDoc PR #14701 (Node @firecrawl/anydoc 0.1.3) for corpus parity. -firecrawl-anydoc==0.1.3 diff --git a/service/Dockerfile b/service/Dockerfile new file mode 100644 index 00000000..27985f22 --- /dev/null +++ b/service/Dockerfile @@ -0,0 +1,13 @@ +FROM oven/bun:1.4.2-slim AS dependencies +WORKDIR /app +COPY service/package.json service/bun.lock ./ +RUN bun install --frozen-lockfile --production + +FROM oven/bun:1.4.2-slim +WORKDIR /app +COPY --from=dependencies --chown=bun:bun /app/node_modules ./node_modules +COPY --chown=bun:bun service/package.json ./package.json +COPY --chown=bun:bun service/src ./src +USER bun +EXPOSE 8001 +CMD ["bun", "run", "src/server.ts"] diff --git a/service/bun.lock b/service/bun.lock new file mode 100644 index 00000000..393feaf8 --- /dev/null +++ b/service/bun.lock @@ -0,0 +1,74 @@ +{ + "lockfileVersion": 2, + "configVersion": 1, + "workspaces": { + "": { + "name": "@librechat/rag-service", + "dependencies": { + "@firecrawl/anydoc": "0.1.3", + "busboy": "1.6.0", + "hono": "4.13.11", + "jose": "6.2.12", + "yauzl": "3.4.0", + "zod": "4.3.6", + }, + "devDependencies": { + "@types/bun": "1.4.2", + "@types/busboy": "1.5.4", + "@types/yauzl": "3.4.0", + "fflate": "0.8.2", + "prettier": "3.8.1", + "typescript": "5.9.3", + }, + }, + }, + "packages": { + "@firecrawl/anydoc": ["@firecrawl/anydoc@0.1.3", "", { "optionalDependencies": { "@firecrawl/anydoc-darwin-arm64": "0.1.3", "@firecrawl/anydoc-darwin-x64": "0.1.3", "@firecrawl/anydoc-linux-arm64-gnu": "0.1.3", "@firecrawl/anydoc-linux-arm64-musl": "0.1.3", "@firecrawl/anydoc-linux-x64-gnu": "0.1.3", "@firecrawl/anydoc-linux-x64-musl": "0.1.3", "@firecrawl/anydoc-win32-x64-msvc": "0.1.3" }, "bin": { "anydoc": "cli.js" } }, "sha512-OuYZWJHxiSNxU414g3l4w0gKb5BTqVZKGCwgmrkn8Mq84lf5z1Y2pHcTWQOWbyN4qD2NX7hQhrnozGhYYql6sw=="], + + "@firecrawl/anydoc-darwin-arm64": ["@firecrawl/anydoc-darwin-arm64@0.1.3", "", { "os": "darwin", "cpu": "arm64" }, "sha512-tCi0pqATVbS4el+lF/XBGvXzMUzc1dLTQd6eZxctLoP5HBLyQsmBQiSDdU+dE5reVpriLdumAo3MRhSJVBfikw=="], + + "@firecrawl/anydoc-darwin-x64": ["@firecrawl/anydoc-darwin-x64@0.1.3", "", { "os": "darwin", "cpu": "x64" }, "sha512-WpegrFgCZfXgbDBsrHrZcUcJNoB31fL7ztGL0x3OCThGNcND/JFOAb0ihS3Y/uwbHvXJDZTt+ZzAQOhSoC/8Rg=="], + + "@firecrawl/anydoc-linux-arm64-gnu": ["@firecrawl/anydoc-linux-arm64-gnu@0.1.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-5/QCKECO2ujVZ7b3JOcNyvZrH1uSB3I/uLmgN05FWEIERA3rrwdzZi/k6K7ULcmSOD4eMJDfd6g41Ka48gW6zw=="], + + "@firecrawl/anydoc-linux-arm64-musl": ["@firecrawl/anydoc-linux-arm64-musl@0.1.3", "", { "os": "linux", "cpu": "arm64" }, "sha512-nf/IakkpFAA3wdIoicohgOW+3D3xDr8gGhU5C+QdWDvRTQjK3Bm2R0DLvZJcnuOonsb01ob+BqFb9Q7PqmCOpw=="], + + "@firecrawl/anydoc-linux-x64-gnu": ["@firecrawl/anydoc-linux-x64-gnu@0.1.3", "", { "os": "linux", "cpu": "x64" }, "sha512-VYnBbTIyRwv3HcWZgEaguPMjRs2CdA5aV+Nx60kCBV4Ykp3GOIdiCxJFE0RfOUQN3nfPV6hcaKtXz+1O4jxX0A=="], + + "@firecrawl/anydoc-linux-x64-musl": ["@firecrawl/anydoc-linux-x64-musl@0.1.3", "", { "os": "linux", "cpu": "x64" }, "sha512-gL3dYfVmpN6ixfdmy2Ifgkdd5s3xrIHuwUlTTsTWR15OkhfL9LuA/VkgDO+p82D3SWAVnCTEWI7scE8nkMLmBQ=="], + + "@firecrawl/anydoc-win32-x64-msvc": ["@firecrawl/anydoc-win32-x64-msvc@0.1.3", "", { "os": "win32", "cpu": "x64" }, "sha512-NwjmXqhrL6xVTLAJeZ0Ie7h7Clm2CwSPuJ9wuZX9fPP5nx/p7GvPRDvtLq6PxtXk2LQFWklfHmDczUQmzMh1Kg=="], + + "@types/bun": ["@types/bun@1.4.2", "", { "dependencies": { "bun-types": "1.4.2" } }, "sha512-GimotNn7+ZV0uVArItBbriZsR1oNf0+WTzPkdcFrzShI7k2norL0uzEaJT8T33dWr7O/c9ZDuAFQrctKCi72oQ=="], + + "@types/busboy": ["@types/busboy@1.5.4", "", { "dependencies": { "@types/node": "*" } }, "sha512-kG7WrUuAKK0NoyxfQHsVE6j1m01s6kMma64E+OZenQABMQyTJop1DumUWcLwAQ2JzpefU7PDYoRDKl8uZosFjw=="], + + "@types/node": ["@types/node@26.6.3", "", { "dependencies": { "undici-types": "~8.9.0" } }, "sha512-dsqMQQoeTLqu9wynDD00q573mNzso3IdQOAfHRJqLCcmCFPoGo9A1bDpUcv/9tnKpErQWv9uKeGfl37EIS02Yg=="], + + "@types/yauzl": ["@types/yauzl@3.4.0", "", { "dependencies": { "@types/node": "*" } }, "sha512-NRPn5w6h8dhcnmx3YIRQcqMywY/+nND/uOkJessedcrowO3C0AssHp3tMJpxKAwOhFOo0OV1y9VtsC5hbKKBAw=="], + + "bun-types": ["bun-types@1.4.2", "", { "dependencies": { "@types/node": "*" } }, "sha512-bxV1FgK7yBIzjRe5zBozIM4Bem11ZJcCXSrjWRG3YWLt8yFDePu4cLjpebO8OvPeIE9trbyPF4fuj3Cia4Fj3w=="], + + "busboy": ["busboy@1.6.0", "", { "dependencies": { "streamsearch": "^1.1.0" } }, "sha512-8SFQbg/0hQ9xy3UNTB0YEnsNBbWfhf7RtnzpL7TkBiTBRfrQ9Fxcnz7VJsleJpyp6rVLvXiuORqjlHi5q+PYuA=="], + + "fflate": ["fflate@0.8.2", "", {}, "sha512-cPJU47OaAoCbg0pBvzsgpTPhmhqI5eJjh/JIu8tPj5q+T7iLvW/JAYUqmE7KOB4R1ZyEhzBaIQpQpardBF5z8A=="], + + "hono": ["hono@4.13.11", "", {}, "sha512-/SMX/RQNJn7oNmFwH6DtwDqcrzZU2otm1FD6OJe2cWF6cfy/F8hq0XVBv7xW3FTJshRQwuBnXHDq2kNsdZnshg=="], + + "jose": ["jose@6.2.12", "", {}, "sha512-9NiFmJEex0sy2Dk58j2UGBSHgUs2ypF9eZSu4L6vjOX3Dp96Sw1F3uL+H+D1sx02jZZdzUT0HgvCy59CuvXcWw=="], + + "pend": ["pend@1.2.0", "", {}, "sha512-F3asv42UuXchdzt+xXqfW1OGlVBe+mxa2mqI0pg5yAHZPvFmY3Y6drSf/GQ1A86WgWEN9Kzh/WrgKa6iGcHXLg=="], + + "prettier": ["prettier@3.8.1", "", { "bin": { "prettier": "bin/prettier.cjs" } }, "sha512-UOnG6LftzbdaHZcKoPFtOcCKztrQ57WkHDeRD9t/PTQtmT0NHSeWWepj6pS0z/N7+08BHFDQVUrfmfMRcZwbMg=="], + + "streamsearch": ["streamsearch@1.1.0", "", {}, "sha512-Mcc5wHehp9aXz1ax6bZUyY5afg9u2rv5cqQI3mRrYkGC8rW2hM02jWuwjtL++LS5qinSyhj2QfLyNsuc+VsExg=="], + + "typescript": ["typescript@5.9.3", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw=="], + + "undici-types": ["undici-types@8.9.0", "", {}, "sha512-KTDyRTYX8sWmKXAikPHHSyc63CRPETMctyjKFupcC6OBLXT3xsN0e9aF7m+mIXutFWpUXuedtowG7iLOzp0kQg=="], + + "yauzl": ["yauzl@3.4.0", "", { "dependencies": { "pend": "~1.2.0" } }, "sha512-jIH9yLR9wqr0wOS0TpBvo/g/2UgZH5qePVbjgRliiF0BYvOZyaBknKsF+x9Iht0O6sqgnB93rCICdOZFecJuDw=="], + + "zod": ["zod@4.3.6", "", {}, "sha512-rftlrkhHZOcjDwkGlnUtZZkvaPHCsDATp4pGpuOOMDaTdDDXF91wuVDJoWoPsKX/3YPQ5fHuF3STjcYyKr+Qhg=="], + } +} diff --git a/service/package.json b/service/package.json new file mode 100644 index 00000000..b220840f --- /dev/null +++ b/service/package.json @@ -0,0 +1,30 @@ +{ + "name": "@librechat/rag-service", + "private": true, + "version": "0.1.0", + "type": "module", + "packageManager": "bun@1.4.2", + "scripts": { + "start": "bun run src/server.ts", + "test": "bun test", + "typecheck": "tsc --noEmit", + "format:check": "prettier --check src test *.json", + "format": "prettier --write src test *.json" + }, + "dependencies": { + "@firecrawl/anydoc": "0.1.3", + "busboy": "1.6.0", + "hono": "4.13.11", + "jose": "6.2.12", + "yauzl": "3.4.0", + "zod": "4.3.6" + }, + "devDependencies": { + "@types/bun": "1.4.2", + "@types/busboy": "1.5.4", + "@types/yauzl": "3.4.0", + "fflate": "0.8.2", + "prettier": "3.8.1", + "typescript": "5.9.3" + } +} diff --git a/service/src/admission.ts b/service/src/admission.ts new file mode 100644 index 00000000..ece93250 --- /dev/null +++ b/service/src/admission.ts @@ -0,0 +1,56 @@ +import { ExtractionError } from "./contract"; + +type Release = () => void; +type Waiter = { + signal: AbortSignal; + resolve: (release: Release) => void; + reject: (error: Error) => void; + abort: () => void; +}; + +export class Admission { + private active = 0; + private readonly waiting: Waiter[] = []; + constructor( + private readonly concurrent: number, + private readonly queued: number, + ) {} + + acquire(signal: AbortSignal): Promise { + signal.throwIfAborted(); + if (this.active < this.concurrent) { + this.active++; + return Promise.resolve(this.release()); + } + if (this.waiting.length >= this.queued) { + return Promise.reject(new ExtractionError("CONCURRENCY_LIMIT")); + } + return new Promise((resolve, reject) => { + const waiter: Waiter = { + signal, + resolve, + reject, + abort: () => { + const index = this.waiting.indexOf(waiter); + if (index >= 0) this.waiting.splice(index, 1); + reject(new ExtractionError("REQUEST_CANCELLED")); + }, + }; + signal.addEventListener("abort", waiter.abort, { once: true }); + this.waiting.push(waiter); + }); + } + + private release(): Release { + let released = false; + return () => { + if (released) return; + released = true; + const next = this.waiting.shift(); + if (next) { + next.signal.removeEventListener("abort", next.abort); + next.resolve(this.release()); + } else this.active--; + }; + } +} diff --git a/service/src/app.ts b/service/src/app.ts new file mode 100644 index 00000000..72453bd5 --- /dev/null +++ b/service/src/app.ts @@ -0,0 +1,128 @@ +import { Hono } from "hono"; +import { jwtVerify } from "jose"; +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import type { ContentfulStatusCode } from "hono/utils/http-status"; +import type { Runner } from "./process"; +import { Admission } from "./admission"; +import { configSchema, type Config } from "./config"; +import { ExtractionError, type ErrorCode } from "./contract"; +import { runWorker } from "./process"; +import { stageUpload } from "./upload"; + +const status: Record = { + EXTRACTION_DISABLED: 404, + EXTRACTION_AUTH_REQUIRED: 401, + EXTRACTION_FORBIDDEN: 403, + UNSUPPORTED_PROFILE: 400, + UNSUPPORTED_DOCUMENT_TYPE: 415, + INVALID_MULTIPART: 400, + PARSER_INPUT_LIMIT: 413, + PARSER_OUTPUT_LIMIT: 413, + ZIP_BOMB: 413, + ARCHIVE_INVALID: 422, + NO_DOCUMENT_TEXT: 422, + PARSE_FAILED: 422, + CONCURRENCY_LIMIT: 429, + PARSER_CRASH: 503, + PARSER_UNAVAILABLE: 503, + PARSER_TIMEOUT: 504, + REQUEST_CANCELLED: 408, +}; + +export function createApp( + input: Partial = {}, + runner: Runner = runWorker, +): Hono { + const config = configSchema.parse(input); + const app = new Hono(); + const admission = new Admission(config.concurrent, config.queued); + app.onError((error, context) => { + const code = + error instanceof ExtractionError ? error.code : "PARSER_UNAVAILABLE"; + return context.json({ detail: { code } }, status[code]); + }); + app.notFound((context) => + context.json({ detail: { code: "EXTRACTION_DISABLED" } }, 404), + ); + app.get("/health", (context) => + context.json({ + status: "UP", + service: "rag-bun", + extraction_profiles: config.enabled ? ["document-v1"] : [], + }), + ); + if (!config.enabled) return app; + const key = new TextEncoder().encode(config.secret); + app.post("/v1/extract", async (context) => { + const authorization = context.req.header("Authorization"); + if (!authorization?.startsWith("Bearer ")) + throw new ExtractionError("EXTRACTION_AUTH_REQUIRED"); + try { + const { payload } = await jwtVerify(authorization.slice(7), key, { + algorithms: ["HS256"], + issuer: config.issuer, + audience: config.audience, + requiredClaims: ["exp", "sub"], + }); + if (typeof payload.sub !== "string" || !payload.sub) + throw new ExtractionError("EXTRACTION_AUTH_REQUIRED"); + if ( + !Array.isArray(payload.scopes) || + !payload.scopes.includes("rag:documents") + ) { + throw new ExtractionError("EXTRACTION_FORBIDDEN"); + } + } catch (error) { + if (error instanceof ExtractionError) throw error; + throw new ExtractionError("EXTRACTION_AUTH_REQUIRED"); + } + const length = context.req.header("Content-Length"); + if ( + length !== undefined && + (!/^\d+$/.test(length) || Number(length) > config.maxBodyBytes) + ) { + throw new ExtractionError("PARSER_INPUT_LIMIT"); + } + const deadline = new AbortController(); + const timer = setTimeout(() => deadline.abort(), config.timeoutMs); + const signal = AbortSignal.any([context.req.raw.signal, deadline.signal]); + let release: (() => void) | undefined; + let directory: string | undefined; + try { + release = await admission.acquire(signal); + signal.throwIfAborted(); + directory = await mkdtemp( + join(config.tempRoot ?? tmpdir(), "rag-extract-"), + ); + signal.throwIfAborted(); + const path = join(directory, "input.docx"); + await stageUpload(context.req.raw, path, config, signal); + const result = await runner( + { + path, + maxOutputBytes: config.maxOutputBytes, + maxEntryBytes: config.maxEntryBytes, + maxArchiveBytes: config.maxArchiveBytes, + maxEntries: config.maxEntries, + }, + signal, + ); + signal.throwIfAborted(); + return context.json(result); + } catch (error) { + if (deadline.signal.aborted) throw new ExtractionError("PARSER_TIMEOUT"); + if (signal.aborted) throw new ExtractionError("REQUEST_CANCELLED"); + throw error; + } finally { + clearTimeout(timer); + try { + if (directory) await rm(directory, { recursive: true, force: true }); + } finally { + release?.(); + } + } + }); + return app; +} diff --git a/service/src/config.ts b/service/src/config.ts new file mode 100644 index 00000000..d57e28c2 --- /dev/null +++ b/service/src/config.ts @@ -0,0 +1,64 @@ +import { z } from "zod"; + +const positive = z.number().int().positive(); +export const configSchema = z + .object({ + enabled: z.boolean().default(false), + secret: z.string().min(32).optional(), + issuer: z.string().min(1).default("librechat"), + audience: z.string().min(1).default("rag-api"), + concurrent: positive.default(2), + queued: z.number().int().nonnegative().default(6), + timeoutMs: positive.max(2_147_483_647).default(30_000), + maxFileBytes: positive.default(15 * 1024 * 1024), + maxBodyBytes: positive.default(16 * 1024 * 1024), + maxOutputBytes: positive.default(15 * 1024 * 1024), + maxEntryBytes: positive.default(25 * 1024 * 1024), + maxArchiveBytes: positive.default(100 * 1024 * 1024), + maxEntries: positive.default(4096), + tempRoot: z.string().min(1).optional(), + }) + .superRefine((config, ctx) => { + if (config.enabled && !config.secret) { + ctx.addIssue({ + code: "custom", + message: "Enabled extraction requires RAG_JWT_SECRET (32+ characters)", + }); + } + if (config.maxBodyBytes <= config.maxFileBytes) { + ctx.addIssue({ + code: "custom", + message: + "Body ceiling must exceed the file ceiling for multipart framing", + }); + } + }); +export type Config = z.infer; + +export function fromEnv(env: NodeJS.ProcessEnv): Config { + const number = (name: string) => + env[name] === undefined ? undefined : Number(env[name]); + const enabled = env.RAG_EXTRACTION_API_ENABLED; + if ( + enabled !== undefined && + !["true", "false", "1", "0"].includes(enabled.toLowerCase()) + ) { + throw new Error("RAG_EXTRACTION_API_ENABLED must be true, false, 1 or 0"); + } + return configSchema.parse({ + enabled: enabled === "1" || enabled?.toLowerCase() === "true", + secret: env.RAG_JWT_SECRET, + issuer: env.RAG_JWT_ISSUER, + audience: env.RAG_JWT_AUDIENCE, + concurrent: number("RAG_EXTRACTION_CONCURRENT"), + queued: number("RAG_EXTRACTION_QUEUED"), + timeoutMs: number("RAG_EXTRACTION_TIMEOUT_MS"), + maxFileBytes: number("RAG_EXTRACTION_MAX_FILE_BYTES"), + maxBodyBytes: number("RAG_EXTRACTION_MAX_BODY_BYTES"), + maxOutputBytes: number("RAG_EXTRACTION_MAX_OUTPUT_BYTES"), + maxEntryBytes: number("RAG_EXTRACTION_MAX_ENTRY_BYTES"), + maxArchiveBytes: number("RAG_EXTRACTION_MAX_ARCHIVE_BYTES"), + maxEntries: number("RAG_EXTRACTION_MAX_ENTRIES"), + tempRoot: env.RAG_EXTRACTION_TEMP_DIR, + }); +} diff --git a/service/src/contract.ts b/service/src/contract.ts new file mode 100644 index 00000000..34e2142a --- /dev/null +++ b/service/src/contract.ts @@ -0,0 +1,55 @@ +import { z } from "zod"; + +export const resultSchema = z + .object({ + profile: z.literal("document-v1"), + text: z.string().min(1), + format: z.literal("markdown"), + completeness: z.enum(["complete", "partial"]), + may_omit_content: z.boolean(), + pages_needing_ocr: z.array(z.number().int().positive()), + truncated: z.literal(false), + parser: z.object({ + name: z.literal("anydoc"), + version: z.literal("0.1.3"), + }), + }) + .strict() + .superRefine((result, ctx) => { + if ((result.completeness === "partial") !== result.may_omit_content) { + ctx.addIssue({ code: "custom", message: "Inconsistent completeness" }); + } + }); +export type ExtractionResult = z.infer; + +export const errorSchema = z.enum([ + "EXTRACTION_DISABLED", + "EXTRACTION_AUTH_REQUIRED", + "EXTRACTION_FORBIDDEN", + "UNSUPPORTED_PROFILE", + "UNSUPPORTED_DOCUMENT_TYPE", + "INVALID_MULTIPART", + "PARSER_INPUT_LIMIT", + "PARSER_OUTPUT_LIMIT", + "ZIP_BOMB", + "ARCHIVE_INVALID", + "NO_DOCUMENT_TEXT", + "PARSE_FAILED", + "CONCURRENCY_LIMIT", + "PARSER_CRASH", + "PARSER_UNAVAILABLE", + "PARSER_TIMEOUT", + "REQUEST_CANCELLED", +]); +export type ErrorCode = z.infer; +export class ExtractionError extends Error { + constructor(readonly code: ErrorCode) { + super(code); + } +} +export const workerResponseSchema = z.discriminatedUnion("ok", [ + z.object({ ok: z.literal(true), result: resultSchema }), + z.object({ ok: z.literal(false), code: errorSchema }), +]); +export const DOCX_TYPE = + "application/vnd.openxmlformats-officedocument.wordprocessingml.document"; diff --git a/service/src/process.ts b/service/src/process.ts new file mode 100644 index 00000000..cc271943 --- /dev/null +++ b/service/src/process.ts @@ -0,0 +1,67 @@ +import { fileURLToPath } from "node:url"; +import type { WorkerRequest } from "./worker"; +import { + ExtractionError, + workerResponseSchema, + type ExtractionResult, +} from "./contract"; + +const workerPath = fileURLToPath(new URL("./worker.ts", import.meta.url)); +export type Runner = ( + request: WorkerRequest, + signal: AbortSignal, +) => Promise; + +export async function runWorker( + request: WorkerRequest, + signal: AbortSignal, + command: readonly string[] = [process.execPath, workerPath], +): Promise { + signal.throwIfAborted(); + const child = Bun.spawn([...command], { + stdin: "pipe", + stdout: "pipe", + stderr: "ignore", + // Native parsing gets no signing key, provider credential or service config. + env: { PATH: process.env.PATH ?? "" }, + }); + const abort = () => { + child.kill("SIGKILL"); + }; + signal.addEventListener("abort", abort, { once: true }); + const reader = child.stdout.getReader(); + const chunks: Uint8Array[] = []; + let size = 0; + try { + signal.throwIfAborted(); + child.stdin.write(JSON.stringify(request)); + child.stdin.end(); + while (true) { + const part = await reader.read(); + if (part.done) break; + size += part.value.byteLength; + if (size > request.maxOutputBytes) + throw new ExtractionError("PARSER_OUTPUT_LIMIT"); + chunks.push(part.value); + } + const exit = await child.exited; + signal.throwIfAborted(); + if (exit !== 0) throw new ExtractionError("PARSER_CRASH"); + const result = workerResponseSchema.safeParse( + JSON.parse(Buffer.concat(chunks, size).toString("utf8")), + ); + if (!result.success) throw new ExtractionError("PARSER_CRASH"); + if (!result.data.ok) throw new ExtractionError(result.data.code); + return result.data.result; + } catch (error) { + child.kill("SIGKILL"); + await child.exited; + await reader.cancel().catch(() => {}); + if (signal.aborted) throw new ExtractionError("REQUEST_CANCELLED"); + if (error instanceof ExtractionError) throw error; + throw new ExtractionError("PARSER_CRASH"); + } finally { + signal.removeEventListener("abort", abort); + reader.releaseLock(); + } +} diff --git a/service/src/server.ts b/service/src/server.ts new file mode 100644 index 00000000..d7375ef4 --- /dev/null +++ b/service/src/server.ts @@ -0,0 +1,29 @@ +import { createApp } from "./app"; +import { fromEnv } from "./config"; + +if (import.meta.main) { + try { + const config = fromEnv(process.env); + const port = Number(process.env.RAG_PORT ?? 8001); + if (!Number.isInteger(port) || port < 1 || port > 65535) + throw new Error("Invalid port"); + const server = Bun.serve({ + hostname: process.env.RAG_HOST ?? "0.0.0.0", + port, + maxRequestBodySize: config.maxBodyBytes, + fetch: createApp(config).fetch, + }); + // Graceful shutdown lets active requests finish; their deadlines stay bounded. + for (const event of ["SIGINT", "SIGTERM"] as const) { + process.once(event, () => { + void server.stop(false); + }); + } + console.info(`RAG Bun service listening on port ${server.port}`); + } catch { + console.error( + "Unable to start RAG Bun service: check runtime configuration", + ); + process.exitCode = 1; + } +} diff --git a/service/src/upload.ts b/service/src/upload.ts new file mode 100644 index 00000000..938b60d0 --- /dev/null +++ b/service/src/upload.ts @@ -0,0 +1,126 @@ +import busboy from "busboy"; +import { createWriteStream } from "node:fs"; +import { basename, extname } from "node:path"; +import { Readable } from "node:stream"; +import { pipeline } from "node:stream/promises"; +import type { Config } from "./config"; +import { DOCX_TYPE, ExtractionError } from "./contract"; + +export async function stageUpload( + request: Request, + path: string, + config: Config, + signal: AbortSignal, +): Promise { + if (!request.body) throw new ExtractionError("INVALID_MULTIPART"); + let parser: ReturnType; + try { + parser = busboy({ + headers: { "content-type": request.headers.get("content-type") ?? "" }, + limits: { + fileSize: config.maxFileBytes, + files: 1, + fields: 1, + parts: 3, + fieldSize: 64, + }, + }); + } catch { + throw new ExtractionError("INVALID_MULTIPART"); + } + const reader = request.body.getReader(); + const abort = () => { + void reader.cancel().catch(() => {}); + }; + signal.addEventListener("abort", abort, { once: true }); + let bytes = 0; + const input = Readable.from( + (async function* () { + try { + while (true) { + signal.throwIfAborted(); + const part = await reader.read(); + if (part.done) break; + bytes += part.value.byteLength; + if (bytes > config.maxBodyBytes) + throw new ExtractionError("PARSER_INPUT_LIMIT"); + yield part.value; + } + } finally { + reader.releaseLock(); + } + })(), + ); + let profile: string | undefined; + let fileCount = 0; + let refusal: ExtractionError | undefined; + const writes: Promise[] = []; + const fail = (error: ExtractionError) => { + refusal ??= error; + parser.destroy(error); + }; + parser.on("field", (name, value, info) => { + if (name !== "profile" || info.valueTruncated || profile !== undefined) { + fail(new ExtractionError("INVALID_MULTIPART")); + } else profile = value; + }); + parser.on("file", (name, stream, info) => { + // Busboy destroys its current file stream when the parser is refused, even + // before a disk pipeline exists. Refused streams need an error listener too. + stream.on("error", () => {}); + fileCount++; + const extension = extname(basename(info.filename)).toLowerCase(); + const generic = [ + "application/octet-stream", + "binary/octet-stream", + ].includes(info.mimeType); + if (name !== "file" || fileCount !== 1) { + stream.resume(); + fail(new ExtractionError("INVALID_MULTIPART")); + return; + } + if (info.mimeType !== DOCX_TYPE && !(generic && extension === ".docx")) { + stream.resume(); + fail(new ExtractionError("UNSUPPORTED_DOCUMENT_TYPE")); + return; + } + stream.once("limit", () => fail(new ExtractionError("PARSER_INPUT_LIMIT"))); + const write = pipeline( + stream, + createWriteStream(path, { flags: "wx", mode: 0o600 }), + { signal }, + ); + // Attach rejection handling at creation, not only after the multipart parser + // finishes, so a storage failure terminates upload and never becomes unhandled. + writes.push( + write.catch((error: Error) => { + parser.destroy(error); + throw error; + }), + ); + void writes.at(-1)?.catch(() => {}); + }); + for (const event of ["filesLimit", "fieldsLimit", "partsLimit"] as const) { + parser.on(event, () => fail(new ExtractionError("INVALID_MULTIPART"))); + } + try { + await pipeline(input, parser, { signal }); + await Promise.all(writes); + signal.throwIfAborted(); + if (fileCount !== 1 || profile === undefined) + throw new ExtractionError("INVALID_MULTIPART"); + if (profile !== "document-v1") + throw new ExtractionError("UNSUPPORTED_PROFILE"); + } catch (error) { + input.destroy(); + parser.destroy(); + await reader.cancel().catch(() => {}); + await Promise.allSettled(writes); + if (refusal) throw refusal; + if (error instanceof ExtractionError) throw error; + if (signal.aborted) throw new ExtractionError("REQUEST_CANCELLED"); + throw new ExtractionError("INVALID_MULTIPART"); + } finally { + signal.removeEventListener("abort", abort); + } +} diff --git a/service/src/worker.ts b/service/src/worker.ts new file mode 100644 index 00000000..62e24655 --- /dev/null +++ b/service/src/worker.ts @@ -0,0 +1,159 @@ +import { readFile } from "node:fs/promises"; +import { createRequire } from "node:module"; +import { open, type Entry, type ZipFile } from "yauzl"; +import { z } from "zod"; +import type { ExtractionResult } from "./contract"; +import { ExtractionError } from "./contract"; + +const requestSchema = z.object({ + path: z.string(), + maxOutputBytes: z.number().int().positive(), + maxEntryBytes: z.number().int().positive(), + maxArchiveBytes: z.number().int().positive(), + maxEntries: z.number().int().positive(), +}); +export type WorkerRequest = z.infer; +const IMAGE = + /\.(?:jpe?g|png|gif|tiff?|bmp|webp|jp2|jpx|avif|heic|heif|emf|wmf|svg)$/i; +const PREVIEW = /^(?:docProps|Thumbnails)\//i; + +function inspectArchive(path: string, limits: WorkerRequest): Promise { + return new Promise((resolve, reject) => { + open( + path, + { lazyEntries: true, validateEntrySizes: true }, + (error, zip) => { + if (error || !zip) { + reject(new ExtractionError("ARCHIVE_INVALID")); + return; + } + let settled = false; + let total = 0; + let media = false; + const names = new Set(); + const fail = (code: "ARCHIVE_INVALID" | "ZIP_BOMB") => { + if (settled) return; + settled = true; + zip.close(); + reject(new ExtractionError(code)); + }; + zip.on("error", () => fail("ARCHIVE_INVALID")); + zip.on("end", () => { + if (settled) return; + if ( + !names.has("[Content_Types].xml") || + !names.has("word/document.xml") + ) { + fail("ARCHIVE_INVALID"); + return; + } + settled = true; + resolve(media); + }); + if (zip.entryCount > limits.maxEntries) { + fail("ZIP_BOMB"); + return; + } + zip.on("entry", (entry: Entry) => { + if (names.has(entry.fileName)) { + fail("ARCHIVE_INVALID"); + return; + } + names.add(entry.fileName); + if (/\/$/.test(entry.fileName)) { + zip.readEntry(); + return; + } + if ( + entry.uncompressedSize > limits.maxEntryBytes || + total + entry.uncompressedSize > limits.maxArchiveBytes + ) { + fail("ZIP_BOMB"); + return; + } + if ( + (IMAGE.test(entry.fileName) && !PREVIEW.test(entry.fileName)) || + /^word\/embeddings\//i.test(entry.fileName) + ) + media = true; + zip.openReadStream(entry, (streamError, stream) => { + if (streamError || !stream) { + fail("ARCHIVE_INVALID"); + return; + } + let bytes = 0; + stream.on("data", (chunk: Buffer) => { + bytes += chunk.byteLength; + total += chunk.byteLength; + if ( + bytes > limits.maxEntryBytes || + total > limits.maxArchiveBytes + ) { + stream.destroy(); + fail("ZIP_BOMB"); + } + }); + stream.on("error", () => fail("ARCHIVE_INVALID")); + stream.on("end", () => { + if (!settled) zip.readEntry(); + }); + }); + }); + zip.readEntry(); + }, + ); + }); +} + +async function extract(request: WorkerRequest): Promise { + const media = await inspectArchive(request.path, request); + // Loading a native binding is itself isolated and only happens after refusal guards. + const require = createRequire(import.meta.url); + let anydoc: typeof import("@firecrawl/anydoc"); + try { + anydoc = require("@firecrawl/anydoc"); + } catch { + throw new ExtractionError("PARSER_UNAVAILABLE"); + } + const bytes = await readFile(request.path); + let text: string; + try { + text = await anydoc.toMarkdownBytes( + bytes, + "docx" as import("@firecrawl/anydoc").Format, + ); + } catch { + throw new ExtractionError("PARSE_FAILED"); + } + if (!text.trim()) throw new ExtractionError("NO_DOCUMENT_TEXT"); + if (Buffer.byteLength(text) > request.maxOutputBytes) + throw new ExtractionError("PARSER_OUTPUT_LIMIT"); + return { + profile: "document-v1", + text, + format: "markdown", + completeness: media ? "partial" : "complete", + may_omit_content: media, + pages_needing_ocr: [], + truncated: false, + parser: { name: "anydoc", version: "0.1.3" }, + }; +} + +if (import.meta.main) { + let maxOutput = 15 * 1024 * 1024; + let serialized: string; + try { + const request = requestSchema.parse(JSON.parse(await Bun.stdin.text())); + maxOutput = request.maxOutputBytes; + serialized = JSON.stringify({ ok: true, result: await extract(request) }); + if (Buffer.byteLength(serialized) > maxOutput) + throw new ExtractionError("PARSER_OUTPUT_LIMIT"); + } catch (error) { + serialized = JSON.stringify({ + ok: false, + code: error instanceof ExtractionError ? error.code : "PARSE_FAILED", + }); + } + await Bun.write(Bun.stdout, serialized); +} diff --git a/service/test/extraction.test.ts b/service/test/extraction.test.ts new file mode 100644 index 00000000..adaa5726 --- /dev/null +++ b/service/test/extraction.test.ts @@ -0,0 +1,387 @@ +import { afterEach, beforeEach, expect, test } from "bun:test"; +import { SignJWT } from "jose"; +import { unzipSync, zipSync } from "fflate"; +import { mkdtemp, readdir, readFile, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { createApp } from "../src/app"; +import { configSchema, fromEnv } from "../src/config"; +import { DOCX_TYPE, resultSchema } from "../src/contract"; +import { runWorker, type Runner } from "../src/process"; + +const fixture = await Bun.file( + new URL("../../tests/fixtures/structured.docx", import.meta.url), +).bytes(); +const expectedText = + "# Quarterly Report\n\nThis document summarizes the results for the period.\n\n## Regional Totals\n\n| | | |\n| --- | --- | --- |\n| Region | Units | Revenue |\n| North | 1200 | 48000 |\n| South | 950 | 38000 |\n| East | 1430 | 57200 |\n\n**Totals are unaudited.**\n"; +const secret = "test-rag-key-with-more-than-32-characters"; +let tempRoot: string; +beforeEach(async () => { + tempRoot = await mkdtemp(join(tmpdir(), "rag-bun-test-")); +}); +afterEach(async () => { + await rm(tempRoot, { recursive: true, force: true }); +}); + +async function token( + options: { + scopes?: string[]; + key?: string; + audience?: string; + expiry?: number; + subject?: string; + } = {}, +) { + return new SignJWT({ scopes: options.scopes ?? ["rag:documents"] }) + .setProtectedHeader({ alg: "HS256" }) + .setIssuer("librechat") + .setAudience(options.audience ?? "rag-api") + .setSubject(options.subject ?? "owner") + .setExpirationTime(options.expiry ?? Math.floor(Date.now() / 1000) + 60) + .sign(new TextEncoder().encode(options.key ?? secret)); +} +function app(overrides: Parameters[0] = {}, runner?: Runner) { + return createApp({ enabled: true, secret, tempRoot, ...overrides }, runner); +} +function form( + bytes: Uint8Array = fixture, + mime = DOCX_TYPE, + name = "report.docx", + profile = "document-v1", +) { + const body = new FormData(); + body.append("profile", profile); + body.append("file", new File([new Uint8Array(bytes)], name, { type: mime })); + return body; +} +async function post( + application = app(), + body: BodyInit = form(), + jwt?: string, + signal?: AbortSignal, +) { + return application.request("/v1/extract", { + method: "POST", + headers: { Authorization: `Bearer ${jwt ?? (await token())}` }, + body, + signal, + }); +} +async function code(response: Response, status: number, error: string) { + expect(response.status).toBe(status); + expect(await response.json()).toEqual({ detail: { code: error } }); +} +async function clean() { + expect(await readdir(tempRoot)).toEqual([]); +} +function extraEntry(name: string, bytes: Uint8Array) { + return zipSync({ ...unzipSync(fixture), [name]: bytes }); +} + +// Same golden output as PR 330's Python prototype and Marco's pinned AnyDoc fixture. +test("real AnyDoc DOCX preserves the exact versioned Markdown contract", async () => { + const response = await post(); + expect(response.status).toBe(200); + expect(await response.json()).toEqual({ + profile: "document-v1", + text: expectedText, + format: "markdown", + completeness: "complete", + may_omit_content: false, + pages_needing_ocr: [], + truncated: false, + parser: { name: "anydoc", version: "0.1.3" }, + }); + await clean(); +}); +test("renamed DOCX follows MIME while generic DOCX follows filename", async () => { + for (const [mime, name] of [ + [DOCX_TYPE, "renamed.csv"], + ["application/octet-stream", "REPORT.DOCX"], + ] as const) { + expect((await post(app(), form(fixture, mime, name))).status).toBe(200); + } + await clean(); +}); +test("embedded image or object cannot claim complete inspection", async () => { + for (const name of ["word/media/scan.png", "word/embeddings/object.bin"]) { + const response = await post( + app(), + form(extraEntry(name, new Uint8Array([1, 2, 3]))), + ); + expect(response.status).toBe(200); + const result = resultSchema.parse(await response.json()); + expect(result.completeness).toBe("partial"); + expect(result.may_omit_content).toBe(true); + } + await clean(); +}); +test("cover thumbnail does not cause paid OCR escalation", async () => { + const response = await post( + app(), + form(extraEntry("docProps/thumbnail.png", new Uint8Array([1]))), + ); + expect(response.status).toBe(200); + expect((await response.json()).may_omit_content).toBe(false); + await clean(); +}); +test("disabled service has no extraction route, even for a malformed body", async () => { + const response = await post(createApp(), "not multipart"); + await code(response, 404, "EXTRACTION_DISABLED"); + await clean(); + const health = await createApp().request("/health"); + expect(health.status).toBe(200); + expect((await health.json()).extraction_profiles).toEqual([]); +}); +test("enabled startup requires a dedicated service key and finite limits", () => { + expect(() => createApp({ enabled: true })).toThrow(); + for (const limits of [ + { timeoutMs: Infinity }, + { concurrent: 0 }, + { queued: -1 }, + { maxBodyBytes: 1 }, + ]) { + expect(() => + configSchema.parse({ enabled: true, secret, ...limits }), + ).toThrow(); + } + expect(() => fromEnv({ RAG_EXTRACTION_API_ENABLED: "yes" })).toThrow(); +}); +test("rejects absent, expired, wrong-audience, and session-key tokens before parsing", async () => { + const application = app(); + for (const jwt of [ + "", + await token({ expiry: 1 }), + await token({ audience: "session" }), + await token({ key: "different-session-signing-key-32-characters" }), + ]) { + await code( + await post(application, "not multipart", jwt), + 401, + "EXTRACTION_AUTH_REQUIRED", + ); + } + const noHeader = await application.request("/v1/extract", { method: "POST" }); + await code(noHeader, 401, "EXTRACTION_AUTH_REQUIRED"); + await clean(); +}); +test("inference scopes never grant document extraction", async () => { + for (const scopes of [[], ["rag:embed"], ["rag:rerank"]]) { + await code( + await post(app(), "not multipart", await token({ scopes })), + 403, + "EXTRACTION_FORBIDDEN", + ); + } + await clean(); +}); +test("legacy id token cannot reach the new service", async () => { + const legacy = await new SignJWT({ id: "owner" }) + .setProtectedHeader({ alg: "HS256" }) + .sign(new TextEncoder().encode(secret)); + await code(await post(app(), "bad", legacy), 401, "EXTRACTION_AUTH_REQUIRED"); + await clean(); +}); +test("unsupported type/profile and malformed multipart never reach native parsing", async () => { + await code( + await post(app(), form(fixture, "application/pdf")), + 415, + "UNSUPPORTED_DOCUMENT_TYPE", + ); + await code( + await post(app(), form(fixture, "text/markdown", "note.md")), + 415, + "UNSUPPORTED_DOCUMENT_TYPE", + ); + await code( + await post(app(), form(fixture, DOCX_TYPE, "report.docx", "raw-v1")), + 400, + "UNSUPPORTED_PROFILE", + ); + await code(await post(app(), "bad"), 400, "INVALID_MULTIPART"); + await clean(); +}); +test("duplicate fields, extra files and missing profile are rejected", async () => { + const duplicate = form(); + duplicate.append("profile", "document-v1"); + await code(await post(app(), duplicate), 400, "INVALID_MULTIPART"); + const files = form(); + files.append("file", new File([fixture], "second.docx", { type: DOCX_TYPE })); + await code(await post(app(), files), 400, "INVALID_MULTIPART"); + const missing = form(); + missing.delete("profile"); + await code(await post(app(), missing), 400, "INVALID_MULTIPART"); + await clean(); +}); +test("invalid archive errors never contain caller text", async () => { + await code( + await post(app(), form(new TextEncoder().encode("private contents"))), + 422, + "ARCHIVE_INVALID", + ); + await clean(); +}); +test("zip bomb, total size and entry counts are hard refusals", async () => { + const bomb = extraEntry("word/bomb.xml", new Uint8Array(20_000)); + await code( + await post(app({ maxEntryBytes: 10_000 }), form(bomb)), + 413, + "ZIP_BOMB", + ); + await code(await post(app({ maxArchiveBytes: 1000 })), 413, "ZIP_BOMB"); + await code(await post(app({ maxEntries: 2 })), 413, "ZIP_BOMB"); + await clean(); +}); +test("empty document returns no text instead of a successful extraction", async () => { + const xml = new TextEncoder().encode( + '', + ); + await code( + await post(app(), form(extraEntry("word/document.xml", xml))), + 422, + "NO_DOCUMENT_TEXT", + ); + await clean(); +}); +test("file size and serialized native output have independent ceilings", async () => { + await code(await post(app({ maxFileBytes: 500 })), 413, "PARSER_INPUT_LIMIT"); + await code( + await post(app({ maxOutputBytes: 100 })), + 413, + "PARSER_OUTPUT_LIMIT", + ); + await clean(); +}); +test("parent caps IPC even if a child ignores its output limit", async () => { + const runner: Runner = (request, signal) => + runWorker(request, signal, [ + process.execPath, + "-e", + 'process.stdout.write("x".repeat(1000))', + ]); + await code( + await post(app({ maxOutputBytes: 100 }, runner)), + 413, + "PARSER_OUTPUT_LIMIT", + ); + await clean(); +}); +test("child crash and malformed response are sanitized and permit retry", async () => { + for (const source of [ + "process.exit(11)", + 'console.log("private native failure")', + 'console.log("null")', + ]) { + const runner: Runner = (request, signal) => + runWorker(request, signal, [process.execPath, "-e", source]); + await code(await post(app({}, runner)), 503, "PARSER_CRASH"); + await clean(); + } + expect((await post()).status).toBe(200); + await clean(); +}); + +const sleepPath = fileURLToPath(new URL("./sleep.fixture.ts", import.meta.url)); +const sleeping: Runner = (request, signal) => + runWorker(request, signal, [process.execPath, sleepPath, request.path]); +async function childPid() { + for (let i = 0; i < 200; i++) { + for (const dir of await readdir(tempRoot)) { + try { + return Number( + await readFile(join(tempRoot, dir, "input.docx.pid"), "utf8"), + ); + } catch { + /* Not started yet. */ + } + } + await Bun.sleep(10); + } + throw new Error("Child never started"); +} +function reaped(pid: number) { + expect(() => process.kill(pid, 0)).toThrow(); +} +test("overall deadline kills and reaps a running native process before cleanup", async () => { + const pending = post(app({ timeoutMs: 400 }, sleeping)); + const pid = await childPid(); + await code(await pending, 504, "PARSER_TIMEOUT"); + reaped(pid); + await clean(); +}); +test("abort kills and reaps the child, cleans temp files and permits the next request", async () => { + let calls = 0; + const runner: Runner = (request, signal) => + ++calls === 1 ? sleeping(request, signal) : runWorker(request, signal); + const application = app({}, runner); + const controller = new AbortController(); + const pending = post(application, form(), await token(), controller.signal); + const pid = await childPid(); + controller.abort(); + await code(await pending, 408, "REQUEST_CANCELLED"); + reaped(pid); + await clean(); + expect((await post(application)).status).toBe(200); + await clean(); +}); +test("overload refuses before reading or staging another request body", async () => { + const application = app({ concurrent: 1, queued: 0 }, sleeping); + const controller = new AbortController(); + const pending = post(application, form(), await token(), controller.signal); + const pid = await childPid(); + let pulls = 0; + const body = new ReadableStream( + { + pull(control) { + pulls++; + control.enqueue(new Uint8Array([1])); + }, + }, + { highWaterMark: 0 }, + ); + const request = new Request("http://test/v1/extract", { + method: "POST", + headers: { + Authorization: `Bearer ${await token()}`, + "Content-Type": "multipart/form-data; boundary=test", + }, + body, + }); + await code(await application.fetch(request), 429, "CONCURRENCY_LIMIT"); + expect(pulls).toBe(0); + controller.abort(); + await pending; + reaped(pid); + await clean(); +}); +test("body limit is counted for chunked requests, not trusted Content-Length", async () => { + const application = app({ maxFileBytes: 100, maxBodyBytes: 200 }); + let pulls = 0; + const prefix = + '--test\r\nContent-Disposition: form-data; name="file"; filename="r.docx"\r\nContent-Type: ' + + DOCX_TYPE + + "\r\n\r\n"; + const body = new ReadableStream( + { + pull(control) { + pulls++; + control.enqueue( + new TextEncoder().encode(pulls === 1 ? prefix : "x".repeat(256)), + ); + }, + }, + { highWaterMark: 0 }, + ); + const request = new Request("http://test/v1/extract", { + method: "POST", + headers: { + Authorization: `Bearer ${await token()}`, + "Content-Type": "multipart/form-data; boundary=test", + }, + body, + }); + await code(await application.fetch(request), 413, "PARSER_INPUT_LIMIT"); + expect(pulls).toBeLessThan(5); + await clean(); +}); diff --git a/service/test/image.sh b/service/test/image.sh new file mode 100644 index 00000000..fd67ca56 --- /dev/null +++ b/service/test/image.sh @@ -0,0 +1,28 @@ +#!/usr/bin/env bash +set -euo pipefail +image=${1:?Pass the locally built image name} +root=$(git rev-parse --show-toplevel) +docker run --rm --entrypoint sh "$image" -c '! command -v python && ! command -v python3' +docker run --rm -i --entrypoint bun "$image" -e ' +import { createApp } from "./src/app.ts"; +import { SignJWT } from "jose"; +const secret = "image-smoke-test-key-not-a-real-service-key"; +const data = await Bun.stdin.bytes(); +const application = createApp({enabled: true, secret}); +const server = Bun.serve({hostname: "127.0.0.1", port: 0, fetch: application.fetch}); +try { + const health = await fetch(`http://127.0.0.1:${server.port}/health`); + if (health.status !== 200) throw new Error("Health failed"); + const token = await new SignJWT({scopes: ["rag:documents"]}).setProtectedHeader({alg: "HS256"}) + .setIssuer("librechat").setAudience("rag-api").setSubject("owner").setExpirationTime("1m") + .sign(new TextEncoder().encode(secret)); + const body = new FormData(); body.append("profile", "document-v1"); + body.append("file", new File([data], "report.docx", {type: "application/vnd.openxmlformats-officedocument.wordprocessingml.document"})); + const response = await fetch(`http://127.0.0.1:${server.port}/v1/extract`, {method: "POST", body, headers: {Authorization: `Bearer ${token}`}}); + const result = await response.json(); + if (response.status !== 200 || !result.text?.includes("Regional Totals") || result.parser?.name !== "anydoc") { + throw new Error(`Native DOCX failed: ${response.status}`); + } + console.log("PASS: Bun listener, health and native DOCX extraction in a Python-free image"); +} finally { await server.stop(true); } +' < "$root/tests/fixtures/structured.docx" diff --git a/service/test/lifecycle.test.ts b/service/test/lifecycle.test.ts new file mode 100644 index 00000000..b2d29388 --- /dev/null +++ b/service/test/lifecycle.test.ts @@ -0,0 +1,125 @@ +import { expect, test } from "bun:test"; +import { SignJWT } from "jose"; +import { mkdtemp, readdir, readFile, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { Admission } from "../src/admission"; +import { createApp } from "../src/app"; +import { DOCX_TYPE } from "../src/contract"; +import { runWorker, type Runner } from "../src/process"; + +const fixture = await Bun.file( + new URL("../../tests/fixtures/structured.docx", import.meta.url), +).bytes(); +const sleepPath = fileURLToPath(new URL("./sleep.fixture.ts", import.meta.url)); +const secret = "lifecycle-test-service-signing-key-32-characters"; + +async function headers() { + const token = await new SignJWT({ scopes: ["rag:documents"] }) + .setProtectedHeader({ alg: "HS256" }) + .setIssuer("librechat") + .setAudience("rag-api") + .setSubject("owner") + .setExpirationTime("1m") + .sign(new TextEncoder().encode(secret)); + return { Authorization: `Bearer ${token}` }; +} +function form() { + const body = new FormData(); + body.append("profile", "document-v1"); + body.append("file", new File([fixture], "report.docx", { type: DOCX_TYPE })); + return body; +} +async function pid(temp: string) { + for (let i = 0; i < 200; i++) { + for (const dir of await readdir(temp)) { + try { + return Number( + await readFile(join(temp, dir, "input.docx.pid"), "utf8"), + ); + } catch { + /* Child is not running yet. */ + } + } + await Bun.sleep(10); + } + throw new Error("Child never started"); +} + +test("cancelled queued work never consumes a slot and remaining work stays FIFO", async () => { + const admission = new Admission(1, 2); + const first = await admission.acquire(new AbortController().signal); + const cancelled = new AbortController(); + const second = admission.acquire(cancelled.signal); + let granted = false; + const third = admission + .acquire(new AbortController().signal) + .then((release) => { + granted = true; + return release; + }); + cancelled.abort(); + await expect(second).rejects.toMatchObject({ code: "REQUEST_CANCELLED" }); + expect(granted).toBe(false); + first(); + const release = await third; + expect(granted).toBe(true); + release(); + release(); + const next = await admission.acquire(new AbortController().signal); + next(); +}); + +test("actual HTTP disconnect kills the native child and cleans up before retry", async () => { + const tempRoot = await mkdtemp(join(tmpdir(), "rag-disconnect-")); + let calls = 0; + const runner: Runner = (request, signal) => + ++calls === 1 + ? runWorker(request, signal, [process.execPath, sleepPath, request.path]) + : runWorker(request, signal); + const application = createApp( + { + enabled: true, + secret, + tempRoot, + concurrent: 1, + queued: 0, + timeoutMs: 3000, + }, + runner, + ); + const server = Bun.serve({ + hostname: "127.0.0.1", + port: 0, + fetch: application.fetch, + }); + const controller = new AbortController(); + try { + const pending = fetch(`http://127.0.0.1:${server.port}/v1/extract`, { + method: "POST", + headers: await headers(), + body: form(), + signal: controller.signal, + }); + void pending.catch(() => {}); + const child = await pid(tempRoot); + controller.abort(); + await expect(pending).rejects.toThrow(); + for (let i = 0; i < 200 && (await readdir(tempRoot)).length; i++) + await Bun.sleep(10); + expect(await readdir(tempRoot)).toEqual([]); + expect(() => process.kill(child, 0)).toThrow(); + const retry = await fetch(`http://127.0.0.1:${server.port}/v1/extract`, { + method: "POST", + headers: await headers(), + body: form(), + }); + expect(retry.status).toBe(200); + expect((await retry.json()).text).toContain("Quarterly Report"); + } finally { + controller.abort(); + await server.stop(true); + await rm(tempRoot, { recursive: true, force: true }); + } +}); diff --git a/service/test/sleep.fixture.ts b/service/test/sleep.fixture.ts new file mode 100644 index 00000000..25fc93a5 --- /dev/null +++ b/service/test/sleep.fixture.ts @@ -0,0 +1,6 @@ +export {}; + +const path = process.argv[2]; +if (!path) throw new Error("Missing fixture path"); +await Bun.write(`${path}.pid`, String(process.pid)); +await Bun.sleep(60_000); diff --git a/service/tsconfig.json b/service/tsconfig.json new file mode 100644 index 00000000..c0d4f1a1 --- /dev/null +++ b/service/tsconfig.json @@ -0,0 +1,15 @@ +{ + "compilerOptions": { + "target": "ES2023", + "lib": ["ES2023", "DOM"], + "module": "ESNext", + "moduleResolution": "Bundler", + "types": ["bun"], + "strict": true, + "noEmit": true, + "noUncheckedIndexedAccess": true, + "allowImportingTsExtensions": true, + "skipLibCheck": true + }, + "include": ["src/**/*.ts", "test/**/*.ts"] +} diff --git a/tests/test_extraction_api.py b/tests/test_extraction_api.py deleted file mode 100644 index 59dc1592..00000000 --- a/tests/test_extraction_api.py +++ /dev/null @@ -1,393 +0,0 @@ -"""Contract tests for the opt-in extraction slice; uses the pinned native wheel.""" - -import asyncio -import io -import os -import subprocess -import sys -import zipfile -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -import httpx -import jwt -import pytest -from fastapi import FastAPI, HTTPException, UploadFile - -from app.middleware import security_middleware -from app.routes import extraction_routes -from app.services import extraction, extraction_worker -from main import app as main_app - -app = FastAPI() -app.middleware("http")(security_middleware) -app.include_router(extraction_routes.router) - -FIXTURE = Path(__file__).parent / "fixtures" / "structured.docx" -DOCX_TYPE = extraction_routes.DOCX_TYPE - - -@pytest.fixture -def configured(monkeypatch, tmp_path): - monkeypatch.setenv("RAG_EXTRACTION_API_ENABLED", "true") - monkeypatch.setenv("JWT_SECRET", "a-test-key-that-is-at-least-32-bytes-long") - monkeypatch.setattr(extraction_routes, "RAG_UPLOAD_DIR", str(tmp_path)) - admission = extraction.ExtractionAdmission() - monkeypatch.setattr(extraction_routes, "_admission", admission) - with ThreadPoolExecutor(max_workers=2) as pool: - monkeypatch.setattr(main_app.state, "thread_pool", pool, raising=False) - yield tmp_path, admission - - -@pytest.fixture -def headers(configured): - token = jwt.encode({"id": "owner"}, os.environ["JWT_SECRET"], algorithm="HS256") - return {"Authorization": f"Bearer {token}"} - - -async def post( - file_bytes, headers, name="report.docx", mime=DOCX_TYPE, profile="document-v1" -): - async with httpx.AsyncClient( - transport=httpx.ASGITransport(app=app), base_url="http://test" - ) as client: - return await client.post( - "/v1/extract", - headers=headers, - data={"profile": profile}, - files={"file": (name, file_bytes, mime)}, - ) - - -def with_entry(original: bytes, name: str, value: bytes) -> bytes: - result = io.BytesIO() - with zipfile.ZipFile(io.BytesIO(original)) as source, zipfile.ZipFile( - result, "w", zipfile.ZIP_DEFLATED - ) as target: - for entry in source.infolist(): - if entry.filename != name: - target.writestr(entry, source.read(entry)) - target.writestr(name, value) - return result.getvalue() - - -async def test_real_docx_from_marcos_pr_returns_markdown_and_provenance( - headers, configured -): - response = await post( - FIXTURE.read_bytes(), headers, name="renamed.csv", mime=DOCX_TYPE - ) - assert response.status_code == 200, response.text - payload = response.json() - assert payload == { - "profile": "document-v1", - "format": "markdown", - "text": ( - "# Quarterly Report\n\nThis document summarizes the results for the period.\n\n" - "## Regional Totals\n\n| | | |\n| --- | --- | --- |\n" - "| Region | Units | Revenue |\n| North | 1200 | 48000 |\n" - "| South | 950 | 38000 |\n| East | 1430 | 57200 |\n\n" - "**Totals are unaudited.**\n" - ), - "completeness": "complete", - "may_omit_content": False, - "pages_needing_ocr": [], - "truncated": False, - "parser": {"name": "anydoc", "version": "0.1.3"}, - } - assert not list(configured[0].iterdir()) - - -async def test_embedded_image_cannot_claim_complete_text(headers, configured): - image_docx = with_entry(FIXTURE.read_bytes(), "word/media/scan.png", b"\x89PNG\r\n") - response = await post( - image_docx, headers, name="report.docx", mime="application/octet-stream" - ) - assert response.status_code == 200, response.text - assert response.json()["completeness"] == "partial" - assert response.json()["may_omit_content"] is True - assert response.json()["pages_needing_ocr"] == [] - assert not list(configured[0].iterdir()) - - -def test_real_app_registers_route_only_when_enabled(): - # A fresh process verifies main.py registration, not just the test router. - script = ( - "from langchain_community.vectorstores.pgvector import PGVector\n" - "from app.services.vector_store.async_pg_vector import AsyncPgVector\n" - "PGVector.__post_init__ = lambda self: None\n" - "AsyncPgVector.__post_init__ = lambda self: None\n" - "from main import app\n" - "import sys\n" - "print(int(any(getattr(r, 'path', None) == '/v1/extract' for r in app.routes)), " - "int('anydoc' in sys.modules))\n" - ) - for enabled, expected in (("false", "0 0"), ("true", "1 0")): - env = { - **os.environ, - "RAG_EXTRACTION_API_ENABLED": enabled, - "OPENAI_API_KEY": "test_key", - } - result = subprocess.run( - [sys.executable, "-c", script], - env=env, - capture_output=True, - text=True, - check=True, - ) - assert result.stdout.strip().splitlines()[-1] == expected, result.stderr - - -async def test_disabled_requires_explicit_opt_in(headers, configured, monkeypatch): - monkeypatch.delenv("RAG_EXTRACTION_API_ENABLED") - response = await post(FIXTURE.read_bytes(), headers) - assert response.status_code == 404 - assert response.json()["detail"]["code"] == "EXTRACTION_DISABLED" - assert not list(configured[0].iterdir()) - - -async def test_requires_verified_identity_and_signing_secret( - headers, configured, monkeypatch -): - missing = await post(FIXTURE.read_bytes(), {}) - assert missing.status_code == 401 - monkeypatch.delenv("JWT_SECRET") - unsigned = await post(FIXTURE.read_bytes(), {}) - assert unsigned.status_code == 401 - assert unsigned.json()["detail"]["code"] == "EXTRACTION_AUTH_REQUIRED" - assert not list(configured[0].iterdir()) - - -@pytest.mark.parametrize( - "name,mime,code", - [ - ("report.md", "text/markdown", "UNSUPPORTED_DOCUMENT_TYPE"), - ("report.docx", "application/pdf", "UNSUPPORTED_DOCUMENT_TYPE"), - ("report.pdf", "application/octet-stream", "UNSUPPORTED_DOCUMENT_TYPE"), - ], -) -async def test_unrelated_formats_never_reach_the_parser( - headers, configured, name, mime, code -): - response = await post(FIXTURE.read_bytes(), headers, name=name, mime=mime) - assert response.status_code == 415 - assert response.json()["detail"]["code"] == code - assert not list(configured[0].iterdir()) - - -async def test_nonfinite_deadline_cannot_disable_worker_timeout( - headers, configured, monkeypatch -): - monkeypatch.setenv("RAG_EXTRACTION_TIMEOUT_SECONDS", "inf") - response = await post(FIXTURE.read_bytes(), headers) - assert response.status_code == 503 - assert response.json()["detail"]["code"] == "PARSER_UNAVAILABLE" - assert not list(configured[0].iterdir()) - - -async def test_unknown_profile_and_invalid_archive_fail_closed(headers, configured): - bad_profile = await post(FIXTURE.read_bytes(), headers, profile="raw-v1") - assert bad_profile.status_code == 400 - assert bad_profile.json()["detail"]["code"] == "UNSUPPORTED_PROFILE" - bad_archive = await post(b"private secret from a forged DOCX", headers) - assert bad_archive.status_code == 422 - assert bad_archive.json() == {"detail": {"code": "ARCHIVE_INVALID"}} - assert not list(configured[0].iterdir()) - - -async def test_refuses_zip_bomb_before_native_parsing(headers, configured): - bomb = with_entry( - FIXTURE.read_bytes(), "word/bomb.xml", b"x" * (25 * 1024 * 1024 + 1) - ) - response = await post(bomb, headers) - assert response.status_code == 413 - assert response.json()["detail"]["code"] == "ZIP_BOMB" - assert not list(configured[0].iterdir()) - - -async def test_empty_docx_reports_no_text_instead_of_success(headers, configured): - result = io.BytesIO() - with zipfile.ZipFile(FIXTURE) as source, zipfile.ZipFile( - result, "w", zipfile.ZIP_DEFLATED - ) as target: - for entry in source.infolist(): - data = source.read(entry) - if entry.filename == "word/document.xml": - data = ( - b'' - b"" - ) - target.writestr(entry, data) - response = await post(result.getvalue(), headers) - assert response.status_code == 422 - assert response.json()["detail"]["code"] == "NO_DOCUMENT_TEXT" - assert not list(configured[0].iterdir()) - - -async def test_archive_entry_count_has_an_independent_limit(headers, configured): - result = io.BytesIO() - with zipfile.ZipFile(FIXTURE) as source, zipfile.ZipFile( - result, "w", zipfile.ZIP_DEFLATED - ) as target: - for entry in source.infolist(): - target.writestr(entry, source.read(entry)) - for index in range(4096): - target.writestr(f"word/noise/{index}", b"") - response = await post(result.getvalue(), headers) - assert response.status_code == 413 - assert response.json()["detail"]["code"] == "ZIP_BOMB" - assert not list(configured[0].iterdir()) - - -async def test_input_limit_is_checked_while_streaming_even_without_size_hint( - configured, monkeypatch -): - monkeypatch.setattr(extraction_routes, "MAX_INPUT_BYTES", 8) - file = UploadFile(file=io.BytesIO(b"0123456789"), filename="report.docx", size=None) - with pytest.raises(HTTPException) as caught: - await extraction_routes._save_bounded(file, configured[0] / "bounded.docx") - assert caught.value.status_code == 413 - assert caught.value.detail == {"code": "PARSER_INPUT_LIMIT"} - - -async def test_worker_output_limit_is_a_refusal(configured, monkeypatch): - monkeypatch.setattr(extraction_worker, "MAX_OUTPUT_BYTES", 10) - with pytest.raises(extraction_worker.ExtractionRefusal) as caught: - extraction_worker.extract(FIXTURE) - assert caught.value.code == "PARSER_OUTPUT_LIMIT" - - -async def test_child_crash_and_malformed_output_are_sanitized( - headers, configured, monkeypatch -): - monkeypatch.setattr( - extraction, - "_command", - lambda path: (sys.executable, "-c", "import sys; sys.exit(11)"), - ) - crash = await post(FIXTURE.read_bytes(), headers) - assert crash.status_code == 503 - assert crash.json() == {"detail": {"code": "PARSER_CRASH"}} - assert not list(configured[0].iterdir()) - monkeypatch.setattr( - extraction, - "_command", - lambda path: (sys.executable, "-c", "print('private secret')"), - ) - invalid = await post(FIXTURE.read_bytes(), headers) - assert invalid.status_code == 503 - assert invalid.json() == {"detail": {"code": "PARSER_CRASH"}} - assert "private secret" not in invalid.text - assert not list(configured[0].iterdir()) - - -async def test_api_kills_worker_that_overproduces_ipc(headers, configured, monkeypatch): - monkeypatch.setattr(extraction, "MAX_IPC_BYTES", 64) - monkeypatch.setattr( - extraction, - "_command", - lambda path: (sys.executable, "-c", "import os; os.write(1, b'x' * 1024)"), - ) - response = await post(FIXTURE.read_bytes(), headers) - assert response.status_code == 413 - assert response.json()["detail"]["code"] == "PARSER_OUTPUT_LIMIT" - assert not list(configured[0].iterdir()) - assert configured[1]._pending == 0 - - -def _sleeping_command(path: Path) -> tuple[str, ...]: - return ( - sys.executable, - "-c", - "import os,sys,time; open(sys.argv[1]+'.pid','w').write(str(os.getpid())); time.sleep(60)", - str(path), - ) - - -async def _await_child(tmp_path: Path) -> Path: - for _ in range(200): - pids = list(tmp_path.glob("*.pid")) - if pids: - return pids[0] - await asyncio.sleep(0.01) - raise AssertionError("parser child did not start") - - -async def _assert_child_reaped( - pid_file: Path, admission: extraction.ExtractionAdmission -) -> None: - pid = int(pid_file.read_text()) - for _ in range(200): - try: - os.kill(pid, 0) - except ProcessLookupError: - if admission._pending == 0 and not list(pid_file.parent.glob("*.docx")): - pid_file.unlink() - return - await asyncio.sleep(0.01) - raise AssertionError("parser child or staging file survived cancellation") - - -async def test_timeout_kills_and_reaps_child(headers, configured, monkeypatch): - monkeypatch.setattr(extraction, "_command", _sleeping_command) - monkeypatch.setenv("RAG_EXTRACTION_TIMEOUT_SECONDS", "0.3") - task = asyncio.create_task(post(FIXTURE.read_bytes(), headers)) - pid_file = await _await_child(configured[0]) - response = await asyncio.wait_for(task, 2) - assert response.status_code == 504 - assert response.json()["detail"]["code"] == "PARSER_TIMEOUT" - await _assert_child_reaped(pid_file, configured[1]) - assert not list(configured[0].iterdir()) - assert configured[1]._pending == 0 - - -async def test_cancel_reaps_child_and_frees_slot_for_next_upload( - headers, configured, monkeypatch -): - original_command = extraction._command - monkeypatch.setattr(extraction, "_command", _sleeping_command) - task = asyncio.create_task(post(FIXTURE.read_bytes(), headers)) - pid_file = await _await_child(configured[0]) - task.cancel() - with pytest.raises(asyncio.CancelledError): - await asyncio.wait_for(task, 2) - await _assert_child_reaped(pid_file, configured[1]) - assert not list(configured[0].iterdir()) - assert configured[1]._pending == 0 - monkeypatch.setattr(extraction, "_command", original_command) - response = await post(FIXTURE.read_bytes(), headers) - assert response.status_code == 200 - - -async def test_busy_parser_refuses_before_staging_anything( - headers, configured, monkeypatch -): - monkeypatch.setattr(extraction, "_command", _sleeping_command) - monkeypatch.setattr( - extraction_routes, "_admission", extraction.ExtractionAdmission(1, 0) - ) - task = asyncio.create_task(post(FIXTURE.read_bytes(), headers)) - pid_file = await _await_child(configured[0]) - busy = await post(FIXTURE.read_bytes(), headers) - assert busy.status_code == 429 - assert busy.json()["detail"]["code"] == "CONCURRENCY_LIMIT" - assert len(list(configured[0].glob("*.docx"))) == 1 - task.cancel() - with pytest.raises(asyncio.CancelledError): - await asyncio.wait_for(task, 2) - await _assert_child_reaped(pid_file, extraction_routes._admission) - assert not list(configured[0].iterdir()) - - -async def test_existing_text_endpoint_still_preserves_raw_markdown(headers, configured): - async with httpx.AsyncClient( - transport=httpx.ASGITransport(app=main_app), base_url="http://test" - ) as client: - response = await client.post( - "/text", - headers=headers, - data={"file_id": "md-test"}, - files={"file": ("readme.md", b"# Raw **Markdown**\n", "text/markdown")}, - ) - assert response.status_code == 200, response.text - assert response.json()["text"] == "# Raw **Markdown**" From 9324808c0db895fb138c1e67d1f79edbfd64c318 Mon Sep 17 00:00:00 2001 From: Lia Date: Wed, 30 Sep 2026 09:53:11 +0000 Subject: [PATCH 4/5] =?UTF-8?q?=F0=9F=94=91=20fix:=20Refuse=20Reusing=20Se?= =?UTF-8?q?ssion=20Keys=20in=20the=20Bun=20Service?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- service/src/config.ts | 3 +++ service/test/extraction.test.ts | 3 +++ 2 files changed, 6 insertions(+) diff --git a/service/src/config.ts b/service/src/config.ts index d57e28c2..9dc14d18 100644 --- a/service/src/config.ts +++ b/service/src/config.ts @@ -36,6 +36,9 @@ export const configSchema = z export type Config = z.infer; export function fromEnv(env: NodeJS.ProcessEnv): Config { + if (env.RAG_JWT_SECRET && env.RAG_JWT_SECRET === env.JWT_SECRET) { + throw new Error("RAG_JWT_SECRET must differ from JWT_SECRET"); + } const number = (name: string) => env[name] === undefined ? undefined : Number(env[name]); const enabled = env.RAG_EXTRACTION_API_ENABLED; diff --git a/service/test/extraction.test.ts b/service/test/extraction.test.ts index adaa5726..ae3270c2 100644 --- a/service/test/extraction.test.ts +++ b/service/test/extraction.test.ts @@ -147,6 +147,9 @@ test("enabled startup requires a dedicated service key and finite limits", () => ).toThrow(); } expect(() => fromEnv({ RAG_EXTRACTION_API_ENABLED: "yes" })).toThrow(); + expect(() => + fromEnv({ RAG_JWT_SECRET: secret, JWT_SECRET: secret }), + ).toThrow(); }); test("rejects absent, expired, wrong-audience, and session-key tokens before parsing", async () => { const application = app(); From c967dee985ea26d1de7dd77fcd1b37cbd05f8c4d Mon Sep 17 00:00:00 2001 From: Lia Date: Wed, 30 Sep 2026 10:06:03 +0000 Subject: [PATCH 5/5] =?UTF-8?q?=F0=9F=9B=A1=EF=B8=8F=20fix:=20Preserve=20P?= =?UTF-8?q?ackage=20Fidelity=20and=20Bound=20the=20Bun=20Request=20Lifecyc?= =?UTF-8?q?le?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- EXTRACTION.md | 13 +++- service/bun.lock | 5 ++ service/package.json | 1 + service/src/config.ts | 2 +- service/src/process.ts | 4 +- service/src/server.ts | 30 +++++++-- service/src/upload.ts | 2 + service/src/worker.ts | 109 ++++++++++++++++++++++++++++---- service/test/extraction.test.ts | 90 +++++++++++++++++++++++++- service/test/lifecycle.test.ts | 29 +++++++++ 10 files changed, 263 insertions(+), 22 deletions(-) diff --git a/EXTRACTION.md b/EXTRACTION.md index 06e46083..5ab576bc 100644 --- a/EXTRACTION.md +++ b/EXTRACTION.md @@ -66,7 +66,10 @@ Successful response: ``` An archive with non-thumbnail artwork or embedded objects is marked `partial` -and `may_omit_content=true`. This is conservative omission detection, **not a +and `may_omit_content=true`. Package relationships and content types identify +artwork even when its filename has no image extension. The root relationship +resolves the main document; a conventional `word/document.xml` path is not +required. This is conservative omission detection, **not a proof that every source element is inspectable**. Never use partial text or a preview as complete content-inspection input. No automatic fallback occurs inside the service, so hard refusals cannot accidentally become paid OCR calls. @@ -110,7 +113,8 @@ or health-check round trip. The Bun listener also enforces the body ceiling. Limits are per service process, not cluster-wide quotas. Queue wait, upload staging and parsing share one deadline. Cancelled queued work is removed; an active child is killed and reaped before cleanup and slot reuse. The child -receives no JWT secret or provider credentials. Both sides cap serialized IPC +receives no JWT secret or provider credentials, and Bun's automatic dotenv +loading is disabled in it. Both sides cap serialized IPC output. Actual decompressed entry bytes are checked before native parsing; metadata-only size claims are not trusted. Graceful shutdown stops accepting requests and allows active operations to finish within their deadlines. @@ -130,3 +134,8 @@ preview/sharing, token scopes, outages and mixed-version behavior end to end. Keep raw Markdown, rich HTML preview and complete inspection as distinct contracts. Expand formats or enable reranking only behind their own fidelity and quality gates. No measured speedup or full service parity is claimed here. + +The listener idle timeout is set above the overall extraction deadline so a +parse lasting more than Bun's default ten seconds is not reset prematurely. +`RAG_EXTRACTION_TIMEOUT_MS` may be at most 254,000, leaving one second within +Bun's 255-second listener limit for the result or typed timeout response. diff --git a/service/bun.lock b/service/bun.lock index 393feaf8..d10a5005 100644 --- a/service/bun.lock +++ b/service/bun.lock @@ -9,6 +9,7 @@ "busboy": "1.6.0", "hono": "4.13.11", "jose": "6.2.12", + "saxes": "6.0.0", "yauzl": "3.4.0", "zod": "4.3.6", }, @@ -61,12 +62,16 @@ "prettier": ["prettier@3.8.1", "", { "bin": { "prettier": "bin/prettier.cjs" } }, "sha512-UOnG6LftzbdaHZcKoPFtOcCKztrQ57WkHDeRD9t/PTQtmT0NHSeWWepj6pS0z/N7+08BHFDQVUrfmfMRcZwbMg=="], + "saxes": ["saxes@6.0.0", "", { "dependencies": { "xmlchars": "^2.2.0" } }, "sha512-xAg7SOnEhrm5zI3puOOKyy1OMcMlIJZYNJY7xLBwSze0UjhPLnWfj2GF2EpT0jmzaJKIWKHLsaSSajf35bcYnA=="], + "streamsearch": ["streamsearch@1.1.0", "", {}, "sha512-Mcc5wHehp9aXz1ax6bZUyY5afg9u2rv5cqQI3mRrYkGC8rW2hM02jWuwjtL++LS5qinSyhj2QfLyNsuc+VsExg=="], "typescript": ["typescript@5.9.3", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw=="], "undici-types": ["undici-types@8.9.0", "", {}, "sha512-KTDyRTYX8sWmKXAikPHHSyc63CRPETMctyjKFupcC6OBLXT3xsN0e9aF7m+mIXutFWpUXuedtowG7iLOzp0kQg=="], + "xmlchars": ["xmlchars@2.2.0", "", {}, "sha512-JZnDKK8B0RCDw84FNdDAIpZK+JuJw+s7Lz8nksI7SIuU3UXJJslUthsi+uWBUYOwPFwW7W7PRLRfUKpxjtjFCw=="], + "yauzl": ["yauzl@3.4.0", "", { "dependencies": { "pend": "~1.2.0" } }, "sha512-jIH9yLR9wqr0wOS0TpBvo/g/2UgZH5qePVbjgRliiF0BYvOZyaBknKsF+x9Iht0O6sqgnB93rCICdOZFecJuDw=="], "zod": ["zod@4.3.6", "", {}, "sha512-rftlrkhHZOcjDwkGlnUtZZkvaPHCsDATp4pGpuOOMDaTdDDXF91wuVDJoWoPsKX/3YPQ5fHuF3STjcYyKr+Qhg=="], diff --git a/service/package.json b/service/package.json index b220840f..b38f9ca5 100644 --- a/service/package.json +++ b/service/package.json @@ -16,6 +16,7 @@ "busboy": "1.6.0", "hono": "4.13.11", "jose": "6.2.12", + "saxes": "6.0.0", "yauzl": "3.4.0", "zod": "4.3.6" }, diff --git a/service/src/config.ts b/service/src/config.ts index 9dc14d18..512fc54c 100644 --- a/service/src/config.ts +++ b/service/src/config.ts @@ -9,7 +9,7 @@ export const configSchema = z audience: z.string().min(1).default("rag-api"), concurrent: positive.default(2), queued: z.number().int().nonnegative().default(6), - timeoutMs: positive.max(2_147_483_647).default(30_000), + timeoutMs: positive.max(254_000).default(30_000), maxFileBytes: positive.default(15 * 1024 * 1024), maxBodyBytes: positive.default(16 * 1024 * 1024), maxOutputBytes: positive.default(15 * 1024 * 1024), diff --git a/service/src/process.ts b/service/src/process.ts index cc271943..a606a9be 100644 --- a/service/src/process.ts +++ b/service/src/process.ts @@ -18,7 +18,9 @@ export async function runWorker( command: readonly string[] = [process.execPath, workerPath], ): Promise { signal.throwIfAborted(); - const child = Bun.spawn([...command], { + const executable = command[0]; + if (!executable) throw new ExtractionError("PARSER_UNAVAILABLE"); + const child = Bun.spawn([executable, "--no-env-file", ...command.slice(1)], { stdin: "pipe", stdout: "pipe", stderr: "ignore", diff --git a/service/src/server.ts b/service/src/server.ts index d7375ef4..fbe136a3 100644 --- a/service/src/server.ts +++ b/service/src/server.ts @@ -1,19 +1,37 @@ +import type { Hono } from "hono"; +import type { Config } from "./config"; import { createApp } from "./app"; import { fromEnv } from "./config"; +export function listen( + app: Hono, + config: Config, + port: number, + hostname: string, +) { + return Bun.serve({ + hostname, + port, + // Bun includes pending handlers in its idle timer (default: 10 seconds). + // Its maximum is 255 seconds; config bounds the request deadline to 254. + idleTimeout: Math.ceil(config.timeoutMs / 1000) + 1, + maxRequestBodySize: config.maxBodyBytes, + fetch: app.fetch, + }); +} + if (import.meta.main) { try { const config = fromEnv(process.env); const port = Number(process.env.RAG_PORT ?? 8001); if (!Number.isInteger(port) || port < 1 || port > 65535) throw new Error("Invalid port"); - const server = Bun.serve({ - hostname: process.env.RAG_HOST ?? "0.0.0.0", + const server = listen( + createApp(config), + config, port, - maxRequestBodySize: config.maxBodyBytes, - fetch: createApp(config).fetch, - }); - // Graceful shutdown lets active requests finish; their deadlines stay bounded. + process.env.RAG_HOST ?? "0.0.0.0", + ); for (const event of ["SIGINT", "SIGTERM"] as const) { process.once(event, () => { void server.stop(false); diff --git a/service/src/upload.ts b/service/src/upload.ts index 938b60d0..c425f2b6 100644 --- a/service/src/upload.ts +++ b/service/src/upload.ts @@ -47,6 +47,8 @@ export async function stageUpload( yield part.value; } } finally { + // Stop unread HTTP input before releasing the reader's lock. + await reader.cancel().catch(() => {}); reader.releaseLock(); } })(), diff --git a/service/src/worker.ts b/service/src/worker.ts index 62e24655..cee316db 100644 --- a/service/src/worker.ts +++ b/service/src/worker.ts @@ -1,6 +1,9 @@ import { readFile } from "node:fs/promises"; import { createRequire } from "node:module"; -import { open, type Entry, type ZipFile } from "yauzl"; +import { posix } from "node:path"; +import type { Readable } from "node:stream"; +import { SaxesParser } from "saxes"; +import { open, type Entry } from "yauzl"; import { z } from "zod"; import type { ExtractionResult } from "./contract"; import { ExtractionError } from "./contract"; @@ -17,6 +20,19 @@ const IMAGE = /\.(?:jpe?g|png|gif|tiff?|bmp|webp|jp2|jpx|avif|heic|heif|emf|wmf|svg)$/i; const PREVIEW = /^(?:docProps|Thumbnails)\//i; +function mainPart(target: string): string { + // OPC targets are package URIs, never filesystem or network locations. + const decoded = decodeURIComponent(target); + if ( + /^[a-z]+:/i.test(decoded) || + decoded.includes("\\") || + decoded.split("/").includes("..") + ) { + throw new ExtractionError("ARCHIVE_INVALID"); + } + return posix.normalize("/" + decoded).slice(1); +} + function inspectArchive(path: string, limits: WorkerRequest): Promise { return new Promise((resolve, reject) => { open( @@ -28,25 +44,35 @@ function inspectArchive(path: string, limits: WorkerRequest): Promise { return; } let settled = false; + let active: Readable | undefined; let total = 0; let media = false; + let main: string | undefined; const names = new Set(); + const imageParts = new Set(); + const imageExtensions = new Set(); const fail = (code: "ARCHIVE_INVALID" | "ZIP_BOMB") => { if (settled) return; settled = true; + active?.destroy(); zip.close(); reject(new ExtractionError(code)); }; zip.on("error", () => fail("ARCHIVE_INVALID")); zip.on("end", () => { if (settled) return; - if ( - !names.has("[Content_Types].xml") || - !names.has("word/document.xml") - ) { + if (!names.has("[Content_Types].xml") || !main || !names.has(main)) { fail("ARCHIVE_INVALID"); return; } + for (const name of names) { + if (PREVIEW.test(name)) continue; + if ( + imageParts.has(name) || + imageExtensions.has(posix.extname(name).slice(1).toLowerCase()) + ) + media = true; + } settled = true; resolve(media); }); @@ -55,6 +81,7 @@ function inspectArchive(path: string, limits: WorkerRequest): Promise { return; } zip.on("entry", (entry: Entry) => { + if (settled) return; if (names.has(entry.fileName)) { fail("ARCHIVE_INVALID"); return; @@ -73,28 +100,90 @@ function inspectArchive(path: string, limits: WorkerRequest): Promise { } if ( (IMAGE.test(entry.fileName) && !PREVIEW.test(entry.fileName)) || - /^word\/embeddings\//i.test(entry.fileName) + /\/embeddings\//i.test(entry.fileName) ) media = true; + const isTypes = entry.fileName === "[Content_Types].xml"; + const isRelationships = entry.fileName.endsWith(".rels"); + let metadata: SaxesParser<{ xmlns: true }> | undefined; + if (isTypes || isRelationships) { + metadata = new SaxesParser({ xmlns: true }); + metadata.on("error", () => fail("ARCHIVE_INVALID")); + // No DTD/entity expansion or external XML resources in the admission guard. + metadata.on("doctype", () => fail("ARCHIVE_INVALID")); + metadata.on("opentag", (tag) => { + const attr = (name: string) => tag.attributes[name]?.value; + if ( + isTypes && + attr("ContentType")?.toLowerCase().startsWith("image/") + ) { + const part = attr("PartName"); + const extension = attr("Extension"); + if (part) imageParts.add(mainPart(part)); + if (extension) imageExtensions.add(extension.toLowerCase()); + } + if (!isRelationships || tag.local !== "Relationship") return; + const type = attr("Type") ?? ""; + if (/\/(image|oleObject|package)$/.test(type)) media = true; + if ( + entry.fileName !== "_rels/.rels" || + !type.endsWith("/officeDocument") + ) + return; + if ( + main || + attr("TargetMode") === "External" || + !attr("Target") + ) { + fail("ARCHIVE_INVALID"); + return; + } + main = mainPart(attr("Target")!); + }); + } zip.openReadStream(entry, (streamError, stream) => { if (streamError || !stream) { fail("ARCHIVE_INVALID"); return; } + active = stream; let bytes = 0; + let decoder: TextDecoder | undefined; stream.on("data", (chunk: Buffer) => { + if (settled) return; bytes += chunk.byteLength; total += chunk.byteLength; if ( bytes > limits.maxEntryBytes || total > limits.maxArchiveBytes ) { - stream.destroy(); fail("ZIP_BOMB"); + return; + } + if (!metadata) return; + try { + decoder ??= new TextDecoder( + chunk[0] === 0xff && chunk[1] === 0xfe + ? "utf-16le" + : chunk[0] === 0xfe && chunk[1] === 0xff + ? "utf-16be" + : "utf-8", + { fatal: true }, + ); + metadata.write(decoder.decode(chunk, { stream: true })); + } catch { + fail("ARCHIVE_INVALID"); } }); stream.on("error", () => fail("ARCHIVE_INVALID")); stream.on("end", () => { + active = undefined; + if (settled) return; + try { + metadata?.write(decoder?.decode() ?? "").close(); + } catch { + fail("ARCHIVE_INVALID"); + } if (!settled) zip.readEntry(); }); }); @@ -107,7 +196,7 @@ function inspectArchive(path: string, limits: WorkerRequest): Promise { async function extract(request: WorkerRequest): Promise { const media = await inspectArchive(request.path, request); - // Loading a native binding is itself isolated and only happens after refusal guards. + // Loading a native binding is itself isolated and follows all refusal guards. const require = createRequire(import.meta.url); let anydoc: typeof import("@firecrawl/anydoc"); try { @@ -141,13 +230,11 @@ async function extract(request: WorkerRequest): Promise { } if (import.meta.main) { - let maxOutput = 15 * 1024 * 1024; let serialized: string; try { const request = requestSchema.parse(JSON.parse(await Bun.stdin.text())); - maxOutput = request.maxOutputBytes; serialized = JSON.stringify({ ok: true, result: await extract(request) }); - if (Buffer.byteLength(serialized) > maxOutput) + if (Buffer.byteLength(serialized) > request.maxOutputBytes) throw new ExtractionError("PARSER_OUTPUT_LIMIT"); } catch (error) { serialized = JSON.stringify({ diff --git a/service/test/extraction.test.ts b/service/test/extraction.test.ts index ae3270c2..7ad6e3c9 100644 --- a/service/test/extraction.test.ts +++ b/service/test/extraction.test.ts @@ -1,6 +1,6 @@ import { afterEach, beforeEach, expect, test } from "bun:test"; import { SignJWT } from "jose"; -import { unzipSync, zipSync } from "fflate"; +import { unzipSync, zipSync, strFromU8, strToU8 } from "fflate"; import { mkdtemp, readdir, readFile, rm } from "node:fs/promises"; import { tmpdir } from "node:os"; import { join } from "node:path"; @@ -117,6 +117,59 @@ test("embedded image or object cannot claim complete inspection", async () => { } await clean(); }); +test("image relationships and content types identify non-image filenames", async () => { + const data = unzipSync(fixture); + data["word/media/image.bin"] = new Uint8Array([1, 2, 3]); + data["[Content_Types].xml"] = strToU8( + strFromU8(data["[Content_Types].xml"]!).replace( + "", + '', + ), + ); + let response = await post(app(), form(zipSync(data))); + expect(response.status).toBe(200); + expect(resultSchema.parse(await response.json()).may_omit_content).toBe(true); + data["[Content_Types].xml"] = unzipSync(fixture)["[Content_Types].xml"]!; + data["word/_rels/document.xml.rels"] = strToU8( + '', + ); + response = await post(app(), form(zipSync(data))); + expect(response.status).toBe(200); + expect(resultSchema.parse(await response.json()).completeness).toBe( + "partial", + ); + await clean(); +}); +test("relationship-resolved main documents do not need a word directory", async () => { + const data = unzipSync(fixture); + for (const name of Object.keys(data)) { + if (!name.startsWith("word/")) continue; + data[name.replace("word/", "content/")] = data[name]!; + delete data[name]; + } + for (const name of ["_rels/.rels", "[Content_Types].xml"]) { + data[name] = strToU8( + strFromU8(data[name]!).replaceAll("word/", "content/"), + ); + } + const response = await post(app(), form(zipSync(data))); + expect(response.status).toBe(200); + const result = resultSchema.parse(await response.json()); + expect(result.text).toBe(expectedText); + await clean(); +}); +test("external main-part relationships and DTD metadata are hard refusals", async () => { + const data = unzipSync(fixture); + data["_rels/.rels"] = strToU8( + '', + ); + await code(await post(app(), form(zipSync(data))), 422, "ARCHIVE_INVALID"); + data["_rels/.rels"] = strToU8( + ']>&x;', + ); + await code(await post(app(), form(zipSync(data))), 422, "ARCHIVE_INVALID"); + await clean(); +}); test("cover thumbnail does not cause paid OCR escalation", async () => { const response = await post( app(), @@ -270,6 +323,36 @@ test("parent caps IPC even if a child ignores its output limit", async () => { ); await clean(); }); +test("native children do not reload secrets from dotenv files", async () => { + await Bun.write( + join(tempRoot, ".env"), + "NATIVE_ENV_CANARY=should-not-reach-parser\n", + ); + const result = { + profile: "document-v1", + text: "isolated", + format: "markdown", + completeness: "complete", + may_omit_content: false, + pages_needing_ocr: [], + truncated: false, + parser: { name: "anydoc", version: "0.1.3" }, + }; + const script = `const result = ${JSON.stringify(result)}; if (process.env.NATIVE_ENV_CANARY) result.text = "leaked"; console.log(JSON.stringify({ok:true,result}));`; + const response = await runWorker( + { + path: "unused", + maxOutputBytes: 4096, + maxEntryBytes: 4096, + maxArchiveBytes: 4096, + maxEntries: 5, + }, + new AbortController().signal, + [process.execPath, "--cwd", tempRoot, "-e", script], + ); + expect(response.text).toBe("isolated"); +}); + test("child crash and malformed response are sanitized and permit retry", async () => { for (const source of [ "process.exit(11)", @@ -361,6 +444,7 @@ test("overload refuses before reading or staging another request body", async () test("body limit is counted for chunked requests, not trusted Content-Length", async () => { const application = app({ maxFileBytes: 100, maxBodyBytes: 200 }); let pulls = 0; + let cancelled = false; const prefix = '--test\r\nContent-Disposition: form-data; name="file"; filename="r.docx"\r\nContent-Type: ' + DOCX_TYPE + @@ -373,6 +457,9 @@ test("body limit is counted for chunked requests, not trusted Content-Length", a new TextEncoder().encode(pulls === 1 ? prefix : "x".repeat(256)), ); }, + cancel() { + cancelled = true; + }, }, { highWaterMark: 0 }, ); @@ -386,5 +473,6 @@ test("body limit is counted for chunked requests, not trusted Content-Length", a }); await code(await application.fetch(request), 413, "PARSER_INPUT_LIMIT"); expect(pulls).toBeLessThan(5); + expect(cancelled).toBe(true); await clean(); }); diff --git a/service/test/lifecycle.test.ts b/service/test/lifecycle.test.ts index b2d29388..577c87b1 100644 --- a/service/test/lifecycle.test.ts +++ b/service/test/lifecycle.test.ts @@ -6,6 +6,8 @@ import { join } from "node:path"; import { fileURLToPath } from "node:url"; import { Admission } from "../src/admission"; import { createApp } from "../src/app"; +import { configSchema } from "../src/config"; +import { listen } from "../src/server"; import { DOCX_TYPE } from "../src/contract"; import { runWorker, type Runner } from "../src/process"; @@ -71,6 +73,33 @@ test("cancelled queued work never consumes a slot and remaining work stays FIFO" next(); }); +test("listener waits beyond ten seconds for the configured extraction deadline", async () => { + const tempRoot = await mkdtemp(join(tmpdir(), "rag-long-parse-")); + const config = configSchema.parse({ + enabled: true, + secret, + tempRoot, + timeoutMs: 14_000, + }); + const runner: Runner = async (request, signal) => { + await Bun.sleep(11_000); + return runWorker(request, signal); + }; + const server = listen(createApp(config, runner), config, 0, "127.0.0.1"); + try { + const response = await fetch(`http://127.0.0.1:${server.port}/v1/extract`, { + method: "POST", + headers: await headers(), + body: form(), + }); + expect(response.status).toBe(200); + expect((await response.json()).text).toContain("Quarterly Report"); + } finally { + await server.stop(true); + await rm(tempRoot, { recursive: true, force: true }); + } +}, 20_000); + test("actual HTTP disconnect kills the native child and cleans up before retry", async () => { const tempRoot = await mkdtemp(join(tmpdir(), "rag-disconnect-")); let calls = 0;