From 8b52b1dcc73926c4aea2a60b79a2ef12d00c59fb Mon Sep 17 00:00:00 2001 From: Pritom Mazumdar Date: Tue, 31 Mar 2026 01:57:30 +0530 Subject: [PATCH 1/5] PDF extraction, /v1/estimate endpoint, LangChain callback, bump to v0.3.1 --- README.md | 194 ++++++++++++++++++++++--------- pyproject.toml | 4 + tests/test_estimate.py | 149 ++++++++++++++++++++++++ tests/test_langchain_callback.py | 118 +++++++++++++++++++ tests/test_pdf.py | 139 ++++++++++++++++++++++ token0/api/v1/chat.py | 27 +++++ token0/api/v1/estimate.py | 120 +++++++++++++++++++ token0/langchain_callback.py | 111 ++++++++++++++++++ token0/litellm_hook.py | 27 +++++ token0/main.py | 2 + token0/optimization/pdf.py | 57 +++++++++ 11 files changed, 896 insertions(+), 52 deletions(-) create mode 100644 tests/test_estimate.py create mode 100644 tests/test_langchain_callback.py create mode 100644 tests/test_pdf.py create mode 100644 token0/api/v1/estimate.py create mode 100644 token0/langchain_callback.py create mode 100644 token0/optimization/pdf.py diff --git a/README.md b/README.md index de34372..e4d51b9 100644 --- a/README.md +++ b/README.md @@ -8,12 +8,13 @@ Send images to LLMs through Token0. Same accuracy. Fraction of the cost. ## Why Token0 Exists -Every time you send an image to GPT-4o, Claude, or Gemini, you're paying for **vision tokens** — and most of them are wasted. +Every time you send an image to GPT-4.1, Claude, or Gemini, you're paying for **vision tokens** — and most of them are wasted. - A 4000x3000 photo costs **~1,590 tokens** on Claude. The model auto-downscales it to 1568px internally — you paid for pixels that got thrown away. -- A screenshot of a document costs **~765 tokens** on GPT-4o as an image. The same information extracted as text costs **~30 tokens**. That's a **25x markup** for the same answer. -- A simple "classify this image" prompt on GPT-4o uses high-detail mode at **1,105 tokens**. Low-detail mode gives the same answer for **85 tokens** — 13x cheaper. -- A 1280x720 image on GPT-4o creates 4 tiles (765 tokens). Resizing to tile boundaries gives 2 tiles (425 tokens) — 44% cheaper with zero quality loss. +- A screenshot of a document costs **~765 tokens** on GPT-4.1 as an image. The same information extracted as text costs **~30 tokens**. That's a **25x markup** for the same answer. +- A simple "classify this image" prompt on GPT-4.1 uses high-detail mode at **1,105 tokens**. Low-detail mode gives the same answer for **85 tokens** — 13x cheaper. +- A 1280x720 image on GPT-4.1 creates 4 tiles (765 tokens). Resizing to tile boundaries gives 2 tiles (425 tokens) — 44% cheaper with zero quality loss. +- A PDF invoice sent as an image costs ~765 tokens. Extracting its text layer costs ~50 tokens — a **15x markup** for the same data. **The problem**: Text token optimization is mature (prompt caching, compression, smart routing). But for images — the modality that costs 2-5x more per token — almost **no optimization tooling exists**. @@ -29,29 +30,31 @@ Your App → Token0 Proxy → [Analyze → Classify → Route → Transform → Database (logs every optimization decision + savings) ``` -Token0 applies **9 optimizations** automatically: +Token0 applies **10 optimizations** automatically: ### Core Optimizations (Free Tier) -**1. Smart Resize** — Auto-downscale images to the max resolution each model actually processes (Claude: 1568px, GPT-4o: 2048px). Most apps send 4000px images that get silently downscaled by the provider. +**1. Smart Resize** — Auto-downscale images to the max resolution each model actually processes (Claude: 1568px, GPT-4.1: 2048px). Most apps send 4000px images that get silently downscaled by the provider. **2. OCR Routing** — Detect when an image is mostly text (screenshots, documents, invoices, receipts) and extract text via OCR instead. Text tokens cost 10-50x less than vision tokens. Uses a multi-signal heuristic (background uniformity, color variance, horizontal line structure, edge density) — validated at 91% accuracy on real-world images. -**3. JPEG Recompression** — Convert PNG screenshots (large files) to optimized JPEG (smaller payload, faster upload) when transparency isn't needed. +**3. PDF Text Layer Extraction** — When a PDF is sent as a content part, extract its text layer using `pypdf` instead of rendering as an image. Typed/digital PDFs almost always have a text layer. A 2-page invoice PDF: ~1,530 vision tokens → ~80 text tokens. Falls back to passthrough for scanned PDFs with no text layer. + +**4. JPEG Recompression** — Convert PNG screenshots (large files) to optimized JPEG (smaller payload, faster upload) when transparency isn't needed. ### Advanced Optimizations -**4. Prompt-Aware Detail Mode** — Analyze the *prompt* to decide detail level, not just the image. "Classify this image" → low detail (85 tokens). "Extract all text" → high detail. A keyword classifier on the prompt text can cut costs 3-13x per image. +**5. Prompt-Aware Detail Mode** — Analyze the *prompt* to decide detail level, not just the image. "Classify this image" → low detail (85 tokens). "Extract all text" → high detail. A keyword classifier on the prompt text can cut costs 3-13x per image. -**5. Tile-Optimized Resize** — OpenAI tiles images into 512x512 blocks. A 1280x720 image creates 4 tiles (765 tokens). Token0 resizes to optimal tile boundaries: 2 tiles (425 tokens) — 44% savings with zero quality loss. +**6. Tile-Optimized Resize** — OpenAI tiles images into 512x512 blocks. A 1280x720 image creates 4 tiles (765 tokens). Token0 resizes to optimal tile boundaries: 2 tiles (425 tokens) — 44% savings with zero quality loss. -**6. Model Cascade** — Not all images need GPT-4o. Token0 auto-routes simple tasks to cheaper models: GPT-4o → GPT-4o-mini (16.7x cheaper), Claude Opus → Claude Haiku (6.25x cheaper). Complex tasks stay on the flagship model. +**7. Model Cascade** — Not all images need GPT-4.1. Token0 auto-routes simple tasks to cheaper models: GPT-4.1 → GPT-4.1-mini (5x cheaper) → GPT-4.1-nano (20x cheaper), Claude Opus → Claude Haiku (6.25x cheaper). Complex tasks stay on the flagship model. -**7. Semantic Response Cache** — Cache responses for similar image+prompt pairs using perceptual image hashing. Repeated or similar queries cost 0 tokens. Effective on repetitive workloads (product classification, document processing). +**8. Semantic Response Cache** — Cache responses for similar image+prompt pairs using perceptual image hashing. Repeated or similar queries cost 0 tokens. Effective on repetitive workloads (product classification, document processing). -**8. QJL-Compressed Fuzzy Cache** — Similar (not just identical) images hit the cache using Quantized Johnson-Lindenstrauss random projection. Compresses 256-bit perceptual hashes to 128-bit binary signatures, matches via Hamming distance. Inspired by Google's TurboQuant (arXiv 2504.19874). **62% additional token savings** on image variations in benchmarks — similar product photos, re-scanned documents, and slightly different angles all hit cache. +**9. QJL-Compressed Fuzzy Cache** — Similar (not just identical) images hit the cache using Quantized Johnson-Lindenstrauss random projection. Compresses 256-bit perceptual hashes to 128-bit binary signatures, matches via Hamming distance. Inspired by Google's TurboQuant (arXiv 2504.19874). **62% additional token savings** on image variations in benchmarks — similar product photos, re-scanned documents, and slightly different angles all hit cache. -**9. Video Optimization** — Automatically extract keyframes from video at 1fps, deduplicate similar consecutive frames using QJL perceptual hashing, detect scene changes via pixel-level diff, and run each keyframe through the full image optimization pipeline. A 60-second video at 30fps (1,800 frames) reduces to ~10 keyframes before being sent to the LLM. **13-45% savings on local models; ~83% projected savings on GPT-4o.** Optional CLIP-based query-frame scoring (Layer 2) ranks frames by relevance to the user's prompt. +**10. Video Optimization** — Automatically extract keyframes from video at 1fps, deduplicate similar consecutive frames using QJL perceptual hashing, detect scene changes via pixel-level diff, and run each keyframe through the full image optimization pipeline. A 60-second video at 30fps (1,800 frames) reduces to ~10 keyframes before being sent to the LLM. **13-45% savings on local models; ~83% projected savings on GPT-4.1.** Optional CLIP-based query-frame scoring (Layer 2) ranks frames by relevance to the user's prompt. --- @@ -143,63 +146,64 @@ Test setup: 3 videos (product showcase, document montage, mixed content), naive **Why moondream shows less video savings:** moondream uses a very small frame encoder — its per-frame token cost is already low, so frame dedup has less absolute impact than on higher-token models. -### GPT-4o Video Extrapolation (ballpark) +### GPT-4.1 Video Extrapolation (ballpark) -Using OpenAI's published tile formula (512px tiles, 170 tokens/tile): +Using OpenAI's published tile formula (512px tiles, 170 tokens/tile) and GPT-4.1 pricing ($2.00/1M tokens): | Scenario | Naive | Token0 | Savings | |---|---|---|---| | 60s video, 30fps (1,800 frames → 1fps → 60 frames → dedup to ~10) | ~25,500 tokens | ~4,250 tokens | **~83%** | -| Monthly cost at 10K videos/day (GPT-4o $2.50/1M tokens) | $19,125/mo | $3,188/mo | **$15,938/mo saved** | +| Monthly cost at 10K videos/day | $15,300/mo | $2,550/mo | **$12,750/mo saved** | ### Anthropic Video Extrapolation (ballpark) -Using Anthropic's pixel formula (tokens ≈ width × height / 750): +Using Anthropic's pixel formula (tokens ≈ width × height / 750) and Claude Sonnet pricing ($3/1M tokens): | Scenario | Naive | Token0 | Savings | |---|---|---|---| | 60s video, 1fps = 60 frames at 1280×720 | ~73,700 tokens | ~12,300 tokens | **~83%** | -| Monthly cost at 1K videos/day (Claude Sonnet $3/1M tokens) | $6,633/mo | $1,107/mo | **$5,526/mo saved** | +| Monthly cost at 1K videos/day | $6,633/mo | $1,107/mo | **$5,526/mo saved** | > These are linear extrapolations from the token formula + observed dedup ratios (60 frames → ~10 keyframes). Actual savings vary by content type — talking-head video deduplicates more aggressively than action scenes. -### GPT-4o Image Cost Projections (v1 vs v2) +### GPT-4.1 Image Cost Projections (v1 vs v2) -Using OpenAI's published token formulas on real images: +Using OpenAI's published token formulas on real images and GPT-4.1 pricing ($2.00/1M input tokens): | Optimization Level | Per-Image Cost | Savings | 100K imgs/day Monthly | |---|---|---|---| -| Direct GPT-4o (no Token0) | $0.002253 | — | $6,758 | -| **Token0 v1** (resize + OCR + basic detail) | $0.000669 | **70.3%** | $2,006 | -| **Token0 v2** (+ prompt-aware + tile resize + cascade) | $0.000025 | **98.9%** | $74 | +| Direct GPT-4.1 (no Token0) | $0.001802 | — | $5,406 | +| **Token0 v1** (resize + OCR + PDF + basic detail) | $0.000535 | **70.3%** | $1,604 | +| **Token0 v2** (+ prompt-aware + tile resize + cascade) | $0.000020 | **98.9%** | $59 | **v2 monthly savings at scale:** | Scale | Direct Cost | Token0 v2 Cost | Monthly Savings | |---|---|---|---| -| 1K images/day | $67.58 | $0.74 | **$66.83** | -| 10K images/day | $675.75 | $7.45 | **$668.30** | -| 100K images/day | $6,757.50 | $74.47 | **$6,683.03** | -| 500K images/day | $33,787.50 | $372.38 | **$33,415.12** | +| 1K images/day | $54.05 | $0.59 | **$53.46** | +| 10K images/day | $540.54 | $5.94 | **$534.60** | +| 100K images/day | $5,406 | $59.46 | **$5,346** | +| 500K images/day | $27,032 | $297 | **$26,735** | -> **Note**: v2 projections include model cascade (simple tasks → GPT-4o-mini at $0.15/1M tokens vs GPT-4o at $2.50/1M). Semantic cache hits (est. 20% on repetitive workloads) would add further savings on top. +> **Note**: v2 projections include model cascade (simple tasks → GPT-4.1-mini at $0.40/1M tokens vs GPT-4.1 at $2.00/1M). Semantic cache hits (est. 20% on repetitive workloads) would add further savings on top. ### Key Findings 1. **OCR routing delivers 47-70% token savings** on text-heavy images across all models tested. -2. **Smart resize saves 1-6 seconds of latency** on large photos — even when local models report flat token counts. -3. **Photos are never falsely OCR-routed** — the multi-signal text detection heuristic correctly identifies photos vs documents at 91% accuracy. -4. **Text-only passthrough adds zero overhead** — 0 extra tokens across all text-only tests. -5. **Prompt-aware detail mode** drops simple queries from 1,105 → 85 tokens (92% savings) on GPT-4o. -6. **Model cascade** routes simple tasks at 16.7x cheaper rates with equivalent quality. -7. **Tile-optimized resize** cuts OpenAI costs by 44% on mid-size images (1280x720) with zero quality loss. -8. **On cloud APIs, total image savings reach 98.9%** when all optimizations are combined with model cascading. -9. **Video deduplication collapses 60-frame clips to ~10 keyframes** — 13-45% savings on local models, ~83% projected on GPT-4o. -10. **Model-aware OCR skip is critical** — ultra-efficient encoders like llama3.2-vision use <50 tokens/image; OCR text output would cost more, not less. +2. **PDF text layer extraction beats OCR** for typed documents — direct text extraction, no OCR model needed, ~15x cheaper than sending as image. +3. **Smart resize saves 1-6 seconds of latency** on large photos — even when local models report flat token counts. +4. **Photos are never falsely OCR-routed** — the multi-signal text detection heuristic correctly identifies photos vs documents at 91% accuracy. +5. **Text-only passthrough adds zero overhead** — 0 extra tokens across all text-only tests. +6. **Prompt-aware detail mode** drops simple queries from 1,105 → 85 tokens (92% savings) on GPT-4.1. +7. **Model cascade** routes simple tasks 5-20x cheaper (GPT-4.1 → GPT-4.1-nano) with equivalent quality. +8. **Tile-optimized resize** cuts OpenAI costs by 44% on mid-size images (1280x720) with zero quality loss. +9. **On cloud APIs, total image savings reach 98.9%** when all optimizations are combined with model cascading. +10. **Video deduplication collapses 60-frame clips to ~10 keyframes** — 13-45% savings on local models, ~83% projected on GPT-4.1. +11. **Model-aware OCR skip is critical** — ultra-efficient encoders like llama3.2-vision use <50 tokens/image; OCR text output would cost more, not less. ### Additional Test Coverage -Token0 includes **148 unit tests** and benchmarks across multiple suites: +Token0 includes **171 unit tests** and benchmarks across multiple suites: | Suite | Tests | What It Validates | |---|---|---| @@ -213,6 +217,9 @@ Token0 includes **148 unit tests** and benchmarks across multiple suites: | `litellm` | 10 | LiteLLM hook: passthrough, optimization, OCR, cascade, async | | `cache` | 23 | QJL fuzzy cache: perceptual hash, JL compression, Hamming distance, fuzzy match | | `video` | 22 | Frame extraction, QJL dedup, scene detection, CLIP scoring, full pipeline | +| `pdf` | 8 | PDF detection, decode, text extraction, token estimation | +| `estimate` | 11 | /v1/estimate endpoint: single image, multiple images, remote URL skip, cost calc | +| `langchain` | 8 | LangChain callback: import, text passthrough, image optimization, role mapping | --- @@ -224,8 +231,6 @@ Token0 includes **148 unit tests** and benchmarks across multiple suites: pip install token0 ``` -Create a `.env` file with your API key: - Add your LLM provider API key to `.env`: ```bash # At least one of these: @@ -264,7 +269,7 @@ client = OpenAI( # Same code, nothing else changes response = client.chat.completions.create( - model="gpt-4o", # or claude-sonnet-4-6, gemini-2.5-flash + model="gpt-4.1", # or claude-sonnet-4-6, gemini-2.5-flash messages=[{ "role": "user", "content": [ @@ -281,13 +286,74 @@ response = client.chat.completions.create( # response.token0.optimizations_applied = ["resize 4000x3000 → 1568x1176", "convert png → jpeg q=85"] ``` +### Pre-Call Cost Estimate + +Check what a request will cost **before** making any LLM call — no API key needed: + +```bash +curl -X POST http://localhost:8000/v1/estimate \ + -H "Content-Type: application/json" \ + -d '{ + "model": "gpt-4.1", + "messages": [{ + "role": "user", + "content": [ + {"type": "text", "text": "Describe this image"}, + {"type": "image_url", "image_url": {"url": "data:image/jpeg;base64,..."}} + ] + }] + }' +``` + +```json +{ + "model": "gpt-4.1", + "provider": "openai", + "images": [{ + "original_tokens": 1105, + "optimized_tokens": 85, + "tokens_saved": 1020, + "cost_saved_usd": 0.00204, + "optimizations": ["prompt-aware → low detail (simple task)"] + }], + "total_original_tokens": 1105, + "total_optimized_tokens": 85, + "total_tokens_saved": 1020, + "total_cost_saved_usd": 0.00204 +} +``` + +### PDF Support + +Send PDFs directly — Token0 extracts the text layer automatically: + +```python +import base64 + +with open("invoice.pdf", "rb") as f: + pdf_b64 = base64.b64encode(f.read()).decode() + +response = client.chat.completions.create( + model="gpt-4.1", + messages=[{ + "role": "user", + "content": [ + {"type": "text", "text": "What is the total amount on this invoice?"}, + {"type": "image_url", "image_url": {"url": f"data:application/pdf;base64,{pdf_b64}"}} + ] + }], + extra_headers={"X-Provider-Key": "sk-..."} +) +# PDF text layer extracted — ~80 tokens instead of ~765 vision tokens +``` + ### Video Support Send a video URL or base64-encoded video — Token0 automatically extracts keyframes, deduplicates, and optimizes before forwarding: ```python response = client.chat.completions.create( - model="gpt-4o", + model="gpt-4.1", messages=[{ "role": "user", "content": [ @@ -298,7 +364,7 @@ response = client.chat.completions.create( extra_headers={"X-Provider-Key": "sk-..."} ) # 1,800 raw frames → ~10 keyframes → optimized images → LLM -# response.token0.tokens_saved = 21,250 (~83% on GPT-4o) +# ~83% savings on GPT-4.1 ``` ### Streaming Support @@ -307,7 +373,7 @@ Token0 supports `stream=true` — images are optimized before streaming begins, ```python stream = client.chat.completions.create( - model="gpt-4o", + model="gpt-4.1", messages=[{ "role": "user", "content": [ @@ -337,7 +403,7 @@ litellm.callbacks = [Token0Hook()] # All your existing litellm calls now get image optimization for free response = litellm.completion( - model="gpt-4o", + model="gpt-4.1", messages=[{ "role": "user", "content": [ @@ -358,6 +424,29 @@ litellm_settings: callbacks: ["token0.litellm_hook.Token0Hook"] ``` +### Use With LangChain + +Already using LangChain? Add Token0 as a callback to any chat model: + +```bash +pip install token0[langchain] +``` + +```python +from token0.langchain_callback import Token0Callback +from langchain_openai import ChatOpenAI + +llm = ChatOpenAI(model="gpt-4.1", callbacks=[Token0Callback()]) + +# All calls through this llm now get image optimization automatically +response = llm.invoke([HumanMessage(content=[ + {"type": "text", "text": "What's in this image?"}, + {"type": "image_url", "image_url": {"url": "data:image/jpeg;base64,..."}} +])]) +``` + +Works with any LangChain chat model — ChatOpenAI, ChatAnthropic, ChatGoogleGenerativeAI, etc. + ### Use With Ollama (free, fully local) ```bash @@ -468,6 +557,7 @@ S3_BUCKET=token0-images | Method | Path | Description | |--------|------|-------------| | POST | `/v1/chat/completions` | Optimized chat completion (OpenAI-compatible, supports `stream=true`) | +| POST | `/v1/estimate` | Pre-call token cost estimator — no LLM call, no API key needed | | GET | `/v1/usage` | Usage and savings dashboard | | GET | `/health` | Health check + storage mode | @@ -475,7 +565,7 @@ S3_BUCKET=token0-images | Header | Required | Description | |--------|----------|-------------| -| `X-Provider-Key` | Yes | Your LLM provider API key (OpenAI/Anthropic/Google/Ollama) | +| `X-Provider-Key` | Yes (chat only) | Your LLM provider API key (OpenAI/Anthropic/Google/Ollama) | | `X-Token0-Key` | No | Token0 API key for usage tracking | ### Token0-Specific Request Parameters @@ -495,20 +585,20 @@ Standard OpenAI-compatible response with an additional `token0` field: { "id": "token0-abc123", "object": "chat.completion", - "model": "gpt-4o-mini", + "model": "gpt-4.1-mini", "choices": [...], "usage": {"prompt_tokens": 85, "completion_tokens": 50, "total_tokens": 135}, "token0": { "original_prompt_tokens_estimate": 1105, "optimized_prompt_tokens": 85, "tokens_saved": 1020, - "cost_saved_usd": 0.002550, + "cost_saved_usd": 0.002040, "optimizations_applied": [ "prompt-aware → low detail (simple task)", - "cascade → gpt-4o-mini (simple task)" + "cascade → gpt-4.1-mini (simple task)" ], "cache_hit": false, - "model_cascaded_to": "gpt-4o-mini" + "model_cascaded_to": "gpt-4.1-mini" } } ``` @@ -519,10 +609,10 @@ Standard OpenAI-compatible response with an additional `token0` field: | Provider | Models | Notes | |---|---|---| -| **OpenAI** | GPT-4o, GPT-4o-mini, GPT-4.1, GPT-4.1-mini, GPT-4.1-nano | Detail mode + tile optimization | +| **OpenAI** | GPT-4.1, GPT-4.1-mini, GPT-4.1-nano, GPT-4o, GPT-4o-mini | Detail mode + tile optimization | | **Anthropic** | Claude Sonnet 4.6, Claude Opus 4.6, Claude Haiku 4.5 | Pixel-based token formula | | **Google** | Gemini 2.5 Flash, Gemini 2.5 Pro | | -| **Ollama** | moondream, llava, llava-llama3, minicpm-v, any vision model | Free, local inference | +| **Ollama** | moondream, llava, llava-llama3, minicpm-v, gemma3, granite3.2-vision, any vision model | Free, local inference | --- diff --git a/pyproject.toml b/pyproject.toml index 20a9674..a50fdaf 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -21,9 +21,13 @@ dependencies = [ "anthropic>=0.40.0", "openai>=1.50.0", "google-genai>=1.0.0", + "pypdf>=4.0.0", ] [project.optional-dependencies] +langchain = [ + "langchain-core>=0.2.0", +] full = [ "asyncpg>=0.30.0", "redis>=5.0.0", diff --git a/tests/test_estimate.py b/tests/test_estimate.py new file mode 100644 index 0000000..09817ca --- /dev/null +++ b/tests/test_estimate.py @@ -0,0 +1,149 @@ +"""Tests for the /v1/estimate endpoint.""" + +import asyncio +import base64 +import io + +from PIL import Image + + +def _make_image_data_uri(width: int = 800, height: int = 600, fmt: str = "JPEG") -> str: + img = Image.new("RGB", (width, height), color=(100, 150, 200)) + buf = io.BytesIO() + img.save(buf, format=fmt) + b64 = base64.b64encode(buf.getvalue()).decode() + mime = "image/jpeg" if fmt == "JPEG" else "image/png" + return f"data:{mime};base64,{b64}" + + +class TestEstimateEndpoint: + def test_estimate_single_image(self): + from token0.api.v1.estimate import EstimateRequest, estimate + from token0.models.request import ContentPart, ImageUrl, Message + + req = EstimateRequest( + model="gpt-4o", + messages=[ + Message( + role="user", + content=[ + ContentPart(type="text", text="What's in this image?"), + ContentPart( + type="image_url", + image_url=ImageUrl(url=_make_image_data_uri(800, 600)), + ), + ], + ) + ], + ) + result = asyncio.run(estimate(req)) + + assert result.model == "gpt-4o" + assert result.provider == "openai" + assert len(result.images) == 1 + assert result.images[0].original_tokens > 0 + assert result.total_original_tokens > 0 + + def test_estimate_returns_savings(self): + from token0.api.v1.estimate import EstimateRequest, estimate + from token0.models.request import ContentPart, ImageUrl, Message + + # Large image — should trigger resize, yielding savings + req = EstimateRequest( + model="gpt-4o", + messages=[ + Message( + role="user", + content=[ + ContentPart( + type="image_url", + image_url=ImageUrl(url=_make_image_data_uri(3000, 2000)), + ) + ], + ) + ], + ) + result = asyncio.run(estimate(req)) + assert result.total_original_tokens >= result.total_optimized_tokens + + def test_estimate_text_only_no_images(self): + from token0.api.v1.estimate import EstimateRequest, estimate + from token0.models.request import Message + + req = EstimateRequest( + model="gpt-4o", + messages=[Message(role="user", content="Just a text message")], + ) + result = asyncio.run(estimate(req)) + assert result.images == [] + assert result.total_original_tokens == 0 + + def test_estimate_remote_url_skipped_with_note(self): + from token0.api.v1.estimate import EstimateRequest, estimate + from token0.models.request import ContentPart, ImageUrl, Message + + req = EstimateRequest( + model="gpt-4o", + messages=[ + Message( + role="user", + content=[ + ContentPart( + type="image_url", + image_url=ImageUrl(url="https://example.com/image.jpg"), + ) + ], + ) + ], + ) + result = asyncio.run(estimate(req)) + assert result.images == [] + assert result.note is not None + assert "remote" in result.note.lower() + + def test_estimate_multiple_images(self): + from token0.api.v1.estimate import EstimateRequest, estimate + from token0.models.request import ContentPart, ImageUrl, Message + + req = EstimateRequest( + model="claude-sonnet-4-6", + messages=[ + Message( + role="user", + content=[ + ContentPart( + type="image_url", + image_url=ImageUrl(url=_make_image_data_uri(800, 600)), + ), + ContentPart( + type="image_url", + image_url=ImageUrl(url=_make_image_data_uri(400, 300)), + ), + ], + ) + ], + ) + result = asyncio.run(estimate(req)) + assert len(result.images) == 2 + assert result.provider == "anthropic" + + def test_estimate_cost_saved_is_non_negative(self): + from token0.api.v1.estimate import EstimateRequest, estimate + from token0.models.request import ContentPart, ImageUrl, Message + + req = EstimateRequest( + model="gpt-4o", + messages=[ + Message( + role="user", + content=[ + ContentPart( + type="image_url", + image_url=ImageUrl(url=_make_image_data_uri(100, 100)), + ) + ], + ) + ], + ) + result = asyncio.run(estimate(req)) + assert result.total_cost_saved_usd >= 0 diff --git a/tests/test_langchain_callback.py b/tests/test_langchain_callback.py new file mode 100644 index 0000000..bfb4c1b --- /dev/null +++ b/tests/test_langchain_callback.py @@ -0,0 +1,118 @@ +"""Tests for the LangChain callback handler.""" + +import base64 +import io + +import pytest +from PIL import Image + + +def _make_image_data_uri(width: int = 800, height: int = 600) -> str: + img = Image.new("RGB", (width, height), color=(100, 150, 200)) + buf = io.BytesIO() + img.save(buf, format="JPEG") + b64 = base64.b64encode(buf.getvalue()).decode() + return f"data:image/jpeg;base64,{b64}" + + +class TestToken0Callback: + def test_import(self): + from token0.langchain_callback import Token0Callback + + cb = Token0Callback() + assert cb is not None + + def test_init_defaults(self): + from token0.langchain_callback import Token0Callback + + cb = Token0Callback() + assert cb.enable_cascade is False + assert cb.detail_override is None + + def test_init_custom(self): + from token0.langchain_callback import Token0Callback + + cb = Token0Callback(enable_cascade=True, detail_override="low") + assert cb.enable_cascade is True + assert cb.detail_override == "low" + + def test_text_only_message_unchanged(self): + """Text-only messages should pass through without modification.""" + from token0.langchain_callback import Token0Callback + + try: + from langchain_core.messages import HumanMessage + except ImportError: + pytest.skip("langchain-core not installed") + + cb = Token0Callback() + msg = HumanMessage(content="Hello, what is 2+2?") + original_content = msg.content + + cb.on_chat_model_start( + serialized={"kwargs": {"model_name": "gpt-4o"}}, + messages=[[msg]], + ) + + assert msg.content == original_content + + def test_image_message_content_is_list(self): + """After optimization, image message content remains a list.""" + from token0.langchain_callback import Token0Callback + + try: + from langchain_core.messages import HumanMessage + except ImportError: + pytest.skip("langchain-core not installed") + + cb = Token0Callback() + content = [ + {"type": "text", "text": "Describe this image"}, + { + "type": "image_url", + "image_url": {"url": _make_image_data_uri(800, 600)}, + }, + ] + msg = HumanMessage(content=content) + + cb.on_chat_model_start( + serialized={"kwargs": {"model_name": "gpt-4o"}}, + messages=[[msg]], + ) + + assert isinstance(msg.content, list) + + def test_empty_serialized_does_not_crash(self): + """Missing model name in serialized should not crash.""" + from token0.langchain_callback import Token0Callback + + try: + from langchain_core.messages import HumanMessage + except ImportError: + pytest.skip("langchain-core not installed") + + cb = Token0Callback() + msg = HumanMessage(content="hello") + + # Should not raise + cb.on_chat_model_start(serialized={}, messages=[[msg]]) + + def test_extract_model_name(self): + from token0.langchain_callback import _extract_model_name + + assert _extract_model_name({"kwargs": {"model_name": "gpt-4o"}}) == "gpt-4o" + model_id = "claude-sonnet-4-6" + assert _extract_model_name({"kwargs": {"model": model_id}}) == model_id + assert _extract_model_name({}) == "" + + def test_role_for_messages(self): + from token0.langchain_callback import _role_for + + try: + from langchain_core.messages import AIMessage, HumanMessage, SystemMessage + except ImportError: + pytest.skip("langchain-core not installed") + + assert _role_for(HumanMessage(content="hi")) == "user" + assert _role_for(AIMessage(content="hi")) == "assistant" + assert _role_for(SystemMessage(content="hi")) == "system" diff --git a/tests/test_pdf.py b/tests/test_pdf.py new file mode 100644 index 0000000..42f8059 --- /dev/null +++ b/tests/test_pdf.py @@ -0,0 +1,139 @@ +"""Tests for PDF text layer extraction.""" + +import base64 + + +def _make_pdf_with_text(text: str = "Invoice Total: $123.45\nDate: 2024-01-01") -> bytes: + """Create a minimal valid PDF with a text layer using reportlab or fpdf2 if available, + otherwise fall back to a hand-crafted minimal PDF.""" + try: + from fpdf import FPDF + + pdf = FPDF() + pdf.add_page() + pdf.set_font("Helvetica", size=12) + pdf.cell(200, 10, text=text) + return bytes(pdf.output()) + except ImportError: + pass + + # Hand-crafted minimal PDF with text layer + content_stream = f"BT /F1 12 Tf 50 750 Td ({text}) Tj ET" + content_len = len(content_stream) + pdf = f"""%PDF-1.4 +1 0 obj << /Type /Catalog /Pages 2 0 R >> endobj +2 0 obj << /Type /Pages /Kids [3 0 R] /Count 1 >> endobj +3 0 obj << /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] + /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >> endobj +4 0 obj << /Length {content_len} >> +stream +{content_stream} +endstream +endobj +5 0 obj << /Type /Font /Subtype /Type1 /BaseFont /Helvetica >> endobj +xref +0 6 +0000000000 65535 f +0000000009 00000 n +0000000058 00000 n +0000000115 00000 n +0000000266 00000 n +0000000{400 + content_len:06d} 00000 n +trailer << /Size 6 /Root 1 0 R >> +startxref +{500 + content_len} +%%EOF""" + return pdf.encode() + + +def _pdf_data_uri(pdf_bytes: bytes) -> str: + b64 = base64.b64encode(pdf_bytes).decode() + return f"data:application/pdf;base64,{b64}" + + +class TestPdfDetection: + def test_is_pdf_data_uri_true(self): + from token0.optimization.pdf import is_pdf_data_uri + + assert is_pdf_data_uri("data:application/pdf;base64,abc123") is True + + def test_is_pdf_data_uri_false_for_image(self): + from token0.optimization.pdf import is_pdf_data_uri + + assert is_pdf_data_uri("data:image/jpeg;base64,abc123") is False + + def test_is_pdf_data_uri_false_for_url(self): + from token0.optimization.pdf import is_pdf_data_uri + + assert is_pdf_data_uri("https://example.com/doc.pdf") is False + + +class TestPdfDecode: + def test_decode_pdf_roundtrip(self): + from token0.optimization.pdf import decode_pdf + + original = b"fake pdf bytes" + b64 = base64.b64encode(original).decode() + uri = f"data:application/pdf;base64,{b64}" + assert decode_pdf(uri) == original + + +class TestPdfTextExtraction: + def test_extract_text_returns_none_for_empty_bytes(self): + from token0.optimization.pdf import extract_pdf_text + + result = extract_pdf_text(b"not a pdf") + assert result is None + + def test_extract_text_with_valid_pdf(self): + from token0.optimization.pdf import extract_pdf_text + + pdf_bytes = _make_pdf_with_text("Hello World Invoice Total 100") + result = extract_pdf_text(pdf_bytes) + # Either extracts text (if pypdf reads it) or returns None (scanned/unreadable) + assert result is None or isinstance(result, str) + + def test_estimate_pdf_tokens(self): + from token0.optimization.pdf import estimate_pdf_tokens + + text = "a" * 400 # 400 chars → ~100 tokens + assert estimate_pdf_tokens(text) == 100 + + def test_estimate_pdf_tokens_minimum(self): + from token0.optimization.pdf import estimate_pdf_tokens + + assert estimate_pdf_tokens("hi") == 10 # minimum floor + + +class TestPdfEstimateEndpoint: + """Test that the /v1/estimate endpoint handles PDF data URIs gracefully.""" + + def test_pdf_uri_skipped_gracefully(self): + """PDF data URIs in estimate endpoint should not crash — skip or extract.""" + from token0.api.v1.estimate import EstimateRequest, estimate + from token0.models.request import ContentPart, ImageUrl, Message + + pdf_bytes = b"not a real pdf" + b64 = base64.b64encode(pdf_bytes).decode() + pdf_uri = f"data:application/pdf;base64,{b64}" + + # This should not raise — malformed PDFs are silently skipped + import asyncio + + req = EstimateRequest( + model="gpt-4o", + messages=[ + Message( + role="user", + content=[ + ContentPart( + type="image_url", + image_url=ImageUrl(url=pdf_uri), + ) + ], + ) + ], + ) + result = asyncio.run(estimate(req)) + assert result.model == "gpt-4o" + assert result.provider == "openai" diff --git a/token0/api/v1/chat.py b/token0/api/v1/chat.py index 83ded00..9f881ca 100644 --- a/token0/api/v1/chat.py +++ b/token0/api/v1/chat.py @@ -178,6 +178,33 @@ def _optimize_messages(request: ChatRequest, prompt_detail: str): elif part.type == "image_url" and part.image_url and request.token0_optimize: image_data = part.image_url.url + + # PDF pre-processing: extract text layer if available + from token0.optimization.pdf import ( + decode_pdf, + estimate_pdf_tokens, + extract_pdf_text, + is_pdf_data_uri, + ) + + if is_pdf_data_uri(image_data): + pdf_bytes = decode_pdf(image_data) + pdf_text = extract_pdf_text(pdf_bytes) + if pdf_text: + token_count = estimate_pdf_tokens(pdf_text) + total_tokens_before += 765 # approx cost of rendering as image + total_tokens_after += token_count + optimizations_applied.append("pdf → text layer extracted") + optimized_parts.append( + {"type": "text", "text": f"[Extracted text from PDF]:\n{pdf_text}"} + ) + continue + # No text layer — pass through to provider (Anthropic/Gemini support native PDF) + optimized_parts.append( + {"type": "image_url", "image_url": {"url": image_data, "detail": None}} + ) + continue + analysis, raw_bytes, pil_image = analyze_image(image_data) if first_pil_image is None: diff --git a/token0/api/v1/estimate.py b/token0/api/v1/estimate.py new file mode 100644 index 0000000..de1fca7 --- /dev/null +++ b/token0/api/v1/estimate.py @@ -0,0 +1,120 @@ +"""POST /v1/estimate — pre-call token cost estimator. + +Takes an image + prompt + model, returns predicted token count and dollar cost +BEFORE making any LLM call. No provider API key required. + +Useful for: +- Understanding image costs before committing to a provider +- Comparing GPT-4o vs Claude vs Gemini costs for a specific image +- Building cost dashboards without proxying real calls +""" + +from fastapi import APIRouter +from pydantic import BaseModel + +from token0.models.request import Message +from token0.optimization.analyzer import analyze_image +from token0.optimization.prompt_classifier import classify_prompt_detail, extract_prompt_text +from token0.optimization.router import get_provider_from_model, plan_optimization +from token0.providers.base import get_cost_per_token + +router = APIRouter() + + +class ImageEstimate(BaseModel): + original_tokens: int + optimized_tokens: int + tokens_saved: int + cost_saved_usd: float + optimizations: list[str] + + +class EstimateRequest(BaseModel): + model: str + messages: list[Message] + + +class EstimateResponse(BaseModel): + model: str + provider: str + images: list[ImageEstimate] + total_original_tokens: int + total_optimized_tokens: int + total_tokens_saved: int + total_cost_saved_usd: float + note: str | None = None + + +@router.post("/estimate", response_model=EstimateResponse) +async def estimate(request: EstimateRequest) -> EstimateResponse: + """Estimate token cost savings for a request without making any LLM call.""" + provider_name = get_provider_from_model(request.model) + # Convert Pydantic messages to dicts for prompt_classifier compatibility + messages_as_dicts = [ + {"role": m.role, "content": m.content if isinstance(m.content, str) + else [{"type": p.type, "text": p.text} for p in m.content]} + for m in request.messages + ] + prompt_text = extract_prompt_text(messages_as_dicts) + prompt_detail = classify_prompt_detail(prompt_text) + + image_estimates: list[ImageEstimate] = [] + skipped_remote = False + + for msg in request.messages: + if isinstance(msg.content, str): + continue + for part in msg.content: + if part.type != "image_url" or not part.image_url: + continue + + url = part.image_url.url + + if not url.startswith("data:"): + skipped_remote = True + continue # can't analyze remote URLs without fetching + + try: + analysis, _, _ = analyze_image(url) + plan = plan_optimization( + analysis, + request.model, + detail_override=part.image_url.detail, + prompt_detail=prompt_detail, + enable_cascade=False, # cascade changes model pricing; excluded for clarity + ) + cost_per_token = get_cost_per_token(request.model, "input") + cost_before = plan.estimated_tokens_before * cost_per_token + cost_after = plan.estimated_tokens_after * cost_per_token + + image_estimates.append( + ImageEstimate( + original_tokens=plan.estimated_tokens_before, + optimized_tokens=plan.estimated_tokens_after, + tokens_saved=max(0, plan.estimated_tokens_before - plan.estimated_tokens_after), # noqa: E501 + cost_saved_usd=round(max(0.0, cost_before - cost_after), 6), + optimizations=plan.reasons, + ) + ) + except Exception: + continue # skip unanalyzable images silently + + total_orig = sum(e.original_tokens for e in image_estimates) + total_opt = sum(e.optimized_tokens for e in image_estimates) + total_saved_tokens = sum(e.tokens_saved for e in image_estimates) + total_saved_usd = round(sum(e.cost_saved_usd for e in image_estimates), 6) + + note = None + if skipped_remote: + note = "Remote image URLs were skipped — provide base64 data URIs for full estimates." + + return EstimateResponse( + model=request.model, + provider=provider_name, + images=image_estimates, + total_original_tokens=total_orig, + total_optimized_tokens=total_opt, + total_tokens_saved=total_saved_tokens, + total_cost_saved_usd=total_saved_usd, + note=note, + ) diff --git a/token0/langchain_callback.py b/token0/langchain_callback.py new file mode 100644 index 0000000..65f4dfa --- /dev/null +++ b/token0/langchain_callback.py @@ -0,0 +1,111 @@ +"""LangChain callback handler that optimizes vision tokens before LLM calls. + +Usage: + from token0.langchain_callback import Token0Callback + from langchain_openai import ChatOpenAI + + llm = ChatOpenAI(model="gpt-4o", callbacks=[Token0Callback()]) + + # All calls through this llm instance now get image optimization + response = llm.invoke([HumanMessage(content=[ + {"type": "text", "text": "What's in this image?"}, + {"type": "image_url", "image_url": {"url": "data:image/jpeg;base64,..."}} + ])]) + +Works with any LangChain chat model (ChatOpenAI, ChatAnthropic, ChatGoogleGenerativeAI, etc.) +No proxy required — runs as a pre-call callback. + +Note: This implementation mutates message content in-place inside on_chat_model_start. +This works because LangChain's callbacks fire synchronously before the model converts +messages to API format. Compatible with langchain-core >= 0.2. +""" + +import logging +from typing import Any + +try: + from langchain_core.callbacks.base import BaseCallbackHandler + from langchain_core.messages import BaseMessage +except ImportError: + raise ImportError( + "langchain-core is required for the Token0Callback integration. " + "Install it with: pip install langchain-core" + ) + +from token0.litellm_hook import _optimize_messages + +logger = logging.getLogger("token0.langchain") + + +def _extract_model_name(serialized: dict) -> str: + """Extract model name from LangChain's serialized callback data.""" + kwargs = serialized.get("kwargs", {}) + return kwargs.get("model_name") or kwargs.get("model") or "" + + +def _role_for(message: BaseMessage) -> str: + """Map LangChain message type to role string.""" + class_name = type(message).__name__ + if "Human" in class_name: + return "user" + if "AI" in class_name or "Assistant" in class_name: + return "assistant" + if "System" in class_name: + return "system" + return "user" + + +class Token0Callback(BaseCallbackHandler): + """LangChain callback that optimizes vision tokens before LLM calls. + + Drop-in for any LangChain chat model: + llm = ChatOpenAI(model="gpt-4o", callbacks=[Token0Callback()]) + + Args: + enable_cascade: Auto-route simple tasks to cheaper models (default: False, + since the LangChain model is already set by the caller). + detail_override: Force "low" or "high" detail mode for OpenAI (default: auto). + """ + + def __init__( + self, + enable_cascade: bool = False, + detail_override: str | None = None, + ): + self.enable_cascade = enable_cascade + self.detail_override = detail_override + + def on_chat_model_start( + self, + serialized: dict[str, Any], + messages: list[list[BaseMessage]], + **kwargs: Any, + ) -> None: + """Optimize images in messages before the LLM call.""" + model = _extract_model_name(serialized) + + for message_list in messages: + for message in message_list: + if not isinstance(message.content, list): + continue + + # Wrap in the dict format _optimize_messages expects + msg_dicts = [{"role": _role_for(message), "content": message.content}] + + optimized_dicts, stats = _optimize_messages( + msg_dicts, + model, + detail_override=self.detail_override, + enable_cascade=self.enable_cascade, + ) + + # Mutate in-place — LangChain reads this before sending to provider + if optimized_dicts: + message.content = optimized_dicts[0]["content"] + + if stats["tokens_saved"] > 0: + logger.info( + "token0: %d tokens saved (%s)", + stats["tokens_saved"], + ", ".join(stats["optimizations"]), + ) diff --git a/token0/litellm_hook.py b/token0/litellm_hook.py index 193c6a0..2cb52ca 100644 --- a/token0/litellm_hook.py +++ b/token0/litellm_hook.py @@ -126,6 +126,33 @@ def _optimize_messages( opt_parts.append(part) continue + # PDF pre-processing: extract text layer if available + from token0.optimization.pdf import ( + decode_pdf, + estimate_pdf_tokens, + extract_pdf_text, + is_pdf_data_uri, + ) + + if is_pdf_data_uri(url): + try: + pdf_bytes = decode_pdf(url) + pdf_text = extract_pdf_text(pdf_bytes) + if pdf_text: + token_count = estimate_pdf_tokens(pdf_text) + total_before += 765 + total_after += token_count + optimizations.append("pdf → text layer extracted") + opt_parts.append( + {"type": "text", "text": f"[Extracted text from PDF]:\n{pdf_text}"} + ) + else: + opt_parts.append(part) # no text layer — passthrough + except Exception: + logger.warning("token0: PDF extraction failed, passing through", exc_info=True) + opt_parts.append(part) + continue + try: analysis, raw_bytes, pil_image = analyze_image(url) plan = plan_optimization( diff --git a/token0/main.py b/token0/main.py index 4d8d94a..91c415e 100644 --- a/token0/main.py +++ b/token0/main.py @@ -4,6 +4,7 @@ from fastapi import FastAPI from token0.api.v1.chat import router as chat_router +from token0.api.v1.estimate import router as estimate_router from token0.api.v1.usage import router as usage_router from token0.config import settings from token0.storage.postgres import close_db, init_db @@ -40,6 +41,7 @@ async def lifespan(app: FastAPI): ) app.include_router(chat_router, prefix="/v1") +app.include_router(estimate_router, prefix="/v1") app.include_router(usage_router, prefix="/v1") diff --git a/token0/optimization/pdf.py b/token0/optimization/pdf.py new file mode 100644 index 0000000..0169bb3 --- /dev/null +++ b/token0/optimization/pdf.py @@ -0,0 +1,57 @@ +"""PDF text layer extraction — route PDFs to text instead of vision tokens. + +When a PDF has a text layer (the vast majority of typed/digital PDFs), +extracting it costs ~10-50x fewer tokens than rendering as an image. +Scanned PDFs (no text layer) fall back to passthrough. +""" + +import base64 +import io +import logging + +logger = logging.getLogger("token0.pdf") + + +def is_pdf_data_uri(url: str) -> bool: + return url.startswith("data:application/pdf;") + + +def decode_pdf(url: str) -> bytes: + """Decode PDF from base64 data URI into raw bytes.""" + _, b64_data = url.split(",", 1) + return base64.b64decode(b64_data) + + +def extract_pdf_text(pdf_bytes: bytes) -> str | None: + """Extract text layer from PDF bytes using pypdf. + + Returns extracted text string, or None if no usable text layer found + (e.g. scanned PDF — caller should fall back to passthrough). + """ + try: + from pypdf import PdfReader + except ImportError: + logger.warning("pypdf not installed — PDF text extraction unavailable. pip install pypdf") + return None + + try: + reader = PdfReader(io.BytesIO(pdf_bytes)) + pages_text = [] + for page in reader.pages: + text = page.extract_text() + if text: + pages_text.append(text.strip()) + + combined = "\n\n".join(pages_text) + if len(combined) < 20: + return None # too little text — probably a scanned PDF + return combined + + except Exception as e: + logger.warning("PDF text extraction failed: %s", e) + return None + + +def estimate_pdf_tokens(text: str) -> int: + """Rough token estimate for extracted PDF text (4 chars ≈ 1 token).""" + return max(10, len(text) // 4) From 3502c9a44dca58ca106ae7ca2380800ea0dc2c86 Mon Sep 17 00:00:00 2001 From: Pritom Mazumdar Date: Tue, 31 Mar 2026 01:57:50 +0530 Subject: [PATCH 2/5] bumped version --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index a50fdaf..c45b1e5 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "token0" -version = "0.3.0" +version = "0.3.1" description = "Open-source API proxy that makes vision LLM calls 5-10x cheaper" readme = "README.md" license = "Apache-2.0" From 9063ba9b6ad03e6b9f6c4a24363c744dbcafb4d5 Mon Sep 17 00:00:00 2001 From: Pritom Mazumdar Date: Tue, 31 Mar 2026 01:59:38 +0530 Subject: [PATCH 3/5] CI fixes --- token0/api/v1/estimate.py | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/token0/api/v1/estimate.py b/token0/api/v1/estimate.py index de1fca7..8247671 100644 --- a/token0/api/v1/estimate.py +++ b/token0/api/v1/estimate.py @@ -51,8 +51,12 @@ async def estimate(request: EstimateRequest) -> EstimateResponse: provider_name = get_provider_from_model(request.model) # Convert Pydantic messages to dicts for prompt_classifier compatibility messages_as_dicts = [ - {"role": m.role, "content": m.content if isinstance(m.content, str) - else [{"type": p.type, "text": p.text} for p in m.content]} + { + "role": m.role, + "content": m.content + if isinstance(m.content, str) + else [{"type": p.type, "text": p.text} for p in m.content], + } for m in request.messages ] prompt_text = extract_prompt_text(messages_as_dicts) @@ -91,7 +95,9 @@ async def estimate(request: EstimateRequest) -> EstimateResponse: ImageEstimate( original_tokens=plan.estimated_tokens_before, optimized_tokens=plan.estimated_tokens_after, - tokens_saved=max(0, plan.estimated_tokens_before - plan.estimated_tokens_after), # noqa: E501 + tokens_saved=max( + 0, plan.estimated_tokens_before - plan.estimated_tokens_after + ), # noqa: E501 cost_saved_usd=round(max(0.0, cost_before - cost_after), 6), optimizations=plan.reasons, ) From bf2d1cb8f73e3b0461c16d31cc73dcaeeac835fc Mon Sep 17 00:00:00 2001 From: Pritom Mazumdar Date: Tue, 31 Mar 2026 02:03:13 +0530 Subject: [PATCH 4/5] CI fixes --- pyproject.toml | 1 + tests/test_langchain_callback.py | 2 ++ token0/langchain_callback.py | 14 ++++++++++---- 3 files changed, 13 insertions(+), 4 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index c45b1e5..27c08e4 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -39,6 +39,7 @@ dev = [ "pytest-cov>=6.0.0", "ruff>=0.8.0", "mypy>=1.13.0", + "langchain-core>=0.2.0", ] [project.scripts] diff --git a/tests/test_langchain_callback.py b/tests/test_langchain_callback.py index bfb4c1b..009ad11 100644 --- a/tests/test_langchain_callback.py +++ b/tests/test_langchain_callback.py @@ -6,6 +6,8 @@ import pytest from PIL import Image +pytest.importorskip("langchain_core", reason="langchain-core not installed") + def _make_image_data_uri(width: int = 800, height: int = 600) -> str: img = Image.new("RGB", (width, height), color=(100, 150, 200)) diff --git a/token0/langchain_callback.py b/token0/langchain_callback.py index 65f4dfa..435484a 100644 --- a/token0/langchain_callback.py +++ b/token0/langchain_callback.py @@ -26,11 +26,12 @@ try: from langchain_core.callbacks.base import BaseCallbackHandler from langchain_core.messages import BaseMessage + + _langchain_available = True except ImportError: - raise ImportError( - "langchain-core is required for the Token0Callback integration. " - "Install it with: pip install langchain-core" - ) + _langchain_available = False + BaseCallbackHandler = object # type: ignore[assignment,misc] + BaseMessage = object # type: ignore[assignment] from token0.litellm_hook import _optimize_messages @@ -72,6 +73,11 @@ def __init__( enable_cascade: bool = False, detail_override: str | None = None, ): + if not _langchain_available: + raise ImportError( + "langchain-core is required for the Token0Callback integration. " + "Install it with: pip install langchain-core" + ) self.enable_cascade = enable_cascade self.detail_override = detail_override From 43300401e10809db613df231b7ea8855017becfa Mon Sep 17 00:00:00 2001 From: Pritom Mazumdar Date: Tue, 31 Mar 2026 02:08:27 +0530 Subject: [PATCH 5/5] CI fixes --- token0/langchain_callback.py | 4 +- token0/litellm_hook.py | 132 +--------------------- token0/optimization/message_optimizer.py | 138 +++++++++++++++++++++++ 3 files changed, 144 insertions(+), 130 deletions(-) create mode 100644 token0/optimization/message_optimizer.py diff --git a/token0/langchain_callback.py b/token0/langchain_callback.py index 435484a..8a4d82a 100644 --- a/token0/langchain_callback.py +++ b/token0/langchain_callback.py @@ -33,7 +33,7 @@ BaseCallbackHandler = object # type: ignore[assignment,misc] BaseMessage = object # type: ignore[assignment] -from token0.litellm_hook import _optimize_messages +from token0.optimization.message_optimizer import optimize_messages logger = logging.getLogger("token0.langchain") @@ -98,7 +98,7 @@ def on_chat_model_start( # Wrap in the dict format _optimize_messages expects msg_dicts = [{"role": _role_for(message), "content": message.content}] - optimized_dicts, stats = _optimize_messages( + optimized_dicts, stats = optimize_messages( msg_dicts, model, detail_override=self.detail_override, diff --git a/token0/litellm_hook.py b/token0/litellm_hook.py index 2cb52ca..f12bd6f 100644 --- a/token0/litellm_hook.py +++ b/token0/litellm_hook.py @@ -21,9 +21,7 @@ "litellm is required for the Token0Hook integration. Install it with: pip install litellm" ) -from token0.optimization.analyzer import analyze_image -from token0.optimization.router import plan_optimization -from token0.optimization.transformer import transform_image +from token0.optimization.message_optimizer import optimize_messages logger = logging.getLogger("token0.litellm") @@ -55,7 +53,7 @@ async def async_pre_call_hook( return data model = data.get("model", "") - optimized_messages, stats = _optimize_messages( + optimized_messages, stats = optimize_messages( messages, model, detail_override=self.detail_override, @@ -83,127 +81,5 @@ async def async_pre_call_hook( return data -def _optimize_messages( - messages: list[dict], - model: str, - detail_override: str | None = None, - enable_cascade: bool = False, -) -> tuple[list[dict], dict]: - """Optimize images in a list of message dicts. - - Returns (optimized_messages, stats_dict). - """ - optimized = [] - total_before = 0 - total_after = 0 - optimizations = [] - recommended_model = None - - for msg in messages: - content = msg.get("content") - - # Text-only message — pass through - if isinstance(content, str) or content is None: - optimized.append(msg) - continue - - # Multi-part content — check for images - if not isinstance(content, list): - optimized.append(msg) - continue - - opt_parts = [] - for part in content: - if part.get("type") != "image_url": - opt_parts.append(part) - continue - - image_url = part.get("image_url", {}) - url = image_url.get("url", "") - - # Only optimize base64 data URIs - if not url.startswith("data:"): - opt_parts.append(part) - continue - - # PDF pre-processing: extract text layer if available - from token0.optimization.pdf import ( - decode_pdf, - estimate_pdf_tokens, - extract_pdf_text, - is_pdf_data_uri, - ) - - if is_pdf_data_uri(url): - try: - pdf_bytes = decode_pdf(url) - pdf_text = extract_pdf_text(pdf_bytes) - if pdf_text: - token_count = estimate_pdf_tokens(pdf_text) - total_before += 765 - total_after += token_count - optimizations.append("pdf → text layer extracted") - opt_parts.append( - {"type": "text", "text": f"[Extracted text from PDF]:\n{pdf_text}"} - ) - else: - opt_parts.append(part) # no text layer — passthrough - except Exception: - logger.warning("token0: PDF extraction failed, passing through", exc_info=True) - opt_parts.append(part) - continue - - try: - analysis, raw_bytes, pil_image = analyze_image(url) - plan = plan_optimization( - analysis, - model, - detail_override=detail_override, - enable_cascade=enable_cascade, - ) - - total_before += plan.estimated_tokens_before - total_after += plan.estimated_tokens_after - optimizations.extend(plan.reasons) - - if plan.recommended_model and recommended_model is None: - recommended_model = plan.recommended_model - - if plan.use_ocr_route: - result = transform_image(plan, analysis, raw_bytes, pil_image) - opt_parts.append( - { - "type": "text", - "text": f"[Extracted text from image]:\n{result['content']}", - } - ) - elif any([plan.resize, plan.recompress_jpeg, plan.force_detail_low]): - result = transform_image(plan, analysis, raw_bytes, pil_image) - detail = "low" if plan.force_detail_low else image_url.get("detail", "auto") - opt_parts.append( - { - "type": "image_url", - "image_url": { - "url": f"data:{result['media_type']};base64,{result['base64']}", - "detail": detail, - }, - } - ) - else: - opt_parts.append(part) - - except Exception: - logger.warning("token0: failed to optimize image, passing through", exc_info=True) - opt_parts.append(part) - - optimized.append({"role": msg["role"], "content": opt_parts}) - - stats = { - "tokens_before": total_before, - "tokens_after": total_after, - "tokens_saved": total_before - total_after, - "optimizations": optimizations, - "recommended_model": recommended_model, - } - - return optimized, stats +# Backwards-compatible alias +_optimize_messages = optimize_messages diff --git a/token0/optimization/message_optimizer.py b/token0/optimization/message_optimizer.py new file mode 100644 index 0000000..08ff7e8 --- /dev/null +++ b/token0/optimization/message_optimizer.py @@ -0,0 +1,138 @@ +"""Shared message optimization logic — no litellm or langchain dependency. + +Used by both litellm_hook.py and langchain_callback.py. +""" + +import logging + +from token0.optimization.analyzer import analyze_image +from token0.optimization.router import plan_optimization +from token0.optimization.transformer import transform_image + +logger = logging.getLogger("token0.optimizer") + + +def optimize_messages( + messages: list[dict], + model: str, + detail_override: str | None = None, + enable_cascade: bool = False, +) -> tuple[list[dict], dict]: + """Optimize images in a list of message dicts. + + Returns (optimized_messages, stats_dict). + """ + optimized = [] + total_before = 0 + total_after = 0 + optimizations = [] + recommended_model = None + + for msg in messages: + content = msg.get("content") + + # Text-only message — pass through + if isinstance(content, str) or content is None: + optimized.append(msg) + continue + + # Multi-part content — check for images + if not isinstance(content, list): + optimized.append(msg) + continue + + opt_parts = [] + for part in content: + if part.get("type") != "image_url": + opt_parts.append(part) + continue + + image_url = part.get("image_url", {}) + url = image_url.get("url", "") + + # Only optimize base64 data URIs + if not url.startswith("data:"): + opt_parts.append(part) + continue + + # PDF pre-processing: extract text layer if available + from token0.optimization.pdf import ( + decode_pdf, + estimate_pdf_tokens, + extract_pdf_text, + is_pdf_data_uri, + ) + + if is_pdf_data_uri(url): + try: + pdf_bytes = decode_pdf(url) + pdf_text = extract_pdf_text(pdf_bytes) + if pdf_text: + token_count = estimate_pdf_tokens(pdf_text) + total_before += 765 + total_after += token_count + optimizations.append("pdf → text layer extracted") + opt_parts.append( + {"type": "text", "text": f"[Extracted text from PDF]:\n{pdf_text}"} + ) + else: + opt_parts.append(part) # no text layer — passthrough + except Exception: + logger.warning("token0: PDF extraction failed, passing through", exc_info=True) + opt_parts.append(part) + continue + + try: + analysis, raw_bytes, pil_image = analyze_image(url) + plan = plan_optimization( + analysis, + model, + detail_override=detail_override, + enable_cascade=enable_cascade, + ) + + total_before += plan.estimated_tokens_before + total_after += plan.estimated_tokens_after + optimizations.extend(plan.reasons) + + if plan.recommended_model and recommended_model is None: + recommended_model = plan.recommended_model + + if plan.use_ocr_route: + result = transform_image(plan, analysis, raw_bytes, pil_image) + opt_parts.append( + { + "type": "text", + "text": f"[Extracted text from image]:\n{result['content']}", + } + ) + elif any([plan.resize, plan.recompress_jpeg, plan.force_detail_low]): + result = transform_image(plan, analysis, raw_bytes, pil_image) + detail = "low" if plan.force_detail_low else image_url.get("detail", "auto") + opt_parts.append( + { + "type": "image_url", + "image_url": { + "url": f"data:{result['media_type']};base64,{result['base64']}", + "detail": detail, + }, + } + ) + else: + opt_parts.append(part) + + except Exception: + logger.warning("token0: failed to optimize image, passing through", exc_info=True) + opt_parts.append(part) + + optimized.append({"role": msg["role"], "content": opt_parts}) + + stats = { + "tokens_before": total_before, + "tokens_after": total_after, + "tokens_saved": total_before - total_after, + "optimizations": optimizations, + "recommended_model": recommended_model, + } + + return optimized, stats