From ff3842d266c10bfbbf3c2e0b0f70f68df45fccc0 Mon Sep 17 00:00:00 2001 From: tonykipkemboi Date: Fri, 5 Sep 2025 11:42:00 -0400 Subject: [PATCH 1/5] docs: add BUILDING_TOOLS.md --- BUILDING_TOOLS.md | 335 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 335 insertions(+) create mode 100644 BUILDING_TOOLS.md diff --git a/BUILDING_TOOLS.md b/BUILDING_TOOLS.md new file mode 100644 index 00000000..2994b918 --- /dev/null +++ b/BUILDING_TOOLS.md @@ -0,0 +1,335 @@ +## Building CrewAI Tools + +This guide shows you how to build high‑quality CrewAI tools that match the patterns in this repository and are ready to be merged. It focuses on: architecture, conventions, environment variables, dependencies, testing, documentation, and a complete example. + +### Who this is for +- Contributors creating new tools under `crewai_tools/tools/*` +- Maintainers reviewing PRs for consistency and DX + +--- + +## Quick‑start checklist +1. Create a new folder under `crewai_tools/tools//` with a `README.md` and a `.py`. +2. Implement a class that ends with `Tool` and subclasses `BaseTool` (or `RagTool` when appropriate). +3. Define a Pydantic `args_schema` with explicit field descriptions and validation. +4. Declare `env_vars` and `package_dependencies` in the class when needed. +5. Lazily initialize clients in `__init__` or `_run` and handle missing credentials with clear errors. +6. Implement `_run(...) -> str | dict` and, if needed, `_arun(...)`. +7. Add tests under `tests/tools/` (unit, no real network calls; mock or record safely). +8. Add a concise tool `README.md` with usage and required env vars. +9. If you add optional dependencies, register them in `pyproject.toml` under `[project.optional-dependencies]` and reference that extra in your tool docs. +10. Run `uv run pytest` and `pre-commit run -a` locally; ensure green. + +--- + +## Tool anatomy and conventions + +### BaseTool pattern +All tools follow this structure: + +```python +from typing import Any, List, Optional, Type + +import os +from pydantic import BaseModel, Field +from crewai.tools import BaseTool, EnvVar + + +class MyToolInput(BaseModel): + """Input schema for MyTool.""" + query: str = Field(..., description="Your input description here") + limit: int = Field(5, ge=1, le=50, description="Max items to return") + + +class MyTool(BaseTool): + name: str = "My Tool" + description: str = "Explain succinctly what this tool does and when to use it." + args_schema: Type[BaseModel] = MyToolInput + + # Only include when applicable + env_vars: List[EnvVar] = [ + EnvVar(name="MY_API_KEY", description="API key for My service", required=True), + ] + package_dependencies: List[str] = ["my-sdk"] + + def __init__(self, **kwargs: Any) -> None: + super().__init__(**kwargs) + # Lazy import to keep base install light + try: + import my_sdk # noqa: F401 + except Exception as exc: + raise ImportError( + "Missing optional dependency 'my-sdk'. Install with: \n" + " uv add crewai-tools --extra my-sdk\n" + "or\n" + " pip install my-sdk\n" + ) from exc + + if "MY_API_KEY" not in os.environ: + raise ValueError("Environment variable MY_API_KEY is required for MyTool") + + def _run(self, query: str, limit: int = 5, **_: Any) -> str: + """Synchronous execution. Return a concise string or JSON string.""" + # Implement your logic here; do not print. Return the content. + # Handle errors gracefully, return clear messages. + return f"Processed {query} with limit={limit}" + + async def _arun(self, *args: Any, **kwargs: Any) -> str: + """Optional async counterpart if your client supports it.""" + # Prefer delegating to _run when the client is thread-safe + return self._run(*args, **kwargs) +``` + +Key points: +- Class name must end with `Tool` to be auto‑discovered by our tooling. +- Use `args_schema` for inputs; always include `description` and validation. +- Validate env vars early and fail with actionable errors. +- Keep outputs deterministic and compact; favor `str` (possibly JSON‑encoded) or small dicts converted to strings. +- Avoid printing; return the final string. + +### Error handling +- Wrap network and I/O with try/except and return a helpful message. See `BraveSearchTool` and others for patterns. +- Validate required inputs and environment configuration with clear messages. +- Keep exceptions user‑friendly; do not leak stack traces. + +### Rate limiting and retries +- If the upstream API enforces request pacing, implement minimal rate limiting (see `BraveSearchTool`). +- Consider idempotency and backoff for transient errors where appropriate. + +### Async support +- Implement `_arun` only if your library has a true async client or your sync calls are thread‑safe. +- Otherwise, delegate `_arun` to `_run` as in multiple existing tools. + +### Returning values +- Return a string (or JSON string) that’s ready to display in an agent transcript. +- If returning structured data, keep it small and human‑readable. Use stable keys and ordering. + +--- + +## RAG tools and adapters + +If your tool is a knowledge source, consider extending `RagTool` and/or creating an adapter. + +- `RagTool` exposes `add(...)` and a `query(question: str) -> str` contract through an `Adapter`. +- See `crewai_tools/tools/rag/rag_tool.py` and adapters like `embedchain_adapter.py` and `lancedb_adapter.py`. + +Minimal adapter example: + +```python +from typing import Any +from pydantic import BaseModel +from crewai_tools.tools.rag.rag_tool import Adapter, RagTool + + +class MemoryAdapter(Adapter): + store: list[str] = [] + + def add(self, text: str, **_: Any) -> None: + self.store.append(text) + + def query(self, question: str) -> str: + # naive demo: return all text containing any word from the question + tokens = set(question.lower().split()) + hits = [t for t in self.store if tokens & set(t.lower().split())] + return "\n".join(hits) if hits else "No relevant content found." + + +class MemoryRagTool(RagTool): + name: str = "In‑memory RAG" + description: str = "Toy RAG that stores text in memory and returns matches." + adapter: Adapter = MemoryAdapter() +``` + +When using external vector DBs (MongoDB, Qdrant, Weaviate), study the existing tools to follow indexing, embedding, and query configuration patterns closely. + +--- + +## Toolkits (multiple related tools) + +Some integrations expose a toolkit (a group of tools) rather than a single class. See Bedrock `browser_toolkit.py` and `code_interpreter_toolkit.py`. + +Guidelines: +- Provide small, focused `BaseTool` classes for each operation (e.g., `navigate`, `click`, `extract_text`). +- Offer a helper `create__toolkit(...) -> Tuple[ToolkitClass, List[BaseTool]]` to create tools and manage resources. +- If you open external resources (browsers, interpreters), support cleanup methods and optionally context manager usage. + +--- + +## Environment variables and dependencies + +### env_vars +- Declare as `env_vars: List[EnvVar]` with `name`, `description`, `required`, and optional `default`. +- Validate presence in `__init__` or on first `_run` call. + +### Dependencies +- List runtime packages in `package_dependencies` on the class. +- If they are genuinely optional, add an extra under `[project.optional-dependencies]` in `pyproject.toml` (e.g., `tavily-python`, `serpapi`, `scrapfly-sdk`). +- Use lazy imports to avoid hard deps for users who don’t need the tool. + +--- + +## Testing + +Place tests under `tests/tools/` and follow these rules: +- Do not hit real external services in CI. Use mocks, fakes, or recorded fixtures where allowed. +- Validate input validation, env var handling, error messages, and happy path output formatting. +- Keep tests fast and deterministic. + +Example skeleton (`tests/tools/my_tool_test.py`): + +```python +import os +import pytest +from crewai_tools.tools.my_tool.my_tool import MyTool + + +def test_requires_env_var(monkeypatch): + monkeypatch.delenv("MY_API_KEY", raising=False) + with pytest.raises(ValueError): + MyTool() + + +def test_happy_path(monkeypatch): + monkeypatch.setenv("MY_API_KEY", "test") + tool = MyTool() + result = tool.run(query="hello", limit=2) + assert "hello" in result +``` + +Run locally: + +```bash +uv run pytest +pre-commit run -a +``` + +--- + +## Documentation + +Each tool must include a `README.md` in its folder with: +- What it does and when to use it +- Required env vars and optional extras (with install snippet) +- Minimal usage example + +Update the root `README.md` only if the tool introduces a new category or notable capability. + +--- + +## Discovery and specs + +Our internal tooling discovers classes whose names end with `Tool`. Keep your class exported from the module path under `crewai_tools/tools/...` to be picked up by scripts like `generate_tool_specs.py`. + +--- + +## Full example: “Weather Search Tool” + +This example demonstrates: `args_schema`, `env_vars`, `package_dependencies`, lazy imports, validation, and robust error handling. + +```python +# file: crewai_tools/tools/weather_tool/weather_tool.py +from typing import Any, List, Optional, Type +import os +import requests +from pydantic import BaseModel, Field +from crewai.tools import BaseTool, EnvVar + + +class WeatherToolInput(BaseModel): + """Input schema for WeatherTool.""" + city: str = Field(..., description="City name, e.g., 'Berlin'") + country: Optional[str] = Field(None, description="ISO country code, e.g., 'DE'") + units: str = Field( + default="metric", + description="Units system: 'metric' or 'imperial'", + pattern=r"^(metric|imperial)$", + ) + + +class WeatherTool(BaseTool): + name: str = "Weather Search" + description: str = ( + "Look up current weather for a city using a public weather API." + ) + args_schema: Type[BaseModel] = WeatherToolInput + + env_vars: List[EnvVar] = [ + EnvVar( + name="WEATHER_API_KEY", + description="API key for the weather service", + required=True, + ), + ] + package_dependencies: List[str] = ["requests"] + + base_url: str = "https://api.openweathermap.org/data/2.5/weather" + + def __init__(self, **kwargs: Any) -> None: + super().__init__(**kwargs) + if "WEATHER_API_KEY" not in os.environ: + raise ValueError("WEATHER_API_KEY is required for WeatherTool") + + def _run(self, city: str, country: Optional[str] = None, units: str = "metric") -> str: + try: + q = f"{city},{country}" if country else city + params = { + "q": q, + "units": units, + "appid": os.environ["WEATHER_API_KEY"], + } + resp = requests.get(self.base_url, params=params, timeout=10) + resp.raise_for_status() + data = resp.json() + + main = data.get("weather", [{}])[0].get("main", "Unknown") + desc = data.get("weather", [{}])[0].get("description", "") + temp = data.get("main", {}).get("temp") + feels = data.get("main", {}).get("feels_like") + city_name = data.get("name", city) + + return ( + f"Weather in {city_name}: {main} ({desc}). " + f"Temperature: {temp}°, feels like {feels}°." + ) + except requests.Timeout: + return "Weather service timed out. Please try again later." + except requests.HTTPError as e: + return f"Weather service error: {e.response.status_code} {e.response.text[:120]}" + except Exception as e: + return f"Unexpected error fetching weather: {e}" +``` + +Folder layout: + +``` +crewai_tools/tools/weather_tool/ + ├─ weather_tool.py + └─ README.md +``` + +And `README.md` should document env vars and usage. + +--- + +## PR checklist +- [ ] Tool lives under `crewai_tools/tools//` +- [ ] Class ends with `Tool` and subclasses `BaseTool` (or `RagTool`) +- [ ] Precise `args_schema` with descriptions and validation +- [ ] `env_vars` declared (if any) and validated +- [ ] `package_dependencies` and optional extras added in `pyproject.toml` (if any) +- [ ] Clear error handling; no prints +- [ ] Unit tests added (`tests/tools/`), fast and deterministic +- [ ] Tool `README.md` with usage and env vars +- [ ] `pre-commit` and `pytest` pass locally + +--- + +## Tips for great DX +- Keep responses short and useful—agents quote your tool output directly. +- Validate early; fail fast with actionable guidance. +- Prefer lazy imports; minimize default install surface. +- Mirror patterns from similar tools in this repo for a consistent developer experience. + +Happy building! + + From 0b0158712dd4dec0745600496995405fff178363 Mon Sep 17 00:00:00 2001 From: tonykipkemboi Date: Mon, 8 Sep 2025 10:31:55 -0400 Subject: [PATCH 2/5] feat(parallel): add ParallelSearchTool (Search API v1beta), tests, README; register exports; regenerate tool.specs.json --- crewai_tools/__init__.py | 1 + crewai_tools/tools/__init__.py | 3 + crewai_tools/tools/parallel_tools/README.md | 153 ++++++++++++++ crewai_tools/tools/parallel_tools/__init__.py | 7 + .../parallel_tools/parallel_search_tool.py | 119 +++++++++++ tests/tools/parallel_search_tool_test.py | 41 ++++ tool.specs.json | 187 ++++++++++++++---- 7 files changed, 477 insertions(+), 34 deletions(-) create mode 100644 crewai_tools/tools/parallel_tools/README.md create mode 100644 crewai_tools/tools/parallel_tools/__init__.py create mode 100644 crewai_tools/tools/parallel_tools/parallel_search_tool.py create mode 100644 tests/tools/parallel_search_tool_test.py diff --git a/crewai_tools/__init__.py b/crewai_tools/__init__.py index f4c03ba0..27d259b3 100644 --- a/crewai_tools/__init__.py +++ b/crewai_tools/__init__.py @@ -93,4 +93,5 @@ YoutubeChannelSearchTool, YoutubeVideoSearchTool, ZapierActionTools, + ParallelSearchTool, ) diff --git a/crewai_tools/tools/__init__.py b/crewai_tools/tools/__init__.py index bf1a166d..ba162145 100644 --- a/crewai_tools/tools/__init__.py +++ b/crewai_tools/tools/__init__.py @@ -121,3 +121,6 @@ ) from .youtube_video_search_tool.youtube_video_search_tool import YoutubeVideoSearchTool from .zapier_action_tool.zapier_action_tool import ZapierActionTools +from .parallel_tools import ( + ParallelSearchTool, +) diff --git a/crewai_tools/tools/parallel_tools/README.md b/crewai_tools/tools/parallel_tools/README.md new file mode 100644 index 00000000..37f41356 --- /dev/null +++ b/crewai_tools/tools/parallel_tools/README.md @@ -0,0 +1,153 @@ +# ParallelSearchTool + +Unified Parallel web search tool using the Parallel Search API (v1beta). Returns ranked results with compressed excerpts optimized for LLMs. + +- **Quickstart**: see the official docs: [Search API Quickstart](https://docs.parallel.ai/search-api/search-quickstart) +- **Processors**: guidance on `base` vs `pro`: [Processors](https://docs.parallel.ai/search-api/processors) + +## Why this tool + +- **Single-call pipeline**: Replaces search → scrape → extract with a single, low‑latency API call. +- **LLM‑ready**: Returns compressed excerpts that feed directly into LLM prompts (fewer tokens, less pre/post‑processing). +- **Flexible**: Control result count and excerpt length; optionally restrict sources via `source_policy`. + +## Environment + +- `PARALLEL_API_KEY` (required) + +Optional (for the agent example): +- `OPENAI_API_KEY` or other LLM provider keys supported by CrewAI + +## Parameters + +- `objective` (str, optional): Natural‑language research goal (≤ 5000 chars) +- `search_queries` (list[str], optional): Up to 5 keyword queries (each ≤ 200 chars) +- `processor` (str, default `base`): `base` (fast/low cost) or `pro` (freshness/quality) +- `max_results` (int, default 10): ≤ 40 (subject to processor limits) +- `max_chars_per_result` (int, default 6000): ≥ 100; values > 30000 not guaranteed +- `source_policy` (dict, optional): Source policy for domain inclusion/exclusion + +Notes: +- API is in beta; default rate limit is 600 RPM. Contact support for production capacity. + +## Direct usage (when published) + +```python +from crewai_tools import ParallelSearchTool + +tool = ParallelSearchTool() +resp_json = tool.run( + objective="When was the United Nations established? Prefer UN's websites.", + search_queries=["Founding year UN", "Year of founding United Nations"], + processor="base", + max_results=5, + max_chars_per_result=1500, +) +print(resp_json) # => {"search_id": ..., "results": [{"url", "title", "excerpts": [...]}, ...]} +``` + +### Parameters you can pass + +Call `run(...)` with any of the following (at least one of `objective` or `search_queries` is required): + +```python +tool.run( + objective: str | None = None, # ≤ 5000 chars + search_queries: list[str] | None = None, # up to 5 items, each ≤ 200 chars + processor: str = "base", # "base" (fast) or "pro" (freshness/quality) + max_results: int = 10, # ≤ 40 (processor limits apply) + max_chars_per_result: int = 6000, # ≥ 100 (values > 30000 not guaranteed) + source_policy: dict | None = None, # optional SourcePolicy config +) +``` + +Example with `source_policy`: + +```python +source_policy = { + "allow": {"domains": ["un.org"]}, + # "deny": {"domains": ["example.com"]}, # optional +} + +resp_json = tool.run( + objective="When was the United Nations established?", + processor="base", + max_results=5, + max_chars_per_result=1500, + source_policy=source_policy, +) +``` + +## Example with agents + +Here’s a minimal example that calls `ParallelSearchTool` to fetch sources and has an LLM produce a short, cited answer. + +```python +import os +from crewai import Agent, Task, Crew, LLM, Process +from crewai_tools import ParallelSearchTool + +# LLM +llm = LLM( + model="gemini/gemini-2.0-flash", + temperature=0.5, + api_key=os.getenv("GEMINI_API_KEY") +) + +# Parallel Search +search = ParallelSearchTool() + +# User query +query = "find all the recent concerns about AI evals? please cite the sources" + +# Researcher agent +researcher = Agent( + role="Web Researcher", + backstory="You are an expert web researcher", + goal="Find cited, high-quality sources and provide a brief answer.", + tools=[search], + llm=llm, + verbose=True, +) + +# Research task +task = Task( + description=f"Research the {query} and produce a short, cited answer.", + expected_output="A concise, sourced answer to the question. The answer should be in this format: [query]: [answer] - [source]", + agent=researcher, + output_file="answer.mdx", +) + +# Crew +crew = Crew( + agents=[researcher], + tasks=[task], + verbose=True, + process=Process.sequential, +) + +# Run the crew +result = crew.kickoff(inputs={'query': query}) +print(result) +``` + +Output from the agent above: + +```md +Recent concerns about AI evaluations include: the rise of AI-related incidents alongside a lack of standardized Responsible AI (RAI) evaluations among major industrial model developers - [https://hai.stanford.edu/ai-index/2025-ai-index-report]; flawed benchmark datasets that fail to account for critical factors, leading to unrealistic estimates of AI model abilities - [https://www.nature.com/articles/d41586-025-02462-5]; the need for multi-metric, context-aware evaluations in medical imaging AI to ensure reliability and clinical relevance - [https://www.sciencedirect.com/science/article/pii/S3050577125000283]; challenges related to data sets (insufficient, imbalanced, or poor quality), communication gaps, and misaligned expectations in AI model training - [https://www.oracle.com/artificial-intelligence/ai-model-training-challenges/]; the argument that LLM agents should be evaluated primarily on their riskiness, not just performance, due to unreliability, hallucinations, and brittleness - [https://www.technologyreview.com/2025/06/24/1119187/fix-ai-evaluation-crisis/]; the fact that the AI industry's embraced benchmarks may be close to meaningless, with top makers of AI models picking and choosing different responsible AI benchmarks, complicating efforts to systematically compare risks and limitations - [https://themarkup.org/artificial-intelligence/2024/07/17/everyone-is-judging-ai-by-these-tests-but-experts-say-theyre-close-to-meaningless]; and the difficulty of building robust and reliable model evaluations, as many existing evaluation suites are limited in their ability to serve as accurate indicators of model capabilities or safety - [https://www.anthropic.com/research/evaluating-ai-systems]. +``` + +Tips: +- Ensure your LLM provider keys are set (e.g., `GEMINI_API_KEY`) and CrewAI model config is in place. +- For longer analyses, raise `max_chars_per_result` or use `processor="pro"` (higher quality, higher latency). + +## Behavior + +- Single‑request web research; no scraping/post‑processing required. +- Returns `search_id` and ranked `results` with compressed `excerpts`. +- Clear error handling on HTTP/timeouts. + +## References + +- Search API Quickstart: https://docs.parallel.ai/search-api/search-quickstart +- Processors: https://docs.parallel.ai/search-api/processors diff --git a/crewai_tools/tools/parallel_tools/__init__.py b/crewai_tools/tools/parallel_tools/__init__.py new file mode 100644 index 00000000..579dc894 --- /dev/null +++ b/crewai_tools/tools/parallel_tools/__init__.py @@ -0,0 +1,7 @@ +from .parallel_search_tool import ParallelSearchTool + +__all__ = [ + "ParallelSearchTool", +] + + diff --git a/crewai_tools/tools/parallel_tools/parallel_search_tool.py b/crewai_tools/tools/parallel_tools/parallel_search_tool.py new file mode 100644 index 00000000..d695bac9 --- /dev/null +++ b/crewai_tools/tools/parallel_tools/parallel_search_tool.py @@ -0,0 +1,119 @@ +import os +from typing import Any, Dict, List, Optional, Type, Annotated + +import requests +from crewai.tools import BaseTool, EnvVar +from pydantic import BaseModel, Field + + +class ParallelSearchInput(BaseModel): + """Input schema for ParallelSearchTool using the Search API (v1beta). + + At least one of objective or search_queries is required. + """ + + objective: Optional[str] = Field( + None, + description="Natural-language goal for the web research (<=5000 chars)", + max_length=5000, + ) + search_queries: Optional[List[Annotated[str, Field(max_length=200)]]] = Field( + default=None, + description="Optional list of keyword queries (<=5 items, each <=200 chars)", + min_length=1, + max_length=5, + ) + processor: str = Field( + default="base", + description="Search processor: 'base' (fast/low cost) or 'pro' (higher quality/freshness)", + pattern=r"^(base|pro)$", + ) + max_results: int = Field( + default=10, + ge=1, + le=40, + description="Maximum number of search results to return (processor limits apply)", + ) + max_chars_per_result: int = Field( + default=6000, + ge=100, + description="Maximum characters per result excerpt (values >30000 not guaranteed)", + ) + source_policy: Optional[Dict[str, Any]] = Field( + default=None, description="Optional source policy configuration" + ) + + +class ParallelSearchTool(BaseTool): + name: str = "Parallel Web Search Tool" + description: str = ( + "Search the web using Parallel's Search API (v1beta). Returns ranked results with " + "compressed excerpts optimized for LLMs." + ) + args_schema: Type[BaseModel] = ParallelSearchInput + + env_vars: List[EnvVar] = [ + EnvVar( + name="PARALLEL_API_KEY", + description="API key for Parallel", + required=True, + ), + ] + package_dependencies: List[str] = ["requests"] + + search_url: str = "https://api.parallel.ai/v1beta/search" + + def _run( + self, + objective: Optional[str] = None, + search_queries: Optional[List[str]] = None, + processor: str = "base", + max_results: int = 10, + max_chars_per_result: int = 6000, + source_policy: Optional[Dict[str, Any]] = None, + **_: Any, + ) -> str: + api_key = os.environ.get("PARALLEL_API_KEY") + if not api_key: + return "Error: PARALLEL_API_KEY environment variable is required" + + if not objective and not search_queries: + return "Error: Provide at least one of 'objective' or 'search_queries'" + + headers = { + "x-api-key": api_key, + "Content-Type": "application/json", + } + + try: + payload: Dict[str, Any] = { + "processor": processor, + "max_results": max_results, + "max_chars_per_result": max_chars_per_result, + } + if objective is not None: + payload["objective"] = objective + if search_queries is not None: + payload["search_queries"] = search_queries + if source_policy is not None: + payload["source_policy"] = source_policy + + request_timeout = 90 if processor == "pro" else 30 + resp = requests.post(self.search_url, json=payload, headers=headers, timeout=request_timeout) + if resp.status_code >= 300: + return f"Parallel Search API error: {resp.status_code} {resp.text[:200]}" + data = resp.json() + return self._format_output(data) + except requests.Timeout: + return "Parallel Search API timeout. Please try again later." + except Exception as exc: # noqa: BLE001 + return f"Unexpected error calling Parallel Search API: {exc}" + + def _format_output(self, result: Dict[str, Any]) -> str: + # Return the full JSON payload (search_id + results) as a compact JSON string + try: + import json + + return json.dumps(result or {}, ensure_ascii=False) + except Exception: + return str(result or {}) diff --git a/tests/tools/parallel_search_tool_test.py b/tests/tools/parallel_search_tool_test.py new file mode 100644 index 00000000..501ce8a0 --- /dev/null +++ b/tests/tools/parallel_search_tool_test.py @@ -0,0 +1,41 @@ +import os +from unittest.mock import patch + +import pytest + +from crewai_tools.tools.parallel_tools.parallel_search_tool import ( + ParallelSearchTool, +) + + +def test_requires_env_var(monkeypatch): + monkeypatch.delenv("PARALLEL_API_KEY", raising=False) + tool = ParallelSearchTool() + result = tool.run(objective="test") + assert "PARALLEL_API_KEY" in result + + +@patch("crewai_tools.tools.parallel_tools.parallel_search_tool.requests.post") +def test_happy_path(mock_post, monkeypatch): + monkeypatch.setenv("PARALLEL_API_KEY", "test") + + mock_post.return_value.status_code = 200 + mock_post.return_value.json.return_value = { + "search_id": "search_123", + "results": [ + { + "url": "https://www.un.org/en/about-us/history-of-the-un", + "title": "History of the United Nations", + "excerpts": [ + "Four months after the San Francisco Conference ended, the United Nations officially began, on 24 October 1945..." + ], + } + ], + } + + tool = ParallelSearchTool() + result = tool.run(objective="When was the UN established?", search_queries=["Founding year UN"]) + assert "search_id" in result + assert "https://www.un.org" in result + + diff --git a/tool.specs.json b/tool.specs.json index c16df5ee..c9cb7a2c 100644 --- a/tool.specs.json +++ b/tool.specs.json @@ -2508,6 +2508,16 @@ "required": false, "title": "Api Key" }, + "client": { + "anyOf": [ + {}, + { + "type": "null" + } + ], + "default": null, + "title": "Client" + }, "content": { "anyOf": [ { @@ -5528,6 +5538,146 @@ "type": "object" } }, + { + "description": "Search the web using Parallel's Search API (v1beta). Returns ranked results with compressed excerpts optimized for LLMs.", + "env_vars": [ + { + "default": null, + "description": "API key for Parallel", + "name": "PARALLEL_API_KEY", + "required": true + } + ], + "humanized_name": "Parallel Web Search Tool", + "init_params_schema": { + "$defs": { + "EnvVar": { + "properties": { + "default": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Default" + }, + "description": { + "title": "Description", + "type": "string" + }, + "name": { + "title": "Name", + "type": "string" + }, + "required": { + "default": true, + "title": "Required", + "type": "boolean" + } + }, + "required": [ + "name", + "description" + ], + "title": "EnvVar", + "type": "object" + } + }, + "properties": { + "search_url": { + "default": "https://api.parallel.ai/v1beta/search", + "title": "Search Url", + "type": "string" + } + }, + "title": "ParallelSearchTool", + "type": "object" + }, + "name": "ParallelSearchTool", + "package_dependencies": [ + "requests" + ], + "run_params_schema": { + "description": "Input schema for ParallelSearchTool using the Search API (v1beta).\n\nAt least one of objective or search_queries is required.", + "properties": { + "max_chars_per_result": { + "default": 6000, + "description": "Maximum characters per result excerpt (values >30000 not guaranteed)", + "minimum": 100, + "title": "Max Chars Per Result", + "type": "integer" + }, + "max_results": { + "default": 10, + "description": "Maximum number of search results to return (processor limits apply)", + "maximum": 40, + "minimum": 1, + "title": "Max Results", + "type": "integer" + }, + "objective": { + "anyOf": [ + { + "maxLength": 5000, + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "description": "Natural-language goal for the web research (<=5000 chars)", + "title": "Objective" + }, + "processor": { + "default": "base", + "description": "Search processor: 'base' (fast/low cost) or 'pro' (higher quality/freshness)", + "pattern": "^(base|pro)$", + "title": "Processor", + "type": "string" + }, + "search_queries": { + "anyOf": [ + { + "items": { + "maxLength": 200, + "type": "string" + }, + "maxItems": 5, + "minItems": 1, + "type": "array" + }, + { + "type": "null" + } + ], + "default": null, + "description": "Optional list of keyword queries (<=5 items, each <=200 chars)", + "title": "Search Queries" + }, + "source_policy": { + "anyOf": [ + { + "additionalProperties": true, + "type": "object" + }, + { + "type": "null" + } + ], + "default": null, + "description": "Optional source policy configuration", + "title": "Source Policy" + } + }, + "title": "ParallelSearchInput", + "type": "object" + } + }, { "description": "", "env_vars": [ @@ -9044,36 +9194,6 @@ ], "title": "EnvVar", "type": "object" - }, - "Vectorizers": { - "description": "The available vectorization modules in Weaviate.\n\nThese modules encode binary data into lists of floats called vectors.\nSee the [docs](https://weaviate.io/developers/weaviate/modules/retriever-vectorizer-modules) for more details.\n\nAttributes:\n `NONE`\n No vectorizer.\n `TEXT2VEC_AWS`\n Weaviate module backed by AWS text-based embedding models.\n `TEXT2VEC_COHERE`\n Weaviate module backed by Cohere text-based embedding models.\n `TEXT2VEC_CONTEXTIONARY`\n Weaviate module backed by Contextionary text-based embedding models.\n `TEXT2VEC_GPT4ALL`\n Weaviate module backed by GPT-4-All text-based embedding models.\n `TEXT2VEC_HUGGINGFACE`\n Weaviate module backed by HuggingFace text-based embedding models.\n `TEXT2VEC_OPENAI`\n Weaviate module backed by OpenAI and Azure-OpenAI text-based embedding models.\n `TEXT2VEC_PALM`\n Weaviate module backed by PaLM text-based embedding models.\n `TEXT2VEC_TRANSFORMERS`\n Weaviate module backed by Transformers text-based embedding models.\n `TEXT2VEC_JINAAI`\n Weaviate module backed by Jina AI text-based embedding models.\n `TEXT2VEC_VOYAGEAI`\n Weaviate module backed by Voyage AI text-based embedding models.\n `TEXT2VEC_WEAVIATE`\n Weaviate module backed by Weaviate's self-hosted text-based embedding models.\n `IMG2VEC_NEURAL`\n Weaviate module backed by a ResNet-50 neural network for images.\n `MULTI2VEC_CLIP`\n Weaviate module backed by a Sentence-BERT CLIP model for images and text.\n `MULTI2VEC_PALM`\n Weaviate module backed by a palm model for images and text.\n `MULTI2VEC_BIND`\n Weaviate module backed by the ImageBind model for images, text, audio, depth, IMU, thermal, and video.\n `MULTI2VEC_VOYAGEAI`\n Weaviate module backed by a Voyage AI multimodal embedding models.\n `REF2VEC_CENTROID`\n Weaviate module backed by a centroid-based model that calculates an object's vectors from its referenced vectors.", - "enum": [ - "none", - "text2vec-aws", - "text2vec-cohere", - "text2vec-contextionary", - "text2vec-databricks", - "text2vec-gpt4all", - "text2vec-huggingface", - "text2vec-mistral", - "text2vec-ollama", - "text2vec-openai", - "text2vec-palm", - "text2vec-transformers", - "text2vec-jinaai", - "text2vec-voyageai", - "text2vec-weaviate", - "img2vec-neural", - "multi2vec-clip", - "multi2vec-cohere", - "multi2vec-jinaai", - "multi2vec-bind", - "multi2vec-palm", - "multi2vec-voyageai", - "ref2vec-centroid" - ], - "title": "Vectorizers", - "type": "string" } }, "description": "Tool to search the Weaviate database", @@ -9153,14 +9273,13 @@ }, "vectorizer": { "anyOf": [ - { - "$ref": "#/$defs/Vectorizers" - }, + {}, { "type": "null" } ], - "default": null + "default": null, + "title": "Vectorizer" }, "weaviate_api_key": { "description": "The API key for the Weaviate cluster", From 49fbb7903fd9c68f353ed586a455d30ff5ff3159 Mon Sep 17 00:00:00 2001 From: tonykipkemboi Date: Mon, 8 Sep 2025 10:41:15 -0400 Subject: [PATCH 3/5] test(parallel): replace URL substring assertion with hostname allowlist (CodeQL) --- tests/tools/parallel_search_tool_test.py | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/tests/tools/parallel_search_tool_test.py b/tests/tools/parallel_search_tool_test.py index 501ce8a0..0d4df60a 100644 --- a/tests/tools/parallel_search_tool_test.py +++ b/tests/tools/parallel_search_tool_test.py @@ -1,4 +1,6 @@ import os +import json +from urllib.parse import urlparse from unittest.mock import patch import pytest @@ -35,7 +37,11 @@ def test_happy_path(mock_post, monkeypatch): tool = ParallelSearchTool() result = tool.run(objective="When was the UN established?", search_queries=["Founding year UN"]) - assert "search_id" in result - assert "https://www.un.org" in result + data = json.loads(result) + assert "search_id" in data + urls = [r.get("url", "") for r in data.get("results", [])] + # Validate host against allowed set instead of substring matching + allowed_hosts = {"www.un.org", "un.org"} + assert any(urlparse(u).netloc in allowed_hosts for u in urls) From 8e5b6d5706f4ebe6ea629291adcef91040994f20 Mon Sep 17 00:00:00 2001 From: tonykipkemboi Date: Mon, 8 Sep 2025 13:16:17 -0400 Subject: [PATCH 4/5] feat(you): add YouSearchTool (You.com Search API), README, tests; export at top level --- crewai_tools/__init__.py | 1 + crewai_tools/tools/__init__.py | 1 + crewai_tools/tools/you_search_tool/README.md | 85 +++++++++++++++++++ .../tools/you_search_tool/you_search_tool.py | 60 +++++++++++++ tests/tools/you_search_tool_test.py | 40 +++++++++ 5 files changed, 187 insertions(+) create mode 100644 crewai_tools/tools/you_search_tool/README.md create mode 100644 crewai_tools/tools/you_search_tool/you_search_tool.py create mode 100644 tests/tools/you_search_tool_test.py diff --git a/crewai_tools/__init__.py b/crewai_tools/__init__.py index 27d259b3..cf4227b8 100644 --- a/crewai_tools/__init__.py +++ b/crewai_tools/__init__.py @@ -86,6 +86,7 @@ TavilyExtractorTool, TavilySearchTool, TXTSearchTool, + YouSearchTool, VisionTool, WeaviateVectorSearchTool, WebsiteSearchTool, diff --git a/crewai_tools/tools/__init__.py b/crewai_tools/tools/__init__.py index ba162145..7cd8ca4a 100644 --- a/crewai_tools/tools/__init__.py +++ b/crewai_tools/tools/__init__.py @@ -121,6 +121,7 @@ ) from .youtube_video_search_tool.youtube_video_search_tool import YoutubeVideoSearchTool from .zapier_action_tool.zapier_action_tool import ZapierActionTools +from .you_search_tool.you_search_tool import YouSearchTool from .parallel_tools import ( ParallelSearchTool, ) diff --git a/crewai_tools/tools/you_search_tool/README.md b/crewai_tools/tools/you_search_tool/README.md new file mode 100644 index 00000000..a96d932f --- /dev/null +++ b/crewai_tools/tools/you_search_tool/README.md @@ -0,0 +1,85 @@ +# YouSearchTool (You.com Search API) + +Simple wrapper for the You.com Search API that returns structured web/news results with snippets and URLs. + +Docs: +- Search API (Trusted, Recent info): https://documentation.you.com/api-modes/search-api#trustworthy-and-recent-informantion-for-your-research +- Quickstart (Search API): https://documentation.you.com/docs/quickstart#search-api + +## What it's for + +Use when you need accurate, up-to-date web snippets and URLs from trusted sources to ground agent answers without extra scraping. Results include long snippets, titles, and links suitable for direct LLM consumption. + +## Environment + +- YOU_API_KEY (required) — used as `X-API-Key` + +## Usage (published) + +```python +from crewai_tools import YouSearchTool + +you = YouSearchTool() +resp = you.run(query="result of the political debate in EU") +print(resp) # JSON string containing results.web/news arrays +``` + +## Example with agents + +Here’s a minimal CrewAI agent that uses `YouSearchTool` as a tool. The agent will invoke the tool directly (no manual `run(...)` calls). + +```python +import os +from crewai import Agent, Task, Crew, LLM, Process +from crewai_tools import YouSearchTool + +# LLM (configure your provider keys, e.g., GEMINI_API_KEY or OPENAI_API_KEY) +llm = LLM( + model="gemini/gemini-2.0-flash", + temperature=0.5, + api_key=os.getenv("GEMINI_API_KEY") +) + +# You.com Web Search Tool +you = YouSearchTool() + +# User query +query = "all the recent CrewAI news" + +researcher = Agent( + role="Senior Web Researcher", + backstory="You are an expert web researcher.", + goal="Find cited, high-quality sources and provide a detailed answer.", + tools=[you], + llm=llm, + verbose=True, +) + +# Research task +task = Task( + description="""Use the You.com Web Search tool to research: {query}. + Provide the answer in detail and cite sources (sources should be in the format of [source](url)).""", + expected_output="A detailed, sourced answer to the question.", + agent=researcher, + output_file="answer.md", +) + +# Crew +crew = Crew( + agents=[researcher], + tasks=[task], + verbose=True, + process=Process.sequential, +) + +# Kickoff the crew +result = crew.kickoff(inputs={'query': query}) +print(result) +``` + +## Notes + +- Endpoint: `GET https://api.ydc-index.io/v1/search` with headers `{ "X-API-Key": YOU_API_KEY }` and params `{ query: ... }` +- Returns results grouped by sections (e.g., `results.web`, `results.news`) with URLs, titles, descriptions, and snippets. + + diff --git a/crewai_tools/tools/you_search_tool/you_search_tool.py b/crewai_tools/tools/you_search_tool/you_search_tool.py new file mode 100644 index 00000000..d633d69f --- /dev/null +++ b/crewai_tools/tools/you_search_tool/you_search_tool.py @@ -0,0 +1,60 @@ +import os +from typing import Any, Dict, Optional, Type + +import requests +from typing import List +from crewai.tools import BaseTool, EnvVar +from pydantic import BaseModel, Field + + +class YouSearchToolSchema(BaseModel): + """Input for You.com Web Search.""" + + query: str = Field(..., description="Search query text") + + +class YouSearchTool(BaseTool): + name: str = "You.com Web Search Tool" + description: str = ( + "Perform a web search using You.com's Search API and return structured results " + "(web/news) with snippets and URLs." + ) + args_schema: Type[BaseModel] = YouSearchToolSchema + + env_vars: List[EnvVar] = [ + EnvVar( + name="YOU_API_KEY", + description="API key for You.com Search API (used as X-API-Key)", + required=True, + ) + ] + package_dependencies: List[str] = ["requests"] + + base_url: str = "https://api.ydc-index.io/v1/search" + + def _run(self, query: str, **_: Any) -> str: + api_key = os.environ.get("YOU_API_KEY") + if not api_key: + return "Error: YOU_API_KEY environment variable is required" + + headers = {"X-API-Key": api_key} + params: Dict[str, Any] = {"query": query} + + try: + resp = requests.get(self.base_url, headers=headers, params=params, timeout=20) + if resp.status_code >= 300: + return f"You.com Search API error: {resp.status_code} {resp.text[:200]}" + data = resp.json() + # Return JSON string so agents can consume directly + try: + import json + + return json.dumps(data, ensure_ascii=False) + except Exception: + return str(data) + except requests.Timeout: + return "You.com Search API timeout. Please try again later." + except Exception as exc: # noqa: BLE001 + return f"Unexpected error calling You.com Search API: {exc}" + + diff --git a/tests/tools/you_search_tool_test.py b/tests/tools/you_search_tool_test.py new file mode 100644 index 00000000..7323f279 --- /dev/null +++ b/tests/tools/you_search_tool_test.py @@ -0,0 +1,40 @@ +from unittest.mock import patch + +import pytest + +from crewai_tools.tools.you_search_tool.you_search_tool import YouSearchTool + + +def test_requires_env_var(monkeypatch): + monkeypatch.delenv("YOU_API_KEY", raising=False) + tool = YouSearchTool() + result = tool.run(query="test") + assert "YOU_API_KEY" in result + + +@patch("crewai_tools.tools.you_search_tool.you_search_tool.requests.get") +def test_happy_path(mock_get, monkeypatch): + monkeypatch.setenv("YOU_API_KEY", "test") + + mock_get.return_value.status_code = 200 + mock_get.return_value.json.return_value = { + "results": { + "web": [ + { + "url": "https://www.europarl.europa.eu/topics/en/topic/state-of-the-eu-debates", + "title": "State of the EU debates", + "snippets": [ + "MEPs will scrutinise the work of the European Commission..." + ], + } + ] + }, + "metadata": {"query": "result of the political debate in EU"}, + } + + tool = YouSearchTool() + result = tool.run(query="result of the political debate in EU") + assert "results" in result + assert "europarl.europa.eu" in result + + From 814b10125e62dd6ec1ab92680f7bc5db64906913 Mon Sep 17 00:00:00 2001 From: tonykipkemboi Date: Mon, 8 Sep 2025 13:28:29 -0400 Subject: [PATCH 5/5] docs(you): add Parameters section and clarify purpose with doc links --- crewai_tools/tools/you_search_tool/README.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/crewai_tools/tools/you_search_tool/README.md b/crewai_tools/tools/you_search_tool/README.md index a96d932f..53b85895 100644 --- a/crewai_tools/tools/you_search_tool/README.md +++ b/crewai_tools/tools/you_search_tool/README.md @@ -14,6 +14,10 @@ Use when you need accurate, up-to-date web snippets and URLs from trusted source - YOU_API_KEY (required) — used as `X-API-Key` +## Parameters + +- `query` (str, required): Natural‑language search query. Returns structured results (web/news) with snippets and URLs. + ## Usage (published) ```python