Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion bin/retrieval_baseline
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,7 @@ from typing import Any

import nltk
from dotenv import load_dotenv
from langchain.retrievers.self_query.base import SelfQueryRetriever
from langchain_classic.retrievers.self_query.base import SelfQueryRetriever
from langchain_chroma.vectorstores import Chroma
from langchain_community.document_loaders.csv_loader import CSVLoader
from langchain_community.retrievers import BM25Retriever
Expand Down
1,450 changes: 807 additions & 643 deletions poetry.lock

Large diffs are not rendered by default.

73 changes: 50 additions & 23 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -14,34 +14,58 @@ packages = [
#lark is needed for langchain -> SelfQueryRetriever
[tool.poetry.dependencies]
python = ">=3.12, <4"
langchain = "^0.3.4"
openai = "^1.12.0"
chromadb = "<1.0"
# Capped, and coupled to the chromadb pin above. chromadb 0.6 calls
# posthog.capture(user_id, name, properties) positionally; posthog 6 changed the
# signature to capture(event, **kwargs), so every telemetry call raises and
# chromadb logs it at ERROR:
langchain = "^1.4.0"
# Declared, not inherited. LangChain 1.0 moved the pre-LCEL chain builders here
# (retrievers/rag_chain.py, the metadata_info modules), so this repository
# imports langchain_classic directly and must depend on it directly. It arrived
# transitively while langchain-community 0.4 was briefly installed, which is
# exactly why a stale virtualenv made the code look fine locally and CI did not.
langchain-classic = "^1.0.8"
# Same: data_generation imports RecursiveCharacterTextSplitter from it.
langchain-text-splitters = "^1.1.2"
# 2.x, required by langchain-openai 1.x (openai>=2.45). Used directly only by
# bin/probe_model_temperature, for models.list().
openai = "^2.45.0"
# Held below 1.0, and langchain-chroma below 1.0 with it. chromadb 1.x opens a
# 0.5/0.6 bundle by migrating its sqlite sysdb in place (9 -> 10), which needs
# write access to the bundle directory. Installed bundles here are root-owned
# 755, so every developer running as themselves gets
#
# Failed to send telemetry event ClientCreateCollectionEvent:
# capture() takes 1 positional argument but 3 were given
# InternalError: error returned from database: (code: 8)
# attempt to write a readonly database
#
# Nothing breaks -- telemetry is off anyway (anonymized_telemetry=False in
# csv_chroma.chroma_settings) -- but it puts several meaningless ERROR lines in
# the log on every startup and every collection open, and an error log nobody
# can act on is one people learn to skip. Lift this with the chromadb 1.x move.
# at startup, before a single question. Migrating is a deliberate operational
# step -- chown every bundle on every host -- not something to slip into a
# dependency upgrade. (The migration itself is sound: verified that chromadb
# 1.5.9 reads a 0.6-written bundle and that 0.6.3 still reads it afterwards, so
# a rollback does not strand the data.)
chromadb = "<1.0"
# chromadb 0.6 calls posthog.capture() positionally; posthog 6 changed the
# signature, so every telemetry call fails and chromadb logs it at ERROR.
# Telemetry is off anyway; this is purely to keep the log readable.
posthog = "<6"
pandas = "^2.2.1"
pyasn1 = "^0.6.4"
langchain-openai = "^0.2.3"
langchain-openai = "^1.6.1"
neo4j = "4.3.6"
python-dotenv = "^1.0.1"
pyfiglet = "^1.0.2"
chainlit = "^2.0.3"
asyncpg = "^0.30.0"
sqlalchemy = "^2.0.30"
langchain-community = "^0.3.3"
langchain-core = "^0.3.13"
langchain-huggingface = "^0.1.0"
# Held at 0.3 deliberately. 0.4 requires langchain-classic and drops
# langchain_community.chat_models.vertexai -- which ragas 0.4.3 still imports
# unconditionally at ragas/llms/base.py:12, so langchain-community 0.4 makes
# `import ragas` a ModuleNotFoundError and takes ./bin/evaluate with it. ragas
# declares langchain-community with no version bound, so the resolver cannot
# see the conflict; only importing it does.
#
# 0.3.31 accepts langchain-core <2.0.0,>=0.3.78, so it runs against core 1.x.
# The app uses it for BM25Retriever, the CSV/directory loaders and
# OpenAICallbackHandler. Lift when ragas stops importing the vertexai module.
langchain-community = "^0.3.31"
langchain-core = "^1.6.2"
langchain-huggingface = "^1.2.2"
# Capped, and coupled to the torch pin below. transformers 5 requires torch>=2.5
# and, finding 2.4.1, disables PyTorch entirely rather than failing:
#
Expand All @@ -59,11 +83,14 @@ torch = [
{ version = "2.4.1+cpu", markers = "sys_platform == 'linux' and platform_machine == 'x86_64'", source = "pytorch_cpu" },
{ version = "2.2.*", markers = "sys_platform != 'linux' or platform_machine != 'x86_64'", source = "PyPI" },
]
langchain-chroma = "0.2.3"
langchain-ollama = "^0.2.0"
# 0.2.x, because 1.x requires chromadb>=1.3.5 -- see the chromadb note above.
# 0.2.3 declares langchain-core>=0.3.52 with no upper bound, so it runs against
# core 1.x.
langchain-chroma = "^0.2.3"
langchain-ollama = "^1.1.0"
lark = "^1.2.2"
langgraph = "^0.2.39"
langgraph-checkpoint-postgres = "^2.0.2"
langgraph = "^1.2.11"
langgraph-checkpoint-postgres = "^3.1.2"
rank-bm25 = "^0.2.2"
psycopg = {extras = ["binary"], version = "^3.2.3"}
pydantic = "^2.10.5"
Expand Down Expand Up @@ -92,8 +119,8 @@ mypy = "^1.13.0"
pandas-stubs = "^2.2.3.241009"
types-requests = "^2.32.0.20241016"
types-pyyaml = "^6.0.12.20241230"
datasets = "^3.2.0"
ragas = "^0.2.11"
datasets = "^4.0.0"
ragas = "^0.4.3"

[[tool.poetry.source]]
name = "PyPI"
Expand Down
7 changes: 6 additions & 1 deletion src/agent/graph.py
Original file line number Diff line number Diff line change
Expand Up @@ -186,7 +186,12 @@ def __del__(self) -> None:
there is. The real fix is an explicit lifecycle -- close the pool from the
application's shutdown hook -- which belongs with the agent-API work.
"""
if self.pool is None:
# getattr, not self.pool: __del__ runs even when __init__ raised part
# way through, and then the attribute does not exist yet. That turned a
# readable startup error into "AttributeError: 'AgentGraph' object has no
# attribute 'pool'" printed from __del__, which is where the real cause
# went missing.
if getattr(self, "pool", None) is None:
return
try:
asyncio.get_running_loop()
Expand Down
16 changes: 12 additions & 4 deletions src/evaluation/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -5,10 +5,18 @@ Scores the answers the chatbot actually gives.
- `evaluator.py` (run it as `./bin/evaluate`) asks the **shipping RAG chain** a set
of questions and scores the answers with ragas: faithfulness, answer relevancy,
context utilization, and context recall when reference answers are supplied.
- `test_generator.py` synthesizes question/answer sets from the example corpora.
**It targets the ragas 0.1 API and does not run against the pinned 0.2** — see
the TODO in the file. `tests/golden/questions.txt` is what the evaluator uses by
default and needs no generation.
`test_generator.py` used to sit beside it and was **deleted** in the LangChain 1.x
upgrade. It targeted the ragas 0.1 API — `from_langchain(generator_llm=,
critic_llm=)`, `generate_with_langchain_docs(test_size=, distributions=)` — none
of which exists in the pinned 0.4, and its own TODO had said so since 0.2. It was
not a port away from working; it had not run for three major versions, and
nothing imported it. `git log -- src/evaluation/test_generator.py` has it.

Generating reference answers is still worth having — it is what would let
`context_recall` run. Rebuilding it against the current ragas synthesizer API is
a real piece of work, not a rename, and belongs with whoever wants that metric.
`tests/golden/questions.txt` is what the evaluator uses by default and needs no
generation.

## What changed, and why the old flags are gone

Expand Down
12 changes: 11 additions & 1 deletion src/evaluation/evaluator.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,7 @@
from dotenv import load_dotenv
from langchain_core.language_models.chat_models import BaseChatModel
from ragas import EvaluationDataset, SingleTurnSample, evaluate
from ragas.dataset_schema import EvaluationResult, MultiTurnSample
from ragas.embeddings import LangchainEmbeddingsWrapper
from ragas.llms import LangchainLLMWrapper
from ragas.metrics import (
Expand Down Expand Up @@ -174,7 +175,7 @@ def score(
)
)

samples = [
samples: list[SingleTurnSample | MultiTurnSample] = [
SingleTurnSample(
user_input=q,
response=a,
Expand All @@ -200,6 +201,15 @@ def score(
llm=judge_llm,
embeddings=judge_embeddings,
)
# ragas 0.4 types evaluate() as EvaluationResult | Executor; it returns an
# Executor only when asked to run asynchronously, which this does not do.
# Checked rather than cast, so a future ragas that changes the default says
# so here instead of failing on the next line with an AttributeError.
if not isinstance(result, EvaluationResult):
raise TypeError(
f"ragas returned {type(result).__name__}, not EvaluationResult. "
"evaluate() now defers by default; this tool expects a completed run."
)
# result.scores is a public field holding one dict of metric -> score per
# sample. The aggregate used to come from result._repr_dict, which is private
# and would break on a ragas upgrade without warning -- in a file whose whole
Expand Down
125 changes: 0 additions & 125 deletions src/evaluation/test_generator.py

This file was deleted.

2 changes: 1 addition & 1 deletion src/retrievers/plantreactome/metadata_info.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
from langchain.chains.query_constructor.base import AttributeInfo
from langchain_classic.chains.query_constructor.schema import AttributeInfo

pathway_id_description = "A Plant Reactome Identifier unique to each pathway. A pathway name may appear multiple times in the dataset\
This ID allows for the specific identification and exploration of each pathway's details within the Plant Reactome Database."
Expand Down
8 changes: 6 additions & 2 deletions src/retrievers/rag_chain.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,9 @@
from langchain.chains.combine_documents import create_stuff_documents_chain
from langchain.chains.retrieval import create_retrieval_chain
# langchain_classic, not langchain: LangChain 1.0 moved the pre-LCEL chain
# builders out of the core package into langchain-classic, which is their
# supported home rather than a deprecation shim. Rewriting this as LCEL is a
# behaviour change and does not belong in a dependency upgrade.
from langchain_classic.chains.combine_documents import create_stuff_documents_chain
from langchain_classic.chains.retrieval import create_retrieval_chain
from langchain_core.language_models.chat_models import BaseChatModel
from langchain_core.prompts import BasePromptTemplate, ChatPromptTemplate
from langchain_core.retrievers import BaseRetriever
Expand Down
2 changes: 1 addition & 1 deletion src/retrievers/reactome/metadata_info.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
from langchain.chains.query_constructor.base import AttributeInfo
from langchain_classic.chains.query_constructor.schema import AttributeInfo

pathway_id_description = "A Reactome Identifier unique to each pathway. A pathway name may appear multiple times in the dataset\
This ID allows for the specific identification and exploration of each pathway's details within the Reactome Database."
Expand Down
2 changes: 1 addition & 1 deletion src/retrievers/uniprot/metadata_info.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
from langchain.chains.query_constructor.base import AttributeInfo
from langchain_classic.chains.query_constructor.schema import AttributeInfo

uniprot_descriptions_info = {
"uniprot_data": "Contains detailed protein information about gene names, protein names, subcellular localizations, family classifications, biological pathway associations, domains, motifs, disease associations, and functional descriptions. ",
Expand Down
22 changes: 18 additions & 4 deletions src/tools/external_search/workflow.py
Original file line number Diff line number Diff line change
@@ -1,10 +1,9 @@
from typing import Literal
from typing import Any, Literal, Protocol

from langchain_core.language_models.chat_models import BaseChatModel
from langchain_core.runnables import Runnable, RunnableConfig
from langgraph.graph import StateGraph
from langgraph.graph.state import CompiledStateGraph
from langgraph.utils.runnable import RunnableLike

from agent.tasks.completeness_grader import (
CompletenessGrade,
Expand All @@ -20,11 +19,26 @@ def decide_next_steps(state: SearchState) -> Literal["perform_web_search", "no_s
return "no_search"


def no_search(_: SearchState) -> SearchState:
def no_search(state: SearchState) -> SearchState:
return SearchState(search_results=[])


def run_completeness_grader(grader: Runnable) -> RunnableLike:
class SearchNode(Protocol):
"""The shape langgraph 1.0 requires of a graph node taking a config.

Spelled out rather than imported. This was
`langgraph.utils.runnable.RunnableLike` -- a private module, and a union too
wide for langgraph 1.0's stricter `add_node`.

It has to be a Protocol and not a Callable alias: langgraph matches nodes
structurally on the *parameter name* `state`, which a Callable alias cannot
express. That is also why `no_search` below takes `state` rather than `_`.
"""

def __call__(self, state: SearchState, config: RunnableConfig) -> Any: ...


def run_completeness_grader(grader: Runnable) -> SearchNode:
async def _run_completeness_grader(
state: SearchState, config: RunnableConfig
) -> SearchState:
Expand Down
Loading
Loading