Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions bin/chat-fastapi.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,9 +15,9 @@

from agent.registry import build_graph, set_graph
from api.answer import router as answer_router
from util.caller_token import load_verifying_key
from util.captcha_scope import is_captcha_exempt
from util.embedding_environment import EmbeddingEnvironment
from util.human_token import load_verifying_key
from util.logging import logging
from util.secrets import SECRET_NAMES, get_secret, load_secrets_to_environ

Expand All @@ -43,7 +43,7 @@ async def lifespan(_app: FastAPI) -> AsyncIterator[None]:
started = time.monotonic()
# Before the graph: a missing verifying key must stop the process, and
# spending 52 seconds building a graph first only delays the failure.
_app.state.human_token_key = load_verifying_key()
_app.state.caller_token_key = load_verifying_key()
# The release the served answers are built from, for FR-007 cache
# invalidation. Read here rather than per request because the graph below is
# built from these same bundles, so this value describes what is served even
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -29,8 +29,8 @@ def main() -> int:
directory = Path(sys.argv[1])
directory.mkdir(parents=True, exist_ok=True)

public_path = directory / "human_token_public.pem"
private_path = directory / "human_token_private.pem"
public_path = directory / "caller_token_public.pem"
private_path = directory / "caller_token_private.pem"

# Refuse rather than overwrite: silently replacing a private key would
# invalidate every token in flight with no way back.
Expand Down Expand Up @@ -61,7 +61,7 @@ def main() -> int:
print(f"private {private_path} (0600, for whoever mints tokens -- D1)")
print()
print("Point the service at the public half:")
print(f" HUMAN_TOKEN_PUBLIC_KEY_PATH=/run/secrets/{public_path.name}")
print(f" CALLER_TOKEN_PUBLIC_KEY_PATH=/run/secrets/{public_path.name}")
return 0


Expand Down
12 changes: 6 additions & 6 deletions compose.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -28,11 +28,11 @@ services:
OAUTH_GOOGLE_CLIENT_SECRET: ${OAUTH_GOOGLE_CLIENT_SECRET}
OPENAI_API_KEY: ${OPENAI_API_KEY}
TAVILY_API_KEY: ${TAVILY_API_KEY}
HUMAN_TOKEN_PUBLIC_KEY_PATH: /run/secrets/HUMAN_TOKEN_PUBLIC_KEY
CALLER_TOKEN_PUBLIC_KEY_PATH: /run/secrets/CALLER_TOKEN_PUBLIC_KEY
# Postgres access when not using Vault (for development)
POSTGRES_PASSWORD: ${POSTGRES_PASSWORD}
secrets:
- HUMAN_TOKEN_PUBLIC_KEY
- CALLER_TOKEN_PUBLIC_KEY
- CHAINLIT_AUTH_SECRET
- CLOUDFLARE_SECRET_KEY
- OAUTH_AUTH0_CLIENT_SECRET
Expand Down Expand Up @@ -72,11 +72,11 @@ services:
CLOUDFLARE_SECRET_KEY: ${CLOUDFLARE_SECRET_KEY}
OPENAI_API_KEY: ${OPENAI_API_KEY}
TAVILY_API_KEY: ${TAVILY_API_KEY}
HUMAN_TOKEN_PUBLIC_KEY_PATH: /run/secrets/HUMAN_TOKEN_PUBLIC_KEY
CALLER_TOKEN_PUBLIC_KEY_PATH: /run/secrets/CALLER_TOKEN_PUBLIC_KEY
# Postgres access when not using Vault (for development)
POSTGRES_PASSWORD: ${POSTGRES_PASSWORD}
secrets:
- HUMAN_TOKEN_PUBLIC_KEY
- CALLER_TOKEN_PUBLIC_KEY
- CLOUDFLARE_SECRET_KEY
- OPENAI_API_KEY
- TAVILY_API_KEY
Expand Down Expand Up @@ -158,9 +158,9 @@ services:
secrets:
# The public half of the answer endpoint's token-verifying keypair.
# External like the rest, so no key material lives in the repository:
# docker secret create HUMAN_TOKEN_PUBLIC_KEY deploy/beta/human_token_public.pem
# docker secret create CALLER_TOKEN_PUBLIC_KEY deploy/beta/caller_token_public.pem
# Generate the pair with ./bin/make-human-token-keypair.py deploy/beta
HUMAN_TOKEN_PUBLIC_KEY:
CALLER_TOKEN_PUBLIC_KEY:
external: true
CHAINLIT_AUTH_SECRET:
external: true
Expand Down
16 changes: 8 additions & 8 deletions docker-compose.yml
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,7 @@ services:
- OAUTH_GOOGLE_CLIENT_SECRET=${OAUTH_GOOGLE_CLIENT_SECRET}
- CHAINLIT_AUTH_SECRET=${CHAINLIT_AUTH_SECRET}
- CHAINLIT_URI=${CHAINLIT_URI}
- HUMAN_TOKEN_PUBLIC_KEY_PATH=${HUMAN_TOKEN_PUBLIC_KEY_PATH}
- CALLER_TOKEN_PUBLIC_KEY_PATH=${CALLER_TOKEN_PUBLIC_KEY_PATH}
- CHAINLIT_URL=${CHAINLIT_URL}
- CHAINLIT_ROOT_PATH=${CHAINLIT_ROOT_PATH}
- TAVILY_API_KEY=${TAVILY_API_KEY}
Expand All @@ -38,7 +38,7 @@ services:
- ./records:/app/records
- ./config.yml:/app/config.yml
secrets:
- human_token_public.pem
- caller_token_public.pem

chainlit-no-login:
image: ${CHAINLIT_IMAGE}
Expand All @@ -55,7 +55,7 @@ services:
- CLOUDFLARE_SECRET_KEY=${CLOUDFLARE_SECRET_KEY}
- CLOUDFLARE_SITE_KEY=${CLOUDFLARE_SITE_KEY}
- CHAINLIT_URI=${CHAINLIT_URI_NO_LOGIN}
- HUMAN_TOKEN_PUBLIC_KEY_PATH=${HUMAN_TOKEN_PUBLIC_KEY_PATH}
- CALLER_TOKEN_PUBLIC_KEY_PATH=${CALLER_TOKEN_PUBLIC_KEY_PATH}
- CHAINLIT_URI_LOGIN=${CHAINLIT_URI}
- CHAINLIT_URL=${CHAINLIT_URL}
- TAVILY_API_KEY=${TAVILY_API_KEY}
Expand All @@ -68,7 +68,7 @@ services:
- ./embeddings:/app/embeddings
- ./config.yml:/app/config.yml
secrets:
- human_token_public.pem
- caller_token_public.pem


postgres:
Expand Down Expand Up @@ -106,12 +106,12 @@ secrets:
# exist, and it declares this as an external secret like every other.
# This file-based form keeps THIS compose file working standalone.
#
# Delivered at /run/secrets/human_token_public.pem, which is what
# HUMAN_TOKEN_PUBLIC_KEY_PATH points at. A secret rather than a bind mount on
# Delivered at /run/secrets/caller_token_public.pem, which is what
# CALLER_TOKEN_PUBLIC_KEY_PATH points at. A secret rather than a bind mount on
# purpose: a bind mount whose source is missing makes Docker create a
# root-owned directory at that path, which then needs sudo to clear and which
# the key generator would refuse to overwrite. A missing secret just errors.
#
# Generate it with ./bin/make-human-token-keypair.py deploy/beta
human_token_public.pem:
file: ./deploy/beta/human_token_public.pem
caller_token_public.pem:
file: ./deploy/beta/caller_token_public.pem
38 changes: 34 additions & 4 deletions specs/010-search-page-answers/contracts/answer_endpoint.md
Original file line number Diff line number Diff line change
Expand Up @@ -11,12 +11,36 @@ POST /chat/api/answer
Content-Type: application/json

{ "question": "what does CDK5 phosphorylate in Alzheimer disease?",
"human_token": "<evidence the caller verified a person>" }
"caller_token": "<a token the website minted for this call>" }
```

`human_token` is D1 and **not yet decided** -- a signed cookie on the shared parent
domain, a short-lived minted token, or a server-side vouch. This repo verifies
evidence; it does not perform the check.
**`caller_token` is settled (D1, 2026-09-18), and it does not assert humanity.**
The field was `human_token` and the premise was wrong: there is no human gate on
the search path and there will not be one -- nobody solves a captcha to run a
search. What the token asserts is *caller identity*.

The website mints it server-side, per request, and verification here is:

| claim | required | checked against |
|---|---|---|
| signature | yes | the public key, EdDSA or RS256 -- never an HS* algorithm |
| `exp` | yes | now; they mint at +120s |
| `aud` | yes | `reactome-chatbot`, overridable with `CALLER_TOKEN_AUDIENCE` |
| `sub` | no, but expected | not validated; used as the rate-limit key |
| `iss` | no | not currently checked -- the signature already identifies the minter |

`sub` is an opaque per-visit id, 128 random bits, not derived from anything about
the reader. The backstop limit keys on it, so repeat questions in one reading
session count as one caller and a freshly minted token does not buy a fresh
allowance.

**A token with no `aud`, or one minted for a different audience, is refused.**
Worth stating because the earlier code refused *every* token carrying an `aud`
claim -- PyJWT rejects one when no audience is expected -- so this had to change
before the first real token could ever have been accepted.

Abuse control is the website's: the panel is opt-in behind a click, so a crawled
search never reaches a model, and their proxy rate limits by address.

## Response: Server-Sent Events

Expand Down Expand Up @@ -72,6 +96,12 @@ produces a `done` with a non-`answered` state. The website renders no panel. The
search page must never be slower or broken because this service is down (FR-006,
SC-004).

**Hanging up cancels the work.** Measured: when the caller closes the connection
mid-stream, `CancelledError` is raised inside the answer generator and nothing
further is produced -- 3 tokens read, 3 produced, then cancelled. So a reader who
navigates away does not cost a full model call, and there is nothing for the
proxy to cancel upstream beyond closing the connection.

**The stream is bounded.** The server gives up after 120 seconds and sends `done`
with `state: failed`. A caller still needs its own timeout -- a dropped connection
sends nothing -- but the server will not hold one open indefinitely. 120s is about
Expand Down
2 changes: 1 addition & 1 deletion specs/010-search-page-answers/plan.md
Original file line number Diff line number Diff line change
Expand Up @@ -89,7 +89,7 @@ specs/010-search-page-answers/
src/
├── api/ # new: the endpoint, its models, SSE framing
├── agent/graph.py # gains a streaming surface
└── util/human_token.py # new: signature verification only
└── util/caller_token.py # new: signature verification only

bin/chat-fastapi.py # mounts the router; middleware ordering matters
tests/api/ # new
Expand Down
2 changes: 1 addition & 1 deletion specs/010-search-page-answers/quickstart.md
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,7 @@ Measured 2026-09-17 across the 15 tracked sweep questions, so Phase 5 has a befo
curl -N -X POST https://beta.reactome.org/chat/api/answer \
-H 'Content-Type: application/json' \
-d '{"question":"what does CDK5 phosphorylate in Alzheimer disease?",
"human_token":"<token>"}'
"caller_token":"<token>"}'
```

`-N` matters. Without it curl buffers and the stream looks like a single slow
Expand Down
39 changes: 25 additions & 14 deletions specs/010-search-page-answers/spec.md
Original file line number Diff line number Diff line change
Expand Up @@ -234,20 +234,31 @@ or dropping it for this path -- is open, and is not a blocker for the handover.

## Decisions

### D1 -- what proves a person is human?

Turnstile already exists here, and a bug in it was fixed on 2026-09-17: a deployment
mounting `CLOUDFLARE_SECRET_KEY` as a Docker secret had the captcha silently
disabled, because the middleware read `os.environ` directly while the value came from
`get_secret`.

What is missing is the handoff. The website needs something to present to this
service. Options: a signed cookie on the shared parent domain (what the chat uses
now); a short-lived token the website mints after its own Turnstile check; or the
website proxying the call and vouching server-side.

**Open.** It is a security boundary and belongs with whoever owns the website's
session model, not with this repo alone.
### D1 -- what does the caller present? **Decided 2026-09-18**

This was "what proves a person is human?", and the question had a false premise.
The website session checked before answering: there is **no human gate on the
search path**, and there is not going to be one -- nobody solves a captcha to run
a search. Their only hCaptcha belongs to the contact form, gating that form and
spent on submit. Every search is anonymous and ungated by design.

So the token cannot honestly assert humanity, and asking it to would have meant
inventing a claim. **It asserts caller identity instead**: minted server-side by
the website, per request, EdDSA, with `iss`, `aud: reactome-chatbot`, `iat`,
`exp` at +120s, and `sub` -- an opaque per-visit id of 128 random bits, not
derived from anything about the reader, held in an HttpOnly cookie. No address,
no IP hash; neither side holds personal data.

This service verifies signature, expiry and **audience**, and keys its backstop
limit on `sub`. The request field is `caller_token`, and the misleading
`caller_token` name is gone from the code as well as this document.

Abuse control moved with the premise. The panel is opt-in behind a click (D5), so
a crawled search never reaches a model, and the website's proxy rate limits by
address. That is a real deterrent rather than a claim we would be inventing.

Turnstile stays where it is, guarding the chat UI. It was never part of this
path, and the two must not be "unified".

### D2 -- which searches get an answer?

Expand Down
9 changes: 6 additions & 3 deletions specs/010-search-page-answers/tasks.md
Original file line number Diff line number Diff line change
Expand Up @@ -27,9 +27,9 @@ website repo entirely. Speed is Phase 5 and does not gate the handover.
arrives, tokens stream, citations resolve to real stable IDs, and `done` carries a
state.

- [x] T007 [P] [US1] Add src/util/human_token.py: verify signature and expiry only, stateless, no consumption tracking
- [x] T008 [P] [US1] Test human_token in tests/util/test_human_token.py: valid, expired, wrong key, tampered payload, absent — every failure refuses
- [x] T009 [US1] Refuse at startup in src/util/human_token.py when the verifying key is missing or unreadable, rather than accepting everything (Principle IV)
- [x] T007 [P] [US1] Add src/util/caller_token.py: verify signature and expiry only, stateless, no consumption tracking
- [x] T008 [P] [US1] Test caller_token in tests/util/test_caller_token.py: valid, expired, wrong key, tampered payload, absent — every failure refuses
- [x] T009 [US1] Refuse at startup in src/util/caller_token.py when the verifying key is missing or unreadable, rather than accepting everything (Principle IV)
- [x] T010 [US1] Add the SSE endpoint in src/api/answer.py implementing contracts/answer_endpoint.md: start, token, citation, done
- [x] T011 [US1] Emit citations from retrieved documents' `st_id` metadata, deduplicated — never by parsing anchors out of the model's prose
- [x] T012 [US1] Mount the router in bin/chat-fastapi.py and let the captcha middleware pass /chat/api/ through, since the endpoint verifies its own caller
Expand All @@ -44,6 +44,9 @@ state.
- [x] T016 [US2] Test that no model call happens for a refused request in tests/api/test_answer_endpoint.py, by asserting on a patched graph rather than on timing (SC-002)
- [x] T017 [P] [US2] Rate limit per token as a backstop in src/util/rate_limit.py; the budget is the website's, enforced before the call reaches here (FR-008). 30 per 10 minutes, keyed on `sub`/`jti` when D1 provides one and a token hash until then (PR #237)
- [x] T017a [US2] Stop paying for a discarded web search: the endpoint took `enable_postprocess` at its default, so every answer ran a Tavily search that `astream_answer` has no event to return (PR #237)
- [x] T021 [US2] Enforce `aud` on the caller token (asked for by the website, D1); the code refused every token carrying one, since PyJWT rejects `aud` when no audience is expected (PR #240)
- [x] T022 Rename human_token -> caller_token everywhere; D1 established the token asserts caller identity, not humanity (PR #240)
- [x] T023 Answer the website's cancellation question: a client hang-up raises CancelledError inside the answer generator and produces nothing further, so they need not cancel upstream (PR #240)
- [ ] T020a Decide what to do about non-reproducible retrieval: three runs of one question shared only 4 of 19 citations (Jaccard 0.26) because query expansion is itself a model call. Affects what FR-007 can cache
- [x] T018 [US2] Return `state: failed` with no partial answer on any internal error, so the page renders no panel (FR-006)
- [x] T011a [US1] Strip inline HTML anchors from the token stream in src/util/anchor_strip.py; the contract promises prose without them and the chat prompt emits them, split across ~20 fragments (PR #236)
Expand Down
10 changes: 5 additions & 5 deletions src/api/answer.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,7 @@

from agent.registry import get_graph
from util.anchor_strip import AnchorStripper
from util.human_token import TokenRejectedError, verify
from util.caller_token import TokenRejectedError, verify
from util.logging import logging
from util.rate_limit import identity_of, limiter_from_env

Expand All @@ -57,7 +57,7 @@

class AnswerRequest(BaseModel):
question: str = Field(min_length=1, max_length=2000)
human_token: str = ""
caller_token: str = ""


def _sse(event: str, payload: dict[str, Any]) -> str:
Expand All @@ -83,21 +83,21 @@ async def body() -> AsyncIterator[str]:

@router.post("/answer")
async def answer(request: Request, body: AnswerRequest) -> StreamingResponse:
verifying_key = getattr(request.app.state, "human_token_key", None)
verifying_key = getattr(request.app.state, "caller_token_key", None)
if not verifying_key:
# Should be unreachable: startup refuses without a key. If it happens,
# refuse rather than answer.
return _refusal("no verifying key on the app")

try:
claims = verify(body.human_token, verifying_key)
claims = verify(body.caller_token, verifying_key)
except TokenRejectedError as rejected:
return _refusal(rejected.reason)

# After verification, so an unsigned token cannot consume someone else's
# budget by claiming their `sub`, and before the graph, so a caller over the
# limit costs nothing.
if not _limiter.allow(identity_of(claims, body.human_token)):
if not _limiter.allow(identity_of(claims, body.caller_token)):
return _refusal("rate limited")

graph = get_graph()
Expand Down
Loading
Loading