diff --git a/.chainlit/config.toml b/.chainlit/config.toml index 9ff01b2..4ae9b2a 100644 --- a/.chainlit/config.toml +++ b/.chainlit/config.toml @@ -20,7 +20,10 @@ allow_origins = ["*"] [features] # Process and display HTML in messages. This can be a security risk (see https://stackoverflow.com/questions/19603097/why-is-it-dangerous-to-render-user-generated-html-or-javascript) -unsafe_allow_html = true +# Off since 2026-10-03. Citations are markdown links now, so nothing needs +# HTML, and with it on any markup that reached a message -- a handed-off +# summary a visitor had steered -- ran in the reader's browser (review, 1b). +unsafe_allow_html = false # Process and display mathematical expressions. This can clash with "$" characters in messages. latex = false diff --git a/bin/chat-chainlit.py b/bin/chat-chainlit.py index b55cf6b..d237d6a 100644 --- a/bin/chat-chainlit.py +++ b/bin/chat-chainlit.py @@ -53,6 +53,7 @@ from util.config_yml import Config from util.config_yml.messages import TriggerEvent from util.logging import logging +from util.markdown import escape_directives from util.orcid_provider import ORCIDOAuthProvider from util.rate_limit import SlidingWindowLimiter, positive_int from util.secrets import ( @@ -525,6 +526,14 @@ async def answer_with_model(content: str, message_id: str) -> None: enable_postprocess=enable_postprocess, ) assistant_message: cl.Message | None = chainlit_cb.final_stream + if assistant_message is not None: + # Once streamed: "CDK5:p25" in a citation or the prose was read as a + # markdown directive and dropped. Escaped in the final message; while + # it streams it may show briefly without (review follow-up, 10-03). + fixed = escape_directives(assistant_message.content) + if fixed != assistant_message.content: + assistant_message.content = fixed + await assistant_message.update() if ( enable_postprocess diff --git a/src/agent/tasks/cross_database/summarize_reactome_uniprot.py b/src/agent/tasks/cross_database/summarize_reactome_uniprot.py index 048d84f..37ead80 100644 --- a/src/agent/tasks/cross_database/summarize_reactome_uniprot.py +++ b/src/agent/tasks/cross_database/summarize_reactome_uniprot.py @@ -22,11 +22,11 @@ - Incorporate or list these citations clearly so the user can trace the information back to each respective database. - Example: - Reactome Citations: - - Apoptosis - - Cell Cycle + - [Apoptosis](https://reactome.org/content/detail/R-HSA-109581) + - [Cell Cycle](https://reactome.org/content/detail/R-HSA-1640170) - UniProt Citations: - - GATA6 - - NR5A2 + - [GATA6](https://www.uniprot.org/uniprotkb/Q92908) + - [NR5A2](https://www.uniprot.org/uniprotkb/O00482) 7. Answer in the Language requested. 8. Write in a conversational and engaging tone suitable for a chatbot. diff --git a/src/api/answer.py b/src/api/answer.py index 92ada65..b944851 100644 --- a/src/api/answer.py +++ b/src/api/answer.py @@ -30,7 +30,7 @@ from agent.registry import get_graph from api.answer_store import StoredAnswer, answers -from util.anchor_strip import AnchorStripper +from util.anchor_strip import AnchorStripper, MarkdownLinkStripper from util.caller_token import TokenRejectedError, verify from util.logging import logging from util.rate_limit import identity_of, limiter_from_env @@ -135,6 +135,9 @@ async def stream() -> AsyncIterator[str]: # separate events, so they come out here -- across fragment boundaries, # because one anchor arrives as twenty-odd fragments. stripper = AnchorStripper() + # And markdown links, which citations use since the chat stopped + # rendering HTML; anchors are still stripped in case one slips through. + links = MarkdownLinkStripper() # And the trailing source list goes too: this caller renders citations # from the `citation` events, so the prose copy is a duplicate -- and a # worse one, since `AnchorStripper` has just taken its links off. @@ -161,7 +164,7 @@ async def stream() -> AsyncIterator[str]: enable_postprocess=False, ): if event.kind == "token": - text = sources.feed(stripper.feed(event.text)) + text = sources.feed(links.feed(stripper.feed(event.text))) if text: tokens_sent += 1 shown.append(text) @@ -232,7 +235,9 @@ async def stream() -> AsyncIterator[str]: # Nothing reads this thread again. A task, not an await: on a # hang-up this runs during cancellation, where awaiting is unsafe. _forget(graph, thread_id) - held = sources.feed(stripper.flush()) + sources.flush() + held = ( + sources.feed(links.feed(stripper.flush()) + links.flush()) + sources.flush() + ) if held: shown.append(held) yield _sse("token", {"text": held}) diff --git a/src/retrievers/plantreactome/prompt.py b/src/retrievers/plantreactome/prompt.py index 8426b50..43535a9 100644 --- a/src/retrievers/plantreactome/prompt.py +++ b/src/retrievers/plantreactome/prompt.py @@ -17,21 +17,21 @@ - If the context does not answer the question, say the answer was not found **in the Plant Reactome pathway content searched**. Do **not** say it is absent from Plant Reactome, and do **not** answer the question. - The search covers indexed pathway content, not everything Plant Reactome holds, so "not in Plant Reactome" is a claim you are not in a position to make. - If part of the question is unfamiliar -- a name, an acronym, a term absent from the context -- say that part was not found rather than describing it. Never assign a role, function or relationship to something the context does not describe. -2. Inline citations required: Every factual statement must include ≥1 inline anchor citation in the format: display_name +2. Inline citations required: Every factual statement must include ≥1 inline citation, as a markdown link, in the format: [display_name](URL) - If multiple entries support the same fact, cite them together (space-separated). 3. Comprehensiveness: Capture all mechanistically relevant details available in PlantReactome, focusing on processes, complexes, regulations, and interactions. 4. Tone & Style: - Write in a clear, engaging, and conversational tone. - Use accessible language while maintaining technical precision. - Ensure the narrative flows logically, presenting background, mechanisms, and significance -5. Source list at the end: After the main narrative, provide a bullet-point list of each unique citation anchor exactly once, in the same Node Name format. +5. Source list at the end: After the main narrative, provide a bullet-point list of each unique citation link exactly once, in the same [Node Name](URL) format. - Head the list with exactly this line and nothing else: `## Sources` Not a variation on it. The search page strips this section by that exact heading, because it renders the citations itself; a different wording leaves the reader a duplicate list. - Examples: - - Mitosis - - Cell Cycle + - [Mitosis](https://plantreactome.gramene.org/content/detail/R-OSA-9640713) + - [Cell Cycle](https://plantreactome.gramene.org/content/detail/R-OSA-9640670) ## Internal QA (silent) - All factual claims are cited correctly. diff --git a/src/retrievers/reactome/prompt.py b/src/retrievers/reactome/prompt.py index b6afa3c..1c41033 100644 --- a/src/retrievers/reactome/prompt.py +++ b/src/retrievers/reactome/prompt.py @@ -17,7 +17,7 @@ - If the context does not answer the question, say the answer was not found **in the Reactome pathway content searched**. Do **not** say it is absent from Reactome, or from the Reactome Knowledgebase, and do **not** answer the question. - That distinction is not pedantry. This search covers pathways, reactions, complexes, proteins, disease variants and the user guide. Reactome also holds curators and authors, literature references, and much else that is not in this index -- so "not in Reactome" is a claim you are not in a position to make, and it has been wrong: a question about a Reactome curator was answered "not currently available in the Reactome Knowledgebase" while that person was in Reactome as a Person record. - If part of the question is unfamiliar -- a name, an acronym, a term absent from the context -- say that part was not found rather than describing it. Never assign a role, function or relationship to something the context does not describe. -2. Inline citations required: Every factual statement must include ≥1 inline anchor citation in the format: display_name +2. Inline citations required: Every factual statement must include ≥1 inline citation, as a markdown link, in the format: [display_name](URL) - If multiple entries support the same fact, cite them together (space-separated). 3. Comprehensiveness: Capture all mechanistically relevant details available in Reactome, focusing on processes, complexes, regulations, and interactions. - When the question asks **which**, or asks you to **list** or **name** specific entities -- variants, complexes, participants, reactions -- name each one in the context individually. Do not answer at the level of the pathway that groups them. @@ -27,14 +27,14 @@ - Write in a clear, engaging, and conversational tone. - Use accessible language while maintaining technical precision. - Ensure the narrative flows logically, presenting background, mechanisms, and significance -5. Source list at the end: After the main narrative, provide a bullet-point list of each unique citation anchor exactly once, in the same Node Name format. +5. Source list at the end: After the main narrative, provide a bullet-point list of each unique citation link exactly once, in the same [Node Name](URL) format. - Head the list with exactly this line and nothing else: `## Sources` Not a variation on it. The search page strips this section by that exact heading, because it renders the citations itself; a different wording leaves the reader a duplicate list. - Examples: - - Apoptosis - - Cell Cycle + - [Apoptosis](https://reactome.org/content/detail/R-HSA-109581) + - [Cell Cycle](https://reactome.org/content/detail/R-HSA-1640170) ## Internal QA (silent) - All factual claims are cited correctly. diff --git a/src/retrievers/uniprot/prompt.py b/src/retrievers/uniprot/prompt.py index b15fb82..1fc8d5c 100644 --- a/src/retrievers/uniprot/prompt.py +++ b/src/retrievers/uniprot/prompt.py @@ -14,10 +14,10 @@ 3. Answer the question comprehensively and accurately, providing useful background information based **only** on the context. 4. keep track of **all** the sources that are directly used to derive the final answer, ensuring **every** piece of information in your response is **explicitly cited**. 5. Create Citations for the sources used to generate the final answer according to the following: - - For UniProt always format citations in the following format: *short_protein_name*. + - For UniProt always format citations in the following format: [*short_protein_name*](citation). Examples: - - GATA6 - - NR5A2 + - [GATA6](https://www.uniprot.org/uniprotkb/Q92908) + - [NR5A2](https://www.uniprot.org/uniprotkb/O00482) 6. Always provide the citations you created in the format requested, in point-form at the end of the response paragraph, ensuring **every piece of information** provided in the final answer is cited. 7. Write in a conversational and engaging tone suitable for a chatbot. diff --git a/src/retrievers/userguide/prompt.py b/src/retrievers/userguide/prompt.py index 9a7f366..a4d5e20 100644 --- a/src/retrievers/userguide/prompt.py +++ b/src/retrievers/userguide/prompt.py @@ -9,7 +9,7 @@ - If the context contains **nothing** relevant, say the user guide does not currently cover that topic. Do **not** guess. - Otherwise answer from what the context does contain, and **do not preface it with a disclaimer**. Never say the guide does not cover something and then describe it anyway — that reads as a denial of a Reactome feature and undersells it. - The user's wording will often not match Reactome's. Asked about "GSEA hosted by Reactome", answer about **ReactomeGSA**: it is the same thing under Reactome's own name. Match on what the user means, not on whether their exact phrase appears. -2. Inline citations required: Every factual statement must include ≥1 inline anchor citation in the format: display_name +2. Inline citations required: Every factual statement must include ≥1 inline citation, as a markdown link, in the format: [display_name](URL) - Use the **exact** URL from the context (the line starting with `URL:`). Copy it verbatim. - Never guess, shorten, or construct URLs from page titles (for example, do not turn "ReactomeGSA" into `/userguide/reactomegsa`). - Use a clear display name (page title or section title). @@ -23,14 +23,14 @@ - Write in a clear, friendly, and conversational tone. - Use accessible language; avoid unnecessary jargon. - Prefer numbered steps for multi-step procedures. -5. Source list at the end: After the main answer, provide a bullet-point list of each unique citation anchor exactly once, in the same display_name format. +5. Source list at the end: After the main answer, provide a bullet-point list of each unique citation link exactly once, in the same [display_name](URL) format. - Head the list with exactly this line and nothing else: `## Sources` Not a variation on it. The search page strips this section by that exact heading, because it renders the citations itself; a different wording leaves the reader a duplicate list. - Examples: - - Pathway Browser - - Searching Reactome + - [Pathway Browser](https://reactome.org/userguide/pathway-browser) + - [Searching Reactome](https://reactome.org/userguide/searching) ## Internal QA (silent) - All factual claims are cited correctly. diff --git a/src/util/anchor_strip.py b/src/util/anchor_strip.py index 0829b97..de776c9 100644 --- a/src/util/anchor_strip.py +++ b/src/util/anchor_strip.py @@ -76,3 +76,73 @@ def flush(self) -> str: """Whatever is still held, emitted as prose. A truncated tag is not one.""" remaining, self._buffer = self._buffer, "" return remaining + + +# The longest link worth holding back for: display names are short, and URLs +# here are under a hundred characters. Past this, a held '[' is prose. +_MAX_LINK = 400 + + +class MarkdownLinkStripper: + """`[text](url)` becomes `text`, across arbitrary fragment boundaries. + + Since citations moved from HTML anchors to markdown links (so the chat can + render without HTML -- review, area 1b), the search page's prose needs + these stripped the way `AnchorStripper` strips anchors. Holds back only + from a '[' that could still become a link; anything else streams straight + through. Call `flush` at the end. + """ + + def __init__(self) -> None: + self._buffer = "" + + def feed(self, text: str) -> str: + self._buffer += text + out: list[str] = [] + while self._buffer: + start = self._buffer.find("[") + if start == -1: + out.append(self._buffer) + self._buffer = "" + break + out.append(self._buffer[:start]) + self._buffer = self._buffer[start:] + state, label, end = self._parse() + if state == "partial": + break # Could still become a link: wait for more. + if state == "prose": + out.append("[") + self._buffer = self._buffer[1:] + continue + out.append(label) + self._buffer = self._buffer[end:] + return "".join(out) + + def _parse(self) -> tuple[str, str, int]: + """("partial" | "prose" | "link", label, end of the link).""" + held = self._buffer + close = held.find("]") + if close == -1: + state = ( + "partial" if len(held) <= _MAX_LINK and "\n" not in held else "prose" + ) + return state, "", 0 + if close + 1 == len(held): + return "partial", "", 0 # '[label]' -- the next character decides. + if held[close + 1] != "(": + return "prose", "", 0 + end = held.find(")", close + 2) + if end == -1: + tail = held[close + 2 :] + if len(held) > _MAX_LINK or any(c.isspace() for c in tail): + return "prose", "", 0 + return "partial", "", 0 + url = held[close + 2 : end] + if not url or any(c.isspace() for c in url): + return "prose", "", 0 + return "link", held[1:close], end + 1 + + def flush(self) -> str: + """Whatever is still held, as prose: a truncated link is not one.""" + remaining, self._buffer = self._buffer, "" + return remaining diff --git a/src/util/markdown.py b/src/util/markdown.py index 5abe466..b9ff86f 100644 --- a/src/util/markdown.py +++ b/src/util/markdown.py @@ -6,13 +6,16 @@ renders with characters silently missing; a `|` splits the cell. """ +import re + SPECIAL = "\\`*_[]<>|~" def escape(text: str) -> str: """Literal text, on one line: a newline would end a table row.""" flat = " ".join(text.splitlines()) - return "".join(f"\\{ch}" if ch in SPECIAL else ch for ch in flat) + escaped = "".join(f"\\{ch}" if ch in SPECIAL else ch for ch in flat) + return escape_directives(escaped) def inert_html(text: str) -> str: @@ -24,4 +27,16 @@ def inert_html(text: str) -> str: the browser of whoever opened their handoff link (review, area 1b). `escape` would also flatten the summary's bold and lists; this only stops tags. """ - return text.replace("<", "\\<") + return escape_directives(text.replace("<", "\\<")) + + +#: A colon that the chat's markdown would read as a directive: `remark-directive` +#: is in Chainlit's renderer, so the ":p25" in "CDK5:p25" was parsed as markup +#: and dropped -- shown as "CDK5", a break, then the rest. Reactome names are +#: full of these (complexes are written A:B). Never "://" in a URL. +_DIRECTIVE_COLON = re.compile(r"(?<=[A-Za-z0-9]):(?=[A-Za-z])(?!//)") + + +def escape_directives(text: str) -> str: + """Keep "A:B" literal in rendered markdown; nothing else changes.""" + return _DIRECTIVE_COLON.sub(r"\\:", text) diff --git a/tests/api/test_answer_endpoint.py b/tests/api/test_answer_endpoint.py index a8c984e..b17259f 100644 --- a/tests/api/test_answer_endpoint.py +++ b/tests/api/test_answer_endpoint.py @@ -708,3 +708,32 @@ def test_the_one_shot_thread_is_deleted_afterwards( ) assert graph.threads, "the graph was never asked" assert graph.forgotten == graph.threads + + +def test_markdown_link_citations_reach_the_page_as_prose( + keys: tuple[str, str], monkeypatch: pytest.MonkeyPatch +) -> None: + # Citations are markdown links since the chat stopped rendering HTML; the + # page gets citations as events, so the prose carries the label only. + graph = _StubGraph( + [ + AnswerEvent(kind="token", text="CDK5 acts in [Apopto"), + AnswerEvent( + kind="token", + text="sis](https://reactome.org/content/detail/R-HSA-109581).", + ), + AnswerEvent(kind="done", state="answered"), + ] + ) + monkeypatch.setattr("api.answer.get_graph", lambda: graph) + private, public = keys + response = _client(public).post( + f"{PREFIX}/answer", + json={"question": "what is CDK5", "caller_token": _token(private)}, + ) + prose = "".join( + json.loads(data)["text"] + for kind, data in _events(response.text) + if kind == "token" + ) + assert prose == "CDK5 acts in Apoptosis." diff --git a/tests/util/test_anchor_strip.py b/tests/util/test_anchor_strip.py index ec61e55..1568935 100644 --- a/tests/util/test_anchor_strip.py +++ b/tests/util/test_anchor_strip.py @@ -107,3 +107,51 @@ def test_an_unclosed_tag_is_not_swallowed_at_the_end() -> None: out = stripper.feed('kept text str: + return "".join(stripper.feed(p) for p in pieces) + stripper.flush() + + +def test_a_link_becomes_its_label() -> None: + assert _through_links(MarkdownLinkStripper(), [LINKED]) == ( + "CDK5 phosphorylates tau Phosphorylation of tau by CDK5 in neurons." + ) + + +def test_a_link_split_at_every_character_still_strips() -> None: + assert _through_links(MarkdownLinkStripper(), list(LINKED)) == ( + "CDK5 phosphorylates tau Phosphorylation of tau by CDK5 in neurons." + ) + + +def test_prose_brackets_are_left_alone_and_not_held() -> None: + stripper = MarkdownLinkStripper() + # Released as soon as the character after ']' says it is not a link. + assert stripper.feed("an array [1, 2] and ") == "an array [1, 2] and " + assert ( + _through_links(MarkdownLinkStripper(), ["see [note] (below)"]) + == "see [note] (below)" + ) + + +def test_a_truncated_link_is_flushed_as_text() -> None: + assert _through_links(MarkdownLinkStripper(), ["end [Apoptosis](https://reac"]) == ( + "end [Apoptosis](https://reac" + ) + + +def test_two_links_side_by_side() -> None: + text = "[A](https://a.example) [B](https://b.example)" + assert _through_links(MarkdownLinkStripper(), list(text)) == "A B" diff --git a/tests/util/test_markdown.py b/tests/util/test_markdown.py new file mode 100644 index 0000000..97daed5 --- /dev/null +++ b/tests/util/test_markdown.py @@ -0,0 +1,34 @@ +"""Text placed into chat markdown, kept literal.""" + +import pytest + +from util.markdown import escape, escape_directives, inert_html + + +@pytest.mark.parametrize( + ("text", "expected"), + [ + # Chainlit's renderer has remark-directive: ":p25" was parsed as markup + # and dropped, so "CDK5:p25 phosphorylates CDC25A" showed as "CDK5", + # a break, then the rest. + ("CDK5:p25 phosphorylates CDC25A", "CDK5\\:p25 phosphorylates CDC25A"), + ( + "[CDK5:p25 x](https://reactome.org/content/detail/R-HSA-1)", + "[CDK5\\:p25 x](https://reactome.org/content/detail/R-HSA-1)", + ), + # Left alone: not directive-shaped. + ("see https://reactome.org/x", "see https://reactome.org/x"), + ("at 12:30", "at 12:30"), + ("ratio 1:2", "ratio 1:2"), + ("Note: this", "Note: this"), + ], +) +def test_directive_colons_are_escaped_and_nothing_else( + text: str, expected: str +) -> None: + assert escape_directives(text) == expected + + +def test_names_in_tables_and_handed_off_summaries_get_it_too() -> None: + assert "CDK5\\:p25" in escape("CDK5:p25 complex") + assert inert_html("**A:B** ") == "**A\\:B** \\"