From 08415013ee4dbd81d70e0d13532d0003ac43692c Mon Sep 17 00:00:00 2001 From: Adam Wright Date: Sun, 20 Sep 2026 17:06:06 +0000 Subject: [PATCH] Adversarial review: I cited a source I had not read The comment on the fragility threshold said Reactome's user guide makes the point about hits resting on few entities. It may well. I did not check before writing it, and I could not check afterwards -- the user guide here is a Chroma database, not text. So the citation is withdrawn rather than left standing, and what is actually true is stated instead: five is chosen, taken from this feature's own spec using 2 of 3 as its example of what is not evidence, and setting it properly is a curator's judgement. An invented citation is worse than none. It survives review precisely because it looks like the thing that ends an argument. The second attack found the design sound, which is worth recording too. A flag that fires on everything is as useless as one that never fires, and this one flagged 12 of 12 pathways on the four-identifier analysis I had been testing with. Against a hundred-gene analysis on beta it flags 0 of 12, with found counts of 13 to 67. So it discriminates: it fires where a hit really does rest on nothing and stays quiet on ordinary inputs. Both ends are now pinned with those real shapes rather than invented ones. Co-Authored-By: Claude Opus 5 --- src/analysis/summarise.py | 20 ++++++++++++++++---- tests/analysis/test_summarise.py | 18 ++++++++++++++++++ 2 files changed, 34 insertions(+), 4 deletions(-) diff --git a/src/analysis/summarise.py b/src/analysis/summarise.py index a8ecd49..248e5c8 100644 --- a/src/analysis/summarise.py +++ b/src/analysis/summarise.py @@ -16,10 +16,22 @@ #: Below this many matched entities, a pathway's p-value rests on so little -#: that it should not be read as evidence however small it is. Reactome's own -#: user guide makes this point and it is the single thing a reader most often -#: gets wrong -- a pathway with 2 of 3 entities found looks like a perfect hit -#: and is nearly meaningless. +#: that it should not be read as evidence however small it is -- a pathway +#: with 2 of 3 entities found looks like a perfect hit and is nearly +#: meaningless. +#: +#: **Five is chosen, not derived.** An earlier version of this comment said +#: Reactome's user guide makes the point; it may, but nobody checked before +#: writing that, so the claim is withdrawn rather than left cited. The +#: number comes from this feature's own spec, which uses 2 of 3 as its +#: example of what is not evidence, and from wanting a margin above it. +#: Setting it properly is a curator's judgement, not a programmer's. +#: +#: It does discriminate, which is the part that was measured. Against beta on +#: 2026-09-20: a four-identifier analysis flagged 12 of 12 shown pathways +#: (found counts of 2), and a hundred-gene analysis flagged 0 of 12 (found +#: counts 13 to 67). So it fires on the inputs where a hit really does rest +#: on nothing and stays quiet on ordinary ones. #: #: Computed here rather than left to the model. Handed a small p-value and a #: small count and asked to be careful, a model describes the p-value. diff --git a/tests/analysis/test_summarise.py b/tests/analysis/test_summarise.py index aad6a35..d8c4ed2 100644 --- a/tests/analysis/test_summarise.py +++ b/tests/analysis/test_summarise.py @@ -192,3 +192,21 @@ def test_the_model_is_told_not_to_repeat_our_field_names() -> None: assert "Never use the word 'fragile'" in STATISTICS_INSTRUCTION assert "internal labels" in STATISTICS_INSTRUCTION + + +def test_the_fragility_threshold_discriminates_on_realistic_inputs() -> None: + # A flag that fires on everything is as useless as one that never fires, + # and this one fired on 12 of 12 pathways for a four-identifier analysis. + # Measured against beta: a hundred-gene analysis flags 0 of 12, with + # found counts of 13 to 67. Both ends pinned here with those real shapes. + tiny = _payload(1e-9, 1e-8, 1e-7) + for pathway in tiny["pathways"]: + pathway["entities"].update({"found": 2, "total": 4}) + realistic = _payload(1e-9, 1e-8, 1e-7) + for pathway, found in zip(realistic["pathways"], (31, 17, 13), strict=True): + pathway["entities"].update({"found": found, "total": 164}) + + tiny_flags = [p["fragile"] for p in prompt_input(tiny)["pathways"]] + real_flags = [p["fragile"] for p in prompt_input(realistic)["pathways"]] + assert all(tiny_flags), "a hit on two entities must be flagged" + assert not any(real_flags), "ordinary hits must not all be flagged"