diff --git a/src/analysis/summarise.py b/src/analysis/summarise.py index a8ecd49..248e5c8 100644 --- a/src/analysis/summarise.py +++ b/src/analysis/summarise.py @@ -16,10 +16,22 @@ #: Below this many matched entities, a pathway's p-value rests on so little -#: that it should not be read as evidence however small it is. Reactome's own -#: user guide makes this point and it is the single thing a reader most often -#: gets wrong -- a pathway with 2 of 3 entities found looks like a perfect hit -#: and is nearly meaningless. +#: that it should not be read as evidence however small it is -- a pathway +#: with 2 of 3 entities found looks like a perfect hit and is nearly +#: meaningless. +#: +#: **Five is chosen, not derived.** An earlier version of this comment said +#: Reactome's user guide makes the point; it may, but nobody checked before +#: writing that, so the claim is withdrawn rather than left cited. The +#: number comes from this feature's own spec, which uses 2 of 3 as its +#: example of what is not evidence, and from wanting a margin above it. +#: Setting it properly is a curator's judgement, not a programmer's. +#: +#: It does discriminate, which is the part that was measured. Against beta on +#: 2026-09-20: a four-identifier analysis flagged 12 of 12 shown pathways +#: (found counts of 2), and a hundred-gene analysis flagged 0 of 12 (found +#: counts 13 to 67). So it fires on the inputs where a hit really does rest +#: on nothing and stays quiet on ordinary ones. #: #: Computed here rather than left to the model. Handed a small p-value and a #: small count and asked to be careful, a model describes the p-value. diff --git a/tests/analysis/test_summarise.py b/tests/analysis/test_summarise.py index aad6a35..d8c4ed2 100644 --- a/tests/analysis/test_summarise.py +++ b/tests/analysis/test_summarise.py @@ -192,3 +192,21 @@ def test_the_model_is_told_not_to_repeat_our_field_names() -> None: assert "Never use the word 'fragile'" in STATISTICS_INSTRUCTION assert "internal labels" in STATISTICS_INSTRUCTION + + +def test_the_fragility_threshold_discriminates_on_realistic_inputs() -> None: + # A flag that fires on everything is as useless as one that never fires, + # and this one fired on 12 of 12 pathways for a four-identifier analysis. + # Measured against beta: a hundred-gene analysis flags 0 of 12, with + # found counts of 13 to 67. Both ends pinned here with those real shapes. + tiny = _payload(1e-9, 1e-8, 1e-7) + for pathway in tiny["pathways"]: + pathway["entities"].update({"found": 2, "total": 4}) + realistic = _payload(1e-9, 1e-8, 1e-7) + for pathway, found in zip(realistic["pathways"], (31, 17, 13), strict=True): + pathway["entities"].update({"found": found, "total": 164}) + + tiny_flags = [p["fragile"] for p in prompt_input(tiny)["pathways"]] + real_flags = [p["fragile"] for p in prompt_input(realistic)["pathways"]] + assert all(tiny_flags), "a hit on two entities must be flagged" + assert not any(real_flags), "ordinary hits must not all be flagged"