From ee43863e0a235823f8f6d1ccfc45e7acd8680fae Mon Sep 17 00:00:00 2001 From: NetDevAutomate Date: Sat, 19 Sep 2026 10:20:12 +0100 Subject: [PATCH 01/32] =?UTF-8?q?test(ci):=20RED=20=E2=80=94=20the=20night?= =?UTF-8?q?ly=20installer=20job=20must=20plant=20a=20harness=20before=20in?= =?UTF-8?q?stall.sh?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The `installer` job in nightly-install.yml (A4, b5339db9, 2026-09-14) has failed every night since it was added (runs 34949338879 .. 35431368317). It isolates HOME to `${{ runner.temp }}/home` so `studyloop install agents` writes into scratch, but an empty HOME has no harness: on the runner no codex/opencode/pi/grok binary is on PATH and no ~/.kiro, ~/.claude, ~/.pi or ~/.grok exists, so `detect_available_agent_tools()` returns [] and `install.sh` exits 1 at "Installing agent definitions" — the script's documented behaviour when no supported AI tool is present. The two tools the job was written to prove had already installed fine by then; the verify step never ran. Reproduced locally with an empty HOME and a PATH holding only uv and the system bins: exit 1, same message. Planting ~/.kiro (or ~/.claude, ~/.pi, ~/.grok — the markers the detector reads without a binary) makes the same command exit 0 and write the links. This test pins the fixture: the run step plants a marker before the script, and a verify step reads what `install agents` wrote into the isolated HOME. Red on the current workflow for that reason (1 failed, 13 passed). --- .../tests/test_ci_workflow_contract.py | 54 +++++++++++++++++++ 1 file changed, 54 insertions(+) diff --git a/packages/studyloop/tests/test_ci_workflow_contract.py b/packages/studyloop/tests/test_ci_workflow_contract.py index ebd150b8b..be7101fcd 100644 --- a/packages/studyloop/tests/test_ci_workflow_contract.py +++ b/packages/studyloop/tests/test_ci_workflow_contract.py @@ -236,3 +236,57 @@ def test_sast_and_pre_commit_bandit_skip_lists_agree() -> None: ) assert sast_skipped == precommit_skipped == ci_standards_skipped + + +NIGHTLY_WORKFLOW = WORKFLOW_DIR / "nightly-install.yml" + + +def _nightly_workflow() -> dict[str, Any]: + return yaml.safe_load(NIGHTLY_WORKFLOW.read_text(encoding="utf-8")) + + +def test_nightly_installer_job_plants_a_harness_before_running_install_sh() -> None: + """The nightly `installer` job (A4, added 2026-09-14) isolates HOME so + `studyloop install agents` writes into scratch -- and had never passed: + an empty HOME has no harness, `detect_available_agent_tools()` finds + nothing, and `install.sh` exits 1 at "Installing agent definitions" + exactly as the README says it should when no supported AI tool exists. + The job died five nights running before its verify step ever ran. + + The fixture must supply the precondition the script documents: at least + one harness marker directory that the detector reads without a binary + (`~/.kiro`, `~/.claude`, `~/.pi`, `~/.grok`), created in the isolated + HOME by the same step that runs the script. And the job must then check + what `install agents` wrote, or the isolation buys nothing. + """ + data = _nightly_workflow() + installer = data["jobs"]["installer"] + steps = installer["steps"] + run_step = next(step for step in steps if step.get("name") == "Run scripts/install.sh") + + assert run_step["env"]["HOME"] == "${{ runner.temp }}/home", ( + "HOME isolation is the point of the job; it must stay" + ) + run = run_step["run"] + assert "./scripts/install.sh --non-interactive --no-smoke" in run + + planted = [ + marker + for marker in ("$HOME/.kiro", "$HOME/.claude", "$HOME/.pi", "$HOME/.grok") + if marker in run + ] + assert planted, ( + "the installer job runs install.sh in an empty HOME; plant at least one " + "directory-detected harness marker first or `install agents` exits 1" + ) + plant_at = min(run.index(marker) for marker in planted) + assert plant_at < run.index("./scripts/install.sh"), "plant the marker BEFORE the script runs" + + verify = next( + (step for step in steps if step.get("name") == "Verify installed agent definitions"), + None, + ) + assert verify is not None, "the job must verify what `install agents` wrote into HOME" + assert verify["env"]["HOME"] == "${{ runner.temp }}/home" + for planted_marker in planted: + assert planted_marker in verify["run"], f"verify step does not look inside {planted_marker}" From 57b38838698098eb53a2babafea9b4f2faf89b5d Mon Sep 17 00:00:00 2001 From: NetDevAutomate Date: Sat, 19 Sep 2026 10:22:05 +0100 Subject: [PATCH 02/32] fix(ci): plant harness markers in the nightly installer job's isolated HOME MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit GREEN for ee43863e. The `installer` job isolates HOME so `install agents` writes into scratch; the fixture now creates ~/.kiro, ~/.claude, ~/.pi and ~/.grok in that HOME first — the four markers detect_available_agent_tools() reads without a binary — so `install.sh` reaches and exercises all four link sets instead of refusing with "No supported AI tools detected". A new step verifies what `install agents` wrote: two artefacts each for kiro, claude and grok, one for pi, every path taken from installers.py (_TOOL_LINKS / _configure_*). Without a read of the isolated HOME the isolation proved nothing. Proved locally the way the runner runs it: full ./scripts/install.sh --non-interactive --no-smoke with HOME, UV_TOOL_DIR and UV_TOOL_BIN_DIR in scratch and a PATH holding only uv and the system bins — exit 0, "Installation complete!", both entry points answer, all seven artefacts present. Contract module 14/14. The nightly run itself can only be observed at 03:30 UTC or via workflow_dispatch on main after merge. --- .github/workflows/nightly-install.yml | 27 ++++++++++++++++++++++++++- 1 file changed, 26 insertions(+), 1 deletion(-) diff --git a/.github/workflows/nightly-install.yml b/.github/workflows/nightly-install.yml index 6c5b2f5de..9fedbc088 100644 --- a/.github/workflows/nightly-install.yml +++ b/.github/workflows/nightly-install.yml @@ -72,11 +72,22 @@ jobs: # (~/.kiro, ~/.claude, ~/.codex, ~/.config/opencode). With a fresh HOME # there is no Python 3.12 either, so this also exercises the script's # `uv python install` fallback every night. + # + # The script refuses, by design, when no supported AI tool is present + # ("No supported AI tools detected"), and the runner has none: no + # harness binary on PATH and nothing in the fresh HOME. So the fixture + # plants the four harness markers `detect_available_agent_tools()` + # reads without a binary -- ~/.kiro, ~/.claude, ~/.pi, ~/.grok -- and + # the install-agents step then exercises all four link sets for real. + # (Without this the job failed every night from 2026-09-15, before its + # verify steps ever ran; pinned by test_ci_workflow_contract.py.) env: HOME: ${{ runner.temp }}/home UV_TOOL_DIR: ${{ runner.temp }}/tools UV_TOOL_BIN_DIR: ${{ runner.temp }}/bin - run: mkdir -p "$HOME" && ./scripts/install.sh --non-interactive --no-smoke + run: | + mkdir -p "$HOME/.kiro" "$HOME/.claude" "$HOME/.pi" "$HOME/.grok" + ./scripts/install.sh --non-interactive --no-smoke - name: Verify installed CLI entry points env: @@ -84,3 +95,17 @@ jobs: run: | "$UV_TOOL_BIN_DIR/studyloop" --version "$UV_TOOL_BIN_DIR/session-export" --help + + - name: Verify installed agent definitions + # One artefact per planted harness, each a path `install agents` + # writes for that tool (installers.py _TOOL_LINKS / _configure_*). + env: + HOME: ${{ runner.temp }}/home + run: | + test -e "$HOME/.kiro/agents/study-mentor.json" + test -e "$HOME/.kiro/agents/study-plan-architect.json" + test -e "$HOME/.claude/agents/socratic-mentor.md" + test -e "$HOME/.claude/agents/study-plan-architect.md" + test -e "$HOME/.pi/agent/AGENTS.md" + test -e "$HOME/.grok/hooks/studyloop.json" + test -e "$HOME/.grok/rules/session-db.md" From 4f8e3e0f3c723b2a60dbd12a49d6c83aeea2f04f Mon Sep 17 00:00:00 2001 From: NetDevAutomate Date: Sat, 19 Sep 2026 10:25:20 +0100 Subject: [PATCH 03/32] =?UTF-8?q?docs(tasks):=20item=207=20step=204=20done?= =?UTF-8?q?=20=E2=80=94=20#25=20(D-D),=20#26=20(D-E)=20filed;=20#21=20stat?= =?UTF-8?q?us=20posted,=20left=20open?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Both issues carry the proposals' content with every cited test id, symbol and path verified against main a03fc9bd before posting. #21 stays open because its definition of done names OpenCode in the core tuple and the harness receipt records OpenCode as preview on two named blockers. --- openspec/changes/plan-integration-followons/tasks.md | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/openspec/changes/plan-integration-followons/tasks.md b/openspec/changes/plan-integration-followons/tasks.md index a8f912e72..8f3b6d44b 100644 --- a/openspec/changes/plan-integration-followons/tasks.md +++ b/openspec/changes/plan-integration-followons/tasks.md @@ -228,7 +228,11 @@ writer, through the existing gate, closes that. Kept out of item 3 so item 3's f byte-equal afterwards; single `main` on both sides. (3) closeout comments — done: posted on #8–#15 from the re-verified draft (`receipts/issue-closeout-draft-2026-09-16.md`, status paragraph records the shas rewritten and the rows overtaken); #8, #9, #11, #12, #13, #14 closed as completed; #10 and #15 open (row 3b); #7 - reopened (auto-closed by PR #20's body) with the parent mapping. PR #20's body left as merged. **Open:** - (4) the two item-6 issues (after T6.4) and #21's closing comment or decision; (5) the local tag + reopened (auto-closed by PR #20's body) with the parent mapping. PR #20's body left as merged. (4) the two + item-6 issues — done 2026-09-19: **#25** (D-D derived plan bias) and **#26** (D-E nudge + retire/snooze), both + `ready-for-agent`, bodies drawn from the proposals with every cited test id, symbol and path checked against + `main` `a03fc9bd`; #21 given a status comment from the harness receipt and **left open** (its DoD names OpenCode + in core; OpenCode stays preview on the `opencode.db` exporter and the no-completed-reply gaps). **Open:** + (5) the local tag `archive/feat-clean-start-2026-09-15` — owner: push or discard; (6) the GitHub Support ticket text — owner; (7) revoke both tokens and delete `~/tmp/.env` — owner (D-J). From ef319a7bd75728be4aba8b10856310dbcbaaaedc Mon Sep 17 00:00:00 2001 From: NetDevAutomate Date: Sat, 19 Sep 2026 12:10:24 +0100 Subject: [PATCH 04/32] =?UTF-8?q?test(now):=20RED=20=E2=80=94=20item=205?= =?UTF-8?q?=20(D-F):=20energy=20demand=20for=20repair=20and=20the=20body-d?= =?UTF-8?q?oubling=20floor?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rubric row 3 was the owner's one "no" on the D-16 walkthrough (2026-09-16): at low energy the engine recommended hands-on repair of a LIVE struggle, because a struggle-repair candidate carried no energy demand of its own — rule 3 only ever gated new milestone work. Recommending that on a low-energy day risks compounding the struggle (RSD). Design §5 plus its three T5.1 amendments (read against decision.py on 2026-09-18) is the spec these six tests pin: - demand is derived in the struggle collector from the collector's own classes: `struggling` seen within 14 days → high (asks for 6/10); older `struggling`, or a row whose only signal is a weak teach-back → medium (4/10); `learning` → low (0/10); carried in the candidate's metadata; - below the capability, repair above its demand is deferred like new work into a NEW additive key `energy_deferred_repairs` (`DeferredRepair`: plan_id/plan_title when plan-related, else None) — `energy_deferred` is milestone-shaped and its three renderers would print "milestone None"; - due recall is never deferred whatever its confidence says; - when nothing plan-related fits and an active plan exists, one `source="body_double"` conversation candidate is synthesised, base below MILESTONE_BASE_SCORE, plan_refs (plan, None), reason naming the deferred items, evidence_command the co-study session door (`studyloop study "" --mode co-study`) — a proposal, never a filter: a real unrelated candidate still wins and it sits beneath; - no active plan → no body double (a draft is not active); the deferral is plan-independent; when every real candidate was deferred the starter stands in and its reason says so rather than "no evidence found yet"; - each renderer gains one line per deferred repair (CLI now, recap sentence, Today card `deferredRepairNotes()`), a body-double primary shows "Sit with the plan" + its door instead of "Record evidence", and the Today card routes it to the Body Double view (`viewForAction`). The struggle collector runs for real over patched `observations.rows`, since injecting candidates through `_due_progress_candidates` would bypass the derivation under test. 6 failed / 40 passed (golden byte-identity green); JS 5 failed / 139 passed — each for the missing name it is written against. --- .../tests/js/today-panel-plan.test.js | 97 +++++++ .../studyloop/tests/test_now_plan_guidance.py | 271 ++++++++++++++++++ 2 files changed, 368 insertions(+) diff --git a/packages/studyloop/tests/js/today-panel-plan.test.js b/packages/studyloop/tests/js/today-panel-plan.test.js index 6de304877..004c20692 100644 --- a/packages/studyloop/tests/js/today-panel-plan.test.js +++ b/packages/studyloop/tests/js/today-panel-plan.test.js @@ -268,3 +268,100 @@ test('warningNotes: absent key renders no warning text', () => { assert.deepEqual(panel.warningNotes(), []); assert.equal(panel.hasPlanContext, false); }); + +/* Item 5 (D-F): a repair the day's energy cannot carry, in its own additive key + (design §5 amendment 2 — `energy_deferred` is milestone-shaped), and the + body-double proposal the engine synthesises when nothing plan-related fits. */ +const DEFERRED_REPAIR_PAYLOAD = { + ...PLAN_PAYLOAD, + primary: { + concept: 'Sit with SQL Windows', + action_type: 'conversation', + estimated_minutes: 25, + reason: 'Nothing plan-related fits low energy today', + source: 'body_double', + evidence_command: 'studyloop study "SQL Windows" --mode co-study', + plan_refs: [{ plan_id: 'sql-windows', milestone_index: null }], + }, + energy_deferred_repairs: [ + { + plan_id: 'sql-windows', + plan_title: 'SQL Windows', + concept: 'window function', + topic: 'sql', + confidence: 'struggling', + energy_demand: 'high', + required_capability: 6, + energy_capability: 3, + reason: 'low energy carries 3/10; repairing a live struggle asks for at least 6/10', + }, + ], +}; + +test('deferredRepairNotes: one readable line per energy-deferred repair', () => { + const panel = todayPanel(); + panel.plan = DEFERRED_REPAIR_PAYLOAD; + + assert.deepEqual(panel.deferredRepairNotes(), [ + 'SQL Windows \u2014 repairing \u201cwindow function\u201d (struggling) waits for more energy ' + + '(asks for 6/10, low energy carries 3/10)', + ]); +}); + +test('deferredRepairNotes: a repair unrelated to any plan names no plan, and counts as plan context alone', () => { + const panel = todayPanel(); + panel.plan = { + ...NO_PLAN_PAYLOAD, + energy: 'low', + energy_deferred_repairs: [ + { + plan_id: null, + plan_title: null, + concept: 'decorators', + topic: 'python', + confidence: 'struggling', + energy_demand: 'high', + required_capability: 6, + energy_capability: 3, + reason: 'low energy carries 3/10', + }, + ], + }; + + assert.deepEqual(panel.deferredRepairNotes(), [ + 'Repairing \u201cdecorators\u201d (struggling) waits for more energy ' + + '(asks for 6/10, low energy carries 3/10)', + ]); + assert.equal(panel.hasPlanContext, true); +}); + +test('deferredRepairNotes: absent key renders nothing, before and after assignment', () => { + const panel = todayPanel(); + + assert.deepEqual(panel.deferredRepairNotes(), []); + + panel.plan = PLAN_PAYLOAD; + + assert.deepEqual(panel.deferredRepairNotes(), []); +}); + +test('a body-double primary starts in the Body Double view; every other action keeps its view', () => { + const panel = todayPanel(); + + assert.equal(panel.viewForAction(DEFERRED_REPAIR_PAYLOAD.primary), 'body-double'); + assert.equal(panel.viewForAction(PLAN_PAYLOAD.primary), 'study-session'); + assert.equal(panel.viewForAction(NO_PLAN_PAYLOAD.primary), 'flashcards'); +}); + +test('the Today card markup renders the deferred repairs beside the deferred milestones', () => { + const html = fs.readFileSync( + new URL('../../src/studyloop/web/static/index.html', import.meta.url), 'utf8', + ); + const start = html.indexOf('class="today-plan-notes"'); + const end = html.indexOf('</div>', html.indexOf('warningNotes()', start)); + const block = html.slice(start, end); + assert.match(block, /x-for="\(note, i\) in deferredRepairNotes\(\)" :key="'r' \+ i"/); + assert.match(block, /Deferred for energy: <span x-text="note">/); + const show = html.slice(html.lastIndexOf('x-show=', start), start); + assert.match(show, /deferredRepairNotes\(\)\.length > 0/, 'the notes block shows for a deferred repair alone'); +}); diff --git a/packages/studyloop/tests/test_now_plan_guidance.py b/packages/studyloop/tests/test_now_plan_guidance.py index 2bd3d24df..42652f5b3 100644 --- a/packages/studyloop/tests/test_now_plan_guidance.py +++ b/packages/studyloop/tests/test_now_plan_guidance.py @@ -1153,3 +1153,274 @@ def test_completion_evidence_cap_keeps_the_counts_and_names_the_overflow(monkeyp assert action.evidence[-1].startswith("… and ") assert action.evidence[-1].endswith(" more") assert f"{action.due_reviews} due reviews" in action.action + + +# --------------------------------------------------------------------------- +# T5.2 — item 5 (D-F): per-item energy demand for repair, and the body-doubling +# floor. Design §5 with its three T5.1 amendments. The struggle collector runs +# for real here — demand is derived in the collector, so injecting candidates +# through ``_due_progress_candidates`` would bypass the very thing under test. +# --------------------------------------------------------------------------- + + +def _struggle( + concept: str, + *, + topic: str = "sql", + confidence: str = "struggling", + days_ago: int = 3, + teachback: int | None = None, +) -> dict: + """One row as ``history.observations.rows`` projects it. + + ``last_seen`` is relative to the frozen clock. + """ + from datetime import timedelta + + seen = (FROZEN_NOW - timedelta(days=days_ago)).isoformat() + return { + "id": f"{topic}/{concept}", + "topic": topic, + "concept": concept, + "confidence": confidence, + "first_seen": seen, + "last_seen": seen, + "session_count": 1, + "notes": None, + "last_teachback_score": teachback, + } + + +def _plant_struggles( + monkeypatch: pytest.MonkeyPatch, *rows: dict, due: tuple[_Candidate, ...] = () +) -> None: + """Silence every collector except the struggle collector, which reads ``rows``.""" + from studyloop.history import observations + + real_collector = decision._struggle_candidates + _patch_collectors(monkeypatch, *due) + monkeypatch.setattr(decision, "_struggle_candidates", real_collector) + monkeypatch.setattr(observations, "rows", lambda conn: [dict(row) for row in rows]) + + +def _row3_plan() -> None: + """Rubric row 3's plan: floor 5, milestone 0 done, milestone 1 ``Frames`` open.""" + _plan( + "sql-windows", + title="SQL Windows", + energy_floor=5, + milestones=[ + Milestone(title="Window basics", done=True, concepts=["window function"]), + Milestone(title="Frames", concepts=["window frame"]), + ], + ) + + +def test_live_struggle_repair_defers_at_low_energy_like_new_work(monkeypatch) -> None: + """Rule 3 extended (design §5, amendment 1 + 2): repair carries a demand of its own. + + ``struggling`` seen within 14 days is ``high`` (asks for 6/10); ``struggling`` + older than that, or a row whose only signal is a weak teach-back, is + ``medium`` (4/10). Below the capability the repair is deferred like new + milestone work — listed, not ranked — in its own additive key, plan-related + or not; the milestone deferral beside it is untouched. + """ + _row3_plan() + _plant_struggles( + monkeypatch, + _struggle("window function", days_ago=3), # live, plan-related → high + _struggle("window frame", days_ago=20), # old, plan-related → medium + _struggle("decorators", topic="python", days_ago=1), # live, unrelated → high + _struggle("closures", topic="python", confidence="confident", days_ago=2, teachback=9), + ) + + low = build_now_plan(energy="low") + + deferred = {item.concept: item for item in low.energy_deferred_repairs} # pyright: ignore[reportAttributeAccessIssue] + assert set(deferred) == {"window function", "window frame", "decorators", "closures"} + assert not any(rec.concept in deferred for rec in _all(low)), "deferred repair is not ranked" + + live = deferred["window function"] + assert isinstance(live, decision.DeferredRepair) # pyright: ignore[reportAttributeAccessIssue] + assert (live.plan_id, live.plan_title, live.topic, live.confidence) == ( + "sql-windows", + "SQL Windows", + "sql", + "struggling", + ) + assert (live.energy_demand, live.required_capability, live.energy_capability) == ("high", 6, 3) + assert "3/10" in live.reason and "6/10" in live.reason + assert ( + deferred["window frame"].energy_demand, + deferred["window frame"].required_capability, + ) == ( + "medium", + 4, + ) + assert deferred["closures"].energy_demand == "medium", ( + "a weak teach-back alone is medium demand" + ) + assert (deferred["decorators"].plan_id, deferred["decorators"].plan_title) == (None, None) + assert deferred["decorators"].energy_demand == "high" + # The milestone deferral is what it was (rule 3's original half). + assert [(d.plan_id, d.milestone_index) for d in low.energy_deferred] == [("sql-windows", 1)] + + payload = low.to_json_dict() + assert [entry["concept"] for entry in payload["energy_deferred_repairs"]] == [ + item.concept + for item in low.energy_deferred_repairs # pyright: ignore[reportAttributeAccessIssue] + ] + assert payload["energy_deferred_repairs"][0]["energy_demand"] in {"high", "medium"} + + # Medium energy (6/10) carries every demand class: nothing deferred, key absent. + medium = build_now_plan(energy="medium") + + assert medium.energy_deferred_repairs == () # pyright: ignore[reportAttributeAccessIssue] + assert "energy_deferred_repairs" not in medium.to_json_dict() + assert any(rec.concept == "window function" for rec in _all(medium)) + + +def test_recovered_repair_stays_eligible_at_low_energy(monkeypatch) -> None: + """A ``learning`` row is ``low`` demand — the gentle review "repair is cheaper than + encoding" was always about — and stays eligible below the plan's floor with its + plan-related ref. Due recall is unaffected whatever its confidence says.""" + _row3_plan() + _plant_struggles(monkeypatch, _struggle("window function", confidence="learning", days_ago=2)) + + low = build_now_plan(energy="low") + + assert low.primary.concept == "window function" + assert low.primary.action_type == "teachback" + assert low.primary.plan_refs == (PlanRef("sql-windows", None),) + assert low.primary.metadata["energy_demand"] == "low" + assert low.energy_deferred_repairs == () # pyright: ignore[reportAttributeAccessIssue] + assert not any(rec.source == "body_double" for rec in _all(low)) + assert [d.milestone_index for d in low.energy_deferred] == [1] + + # A due row on a plan concept, even one recorded as struggling, is recall, + # not repair: it is never deferred and nothing is synthesised beside it. + import dataclasses + + due = dataclasses.replace( + _candidate("window frame", topic="sql", score=100), + metadata={"confidence": "struggling", "days_ago": 6}, + ) + _plant_struggles(monkeypatch, _struggle("window function", days_ago=3), due=(due,)) + + low = build_now_plan(energy="low") + + assert low.primary.concept == "window frame" + assert low.primary.plan_refs == (PlanRef("sql-windows", None),) + assert [d.concept for d in low.energy_deferred_repairs] == ["window function"] # pyright: ignore[reportAttributeAccessIssue] + assert not any(rec.source == "body_double" for rec in _all(low)) + + +def test_body_double_candidate_is_synthesised_when_nothing_plan_related_fits(monkeypatch) -> None: + """Rubric row 3b's world: the plan's milestone is deferred and its only repair is a + live struggle, so nothing plan-related fits low energy. The engine proposes sitting + with the plan — a body-double session — naming what it stands in for, through the + session door (amendment 3), never the least-bad task.""" + _row3_plan() + _plant_struggles(monkeypatch, _struggle("window function", days_ago=3)) + + low = build_now_plan(energy="low") + + primary = low.primary + assert primary.source == "body_double" + assert primary.action_type == "conversation" + assert primary.plan_refs == (PlanRef("sql-windows", None),) + assert primary.evidence_command == 'studyloop study "SQL Windows" --mode co-study' + assert "Frames" in primary.reason and "window function" in primary.reason + assert low.starter is False + assert low.alternates == [] + assert [d.concept for d in low.energy_deferred_repairs] == ["window function"] # pyright: ignore[reportAttributeAccessIssue] + assert [d.milestone_index for d in low.energy_deferred] == [1] + assert decision.BODY_DOUBLE_BASE_SCORE < decision.MILESTONE_BASE_SCORE # pyright: ignore[reportAttributeAccessIssue] + + golden_keys = list(json.loads(GOLDEN.read_text(encoding="utf-8"))) + payload = low.to_json_dict() + assert list(payload) == [ + *golden_keys, + "active_plans", + "energy_deferred", + "energy_deferred_repairs", + ] + assert payload["primary"]["source"] == "body_double" + + +def test_body_double_is_a_proposal_not_a_filter(monkeypatch) -> None: + """An unrelated real candidate still wins; the body-double proposal sits beneath it + as an alternate, base score below any real candidate's.""" + _row3_plan() + _plant_struggles( + monkeypatch, + _struggle("window function", days_ago=3), + due=(_candidate("decorators", topic="python", score=100),), + ) + + low = build_now_plan(energy="low") + + assert low.primary.concept == "decorators" + assert low.primary.plan_refs == () + assert [rec.source for rec in low.alternates] == ["body_double"] + assert low.alternates[0].score < low.primary.score + assert "Frames" in low.alternates[0].reason + + +def test_body_double_never_appears_without_an_active_plan(monkeypatch) -> None: + """No active plan, no plan to sit with: the deferral still happens (plan-independent, + ``plan_id`` ``None``), the golden world stays untouched, and a non-active plan is + not an active plan.""" + _plant_struggles(monkeypatch, _struggle("decorators", topic="python", days_ago=1)) + + low = build_now_plan(energy="low") + + assert not any(rec.source == "body_double" for rec in _all(low)) + assert [(d.concept, d.plan_id, d.plan_title) for d in low.energy_deferred_repairs] == [ # pyright: ignore[reportAttributeAccessIssue] + ("decorators", None, None) + ] + # Every real candidate was deferred: the starter stands in, and says why. + assert low.starter is True + assert "defer" in low.primary.reason.lower() + + _plan("draft-plan", status="draft") + + low = build_now_plan(energy="low") + + assert not any(rec.source == "body_double" for rec in _all(low)) + assert "active_plans" not in low.to_json_dict() + + +def test_cli_now_and_recap_render_deferred_repairs_and_the_body_double_door( + monkeypatch, +) -> None: + """Amendment 2's "readable off the top" rule: each renderer gains one line per + deferred repair, and a body-double primary shows its door, not "record evidence".""" + from click.testing import CliRunner + + from studyloop.cli import cli + from studyloop.learning import recap + + _row3_plan() + _plant_struggles(monkeypatch, _struggle("window function", days_ago=3)) + + rich = CliRunner().invoke(cli, ["now", "--energy", "low"]) + as_json = CliRunner().invoke(cli, ["now", "--energy", "low", "--json"]) + + assert rich.exit_code == 0, rich.output + flat = " ".join(rich.output.split()) + assert "Deferred for energy" in flat and "Frames" in flat # the milestone line stays + assert "window function" in flat and "6/10" in flat # …and the repair has its own line + assert "Sit with the plan" in flat + assert "--mode co-study" in flat + assert "Record evidence" not in flat + assert as_json.exit_code == 0, as_json.output + payload = json.loads(as_json.output) + assert payload["primary"]["source"] == "body_double" + assert payload["energy_deferred_repairs"][0]["concept"] == "window function" + assert payload["energy_deferred_repairs"][0]["required_capability"] == 6 + + context = recap._plan_context(build_now_plan(energy="low")) + + assert "Frames" in context + assert "window function" in context and "6 of 10" in context From 326abcf92dbcf88280a30c110436d48e9b262a48 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sat, 19 Sep 2026 12:36:28 +0100 Subject: [PATCH 05/32] =?UTF-8?q?feat(now):=20item=205=20(D-F)=20=E2=80=94?= =?UTF-8?q?=20energy=20demand=20for=20repair,=20and=20the=20body-doubling?= =?UTF-8?q?=20floor?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit GREEN for ef319a7b. Rubric row 3 (owner: "no") said a struggle-repair task carried no energy demand of its own, so low energy recommended hands-on repair of a LIVE struggle. Now: Engine (learning/decision.py) - `_struggle_candidates` derives `energy_demand` once, from its own row classes, and carries it in metadata: `struggling` seen within LIVE_STRUGGLE_DAYS (14) → high (6/10); older `struggling` or a weak teach-back alone → medium (4/10); `learning` → low (0/10). An unreadable `last_seen` on a struggling row is read as live — the cautious side. - `_defer_repairs` (rule 3 extended) runs before rule 6 so a deferred repair no longer "represents" a milestone; repair above its demand is listed in the NEW additive key `energy_deferred_repairs` (`DeferredRepair`, plan_id/plan_title when plan-related else None) and never ranked. Due recall is never deferred whatever its confidence. Plan-independent. - `_body_double_candidate`: when nothing plan-related fits and a matchable active plan exists, one `source="body_double"` conversation candidate, base 30 (< MILESTONE_BASE_SCORE 48; +12 bias = 42 < any real candidate), plan_refs (plan, None) per plan, reason naming every deferred milestone and repair, evidence_command the co-study door `studyloop study "<title>" --mode co-study` set explicitly (T5.1 amendment 3: `_evidence_command` would answer with a progress write). - Starter after a deferral says the energy deferred the repair work, not "no evidence found yet" (false). Golden world defers nothing: unchanged. Renderers — one line per deferred repair, beside the milestone line - cli/_now.py: "Deferred for energy: … repairing “x” (confidence) asks for N/10; low energy carries 3/10"; a body-double primary's command is labelled "Sit with the plan", not "Record evidence". - learning/recap.py: a sentence per deferred repair in `_plan_context`. - today-panel.js: `deferredRepairNotes()`, counted in `hasPlanContext`; `viewForAction(rec)` starts a body-double primary in the Body Double view; index.html renders the lines and shows the block for them alone. Decisions taken at GREEN (design §5, recorded): deferral is plan-independent; rule 8's guaranteed slot below the floor is the body-double proposal (test_preserves_one_plan_backed_action_when_energy_allows updated — it advertises no work the energy cannot carry, primary untouched); honest starter; body-double shape; a deferred repair does not represent a milestone. Spec/docs: MODIFIED "The now engine is plan-aware with tested ranking rules" in the active-learning-decisions delta (rule 2's repair half, the body-double clause after rule 5, rule 7's slot, the new key, every current scenario carried by name plus five new ones); docs/study-plans.md "Plan-aware now" and docs/cli-reference.md say the same. Rubric row 3b written (three readings printed from the real engine; verdict PENDING for the owner). tasks.md T5.2/T5.3 ticked. Verification: test_now_plan_guidance 46/46 (golden byte-identical), test_learning_decision 5/5, JS 144/144 (+5), e2e plan journeys 20/20, docs contract 39/39, mkdocs --strict 0, openspec valid, ruff/pyright clean. Full suite vs clean main control: item5 − control = ∅, control − item5 = ∅ (44 shared sandbox-environmental ids, identical to the committed set); receipt docs/architecture/plan-integration/receipts/full-suite-control-item5-2026-09-19.md. --- .../full-suite-control-item5-2026-09-19.md | 36 +++ .../receipts/now-rubric-2026-09-16.md | 10 +- docs/cli-reference.md | 2 +- docs/study-plans.md | 11 +- .../plan-integration-followons/design.md | 23 ++ .../specs/active-learning-decisions/spec.md | 229 +++++++++++++++++ .../plan-integration-followons/tasks.md | 22 +- packages/studyloop/src/studyloop/cli/_now.py | 15 +- .../src/studyloop/learning/decision.py | 238 +++++++++++++++++- .../studyloop/src/studyloop/learning/recap.py | 7 + .../src/studyloop/web/static/index.html | 7 +- .../web/static/js/components/today-panel.js | 26 +- .../studyloop/tests/test_learning_decision.py | 2 +- .../studyloop/tests/test_now_plan_guidance.py | 32 ++- 14 files changed, 626 insertions(+), 34 deletions(-) create mode 100644 docs/architecture/plan-integration/receipts/full-suite-control-item5-2026-09-19.md diff --git a/docs/architecture/plan-integration/receipts/full-suite-control-item5-2026-09-19.md b/docs/architecture/plan-integration/receipts/full-suite-control-item5-2026-09-19.md new file mode 100644 index 000000000..0f415e32e --- /dev/null +++ b/docs/architecture/plan-integration/receipts/full-suite-control-item5-2026-09-19.md @@ -0,0 +1,36 @@ +# Full-suite matched control — item 5 (D-F) — 2026-09-19 + +Two full `packages/studyloop/tests` runs in parallel, same machine, same +sandbox, `-q -p no:cacheprovider -rfE`: + +| Tree | Worktree | Result | +| --- | --- | --- | +| **item 5** (`feat/energy-demand-body-double`, RED `ef319a7b` + GREEN working tree) | `studyloop-wt/item5` | 30 failed, **5126 passed**, 4 skipped, 804 deselected, 14 errors (10:17) | +| **control** (`main` `4f8e3e0f`, detached) | `studyloop-wt/ctrl-item5` | 30 failed, 5120 passed, 4 skipped, 804 deselected, 14 errors (10:24) | + +Both worktrees were `uv sync --all-packages --group dev` and each proved to +import `studyloop` from its own tree before the run. + +## Sorted failing-id sets + +- item5 ∖ control = **∅** — zero regressions. +- control ∖ item5 = **∅** — nothing item 5 fixed by accident, and the six + new tests account for the passed-count difference (+6). +- item5 ∖ committed environmental set (`full-suite-control-item4-2026-09-18.md`, + 44 ids + item 4's seven then-REDs) = **∅**. The seven ids on the other side of + that comparison are item 4's REDs, green since `82293293`. + +The 44 shared ids are the sandbox-environmental set the item-4 receipt lists +by name (journeys world guards, acceptance isolation, second-brain CLI/doctor, +harness-matrix live mechanics, obsidian vault isolation, fresh-install scope); +unchanged here, byte for byte. + +## Scoped gates on the same tree + +- `test_now_plan_guidance.py` 46/46 (six REDs flipped; golden `now_plan_no_active.json` byte-identical); + `test_learning_decision.py` 5/5 (one stub updated to the starter's new keyword). +- JS `node --test packages/studyloop/tests/js/*.test.js` 144/144 (+5). +- e2e `test_journey_study_plan.py` + `test_plans_api.py` 20/20 (browser). +- `test_docs_plan_integration_contract.py` + `test_ci_workflow_contract.py` 39/39. +- `mkdocs build --strict` exit 0; `openspec validate plan-integration-followons` valid. +- ruff check / ruff format --check / pyright: clean on every touched file. diff --git a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md index ff46f1f47..759ce0e94 100644 --- a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md +++ b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md @@ -1,6 +1,6 @@ # Plan-aware `now` — five-scenario human rubric (D-16) · 2026-09-16 -**Status: owner verdicts RECORDED 2026-09-16 (interactive walkthrough with the coordinator).** Scenarios 1, 2 and the primary of 4: yes. Scenario 3: **no** (finding). Scenario 4 completion action: **no as phrased** (finding). Scenario 5: verified. **Row 4b added and scored 2026-09-18** (item 4 / D-G, tree `feat/plan-close`; outputs emitted from the tree committed as `82293293`): scenario 4 re-run with the end assessment planted, both proposals recorded as emitted; owner verdict **yes** on both readings — (a) *extend* with named evidence is a proposal the owner would walk, (b) a clean review's *close* is one the owner would agree to. This closes row 4's "no as phrased" finding; row 4 is kept as the record of the original finding. This receipt was produced unattended +**Status: owner verdicts RECORDED 2026-09-16 (interactive walkthrough with the coordinator).** Scenarios 1, 2 and the primary of 4: yes. Scenario 3: **no** (finding). Scenario 4 completion action: **no as phrased** (finding). Scenario 5: verified. **Row 3b added 2026-09-19** (item 5 / D-F, tree `feat/energy-demand-body-double`; outputs emitted from the tree the GREEN commit records): scenario 3 re-run with the struggle collector live, three readings printed, verdict `PENDING` for the owner. **Row 4b added and scored 2026-09-18** (item 4 / D-G, tree `feat/plan-close`; outputs emitted from the tree committed as `82293293`): scenario 4 re-run with the end assessment planted, both proposals recorded as emitted; owner verdict **yes** on both readings — (a) *extend* with named evidence is a proposal the owner would walk, (b) a clean review's *close* is one the owner would agree to. This closes row 4's "no as phrased" finding; row 4 is kept as the record of the original finding. This receipt was produced unattended overnight. Every scenario below was *run* on frozen fixtures and the primary and its rationale are recorded exactly as the engine emitted them; the "would I do the primary?" column is a human judgement that only the owner can @@ -30,6 +30,7 @@ learning"). | 1 | Matching due | Active plan `sql-windows` ("SQL Windows", topics `[sql]`, next milestone 0 concepts `[window function]`). Due items: `decorators`/python base 102, `window function`/sql base 100. | **`window function`** (sql, recall, score 130, `plan_refs=[(sql-windows, 0)]`); alternate `decorators` (120, no refs). | Rule 5: both are due items two points apart — one urgency class — so the plan-related one takes the +12 bias and wins; the unrelated due item is *kept* as an alternate (bias, not filter). Rule 7 names the milestone the action advances. | **yes** — owner, 2026-09-16: "window function is the logical step before decorating it" (a prerequisite-order argument; see F2 follow-on). | | 2 | Urgent-unrelated wins | Same plan. One due item: `decorators`/python, base 100 (an overdue spaced-repetition review). Nothing represents milestone 0. | **`decorators`** (118, no refs); alternate `window function` (60, `source=study_plan:sql-windows:0`, `plan_refs=[(sql-windows, 0)]`, reason "Next milestone 1/1 of plan 'SQL Windows': Window basics"). | Rule 5: the unrelated candidate is in a more urgent class (due review) and wins outright — the bias cannot lift new-milestone work over it. Rule 6: the plan's next milestone was unrepresented, so it was synthesised at base 48 + bias 12 = 60 and appears as the plan-backed alternate (rule 8 satisfied without any swap). | **yes** — owner, 2026-09-16: clear the overdue review first. Note for follow-on: an overdue item *unrelated* to the plan must not sit as an alternate indefinitely — propose it explicitly (age-aware nudge) or let the learner retire it. | | 3 | Energy-deferred | Plan `sql-windows` with `energy_floor: 5`; milestone 0 `Window basics` **done** (concepts `[window function]`), milestone 1 `Frames` (concepts `[window frame]`). One struggle-repair item `window function`/sql, `hands-on`, base 82. **Energy `low`** (capability 3/10). | **`window function`** (hands-on, score 80, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, milestone 1, floor 5, capability 3)]`; JSON gains `energy_deferred`. | Rule 3: capability 3 < floor 5, so the *new* milestone (Frames) is deferred and named, not synthesised; the plan-related repair on a finished milestone's concept stays eligible and keeps its ref (`None`: plan-related repair, not the next milestone). Score = 82 + 12 bias − 14 (hands-on at low energy). | **no** — owner, 2026-09-16: a struggle-repair task has no energy demand of its own; recommending hands-on repair of a *live* struggle on a low-energy day risks compounding the struggle and damaging confidence (RSD). Finding for council: (1) derive a per-item energy demand for repair from struggle recency/teach-back — at low energy a live struggle defers like new work, a recovered one stays eligible as gentle review; (2) when nothing plan-related fits the day's capability, synthesise a body-doubling / open-session candidate (feature exists: ADR-0001/0003, `web/routes/body_double.py`) naming the deferred items, instead of the least-bad task. | +| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered). | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason *"Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."*; no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone. | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 < any real candidate — a proposal, not a filter) carrying the co-study session door. | **PENDING** — owner: (a) would you sit with the plan rather than repair the live struggle today? (b) is the body-double proposal right to sit beneath the unrelated due recall? (c) is the gentle teach-back on a recovered concept one you would do at low energy? | | 4 | Fully-checked | Plan `done-plan` ("Done Plan"), milestones A and B both done. One due item `decorators`/python base 100. | **`decorators`** (118, no refs); `completion_actions=[(done-plan, "Every milestone of 'Done Plan' is checked off — close the plan or extend it with a follow-on mission.")]`; no `study_plan:` candidate anywhere; JSON gains `active_plans` + `completion_actions`. | Rule 9: a fully-checked plan is reported as a completion action and is neither matched (no bias, no refs) nor synthesised. | **primary yes / completion action no as phrased** — owner, 2026-09-16: the completion action must be contextual and consensual. Run the end assessment (`assess(phase="end")`: due reviews, struggles, unverified milestones on the plan's concepts). If outstanding work touches the plan's concepts (or their prerequisites — F2 concept edges), propose *extend* and name the evidence; if clean, propose *close* and ask the learner to agree ("anything you are not comfortable with?"). Status never changes automatically (#7). Natural vehicle: architect with `purpose=planning` and the assessment in the brief (`plan close <id>`, sibling of `plan repair <id>`). Finding for council. | | 4b | Fully-checked — **re-run after D-G (item 4, 2026-09-18)** | Row 4's fixture (`done-plan`, milestones A `[alpha]` and B `[beta]` both done; one due item `decorators`/python base 100), plus the end assessment's readers planted: **(a)** one due review on plan concept `alpha` (`overdue`) with session mentions backing both concepts; **(b)** no due rows, same mentions. | Primary unchanged in both: **`decorators`** (118, no refs); no `study_plan:` candidate. **(a)** `completion_actions=[(done-plan, due 1 / struggles 0 / unverified 0, proposal **extend**, evidence `["Due review: alpha — overdue"]`)]`, sentence: "Every milestone of 'Done Plan' is checked off, and the closing review proposes extending the plan — 1 due review, 0 struggles and 0 unverified milestones on its concepts. Walk the evidence with the architect: studyloop plan close done-plan." **(b)** counts 0/0/0, proposal **close**, evidence `[]`, sentence: "Every milestone of 'Done Plan' is checked off and the closing review is clean — it proposes closing the plan. Close it with the architect when you agree: studyloop plan close done-plan." No warnings; JSON gains the five keys only inside the entry. | Rule 9 as before for the ranking. The completion action is now the end assessment read as a preview (`assess(phase="end", record=False)`, one call, no write, no status change): `extend` iff any of the three counts on the plan's own concepts is above zero, else `close`; due counts only rows naming a concept (the scheduler's "new topic" row is excluded — owner decision 2026-09-17). `plan close done-plan` launches the architect with the same review as the brief's first section; status moves only when the learner agrees. | **yes / yes** — owner, 2026-09-18, answering the two questions as posed: (a) **yes**, a proposal the owner would walk; (b) **yes**, a close the owner would agree to. No further line given. Closes row 4's "no as phrased" finding; status still moves only when the learner agrees in the architect conversation. | | 5 | No-plan identical | No plan documents at all; no collector candidates. | **`one tiny recall loop`** (python, recall, 28, `source=starter`) — the starter. | D-5: `serialise(plan) == golden` → **byte-identical** (`True` in the run); the JSON key list is exactly the golden's — no additive key is present. | **verified** — owner walkthrough 2026-09-16: nothing to judge; the byte-identical golden is the acceptance. | @@ -47,7 +48,12 @@ The five rows correspond to `test_fully_checked_active_plan_emits_completion_not_candidate` and `test_no_active_plans_json_byte_identical_to_golden`; the primaries above are what those tests assert, printed from a throwaway driver over the same -fixtures. Row 4b corresponds to +fixtures. Row 3b corresponds to +`test_body_double_candidate_is_synthesised_when_nothing_plan_related_fits` +(reading a), `test_body_double_is_a_proposal_not_a_filter` (reading b) and +`test_recovered_repair_stays_eligible_at_low_energy` (reading c), printed on +2026-09-19 from a throwaway driver over the module's `_plant_struggles` +fixture with the struggle collector running for real. Row 4b corresponds to `test_completion_action_carries_the_end_assessment_and_proposes_extend_when_concepts_are_due` (reading a) and `test_completion_action_proposes_close_when_the_assessment_is_clean` (reading b), printed the same way on 2026-09-18 with the end assessment's diff --git a/docs/cli-reference.md b/docs/cli-reference.md index 90078af67..383e33cd9 100644 --- a/docs/cli-reference.md +++ b/docs/cli-reference.md @@ -259,7 +259,7 @@ studyloop now --speak Default ranking is due review first, then struggling or low teach-back score, then active-course continuity, then modality match. Low energy suppresses hard context switching. -With an active study plan the same engine is plan-aware (see [Study Plans](study-plans.md#plan-aware-now)): plan-related work gets a bounded bias within its urgency class, a ready plan's next milestone is suggested when it is within your energy and no gathered candidate represents it, and a milestone above your current energy is deferred with a reason. The panel names the plan and milestone an action advances; `--json` adds `active_plans`, `energy_deferred`, `completion_actions`, `warnings` and per-action `plan_refs` only when each is non-empty, so the no-plan output is unchanged and a plan that cannot be read shows up as a warning rather than a failure. +With an active study plan the same engine is plan-aware (see [Study Plans](study-plans.md#plan-aware-now)): plan-related work gets a bounded bias within its urgency class, a ready plan's next milestone is suggested when it is within your energy and no gathered candidate represents it, and a milestone above your current energy is deferred with a reason. Repair of a live struggle is deferred the same way at low energy (an older struggle or weak teach-back needs medium; a concept still being learned is never deferred, nor is a due review), and when nothing plan-related fits the day the panel proposes sitting with the plan — its command is the co-study door, `studyloop study "<plan>" --mode co-study`, labelled "Sit with the plan" rather than "Record evidence". The panel names the plan and milestone an action advances and prints one "Deferred for energy" line per deferred milestone and per deferred repair; `--json` adds `active_plans`, `energy_deferred`, `energy_deferred_repairs`, `completion_actions`, `warnings` and per-action `plan_refs` only when each is non-empty, so the no-plan output is unchanged and a plan that cannot be read shows up as a warning rather than a failure. `studyloop chat-note` turns one markdown/text note into a compact Socratic context pack. V1 prints or speaks the mentor prompt; it does not run a separate chat backend. diff --git a/docs/study-plans.md b/docs/study-plans.md index 2c072c761..755a87130 100644 --- a/docs/study-plans.md +++ b/docs/study-plans.md @@ -219,7 +219,16 @@ action that advances a plan's next milestone is named with the plan and the milestone it serves; a **ready** plan whose next milestone is within your current energy gets that milestone suggested even when no other evidence points at it; and a plan whose energy floor is above your current energy has -that milestone deferred with a reason rather than dropped. An active plan that +that milestone deferred with a reason rather than dropped. Repair has an +energy demand of its own: on a low-energy day a **live** struggle (recorded +as struggling within the last two weeks) is deferred like new work — listed, +not recommended — an older struggle or a weak teach-back needs medium energy, +and a concept you are still learning is the gentle review that stays +available at any energy; due reviews are never deferred. When nothing +plan-related fits the day's energy, the recommendation is to **sit with the +plan** — a body-double session, no new material, no repair — with the +deferred items named; a real, unrelated action still outranks that proposal +when one exists. An active plan that is **not ready** — a hand edit removed its mission or its milestones — is listed with a warning naming what to repair; it still biases related work, but no milestone is suggested for it until it is paused or repaired. A plan diff --git a/openspec/changes/plan-integration-followons/design.md b/openspec/changes/plan-integration-followons/design.md index c847d6967..cf322f2ad 100644 --- a/openspec/changes/plan-integration-followons/design.md +++ b/openspec/changes/plan-integration-followons/design.md @@ -268,6 +268,29 @@ Rule 3's *deferral* of repair is the change; rule 3's *eligibility* of plan-rela no-plan golden stays byte-identical because a body-double candidate requires an active plan and the golden world has none; `INTERLEAVE_RATIOS["low"]` unchanged. +**Decisions taken at GREEN (2026-09-19), each a test in `test_now_plan_guidance.py`:** + +1. **The deferral is plan-independent.** A live struggle is a live struggle whether or not a plan names it + (amendment 2's `plan_id … else None` already said so); the finding was about the learner's day, not the + plan. The body double, by contrast, *requires* a matchable active plan — it is "sit with the plan". +2. **Rule 8's guaranteed slot below the floor is the body-double proposal.** It carries `plan_refs`, so where + four unrelated due items outrank everything at low energy the second alternate is now the proposal, not a + third unrelated item. It advertises no work the energy cannot carry — the property rule 8's docstring + protects — and the primary is untouched. `test_preserves_one_plan_backed_action_when_energy_allows` says so. +3. **The starter tells the truth after a deferral.** With no plan and every real candidate deferred, the starter + stands in; its reason now says the energy deferred the repair work rather than "no learning evidence found + yet", which would be false. The golden world defers nothing, so its sentence is unchanged. +4. **Body-double shape.** Base `BODY_DOUBLE_BASE_SCORE = 30` (+12 bias = 42 < practice 48, milestone 48); + concept `Sit with <title>` (one plan) / `Sit with your plans`; `plan_refs` for every matchable plan; reason + naming each deferred milestone and repair; command `studyloop study "<first title>" --mode co-study`. The + Today card starts it in the Body Double view (`viewForAction`); the CLI labels the command "Sit with the plan". +5. **A deferred repair does not "represent" a milestone** (rule 6 runs after the deferral), so an eligible + milestone whose only collected representative was a deferred live struggle is synthesised as a conversation — + the learner can still talk about it. + +Known edge, not solved here: a `struggling` row whose `last_seen` cannot be parsed is read as live (`high`) — +the cautious side; `_days_since` returns `None` and the demand falls to `high`. + ## 6. Verification `scripts/verify/plan_integration.py` gains registered checks for: the two architect grants (the ten names in diff --git a/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md b/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md index 2705937c8..0151dae71 100644 --- a/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md +++ b/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md @@ -96,3 +96,232 @@ print each evidence line beneath it, and none SHALL re-rank. `plan close <id>` still launches the architect, its brief's fourth line is `Proposal: unassessed — the review is partial`, the gap is among the first section's lines, and its status line does not say the review proposes + +## MODIFIED Requirements + +### Requirement: The now engine is plan-aware with tested ranking rules +`studyloop.learning.decision.build_now_plan` SHALL remain the only ranker of +study actions and SHALL consume active plans through exactly one call to +`PlanApplication().get_active_guidance(today=…)`, where `today` is the date +of the same instant `generated_at` records. It SHALL apply these rules, in +this order (the plan-application-seam design, §3; decision D-5 of its +council plan; item 5 / D-F of the follow-ons for rule 2's repair half and +the body-doubling floor): + +1. Candidates are collected as before; a failure to read plans at all SHALL + degrade to a `warnings` entry, never a failed recommendation, and SHALL be + logged with its traceback on `studyloop.learning.decision` so a + programming error cannot hide behind the learner-facing warning. +2. The energy capability is `low|medium|high → 3|6|10`. For an active plan + whose `energy_floor` exceeds it, the next milestone SHALL be listed in + `energy_deferred` and SHALL NOT become a candidate. Plan-related due recall + stays eligible and plan-related whatever its recorded confidence. A + struggle **repair** carries an energy demand of its own, derived once in + the struggle collector from its own row classes and carried in the + candidate's `metadata["energy_demand"]`: `struggling` seen within 14 days → + `high` (asks for 6/10); `struggling` older than that, or a row whose only + signal is a weak teach-back → `medium` (4/10); `learning` → `low` (0/10). + A repair whose demand exceeds the capability SHALL be deferred exactly + like new milestone work — listed in `energy_deferred_repairs` as a + `DeferredRepair` (`plan_id`/`plan_title` when plan-related, else `None`, + `concept`, `topic`, `confidence`, `energy_demand`, `required_capability`, + `energy_capability`, `reason` naming the struggle) and never ranked; the + deferral does not depend on a plan existing. A `low`-demand repair is + always carried. `energy_deferred` stays milestone-shaped; a repair is never + folded into it. +3. A candidate is plan-related when `normalise_match_key` of its concept, + topic or course **equals** one of the plan's `match_keys`; no substring + test. It names the plan's next milestone (`milestone_index`) only when the + key equals one of that milestone's concepts **and** the plan is eligible — + ready and within the energy capability; a topic or finished-milestone + match, or any match on an energy-deferred or active-but-unready plan, + carries `milestone_index = None` (plan-related repair), so a payload never + names a milestone it also reports as deferred or that the seam would + refuse to tick. +4. Scoring is today's scoring plus one bounded bias for plan-related + candidates: within one urgency class plan-related beats unrelated, and a + globally more-urgent unrelated candidate still wins — a bias, not a filter. +5. When no collected candidate represents an eligible (ready, energy-permitted) + plan's next milestone, one `conversation` candidate SHALL be synthesised + for it (source `study_plan:<plan_id>:<index>`, concept = the milestone's + first concept or its title, topic = the plan's first topic), scored below + every due and repair class. A learner with an active plan and no evidence + is therefore sent to the plan, and `starter` is `false`. A deferred repair + (rule 2) does not "represent" a milestone. When, after rules 2 and 5, **no + candidate is plan-related** and at least one matchable active plan exists, + one **body-double** candidate SHALL be synthesised instead of leaving the + plan to the least-bad task: `source = "body_double"`, `action_type = + "conversation"`, base score below the synthesised-milestone base so every + real candidate outranks it (a proposal, never a filter), `plan_refs` + `(plan_id, None)` for every matchable plan, a reason naming the deferred + milestones and repairs it stands in for, and `evidence_command` the + co-study session door — `studyloop study "<plan title>" --mode co-study` — + set explicitly, never a progress write. No active plan (a draft is not + one) → no body-double candidate; when every real candidate was deferred + and no plan exists, the starter stands in and its reason says the energy + deferred the repair work, not that no evidence exists. +6. After de-duplication every matching `PlanRef(plan_id, milestone_index)` + SHALL be attached to each ranked action, ordered by target urgency + (`overdue`, `soon`, `later`, `undated`) → most recent `updated` → `plan_id`, + keeping the most specific milestone per plan. +7. When primary + alternates hold no plan-backed action and an eligible one + whose estimate fits the requested time exists further down, it SHALL + replace the last alternate only; the primary is never re-ranked by plans. + Below a plan's floor that plan-backed action is the body-double proposal + (it advertises no work the energy cannot carry), never the deferred + milestone. +8. A fully-checked active plan SHALL appear in `completion_actions` and SHALL + be neither matched nor synthesised. An active-but-unready plan SHALL be + listed and matched (bias and a `milestone_index = None` reference) but + never synthesised and never named as a milestone, with a warning naming + its blockers. + +`NowPlan` gains `active_plans` (ordered as rule 6), `energy_deferred`, +`energy_deferred_repairs`, `completion_actions` and `warnings`; +`LearningRecommendation` gains `plan_refs: tuple[PlanRef, ...] = ()`. +`to_json_dict()` SHALL omit each of these when empty, so a learner with no +active plan receives the pre-#10 payload **byte for byte** — pinned by +`tests/golden/now_plan_no_active.json`, captured before any of this shipped. +Renderers (`studyloop now`, `GET /api/now`, the Today card, the daily recap in +its JSON, spoken and Rich-panel forms) SHALL show plan relevance, energy +deferral — one line per deferred milestone **and** one per deferred repair — +and the engine's warnings from these fields, SHALL label a body-double +primary's command as the session door it is ("Sit with the plan", and the +Today card starts it in the Body Double view) rather than as evidence to +record, SHALL escape learner-authored text before any markup (Rich or HTML), +and SHALL NOT re-rank. Ranking tests prove ranking compliance, not learner +benefit (D-16); a five-scenario human rubric receipt accompanies the change, +with row 3b re-run after this requirement's repair half. + +#### Scenario: No active plan is byte-identical to the golden +- **WHEN** no active plan exists (an empty plans directory, or only a draft) + and `build_now_plan()` runs with a frozen clock in an empty world +- **THEN** the serialised `to_json_dict()` equals + `tests/golden/now_plan_no_active.json` byte for byte, and no + `active_plans`, `energy_deferred`, `energy_deferred_repairs`, + `completion_actions`, `warnings` or `plan_refs` key is present + +#### Scenario: Matching due concept outranks unrelated of the same urgency +- **WHEN** an active plan's milestone names `window function` and two due + items are two points apart, `decorators` (unrelated) ahead +- **THEN** `window function` is primary with `plan_refs == (PlanRef(plan, 0),)` + and `decorators` is the first alternate with no refs + +#### Scenario: A more-urgent unrelated item still wins +- **WHEN** the only collected candidate is an unrelated due item and the + plan's next milestone is unrepresented +- **THEN** the due item is primary and the synthesised milestone + (`study_plan:<id>:0`) is an alternate with a lower score + +#### Scenario: Energy below the floor defers the milestone, keeps repair +- **WHEN** energy is `low` (3/10), the plan's `energy_floor` is 5, its next + milestone is `Frames` and a `learning` row on a finished milestone's + concept is collected by the struggle collector (gentle repair) +- **THEN** that repair is primary (`teachback`, `energy_demand == "low"`) with + `PlanRef(plan, None)`, `energy_deferred` names `(plan, 1, 5, 3)`, + `energy_deferred_repairs` is empty, and no `study_plan:` or `body_double` + candidate exists; at `medium` energy nothing is deferred and the milestone + is synthesised + +#### Scenario: A live struggle's repair defers at low energy like new work +- **WHEN** energy is `low` and the struggle collector holds a `struggling` row + seen 3 days ago on a plan concept, a `struggling` row seen 20 days ago on + another, a `struggling` row seen 1 day ago unrelated to any plan, and a + `confident` row kept only for a teach-back score of 9 +- **THEN** none of the four is ranked; `energy_deferred_repairs` names all + four — the live plan-related one `("high", 6, 3)` with the plan's id and + title, the 20-day one `medium` (4), the weak-teach-back one `medium`, the + unrelated one with `plan_id` and `plan_title` `None` — `energy_deferred` + still names the milestone alone, and at `medium` energy the key is absent + and the live repair is ranked again + +#### Scenario: Due recall is never deferred +- **WHEN** energy is `low`, a due row on a plan concept is collected with + `confidence == "struggling"` and a live struggle repair is also collected +- **THEN** the due row is primary with `PlanRef(plan, None)`, the repair is + in `energy_deferred_repairs`, and no `body_double` candidate exists + +#### Scenario: Nothing plan-related fits, so the engine proposes sitting with the plan +- **WHEN** energy is `low`, the plan's `energy_floor` is 5 (milestone + deferred) and its only repair is a live struggle (deferred) +- **THEN** the primary is `source == "body_double"`, `action_type == + "conversation"`, `plan_refs == (PlanRef(plan, None),)`, `evidence_command == + 'studyloop study "<plan title>" --mode co-study'`, its reason names the + deferred milestone and the deferred repair, `starter` is `false`, and the + JSON keys are the golden's then `active_plans`, `energy_deferred`, + `energy_deferred_repairs` + +#### Scenario: The body-double candidate is a proposal, not a filter +- **WHEN** the same world also collects an unrelated due item +- **THEN** the due item is primary with no refs and the body-double + candidate is the only alternate, with a lower score + +#### Scenario: No active plan, no body double +- **WHEN** no plan document exists (or only a draft) and a live unrelated + struggle is collected at `low` energy +- **THEN** no `body_double` candidate exists, `energy_deferred_repairs` names + the struggle with `plan_id None`, `starter` is `true` and the starter's + reason says the energy deferred the repair work + +#### Scenario: A deferred milestone is never named by a reference +- **WHEN** energy is `low`, the plan's `energy_floor` is 5 and the only + collected candidate's concept equals the next milestone's concept +- **THEN** the candidate is primary with `PlanRef(plan, None)` while + `energy_deferred` names that milestone; at `medium` energy the same + candidate carries `PlanRef(plan, 0)` and nothing is deferred + +#### Scenario: An unready active plan is matched but never named +- **WHEN** an active plan has no mission and no success criteria (unready) + and a collected candidate equals its next milestone's concept +- **THEN** the candidate is primary with `PlanRef(plan, None)`, no + `study_plan:` candidate exists, the plan's `active_plans` entry has + `ready == False` and `eligible == False`, and one warning names the plan, + its blockers and "pause or repair" + +#### Scenario: No substring matching +- **WHEN** a milestone titled `Window functions deep dive` has no concepts + and candidates `window functions deep dive tutorial`, `window` and + `joins`/`SQL` are collected +- **THEN** only `joins` is plan-related (`PlanRef(plan, None)` via the topic + `sql`, casefolded); the other two carry no refs + +#### Scenario: Every matching plan is referenced, in order +- **WHEN** six active plans (overdue, soon, later, three undated with + distinct and tied `updated`) all name the primary's concept +- **THEN** `plan_refs` lists all six ordered overdue → soon → later → undated + by latest `updated` then `plan_id`, and `active_plans` is in the same order + +#### Scenario: A plan-backed action is preserved when energy allows +- **WHEN** four unrelated due items outrank everything and the plan's + `energy_floor` is 5 +- **THEN** at `medium` energy the synthesised milestone replaces the second + alternate (the primary and first alternate are unchanged); at `low` energy + the primary and first alternate are the two best unrelated items, the + second alternate is the body-double proposal with `PlanRef(plan, None)`, + no `study_plan:` candidate exists and `energy_deferred` names the milestone + +#### Scenario: Fully-checked plan emits a completion action +- **WHEN** an active plan's every milestone is done and an unrelated due item + is collected +- **THEN** `completion_actions` names the plan, the due item is primary with + no refs, no `study_plan:` candidate exists, and the plan's `active_plans` + entry has `next_milestone_index == None` + +#### Scenario: Renderers show, never re-rank +- **WHEN** `studyloop now --energy low`, `GET /api/now?energy=low` and the + daily recap run against the energy-deferral fixture, and against the + live-struggle fixture +- **THEN** each names the primary the engine chose, the plan it advances, the + deferred milestone and — for the live-struggle fixture — one line per + deferred repair with its demand and the day's capability; a body-double + primary is labelled "Sit with the plan" with its `--mode co-study` door + and never "Record evidence"; with no plan the CLI panel prints no plan + lines, `GET /api/now` equals the golden, and the recap's `plan_context` is + absent from its JSON, its spoken text and the `recap today` panel + +#### Scenario: Learner-authored text is data to every renderer +- **WHEN** an active plan's title, topic or milestone text contains Rich + markup, HTML or shell punctuation (`Plan [/bold]`, `<script>…`, `"; rm -rf ~`) +- **THEN** `build_now_plan` ranks and serialises it unchanged and writes + nothing to the document; `studyloop now` exits 0 and shows the text + literally; the Today card renders it through `x-text` diff --git a/openspec/changes/plan-integration-followons/tasks.md b/openspec/changes/plan-integration-followons/tasks.md index 8f3b6d44b..da05fd770 100644 --- a/openspec/changes/plan-integration-followons/tasks.md +++ b/openspec/changes/plan-integration-followons/tasks.md @@ -199,14 +199,28 @@ writer, through the existing gate, closes that. Kept out of item 3 so item 3's f and its three renderers are milestone-shaped; the body-double door is `studyloop study … --mode co-study` / the Body Double view's session start, not the read-only `body_double.py` focus route, so the candidate sets its `evidence_command` explicitly. T5.2's RED names hold; a sixth test pins the new key's rendering.) -- [ ] **T5.2** RED `tests/test_now_plan_guidance.py`: `test_live_struggle_repair_defers_at_low_energy_like_new_work`, +- [x] **T5.2** RED `tests/test_now_plan_guidance.py`: `test_live_struggle_repair_defers_at_low_energy_like_new_work`, `test_recovered_repair_stays_eligible_at_low_energy`, `test_body_double_candidate_is_synthesised_when_nothing_plan_related_fits`, `test_body_double_is_a_proposal_not_a_filter`, `test_body_double_never_appears_without_an_active_plan`, - golden byte-identity still green. -- [ ] **T5.3** GREEN: `energy_demand`, rule 3 extension, `body_double` synthesis, CLI/Today "sit with the plan" line. + golden byte-identity still green. (2026-09-19, `ef319a7b` on `feat/energy-demand-body-double`: the five plus + the sixth, `test_cli_now_and_recap_render_deferred_repairs_and_the_body_double_door`, and five JS pins in + `tests/js/today-panel-plan.test.js` (`deferredRepairNotes`, `viewForAction`, markup). The struggle collector + runs for real over patched `observations.rows`, since demand is derived in the collector. 6 red / 40 green, + JS 5 red / 139 green, each for its missing name.) +- [x] **T5.3** GREEN: `energy_demand`, rule 3 extension, `body_double` synthesis, CLI/Today "sit with the plan" line. + (2026-09-19: `DeferredRepair` + `energy_deferred_repairs`, `ENERGY_DEMAND_CAPABILITY`/`LIVE_STRUGGLE_DAYS`, + `_energy_demand` in the collector, `_defer_repairs` before rule 6, `_body_double_candidate` after it, + honest starter after a deferral; CLI `now` + recap + Today card lines; MODIFIED requirement in the + `active-learning-decisions` delta; docs `study-plans.md` "Plan-aware now" + `cli-reference.md`; design §5 + "Decisions taken at GREEN" 1-5. Module 46/46, JS 144/144, e2e plan journeys 20/20, docs contract 39/39, + `mkdocs --strict` exit 0, `openspec validate` valid; full suite vs a clean `main` control — see the GREEN + commit's receipt.) - [ ] ⚖ **T5.4** Council review 7 (`review7`, same seats or `kimi-k2-thinking` third); arbitration; corrections. - [ ] **T5.5** Rubric row **3b** for the owner (`PENDING`); the change is archived only after the owner scores 3b - (4b was scored **yes / yes** on 2026-09-18, so 3b is the one row still outstanding). + (4b was scored **yes / yes** on 2026-09-18, so 3b is the one row still outstanding). (2026-09-19: row 3b + written into `receipts/now-rubric-2026-09-16.md` with three readings printed from the real engine — (a) live + struggle → body-double primary, (b) plus an unrelated due recall → due recall primary, proposal beneath, + (c) recovered `learning` → gentle teach-back primary — verdict `PENDING`.) ## Item 6 — proposals (no code) diff --git a/packages/studyloop/src/studyloop/cli/_now.py b/packages/studyloop/src/studyloop/cli/_now.py index 1d9179060..48cf89e6b 100644 --- a/packages/studyloop/src/studyloop/cli/_now.py +++ b/packages/studyloop/src/studyloop/cli/_now.py @@ -46,6 +46,9 @@ def _render_plan(plan) -> None: primary = plan.primary plans = _active_plans(plan) plan_line = _plan_line(primary, plans) + # A body-double primary (design §5) carries the co-study session door, not a + # progress write: label it as the door it is. + door = "Sit with the plan" if primary.source == "body_double" else "Record evidence" body = ( f"[bold]{escape(primary.concept)}[/bold]\n" f"Topic: [cyan]{escape(primary.topic)}[/cyan]\n" @@ -54,7 +57,7 @@ def _render_plan(plan) -> None: f"Why: {escape(primary.reason)}\n" f"Source: [dim]{escape(primary.source)}[/dim]\n" + (f"Plan: [magenta]{escape(plan_line)}[/magenta]\n" if plan_line else "") - + f"\n[bold]Record evidence:[/bold]\n{escape(primary.evidence_command)}" + + f"\n[bold]{door}:[/bold]\n{escape(primary.evidence_command)}" ) console.print(Panel(body, title="Study Now", border_style="cyan")) @@ -69,6 +72,16 @@ def _render_plan(plan) -> None: f"energy {deferred.energy_floor}/10; {plan.energy} energy carries " f"{deferred.energy_capability}/10. Plan-related review and repair stay available." ) + # One line per deferred repair (design §5, amendment 2): its own key, its + # own sentence — a repair has no milestone number to print. + for repair in getattr(plan, "energy_deferred_repairs", ()): + where = f"{escape(repair.plan_title)} — " if repair.plan_title else "" + console.print( + f"[yellow]Deferred for energy:[/yellow] {where}repairing " + f"“{escape(repair.concept)}” ({escape(repair.confidence)}) asks for " + f"{repair.required_capability}/10; {plan.energy} energy carries " + f"{repair.energy_capability}/10. Due recall and gentle review stay available." + ) for completion in getattr(plan, "completion_actions", ()): # "Closing review", not "Plan complete": the status is still active until # the learner agrees with the architect (council review 6, F6). diff --git a/packages/studyloop/src/studyloop/learning/decision.py b/packages/studyloop/src/studyloop/learning/decision.py index 37132cbd5..0cf570367 100644 --- a/packages/studyloop/src/studyloop/learning/decision.py +++ b/packages/studyloop/src/studyloop/learning/decision.py @@ -53,10 +53,31 @@ #: Design §3 rule 3 — what each self-reported energy level can carry, on the #: 1-10 scale a plan's ``energy_floor`` uses. Below a plan's floor, *new* -#: milestone work is deferred; plan-related due recall and struggle repair -#: stay eligible, because repair is cheaper than encoding. +#: milestone work is deferred; plan-related due recall stays eligible, and so +#: does struggle repair whose own demand (below) the energy can carry. ENERGY_CAPABILITY: dict[EnergyLevel, int] = {"low": 3, "medium": 6, "high": 10} +EnergyDemand = Literal["low", "medium", "high"] + +#: Design §5 (item 5, D-F) — the capability a struggle repair asks for, by the +#: demand class the struggle collector derives from its own row classes: a +#: ``struggling`` row seen within ``LIVE_STRUGGLE_DAYS`` is ``high`` (a live +#: struggle; hands-on repair on a low-energy day risks compounding it — rubric +#: row 3, the owner's one "no"); ``struggling`` older than that, or a row whose +#: only signal is a weak teach-back, is ``medium``; ``learning`` is ``low`` — +#: the gentle review "repair is cheaper than encoding" was always about. +#: Compared with ``ENERGY_CAPABILITY``: ``low`` (3) carries only low demand, +#: ``medium`` (6) carries every class. +ENERGY_DEMAND_CAPABILITY: dict[EnergyDemand, int] = {"high": 6, "medium": 4, "low": 0} +LIVE_STRUGGLE_DAYS = 14 + +#: Base score of the synthesised body-double candidate (design §5): below +#: ``MILESTONE_BASE_SCORE`` so every real candidate — due, repair, practice, +#: continuity, a synthesised milestone — outranks it. A proposal, never a +#: filter; the plan bias then lifts it over nothing but the starter. +BODY_DOUBLE_BASE_SCORE = 30 +BODY_DOUBLE_SOURCE = "body_double" + #: Rule 5 — the bias a plan-related candidate receives. Large enough to decide #: a near-tie inside one urgency class (two due items a few days apart), small #: enough that a clearly more-urgent unrelated candidate (a struggling repair, @@ -129,6 +150,32 @@ def to_json_dict(self) -> dict: return asdict(self) +@dataclass(frozen=True) +class DeferredRepair: + """A struggle repair the current energy cannot carry (rule 3 extended, design §5). + + Its own type, not a :class:`DeferredMilestone`: that one has a mandatory + ``milestone_index`` and its three renderers print ``milestone N`` — a repair + folded into it would read "milestone None" (T5.1 amendment 2). ``plan_id`` + and ``plan_title`` are set when the struggle's concept, topic or course + matches an active plan, else ``None``: the deferral does not depend on a + plan — a live struggle is a live struggle whether or not a plan names it. + """ + + plan_id: str | None + plan_title: str | None + concept: str + topic: str + confidence: str + energy_demand: EnergyDemand + required_capability: int + energy_capability: int + reason: str + + def to_json_dict(self) -> dict: + return asdict(self) + + @dataclass(frozen=True) class CompletionAction: """What to do about an active plan whose every milestone is checked (rule 9). @@ -203,6 +250,7 @@ class NowPlan: starter: bool = False active_plans: tuple[ActivePlanSummary, ...] = () energy_deferred: tuple[DeferredMilestone, ...] = () + energy_deferred_repairs: tuple[DeferredRepair, ...] = () completion_actions: tuple[CompletionAction, ...] = () warnings: tuple[str, ...] = () @@ -224,6 +272,10 @@ def to_json_dict(self) -> dict: data["active_plans"] = [item.to_json_dict() for item in self.active_plans] if self.energy_deferred: data["energy_deferred"] = [item.to_json_dict() for item in self.energy_deferred] + if self.energy_deferred_repairs: + data["energy_deferred_repairs"] = [ + item.to_json_dict() for item in self.energy_deferred_repairs + ] if self.completion_actions: data["completion_actions"] = [item.to_json_dict() for item in self.completion_actions] if self.warnings: @@ -385,6 +437,7 @@ def _struggle_candidates(time_minutes: int) -> list[_Candidate]: conn.close() candidates: list[_Candidate] = [] + today = datetime.now(UTC).date() for row in rows: row_keys = set(row.keys()) concept = str(row["concept"]) @@ -420,12 +473,45 @@ def _struggle_candidates(time_minutes: int) -> list[_Candidate]: "confidence": confidence, "last_teachback_score": teachback_score, "session_count": row["session_count"], + # Design §5: derived once, here, from the collector's own + # classes; the deferral and every renderer read this value. + "energy_demand": _energy_demand( + confidence, row.get("last_seen") if "last_seen" in row_keys else None, today + ), }, ) ) return candidates +def _energy_demand(confidence: str | None, last_seen: object, today: date) -> EnergyDemand: + """The capability class a repair asks for (design §5, T5.1 amendment 1). + + ``struggling`` seen within :data:`LIVE_STRUGGLE_DAYS` is a live struggle — + ``high``; a ``struggling`` row older than that, or one the collector kept + only for its weak teach-back, is ``medium``; ``learning`` is ``low``. An + unreadable ``last_seen`` on a ``struggling`` row is read as live: the + cautious side is the one the finding asks for. + """ + if confidence == "learning": + return "low" + if confidence != "struggling": + return "medium" + seen = _days_since(last_seen, today) + if seen is None or seen <= LIVE_STRUGGLE_DAYS: + return "high" + return "medium" + + +def _days_since(stamp: object, today: date) -> int | None: + if not isinstance(stamp, str) or not stamp: + return None + try: + return (today - datetime.fromisoformat(stamp).date()).days + except ValueError: + return None + + def _due_card_candidates(time_minutes: int) -> list[_Candidate]: try: from studyloop.services.review import list_course_summaries @@ -566,7 +652,7 @@ def _transfer_candidates(time_minutes: int) -> list[_Candidate]: return candidates -def _starter_candidate(time_minutes: int) -> _Candidate: +def _starter_candidate(time_minutes: int, *, after_deferral: bool = False) -> _Candidate: try: from studyloop.topics import get_topics @@ -579,11 +665,20 @@ def _starter_candidate(time_minutes: int) -> _Candidate: else: topic = "python" display = "Python" + # "No learning evidence" would be false when evidence exists and today's + # energy deferred all of it (design §5); say what happened instead. The + # golden world has nothing to defer, so its sentence is unchanged. + reason = ( + "Today's energy deferred the repair work it cannot carry; " + "start with one small retrieval signal instead" + if after_deferral + else "No learning evidence found yet; start by creating one small retrieval signal" + ) return _Candidate( concept="one tiny recall loop", topic=topic, course=topic, - reason="No learning evidence found yet; start by creating one small retrieval signal", + reason=reason, action_type="recall", estimated_minutes=_estimate_minutes("recall", time_minutes, 10), source="starter", @@ -967,6 +1062,18 @@ def milestone_candidates( synthesised.append(_milestone_candidate(plan, milestone, time_minutes)) return synthesised + def first_match(self, candidate: _Candidate) -> ActivePlanGuidance | None: + """The first matchable plan (plan order) this candidate's keys equal, if any.""" + keys = _candidate_keys(candidate) + for plan in self.matchable: + if keys & frozenset(plan.match_keys): + return plan + return None + + def is_plan_related(self, candidate: _Candidate) -> bool: + """Rule 5's test, before scoring: a ref already attached, or a key match.""" + return bool(candidate.plan_refs) or self.first_match(candidate) is not None + def attach_refs(self, candidate: _Candidate) -> _Candidate: """Rule 7: every matching plan, most specific milestone per plan, in plan order. @@ -1048,6 +1155,105 @@ def _milestone_candidate( ) +def _defer_repairs( + candidates: list[_Candidate], *, energy: EnergyLevel, plans: _PlanContext +) -> tuple[list[_Candidate], tuple[DeferredRepair, ...]]: + """Rule 3 extended (design §5): repair above its own energy demand is deferred like new work. + + Only a candidate carrying ``energy_demand`` — the struggle collector's — is + judged. Due recall is never deferred whatever its confidence says, and a + ``learning`` repair (``low`` demand) is always carried. Plan-independent: + the entry names the plan when one matches, else ``None``. + """ + capability = ENERGY_CAPABILITY[energy] + kept: list[_Candidate] = [] + deferred: list[DeferredRepair] = [] + for candidate in candidates: + demand = candidate.metadata.get("energy_demand") + if demand not in ENERGY_DEMAND_CAPABILITY: + kept.append(candidate) + continue + required = ENERGY_DEMAND_CAPABILITY[demand] + if capability >= required: + kept.append(candidate) + continue + plan = plans.first_match(candidate) + confidence = str(candidate.metadata.get("confidence") or "struggling") + if demand == "high": + what = "a live struggle" + elif confidence == "struggling": + what = "an older struggle" + else: + what = "a weak teach-back" + deferred.append( + DeferredRepair( + plan_id=plan.plan.plan_id if plan is not None else None, + plan_title=plan.plan.title if plan is not None else None, + concept=candidate.concept, + topic=candidate.topic, + confidence=confidence, + energy_demand=demand, + required_capability=required, + energy_capability=capability, + reason=( + f"{energy} energy carries {capability}/10; repairing " + f"{candidate.concept!r} ({what}) asks for at least {required}/10 — " + "deferred like new work; due recall and gentle review stay available" + ), + ) + ) + return kept, tuple(deferred) + + +def _body_double_candidate( + plans: _PlanContext, + candidates: list[_Candidate], + deferred_repairs: tuple[DeferredRepair, ...], + *, + energy: EnergyLevel, + time_minutes: int, +) -> _Candidate | None: + """Design §5's floor: nothing plan-related fits and an active plan exists → sit with it. + + One ``source="body_double"`` conversation candidate, base below every real + candidate's (a proposal, not a filter), ``plan_refs`` ``(plan, None)`` for + every matchable plan, reason naming what it stands in for, and the co-study + session door as its command (T5.1 amendment 3): ``_evidence_command`` has + no branch for it and would answer with a progress *write*, not a door. + """ + if not plans.matchable or any(plans.is_plan_related(c) for c in candidates): + return None + named = [plan.plan for plan in plans.matchable] + first = named[0] + titles = " and ".join(plan.title for plan in named) + items = [ + f"milestone {d.milestone_index + 1} “{d.title}” of {d.plan_title}" for d in plans.deferred + ] + [f"repair of “{d.concept}”" for d in deferred_repairs] + deferred_note = f" — deferred: {'; '.join(items)}" if items else "" + topic = first.topics[0] if first.topics else "study" + safe_title = first.title.replace('"', '\\"') + return _Candidate( + concept=f"Sit with {first.title}" if len(named) == 1 else "Sit with your plans", + topic=topic, + course=None, + reason=( + f"Nothing plan-related fits {energy} energy today{deferred_note}. " + f"Sit with {titles} instead: a body-double session, no new material, no repair." + ), + action_type="conversation", + estimated_minutes=_estimate_minutes("conversation", time_minutes, 25), + source=BODY_DOUBLE_SOURCE, + evidence_command=f'studyloop study "{safe_title}" --mode co-study', + score=BODY_DOUBLE_BASE_SCORE, + metadata={ + "plan_id": first.plan_id, + "deferred_milestones": len(plans.deferred), + "deferred_repairs": len(deferred_repairs), + }, + plan_refs=tuple(PlanRef(plan.plan_id, None) for plan in named), + ) + + def _guarantee_plan_backed(ranked: list[_Candidate], time_minutes: int) -> list[_Candidate]: """Rule 8: ≥ 1 plan-backed action among primary + alternates when time permits. @@ -1084,11 +1290,14 @@ def build_now_plan( Order of operations is design §3's: guidance is read once (1), candidates are collected as before (2), the energy capability decides which next - milestones are eligible (3), matching is key equality (4), scoring is - today's plus the plan bias (5), an unrepresented eligible milestone is - synthesised (6), then de-duplication and reference attachment (7), the - plan-backed guarantee (8), with fully-checked plans reported as - completion actions rather than candidates (9). + milestones are eligible and — since design §5 — which struggle repairs + are carried, the rest deferred beside them (3), matching is key equality + (4), scoring is today's plus the plan bias (5), an unrepresented eligible + milestone is synthesised (6) — and when nothing plan-related fits an + active plan, one body-double proposal is (§5) — then de-duplication and + reference attachment (7), the plan-backed guarantee (8), with + fully-checked plans reported as completion actions rather than + candidates (9). """ time_minutes = max(5, min(int(time_minutes), 180)) now = datetime.now(UTC) @@ -1103,11 +1312,19 @@ def build_now_plan( ] if interleave == "adaptive" and energy != "low": candidates.extend(_transfer_candidates(time_minutes)) + # Rule 3 extended: before rule 6 reads what is "represented", so a deferred + # repair does not stand in for the milestone it can no longer carry. + candidates, deferred_repairs = _defer_repairs(candidates, energy=energy, plans=plans) candidates.extend(plans.milestone_candidates(candidates, time_minutes)) + body_double = _body_double_candidate( + plans, candidates, deferred_repairs, energy=energy, time_minutes=time_minutes + ) + if body_double is not None: + candidates.append(body_double) starter = False if not candidates: - candidates = [_starter_candidate(time_minutes)] + candidates = [_starter_candidate(time_minutes, after_deferral=bool(deferred_repairs))] starter = True ranked = _dedupe( @@ -1137,6 +1354,7 @@ def build_now_plan( interleave_ratio=INTERLEAVE_RATIOS[energy] if interleave == "adaptive" else {}, active_plans=plans.summaries, energy_deferred=plans.deferred, + energy_deferred_repairs=deferred_repairs, completion_actions=plans.completions, warnings=plans.warnings, ) diff --git a/packages/studyloop/src/studyloop/learning/recap.py b/packages/studyloop/src/studyloop/learning/recap.py index b9f995fdc..58618fd20 100644 --- a/packages/studyloop/src/studyloop/learning/recap.py +++ b/packages/studyloop/src/studyloop/learning/recap.py @@ -65,6 +65,13 @@ def _plan_context(plan) -> str: f"{deferred.title}, waits for more energy: it needs {deferred.energy_floor} of 10 " f"and today's energy carries {deferred.energy_capability}." ) + for repair in getattr(plan, "energy_deferred_repairs", ()): + where = f" of {repair.plan_title}" if repair.plan_title else "" + sentences.append( + f"Repairing {repair.concept}{where} waits for more energy: it asks for " + f"{repair.required_capability} of 10 and today's energy carries " + f"{repair.energy_capability}." + ) for completion in getattr(plan, "completion_actions", ()): sentences.append(completion.action) return " ".join(sentences) diff --git a/packages/studyloop/src/studyloop/web/static/index.html b/packages/studyloop/src/studyloop/web/static/index.html index c23e0982e..e7885fea8 100644 --- a/packages/studyloop/src/studyloop/web/static/index.html +++ b/packages/studyloop/src/studyloop/web/static/index.html @@ -1104,12 +1104,17 @@ <h3 class="today-concept" x-text="plan?.primary?.concept"></h3> cannot carry, and plans whose every milestone is checked. Not a `.today-card` on purpose: the browser smoke test addresses the single action card by that class. --> - <div x-show="!loading && plan && (deferredNotes().length > 0 || completionNotes().length > 0 || warningNotes().length > 0)" + <div x-show="!loading && plan && (deferredNotes().length > 0 || deferredRepairNotes().length > 0 || completionNotes().length > 0 || warningNotes().length > 0)" class="today-plan-notes"> <p class="today-parked-label">Your plans</p> <template x-for="(note, i) in deferredNotes()" :key="'d' + i"> <p class="today-meta">Deferred for energy: <span x-text="note"></span></p> </template> + <!-- Struggle repair today's energy cannot carry (design §5): its own + key and line — a repair has no milestone number to print. --> + <template x-for="(note, i) in deferredRepairNotes()" :key="'r' + i"> + <p class="today-meta">Deferred for energy: <span x-text="note"></span></p> + </template> <!-- One block per finished plan (council review 6, F6): the closing review's sentence with ITS OWN evidence lines beneath it, keyed by plan id, so two finished plans never share one flat list. The label diff --git a/packages/studyloop/src/studyloop/web/static/js/components/today-panel.js b/packages/studyloop/src/studyloop/web/static/js/components/today-panel.js index ceca60239..0e87dd534 100644 --- a/packages/studyloop/src/studyloop/web/static/js/components/today-panel.js +++ b/packages/studyloop/src/studyloop/web/static/js/components/today-panel.js @@ -114,7 +114,15 @@ export function todayPanel() { }, startAction(rec) { - Alpine.store('nav').go(this._viewFor(rec.action_type)); + Alpine.store('nav').go(this.viewForAction(rec)); + }, + + /* The view an action starts in. A body-double proposal (design §5) is a + session in the Body Double view, whatever its action_type says; every + other action keeps the action_type mapping above. */ + viewForAction(rec) { + if (rec && rec.source === 'body_double') return 'body-double'; + return this._viewFor(rec && rec.action_type); }, /* ---- Plan relevance (issue #10) — rendering of what /api/now ranked. ---- @@ -162,6 +170,21 @@ export function todayPanel() { ); }, + /* One line per struggle repair today's energy cannot carry (design §5, + amendment 2): its own key, its own sentence — a repair has no milestone + number. A repair unrelated to any plan names none. */ + deferredRepairNotes() { + const repairs = (this.plan && this.plan.energy_deferred_repairs) || []; + const energy = (this.plan && this.plan.energy) || 'current'; + return repairs.map((r) => { + const head = r.plan_title + ? `${r.plan_title} \u2014 repairing` + : 'Repairing'; + return `${head} \u201c${r.concept}\u201d (${r.confidence}) waits for more energy ` + + `(asks for ${r.required_capability}/10, ${energy} energy carries ${r.energy_capability}/10)`; + }); + }, + /* One block per finished plan (council review 6, F6): the closing review's sentence and ITS evidence lines, keyed by plan_id, in the engine's order. With two finished plans a flat list of lines lost the plan each belonged @@ -199,6 +222,7 @@ export function todayPanel() { return ( this.planLabel(this.plan && this.plan.primary) !== '' || this.deferredNotes().length > 0 + || this.deferredRepairNotes().length > 0 || this.completionNotes().length > 0 || this.warningNotes().length > 0 ); diff --git a/packages/studyloop/tests/test_learning_decision.py b/packages/studyloop/tests/test_learning_decision.py index 347aa4caa..807a57430 100644 --- a/packages/studyloop/tests/test_learning_decision.py +++ b/packages/studyloop/tests/test_learning_decision.py @@ -48,7 +48,7 @@ def test_no_data_returns_starter_recommendation(monkeypatch) -> None: monkeypatch.setattr( decision, "_starter_candidate", - lambda time_minutes: _candidate("starter", score=10), + lambda time_minutes, after_deferral=False: _candidate("starter", score=10), ) plan = build_now_plan() diff --git a/packages/studyloop/tests/test_now_plan_guidance.py b/packages/studyloop/tests/test_now_plan_guidance.py index 42652f5b3..65ebcca7b 100644 --- a/packages/studyloop/tests/test_now_plan_guidance.py +++ b/packages/studyloop/tests/test_now_plan_guidance.py @@ -335,7 +335,13 @@ def test_synthesizes_milestone_when_no_candidate_represents_it(monkeypatch) -> N def test_preserves_one_plan_backed_action_when_energy_allows(monkeypatch) -> None: - """Rule 8: ≥ 1 eligible plan-backed action in primary + alternates when energy permits.""" + """Rule 8: ≥ 1 eligible plan-backed action in primary + alternates when energy permits. + + Below the floor the milestone is deferred, never synthesised; since design §5 + the plan-backed slot rule 8 keeps is then the body-double proposal — sitting + with the plan asks for no energy the day cannot carry — and the primary and + first alternate stay the real, higher-ranked candidates. + """ _plan( "sql-windows", energy_floor=5, milestones=[Milestone("Frames", concepts=["window frame"])] ) @@ -350,7 +356,10 @@ def test_preserves_one_plan_backed_action_when_energy_allows(monkeypatch) -> Non low = build_now_plan(energy="low") - assert [rec.concept for rec in _all(low)] == ["due 0", "due 1", "due 2"] + assert [rec.concept for rec in _all(low)] == ["due 0", "due 1", "Sit with Sql Windows"] + assert low.alternates[1].source == "body_double" + assert low.alternates[1].plan_refs == (PlanRef("sql-windows", None),) + assert not any(rec.source.startswith("study_plan:") for rec in _all(low)) assert [d.milestone_index for d in low.energy_deferred] == [0] @@ -1236,12 +1245,12 @@ def test_live_struggle_repair_defers_at_low_energy_like_new_work(monkeypatch) -> low = build_now_plan(energy="low") - deferred = {item.concept: item for item in low.energy_deferred_repairs} # pyright: ignore[reportAttributeAccessIssue] + deferred = {item.concept: item for item in low.energy_deferred_repairs} assert set(deferred) == {"window function", "window frame", "decorators", "closures"} assert not any(rec.concept in deferred for rec in _all(low)), "deferred repair is not ranked" live = deferred["window function"] - assert isinstance(live, decision.DeferredRepair) # pyright: ignore[reportAttributeAccessIssue] + assert isinstance(live, decision.DeferredRepair) assert (live.plan_id, live.plan_title, live.topic, live.confidence) == ( "sql-windows", "SQL Windows", @@ -1267,15 +1276,14 @@ def test_live_struggle_repair_defers_at_low_energy_like_new_work(monkeypatch) -> payload = low.to_json_dict() assert [entry["concept"] for entry in payload["energy_deferred_repairs"]] == [ - item.concept - for item in low.energy_deferred_repairs # pyright: ignore[reportAttributeAccessIssue] + item.concept for item in low.energy_deferred_repairs ] assert payload["energy_deferred_repairs"][0]["energy_demand"] in {"high", "medium"} # Medium energy (6/10) carries every demand class: nothing deferred, key absent. medium = build_now_plan(energy="medium") - assert medium.energy_deferred_repairs == () # pyright: ignore[reportAttributeAccessIssue] + assert medium.energy_deferred_repairs == () assert "energy_deferred_repairs" not in medium.to_json_dict() assert any(rec.concept == "window function" for rec in _all(medium)) @@ -1293,7 +1301,7 @@ def test_recovered_repair_stays_eligible_at_low_energy(monkeypatch) -> None: assert low.primary.action_type == "teachback" assert low.primary.plan_refs == (PlanRef("sql-windows", None),) assert low.primary.metadata["energy_demand"] == "low" - assert low.energy_deferred_repairs == () # pyright: ignore[reportAttributeAccessIssue] + assert low.energy_deferred_repairs == () assert not any(rec.source == "body_double" for rec in _all(low)) assert [d.milestone_index for d in low.energy_deferred] == [1] @@ -1311,7 +1319,7 @@ def test_recovered_repair_stays_eligible_at_low_energy(monkeypatch) -> None: assert low.primary.concept == "window frame" assert low.primary.plan_refs == (PlanRef("sql-windows", None),) - assert [d.concept for d in low.energy_deferred_repairs] == ["window function"] # pyright: ignore[reportAttributeAccessIssue] + assert [d.concept for d in low.energy_deferred_repairs] == ["window function"] assert not any(rec.source == "body_double" for rec in _all(low)) @@ -1333,9 +1341,9 @@ def test_body_double_candidate_is_synthesised_when_nothing_plan_related_fits(mon assert "Frames" in primary.reason and "window function" in primary.reason assert low.starter is False assert low.alternates == [] - assert [d.concept for d in low.energy_deferred_repairs] == ["window function"] # pyright: ignore[reportAttributeAccessIssue] + assert [d.concept for d in low.energy_deferred_repairs] == ["window function"] assert [d.milestone_index for d in low.energy_deferred] == [1] - assert decision.BODY_DOUBLE_BASE_SCORE < decision.MILESTONE_BASE_SCORE # pyright: ignore[reportAttributeAccessIssue] + assert decision.BODY_DOUBLE_BASE_SCORE < decision.MILESTONE_BASE_SCORE golden_keys = list(json.loads(GOLDEN.read_text(encoding="utf-8"))) payload = low.to_json_dict() @@ -1376,7 +1384,7 @@ def test_body_double_never_appears_without_an_active_plan(monkeypatch) -> None: low = build_now_plan(energy="low") assert not any(rec.source == "body_double" for rec in _all(low)) - assert [(d.concept, d.plan_id, d.plan_title) for d in low.energy_deferred_repairs] == [ # pyright: ignore[reportAttributeAccessIssue] + assert [(d.concept, d.plan_id, d.plan_title) for d in low.energy_deferred_repairs] == [ ("decorators", None, None) ] # Every real candidate was deferred: the starter stands in, and says why. From 8e9cbdf559827aa67d3b45fe9b25b9485e208e3d Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sat, 19 Sep 2026 13:03:39 +0100 Subject: [PATCH 06/32] fix(now): the body double names ready plans only MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An active-but-unready plan is matched but never synthesised (spec rule 8), and the body-double proposal is a synthesis: with only a husk active and nothing plan-related fitting the day, nothing is proposed to sit with — the warning beside it already says "pause or repair". With a ready plan beside the husk the proposal names the ready one alone. Design §5 decision 6. --- .../src/studyloop/learning/decision.py | 7 +++- .../studyloop/tests/test_now_plan_guidance.py | 36 +++++++++++++++++++ 2 files changed, 42 insertions(+), 1 deletion(-) diff --git a/packages/studyloop/src/studyloop/learning/decision.py b/packages/studyloop/src/studyloop/learning/decision.py index 0cf570367..2f0960e30 100644 --- a/packages/studyloop/src/studyloop/learning/decision.py +++ b/packages/studyloop/src/studyloop/learning/decision.py @@ -1223,7 +1223,12 @@ def _body_double_candidate( """ if not plans.matchable or any(plans.is_plan_related(c) for c in candidates): return None - named = [plan.plan for plan in plans.matchable] + # An active-but-unready plan is matched but never synthesised (spec rule 8); + # the body double is a synthesis, so only ready plans are sat with. The + # warning beside it already says "pause or repair". + named = [plan.plan for plan in plans.matchable if plan.readiness.ready] + if not named: + return None first = named[0] titles = " and ".join(plan.title for plan in named) items = [ diff --git a/packages/studyloop/tests/test_now_plan_guidance.py b/packages/studyloop/tests/test_now_plan_guidance.py index 65ebcca7b..4027f3d0a 100644 --- a/packages/studyloop/tests/test_now_plan_guidance.py +++ b/packages/studyloop/tests/test_now_plan_guidance.py @@ -1375,6 +1375,42 @@ def test_body_double_is_a_proposal_not_a_filter(monkeypatch) -> None: assert "Frames" in low.alternates[0].reason +def test_body_double_is_never_synthesised_for_an_unready_plan(monkeypatch) -> None: + """An active-but-unready plan is matched but never synthesised (spec rule 8), and the + body double is a synthesis: with only a husk active and nothing plan-related fitting, + the engine proposes nothing to sit with — the warning already says repair it.""" + husk = StudyPlan( + plan_id="husk", + title="Husk", + status="active", + created="2026-08-01T00:00:00+00:00", + updated="2026-09-01T00:00:00+00:00", + topics=["sql"], + milestones=[Milestone(title="Frames", concepts=["window frame"])], + ) + store.create_plan(husk) # no mission, no success criteria: unready + _plant_struggles(monkeypatch, _struggle("window function", days_ago=3)) + + low = build_now_plan(energy="low") + + assert not any(rec.source == "body_double" for rec in _all(low)) + assert [d.concept for d in low.energy_deferred_repairs] == ["window function"] + assert low.starter is True + assert any("husk" in w for w in low.warnings) + + # A ready plan beside the husk: the proposal names the ready one only; rule 7 + # may still attach a `(husk, None)` ref because the topics match. + _row3_plan() + + low = build_now_plan(energy="low") + + assert low.primary.source == "body_double" + assert low.primary.concept == "Sit with SQL Windows" + assert PlanRef("sql-windows", None) in low.primary.plan_refs + assert low.primary.evidence_command == 'studyloop study "SQL Windows" --mode co-study' + assert "Husk" not in low.primary.reason + + def test_body_double_never_appears_without_an_active_plan(monkeypatch) -> None: """No active plan, no plan to sit with: the deferral still happens (plan-independent, ``plan_id`` ``None``), the golden world stays untouched, and a non-active plan is From 6d5a2d0e7aa01cc006601a632d9220e2156c8e58 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sat, 19 Sep 2026 13:05:08 +0100 Subject: [PATCH 07/32] docs(spec): the no-plan payload is byte-identical only when nothing is deferred MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Design §5 decision 1 makes repair deferral plan-independent, so D-5's "a learner with no active plan receives the pre-#10 payload byte for byte" now holds for a no-plan learner with NOTHING deferred; one with a live struggle deferred at low energy receives `energy_deferred_repairs` (and the starter, if nothing else was collected) where they used to receive the hands-on repair. The MODIFIED requirement, docs/cli-reference.md and docs/study-plans.md now say exactly that instead of "unchanged"; design §5 records the consequence and puts the scope question to council review 7. --- docs/cli-reference.md | 2 +- docs/study-plans.md | 4 +++- openspec/changes/plan-integration-followons/design.md | 10 ++++++++++ .../specs/active-learning-decisions/spec.md | 8 ++++++-- 4 files changed, 20 insertions(+), 4 deletions(-) diff --git a/docs/cli-reference.md b/docs/cli-reference.md index 383e33cd9..c633011dc 100644 --- a/docs/cli-reference.md +++ b/docs/cli-reference.md @@ -259,7 +259,7 @@ studyloop now --speak Default ranking is due review first, then struggling or low teach-back score, then active-course continuity, then modality match. Low energy suppresses hard context switching. -With an active study plan the same engine is plan-aware (see [Study Plans](study-plans.md#plan-aware-now)): plan-related work gets a bounded bias within its urgency class, a ready plan's next milestone is suggested when it is within your energy and no gathered candidate represents it, and a milestone above your current energy is deferred with a reason. Repair of a live struggle is deferred the same way at low energy (an older struggle or weak teach-back needs medium; a concept still being learned is never deferred, nor is a due review), and when nothing plan-related fits the day the panel proposes sitting with the plan — its command is the co-study door, `studyloop study "<plan>" --mode co-study`, labelled "Sit with the plan" rather than "Record evidence". The panel names the plan and milestone an action advances and prints one "Deferred for energy" line per deferred milestone and per deferred repair; `--json` adds `active_plans`, `energy_deferred`, `energy_deferred_repairs`, `completion_actions`, `warnings` and per-action `plan_refs` only when each is non-empty, so the no-plan output is unchanged and a plan that cannot be read shows up as a warning rather than a failure. +With an active study plan the same engine is plan-aware (see [Study Plans](study-plans.md#plan-aware-now)): plan-related work gets a bounded bias within its urgency class, a ready plan's next milestone is suggested when it is within your energy and no gathered candidate represents it, and a milestone above your current energy is deferred with a reason. Repair of a live struggle is deferred the same way at low energy (an older struggle or weak teach-back needs medium; a concept still being learned is never deferred, nor is a due review), and when nothing plan-related fits the day the panel proposes sitting with the plan — its command is the co-study door, `studyloop study "<plan>" --mode co-study`, labelled "Sit with the plan" rather than "Record evidence". The panel names the plan and milestone an action advances and prints one "Deferred for energy" line per deferred milestone and per deferred repair; `--json` adds `active_plans`, `energy_deferred`, `energy_deferred_repairs`, `completion_actions`, `warnings` and per-action `plan_refs` only when each is non-empty, so with no plan and nothing deferred the output is unchanged (the deferral of a live struggle at low energy is the one thing that happens without a plan), and a plan that cannot be read shows up as a warning rather than a failure. `studyloop chat-note` turns one markdown/text note into a compact Socratic context pack. V1 prints or speaks the mentor prompt; it does not run a separate chat backend. diff --git a/docs/study-plans.md b/docs/study-plans.md index 755a87130..12f69e7ab 100644 --- a/docs/study-plans.md +++ b/docs/study-plans.md @@ -252,7 +252,9 @@ action keeps its plain sentence and a warning says why — a failure is never shown as a clean slate. This is plan-aware guidance with tested ranking rules — a bias, not a filter: an overdue review or a fresh struggle on an unrelated topic can still outrank new milestone work. With no active plan the recommendation is -unchanged; a plan that cannot be read adds a warning and nothing else. The +unchanged — except that a live struggle is deferred at low energy whether or +not a plan names it; a plan that cannot be read adds a warning and nothing +else. The ranking rules are tested; whether the primary is the action *you* would take is a separate judgement. Five frozen scenarios and the engine's primaries are in the project's rubric receipt diff --git a/openspec/changes/plan-integration-followons/design.md b/openspec/changes/plan-integration-followons/design.md index cf322f2ad..a72d50676 100644 --- a/openspec/changes/plan-integration-followons/design.md +++ b/openspec/changes/plan-integration-followons/design.md @@ -273,6 +273,16 @@ has none; `INTERLEAVE_RATIOS["low"]` unchanged. 1. **The deferral is plan-independent.** A live struggle is a live struggle whether or not a plan names it (amendment 2's `plan_id … else None` already said so); the finding was about the learner's day, not the plan. The body double, by contrast, *requires* a matchable active plan — it is "sit with the plan". + Consequence, stated rather than hidden: D-5's "a learner with no active plan receives the pre-#10 payload + byte for byte" now holds for a no-plan learner **with nothing deferred**; a no-plan learner whose live + struggle is deferred at low energy gets `energy_deferred_repairs` (and the starter if nothing else was + collected) where they used to get the hands-on repair. The golden world defers nothing and is unchanged. + The spec delta and both docs say so in those words. **A council question (review 7):** is that the right + scope for D-F, or should the repair half be gated on an active plan? +6. **The body double names ready plans only.** An active-but-unready plan is matched but never synthesised + (spec rule 8), and the body double is a synthesis; with only a husk active and nothing plan-related fitting, + nothing is proposed to sit with — the warning beside it already says "pause or repair" + (`test_body_double_is_never_synthesised_for_an_unready_plan`). 2. **Rule 8's guaranteed slot below the floor is the body-double proposal.** It carries `plan_refs`, so where four unrelated due items outrank everything at low energy the second alternate is now the proposal, not a third unrelated item. It advertises no work the energy cannot carry — the property rule 8's docstring diff --git a/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md b/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md index 0151dae71..d5ffedade 100644 --- a/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md +++ b/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md @@ -180,8 +180,12 @@ the body-doubling floor): `energy_deferred_repairs`, `completion_actions` and `warnings`; `LearningRecommendation` gains `plan_refs: tuple[PlanRef, ...] = ()`. `to_json_dict()` SHALL omit each of these when empty, so a learner with no -active plan receives the pre-#10 payload **byte for byte** — pinned by -`tests/golden/now_plan_no_active.json`, captured before any of this shipped. +active plan **and nothing deferred** receives the pre-#10 payload **byte for +byte** — pinned by `tests/golden/now_plan_no_active.json`, captured before any +of this shipped. The one plan-independent change is rule 2's repair half: a +learner with no plan whose live struggle is deferred at low energy receives +`energy_deferred_repairs` (and the starter, if nothing else was collected) +where they used to receive the hands-on repair itself. Renderers (`studyloop now`, `GET /api/now`, the Today card, the daily recap in its JSON, spoken and Rich-panel forms) SHALL show plan relevance, energy deferral — one line per deferred milestone **and** one per deferred repair — From 0fdda5c7761c437ab8a6a12ff79a6ce14b84a1a6 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sat, 19 Sep 2026 13:08:29 +0100 Subject: [PATCH 08/32] =?UTF-8?q?docs(council):=20brief=20for=20review=207?= =?UTF-8?q?=20=E2=80=94=20item=205=20(D-F),=20reviewed=20tree=206d5a2d0e?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Binding: D-F verbatim, rubric row 3 verbatim, design §5 with its T5.1 amendments and the six GREEN decisions, hard rules verified on the tree. Then the four commits, the agent's own T5.x report, every diff in the range in full, rubric row 3b as written, the control receipt, reference facts, and ten numbered check questions (scope of the plan-independent deferral is asked outright as (a)). --- .../council/brief-review7-2026-09-19.md | 1734 +++++++++++++++++ 1 file changed, 1734 insertions(+) create mode 100644 docs/architecture/plan-integration/council/brief-review7-2026-09-19.md diff --git a/docs/architecture/plan-integration/council/brief-review7-2026-09-19.md b/docs/architecture/plan-integration/council/brief-review7-2026-09-19.md new file mode 100644 index 000000000..bbbefd617 --- /dev/null +++ b/docs/architecture/plan-integration/council/brief-review7-2026-09-19.md @@ -0,0 +1,1734 @@ +# Council review 7 — item 5 (D-F): per-item energy demand for repair, and the body-doubling floor + +**Date:** 2026-09-19 · **Branch:** `feat/energy-demand-body-double`, reviewed tree `6d5a2d0e` (four commits on +`main` `4f8e3e0f`; `main` carries items 1–4 and the follow-on chores, CI fully green on `a03fc9bd` and +`4f8e3e0f`). **Reviewed range:** `4f8e3e0f..6d5a2d0e` — 4 commits, 15 files, +1,042/−25. **You are one +independent seat**; no other seat's answer is visible. You have no tools — this brief is the complete evidence +base. One implementing agent worked between owner checkpoints; your findings gate the merge of item 5 to `main` +and the owner's scoring of rubric row 3b (T5.5), after which the change is archived. + +Item 5 answers the owner's one **"no"** on the D-16 rubric walkthrough (row 3, 2026-09-16): at low energy the +engine recommended hands-on repair of a *live* struggle because a struggle-repair candidate carried no energy +demand of its own — rule 3 only gated new milestone work. This review is about one ranker change and its +renderers; nothing else moved in the range. + +## 0. What you are reviewing against (binding) + +### Owner decision (HANDOFF-2026-09-16.md §2, verbatim) + +> | D-F | Scenario 3 (**no**): a struggle-repair task has no energy demand; hands-on repair of a live struggle on a +> low-energy day compounds the struggle (RSD). Derive per-item energy demand from struggle recency / teach-back; +> when nothing plan-related fits the day's capability, synthesise a **body-doubling / open-session** candidate +> (feature exists: ADR-0001/0003, `web/routes/body_double.py`) naming the deferred items. | + +### Rubric row 3 (`receipts/now-rubric-2026-09-16.md`, the finding, verbatim) + +> | 3 | Energy-deferred | Plan `sql-windows` with `energy_floor: 5`; milestone 0 `Window basics` **done** (concepts +> `[window function]`), milestone 1 `Frames` (concepts `[window frame]`). One struggle-repair item `window +> function`/sql, `hands-on`, base 82. **Energy `low`** (capability 3/10). | **`window function`** (hands-on, score +> 80, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, milestone 1, floor 5, +> capability 3)]`; JSON gains `energy_deferred`. | Rule 3: capability 3 < floor 5, so the *new* milestone (Frames) +> is deferred and named, not synthesised; the plan-related repair on a finished milestone's concept stays +> eligible and keeps its ref (`None`: plan-related repair, not the next milestone). Score = 82 + 12 bias − 14 +> (hands-on at low energy). | **no** — owner, 2026-09-16: a struggle-repair task has no energy demand of its own; +> recommending hands-on repair of a *live* struggle on a low-energy day risks compounding the struggle and +> damaging confidence (RSD). Finding for council: (1) derive a per-item energy demand for repair from struggle +> recency/teach-back — at low energy a live struggle defers like new work, a recovered one stays eligible as +> gentle review; (2) when nothing plan-related fits the day's capability, synthesise a body-doubling / +> open-session candidate (feature exists: ADR-0001/0003, `web/routes/body_double.py`) naming the deferred items, +> instead of the least-bad task. | + +### Owner's standing rule (2026-09-18, design preamble) + +Nothing in the planning/architect flow may be kiro-cli specific; every process and steering surface applies +across all supported harnesses. (Item 5 touches no harness-specific surface; the co-study door is the CLI's +`studyloop study … --mode co-study` and the Web's Body Double view, both harness-neutral.) + +### Design §5 — verbatim as it stands at the reviewed tree (`openspec/changes/plan-integration-followons/design.md`) + +#### (design.md §5) Item 5 — per-item energy demand and the body-doubling floor (D-F) — designed here, reviewed separately + +*(One page, written before item 5's RED; see tasks T5.\*.)* + +- **Energy demand per candidate.** `_struggle_candidates` derives `energy_demand ∈ {low, medium, high}` from + struggle state: `confidence == "struggling"` (a live struggle, ≤ 14 days) → `high`; `struggling` older than + 14 days or a weak teach-back → `medium`; recovered / gentle review → `low`. Demand maps to a required + capability (`high` → 6, `medium` → 4, `low` → 0) compared with `ENERGY_CAPABILITY[energy]`. +- **Rule 3 extended.** Below capability, *repair* above demand is deferred exactly like new milestone work and + listed in `energy_deferred` with a reason naming the struggle; recovered repair stays eligible as gentle + review. Due recall (`source=study_progress` due rows) is unaffected. +- **Body-doubling floor.** When the eligible plan-related set is empty **and** at least one active plan exists, + synthesise one candidate: `source="body_double"`, `action_type="conversation"`, low base score (below any + real candidate), reason naming the deferred items, `plan_refs` for each named plan with `milestone_index + None`, and an `evidence_command` that opens the existing body-double session route (`studyloop study + --mode co-study` / `web/routes/body_double.py`). A proposal, not a filter: real candidates still rank above + it. +- **No-plan output byte-identical to the golden**; `INTERLEAVE_RATIOS["low"]` unchanged (the design does not + call for it). +- **Rubric row 3b** (owner scores): scenario 3's fixture at low energy now yields the deferred repair named in + `energy_deferred` and a body-double primary (or the due recall if one exists). + +**T5.1 review against the code (2026-09-18, tree `7208eb67`) — three amendments, each from reading +`learning/decision.py`, not the text above:** + +1. **Demand classes are the struggle collector's classes.** `_struggle_candidates` emits a row only when + `confidence in ("struggling", "learning")` or `last_teachback_score < 14`; "recovered / gentle review" is not a + row it produces. So: `struggling` with `last_seen` ≤ 14 days → `high`; `struggling` older than 14 days, or any + row whose only signal is a weak teach-back → `medium`; `learning` → `low`. Demand is derived once, in the + collector, and carried in the candidate's `metadata` beside `confidence` so the scorer and the renderers read + one value. Required capability `high → 6`, `medium → 4`, `low → 0` stands (the `low` class is what "repair is + cheaper than encoding" was always about). +2. **`energy_deferred` is milestone-shaped and cannot carry a repair as it is.** `DeferredMilestone` has a + mandatory `milestone_index`, and all three renderers (`cli/_now.py`, `learning/recap.py`, + `today-panel.js::deferredNotes`) print `milestone {index + 1} "{title}" needs energy {floor}/10`. A deferred + repair gets its own frozen `DeferredRepair` (`plan_id`/`plan_title` when plan-related, else `None`, `concept`, + `topic`, `confidence`, `energy_demand`, `required_capability`, `energy_capability`, `reason` naming the + struggle), carried in a **new additive key `energy_deferred_repairs`** — not folded into `energy_deferred`, + whose consumers would print "milestone None". Same "readable off the top" rule as the closing review's + evidence lines: each renderer gains one line per deferred repair. +3. **The body-double door is a session start, not `web/routes/body_double.py`.** That route is the read-only focus + reader (`GET /api/body-double/focus`). The session door is `studyloop study "<topic>" --mode co-study` on the + CLI and a session start from the Body Double view (origin `body-double`) on the Web. `_evidence_command` has + no branch for a `conversation` candidate and would fall through to `studyloop progress … -c learning`, which is + a write, not a door — so the body-double candidate carries `evidence_command = 'studyloop study "<plan title>" + --mode co-study'` set explicitly, and `_evidence_command` is not asked to guess. `source="body_double"`, + `action_type="conversation"`, base score below `MILESTONE_BASE_SCORE` (48) so any real candidate outranks it. + +Rule 3's *deferral* of repair is the change; rule 3's *eligibility* of plan-related due recall is untouched. The +no-plan golden stays byte-identical because a body-double candidate requires an active plan and the golden world +has none; `INTERLEAVE_RATIOS["low"]` unchanged. + +**Decisions taken at GREEN (2026-09-19), each a test in `test_now_plan_guidance.py`:** + +1. **The deferral is plan-independent.** A live struggle is a live struggle whether or not a plan names it + (amendment 2's `plan_id … else None` already said so); the finding was about the learner's day, not the + plan. The body double, by contrast, *requires* a matchable active plan — it is "sit with the plan". + Consequence, stated rather than hidden: D-5's "a learner with no active plan receives the pre-#10 payload + byte for byte" now holds for a no-plan learner **with nothing deferred**; a no-plan learner whose live + struggle is deferred at low energy gets `energy_deferred_repairs` (and the starter if nothing else was + collected) where they used to get the hands-on repair. The golden world defers nothing and is unchanged. + The spec delta and both docs say so in those words. **A council question (review 7):** is that the right + scope for D-F, or should the repair half be gated on an active plan? +6. **The body double names ready plans only.** An active-but-unready plan is matched but never synthesised + (spec rule 8), and the body double is a synthesis; with only a husk active and nothing plan-related fitting, + nothing is proposed to sit with — the warning beside it already says "pause or repair" + (`test_body_double_is_never_synthesised_for_an_unready_plan`). +2. **Rule 8's guaranteed slot below the floor is the body-double proposal.** It carries `plan_refs`, so where + four unrelated due items outrank everything at low energy the second alternate is now the proposal, not a + third unrelated item. It advertises no work the energy cannot carry — the property rule 8's docstring + protects — and the primary is untouched. `test_preserves_one_plan_backed_action_when_energy_allows` says so. +3. **The starter tells the truth after a deferral.** With no plan and every real candidate deferred, the starter + stands in; its reason now says the energy deferred the repair work rather than "no learning evidence found + yet", which would be false. The golden world defers nothing, so its sentence is unchanged. +4. **Body-double shape.** Base `BODY_DOUBLE_BASE_SCORE = 30` (+12 bias = 42 < practice 48, milestone 48); + concept `Sit with <title>` (one plan) / `Sit with your plans`; `plan_refs` for every matchable plan; reason + naming each deferred milestone and repair; command `studyloop study "<first title>" --mode co-study`. The + Today card starts it in the Body Double view (`viewForAction`); the CLI labels the command "Sit with the plan". +5. **A deferred repair does not "represent" a milestone** (rule 6 runs after the deferral), so an eligible + milestone whose only collected representative was a deferred live struggle is synthesised as a conversation — + the learner can still talk about it. + +Known edge, not solved here: a `struggling` row whose `last_seen` cannot be parsed is read as live (`high`) — +the cautious side; `_days_since` returns `None` and the demand falls to `high`. + +### Hard rules for this batch (verified on `6d5a2d0e` before this brief was written) + +- TDD: the RED commit `ef319a7b` (six Python tests, five JS tests, each red for its missing name: 6 failed / 40 + passed; JS 5 failed / 139 passed) precedes GREEN `326abcf9`; the two corrections `8e9cbdf5` (test + fix in one + commit) and `6d5a2d0e` (wording only) follow. +- The seam adds **no new writer**: item 5 is a ranker change; `build_now_plan` still reads plans through exactly + one `PlanApplication().get_active_guidance()` call plus the item-4 preview read for fully-checked plans; the + struggle collector reads `history.observations.rows` as before and writes nothing. +- Golden `tests/golden/now_plan_no_active.json` sha256 + `ec451ce8857c8a72e398e3054e3c060cd3b5b13ecb29e29cba3822dc192503c0` **unchanged** (measured); + `test_no_active_plans_json_byte_identical_to_golden` green. Architecture guard `test_architecture_plan_seam.py` + **30 passed**. MCP production inventory unchanged (`PRODUCTION_TOOL_COUNT = 32`). +- `test_now_plan_guidance.py` **47 passed** (six REDs flipped, one existing pin updated — + `test_preserves_one_plan_backed_action_when_energy_allows`, see design §5 decision 2 — and one correction + test added); plan suites (`test_now_plan_guidance`, `test_learning_decision`, `test_cli_plan_seam`, + `test_plan_application`, `test_docs_plan_integration_contract`) **189 passed**; JS **144 passed** (+5); e2e + browser journeys `test_journey_study_plan.py` + `test_plans_api.py` **20 passed**; `mkdocs build --strict` exit + 0; `openspec validate plan-integration-followons` valid; ruff check / ruff format --check / pyright clean. +- Full suite at GREEN `326abcf9`, matched control on a clean `main` `4f8e3e0f` worktree, both importing from + their own tree: item5 30 failed / **5126 passed** / 14 errors; control 30 failed / 5120 passed / 14 errors; + **item5 − control = ∅, control − item5 = ∅**; the 44 shared ids are byte-identical to the committed + sandbox-environmental set (`receipts/full-suite-control-item4-2026-09-18.md`). Receipt: + `receipts/full-suite-control-item5-2026-09-19.md` (§5 below). The two corrections after GREEN were verified + with the plan suites, not a second full run. +- CI has **not** seen this branch; it is pushed after this review, as with items 3/3b/4. +- The recap (`build_daily_recap`) calls `build_now_plan()` at its default `medium` energy (6/10), which carries + every demand class; a deferred repair can therefore reach the recap's sentence builder `_plan_context(plan)` + only for a low-energy plan handed to it — the renderer exists and is pinned through `_plan_context` directly. + +## 1. Commits in the range (oldest first) + +| Commit | RED/GREEN/other | Files | +| --- | --- | --- | +| `ef319a7b` | **RED** — six Python tests + five JS tests | `tests/test_now_plan_guidance.py`, `tests/js/today-panel-plan.test.js` | +| `326abcf9` | **GREEN** — engine, three renderers, spec delta (MODIFIED requirement), docs, rubric row 3b, task ticks, control receipt | 14 files | +| `8e9cbdf5` | correction (agent's own re-read, before this brief): the body double names **ready** plans only; test + fix | `learning/decision.py`, `tests/test_now_plan_guidance.py` | +| `6d5a2d0e` | wording: the no-plan payload is byte-identical only when nothing is deferred | spec delta, `docs/cli-reference.md`, `docs/study-plans.md`, `design.md` | + +## 2. The agent's own implementation report (verbatim from `tasks.md`, T5.1–T5.5) + +#### (tasks.md) Item 5 — energy demand + body-doubling floor (D-F) · own round + +- [x] **T5.1** Design §5 reviewed against the code (`_struggle_candidates`, `_score_candidates`, rule 3) — amend if + the code contradicts it. (2026-09-18: three amendments recorded under §5 — the demand classes are the struggle + collector's own (`struggling` fresh/old, weak teach-back, `learning`; no "recovered" row exists); a deferred + repair needs its own `DeferredRepair` in a new additive `energy_deferred_repairs` key because `DeferredMilestone` + and its three renderers are milestone-shaped; the body-double door is `studyloop study … --mode co-study` / + the Body Double view's session start, not the read-only `body_double.py` focus route, so the candidate sets + its `evidence_command` explicitly. T5.2's RED names hold; a sixth test pins the new key's rendering.) +- [x] **T5.2** RED `tests/test_now_plan_guidance.py`: `test_live_struggle_repair_defers_at_low_energy_like_new_work`, + `test_recovered_repair_stays_eligible_at_low_energy`, `test_body_double_candidate_is_synthesised_when_nothing_plan_related_fits`, + `test_body_double_is_a_proposal_not_a_filter`, `test_body_double_never_appears_without_an_active_plan`, + golden byte-identity still green. (2026-09-19, `ef319a7b` on `feat/energy-demand-body-double`: the five plus + the sixth, `test_cli_now_and_recap_render_deferred_repairs_and_the_body_double_door`, and five JS pins in + `tests/js/today-panel-plan.test.js` (`deferredRepairNotes`, `viewForAction`, markup). The struggle collector + runs for real over patched `observations.rows`, since demand is derived in the collector. 6 red / 40 green, + JS 5 red / 139 green, each for its missing name.) +- [x] **T5.3** GREEN: `energy_demand`, rule 3 extension, `body_double` synthesis, CLI/Today "sit with the plan" line. + (2026-09-19: `DeferredRepair` + `energy_deferred_repairs`, `ENERGY_DEMAND_CAPABILITY`/`LIVE_STRUGGLE_DAYS`, + `_energy_demand` in the collector, `_defer_repairs` before rule 6, `_body_double_candidate` after it, + honest starter after a deferral; CLI `now` + recap + Today card lines; MODIFIED requirement in the + `active-learning-decisions` delta; docs `study-plans.md` "Plan-aware now" + `cli-reference.md`; design §5 + "Decisions taken at GREEN" 1-5. Module 46/46, JS 144/144, e2e plan journeys 20/20, docs contract 39/39, + `mkdocs --strict` exit 0, `openspec validate` valid; full suite vs a clean `main` control — see the GREEN + commit's receipt.) +- [ ] ⚖ **T5.4** Council review 7 (`review7`, same seats or `kimi-k2-thinking` third); arbitration; corrections. +- [ ] **T5.5** Rubric row **3b** for the owner (`PENDING`); the change is archived only after the owner scores 3b + (4b was scored **yes / yes** on 2026-09-18, so 3b is the one row still outstanding). (2026-09-19: row 3b + written into `receipts/now-rubric-2026-09-16.md` with three readings printed from the real engine — (a) live + struggle → body-double primary, (b) plus an unrelated due recall → due recall primary, proposal beneath, + (c) recovered `learning` → gentle teach-back primary — verdict `PENDING`.) + +## 3. The diffs — `4f8e3e0f..6d5a2d0e`, every file, in full + +```diff +diff --git a/docs/architecture/plan-integration/receipts/full-suite-control-item5-2026-09-19.md b/docs/architecture/plan-integration/receipts/full-suite-control-item5-2026-09-19.md +new file mode 100644 +index 00000000..0f415e32 +--- /dev/null ++++ b/docs/architecture/plan-integration/receipts/full-suite-control-item5-2026-09-19.md +@@ -0,0 +1,36 @@ ++# Full-suite matched control — item 5 (D-F) — 2026-09-19 ++ ++Two full `packages/studyloop/tests` runs in parallel, same machine, same ++sandbox, `-q -p no:cacheprovider -rfE`: ++ ++| Tree | Worktree | Result | ++| --- | --- | --- | ++| **item 5** (`feat/energy-demand-body-double`, RED `ef319a7b` + GREEN working tree) | `studyloop-wt/item5` | 30 failed, **5126 passed**, 4 skipped, 804 deselected, 14 errors (10:17) | ++| **control** (`main` `4f8e3e0f`, detached) | `studyloop-wt/ctrl-item5` | 30 failed, 5120 passed, 4 skipped, 804 deselected, 14 errors (10:24) | ++ ++Both worktrees were `uv sync --all-packages --group dev` and each proved to ++import `studyloop` from its own tree before the run. ++ ++## Sorted failing-id sets ++ ++- item5 ∖ control = **∅** — zero regressions. ++- control ∖ item5 = **∅** — nothing item 5 fixed by accident, and the six ++ new tests account for the passed-count difference (+6). ++- item5 ∖ committed environmental set (`full-suite-control-item4-2026-09-18.md`, ++ 44 ids + item 4's seven then-REDs) = **∅**. The seven ids on the other side of ++ that comparison are item 4's REDs, green since `82293293`. ++ ++The 44 shared ids are the sandbox-environmental set the item-4 receipt lists ++by name (journeys world guards, acceptance isolation, second-brain CLI/doctor, ++harness-matrix live mechanics, obsidian vault isolation, fresh-install scope); ++unchanged here, byte for byte. ++ ++## Scoped gates on the same tree ++ ++- `test_now_plan_guidance.py` 46/46 (six REDs flipped; golden `now_plan_no_active.json` byte-identical); ++ `test_learning_decision.py` 5/5 (one stub updated to the starter's new keyword). ++- JS `node --test packages/studyloop/tests/js/*.test.js` 144/144 (+5). ++- e2e `test_journey_study_plan.py` + `test_plans_api.py` 20/20 (browser). ++- `test_docs_plan_integration_contract.py` + `test_ci_workflow_contract.py` 39/39. ++- `mkdocs build --strict` exit 0; `openspec validate plan-integration-followons` valid. ++- ruff check / ruff format --check / pyright: clean on every touched file. +diff --git a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md +index ff46f1f4..759ce0e9 100644 +--- a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md ++++ b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md +@@ -1,6 +1,6 @@ + # Plan-aware `now` — five-scenario human rubric (D-16) · 2026-09-16 + +-**Status: owner verdicts RECORDED 2026-09-16 (interactive walkthrough with the coordinator).** Scenarios 1, 2 and the primary of 4: yes. Scenario 3: **no** (finding). Scenario 4 completion action: **no as phrased** (finding). Scenario 5: verified. **Row 4b added and scored 2026-09-18** (item 4 / D-G, tree `feat/plan-close`; outputs emitted from the tree committed as `82293293`): scenario 4 re-run with the end assessment planted, both proposals recorded as emitted; owner verdict **yes** on both readings — (a) *extend* with named evidence is a proposal the owner would walk, (b) a clean review's *close* is one the owner would agree to. This closes row 4's "no as phrased" finding; row 4 is kept as the record of the original finding. This receipt was produced unattended ++**Status: owner verdicts RECORDED 2026-09-16 (interactive walkthrough with the coordinator).** Scenarios 1, 2 and the primary of 4: yes. Scenario 3: **no** (finding). Scenario 4 completion action: **no as phrased** (finding). Scenario 5: verified. **Row 3b added 2026-09-19** (item 5 / D-F, tree `feat/energy-demand-body-double`; outputs emitted from the tree the GREEN commit records): scenario 3 re-run with the struggle collector live, three readings printed, verdict `PENDING` for the owner. **Row 4b added and scored 2026-09-18** (item 4 / D-G, tree `feat/plan-close`; outputs emitted from the tree committed as `82293293`): scenario 4 re-run with the end assessment planted, both proposals recorded as emitted; owner verdict **yes** on both readings — (a) *extend* with named evidence is a proposal the owner would walk, (b) a clean review's *close* is one the owner would agree to. This closes row 4's "no as phrased" finding; row 4 is kept as the record of the original finding. This receipt was produced unattended + overnight. Every scenario below was *run* on frozen fixtures and the primary + and its rationale are recorded exactly as the engine emitted them; the + "would I do the primary?" column is a human judgement that only the owner can +@@ -30,6 +30,7 @@ learning"). + | 1 | Matching due | Active plan `sql-windows` ("SQL Windows", topics `[sql]`, next milestone 0 concepts `[window function]`). Due items: `decorators`/python base 102, `window function`/sql base 100. | **`window function`** (sql, recall, score 130, `plan_refs=[(sql-windows, 0)]`); alternate `decorators` (120, no refs). | Rule 5: both are due items two points apart — one urgency class — so the plan-related one takes the +12 bias and wins; the unrelated due item is *kept* as an alternate (bias, not filter). Rule 7 names the milestone the action advances. | **yes** — owner, 2026-09-16: "window function is the logical step before decorating it" (a prerequisite-order argument; see F2 follow-on). | + | 2 | Urgent-unrelated wins | Same plan. One due item: `decorators`/python, base 100 (an overdue spaced-repetition review). Nothing represents milestone 0. | **`decorators`** (118, no refs); alternate `window function` (60, `source=study_plan:sql-windows:0`, `plan_refs=[(sql-windows, 0)]`, reason "Next milestone 1/1 of plan 'SQL Windows': Window basics"). | Rule 5: the unrelated candidate is in a more urgent class (due review) and wins outright — the bias cannot lift new-milestone work over it. Rule 6: the plan's next milestone was unrepresented, so it was synthesised at base 48 + bias 12 = 60 and appears as the plan-backed alternate (rule 8 satisfied without any swap). | **yes** — owner, 2026-09-16: clear the overdue review first. Note for follow-on: an overdue item *unrelated* to the plan must not sit as an alternate indefinitely — propose it explicitly (age-aware nudge) or let the learner retire it. | + | 3 | Energy-deferred | Plan `sql-windows` with `energy_floor: 5`; milestone 0 `Window basics` **done** (concepts `[window function]`), milestone 1 `Frames` (concepts `[window frame]`). One struggle-repair item `window function`/sql, `hands-on`, base 82. **Energy `low`** (capability 3/10). | **`window function`** (hands-on, score 80, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, milestone 1, floor 5, capability 3)]`; JSON gains `energy_deferred`. | Rule 3: capability 3 < floor 5, so the *new* milestone (Frames) is deferred and named, not synthesised; the plan-related repair on a finished milestone's concept stays eligible and keeps its ref (`None`: plan-related repair, not the next milestone). Score = 82 + 12 bias − 14 (hands-on at low energy). | **no** — owner, 2026-09-16: a struggle-repair task has no energy demand of its own; recommending hands-on repair of a *live* struggle on a low-energy day risks compounding the struggle and damaging confidence (RSD). Finding for council: (1) derive a per-item energy demand for repair from struggle recency/teach-back — at low energy a live struggle defers like new work, a recovered one stays eligible as gentle review; (2) when nothing plan-related fits the day's capability, synthesise a body-doubling / open-session candidate (feature exists: ADR-0001/0003, `web/routes/body_double.py`) naming the deferred items, instead of the least-bad task. | ++| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered). | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason *"Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."*; no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone. | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 < any real candidate — a proposal, not a filter) carrying the co-study session door. | **PENDING** — owner: (a) would you sit with the plan rather than repair the live struggle today? (b) is the body-double proposal right to sit beneath the unrelated due recall? (c) is the gentle teach-back on a recovered concept one you would do at low energy? | + | 4 | Fully-checked | Plan `done-plan` ("Done Plan"), milestones A and B both done. One due item `decorators`/python base 100. | **`decorators`** (118, no refs); `completion_actions=[(done-plan, "Every milestone of 'Done Plan' is checked off — close the plan or extend it with a follow-on mission.")]`; no `study_plan:` candidate anywhere; JSON gains `active_plans` + `completion_actions`. | Rule 9: a fully-checked plan is reported as a completion action and is neither matched (no bias, no refs) nor synthesised. | **primary yes / completion action no as phrased** — owner, 2026-09-16: the completion action must be contextual and consensual. Run the end assessment (`assess(phase="end")`: due reviews, struggles, unverified milestones on the plan's concepts). If outstanding work touches the plan's concepts (or their prerequisites — F2 concept edges), propose *extend* and name the evidence; if clean, propose *close* and ask the learner to agree ("anything you are not comfortable with?"). Status never changes automatically (#7). Natural vehicle: architect with `purpose=planning` and the assessment in the brief (`plan close <id>`, sibling of `plan repair <id>`). Finding for council. | + | 4b | Fully-checked — **re-run after D-G (item 4, 2026-09-18)** | Row 4's fixture (`done-plan`, milestones A `[alpha]` and B `[beta]` both done; one due item `decorators`/python base 100), plus the end assessment's readers planted: **(a)** one due review on plan concept `alpha` (`overdue`) with session mentions backing both concepts; **(b)** no due rows, same mentions. | Primary unchanged in both: **`decorators`** (118, no refs); no `study_plan:` candidate. **(a)** `completion_actions=[(done-plan, due 1 / struggles 0 / unverified 0, proposal **extend**, evidence `["Due review: alpha — overdue"]`)]`, sentence: "Every milestone of 'Done Plan' is checked off, and the closing review proposes extending the plan — 1 due review, 0 struggles and 0 unverified milestones on its concepts. Walk the evidence with the architect: studyloop plan close done-plan." **(b)** counts 0/0/0, proposal **close**, evidence `[]`, sentence: "Every milestone of 'Done Plan' is checked off and the closing review is clean — it proposes closing the plan. Close it with the architect when you agree: studyloop plan close done-plan." No warnings; JSON gains the five keys only inside the entry. | Rule 9 as before for the ranking. The completion action is now the end assessment read as a preview (`assess(phase="end", record=False)`, one call, no write, no status change): `extend` iff any of the three counts on the plan's own concepts is above zero, else `close`; due counts only rows naming a concept (the scheduler's "new topic" row is excluded — owner decision 2026-09-17). `plan close done-plan` launches the architect with the same review as the brief's first section; status moves only when the learner agrees. | **yes / yes** — owner, 2026-09-18, answering the two questions as posed: (a) **yes**, a proposal the owner would walk; (b) **yes**, a close the owner would agree to. No further line given. Closes row 4's "no as phrased" finding; status still moves only when the learner agrees in the architect conversation. | + | 5 | No-plan identical | No plan documents at all; no collector candidates. | **`one tiny recall loop`** (python, recall, 28, `source=starter`) — the starter. | D-5: `serialise(plan) == golden` → **byte-identical** (`True` in the run); the JSON key list is exactly the golden's — no additive key is present. | **verified** — owner walkthrough 2026-09-16: nothing to judge; the byte-identical golden is the acceptance. | +@@ -47,7 +48,12 @@ The five rows correspond to + `test_fully_checked_active_plan_emits_completion_not_candidate` and + `test_no_active_plans_json_byte_identical_to_golden`; the primaries above are + what those tests assert, printed from a throwaway driver over the same +-fixtures. Row 4b corresponds to ++fixtures. Row 3b corresponds to ++`test_body_double_candidate_is_synthesised_when_nothing_plan_related_fits` ++(reading a), `test_body_double_is_a_proposal_not_a_filter` (reading b) and ++`test_recovered_repair_stays_eligible_at_low_energy` (reading c), printed on ++2026-09-19 from a throwaway driver over the module's `_plant_struggles` ++fixture with the struggle collector running for real. Row 4b corresponds to + `test_completion_action_carries_the_end_assessment_and_proposes_extend_when_concepts_are_due` + (reading a) and `test_completion_action_proposes_close_when_the_assessment_is_clean` + (reading b), printed the same way on 2026-09-18 with the end assessment's +diff --git a/docs/cli-reference.md b/docs/cli-reference.md +index 90078af6..c633011d 100644 +--- a/docs/cli-reference.md ++++ b/docs/cli-reference.md +@@ -259,7 +259,7 @@ studyloop now --speak + + Default ranking is due review first, then struggling or low teach-back score, then active-course continuity, then modality match. Low energy suppresses hard context switching. + +-With an active study plan the same engine is plan-aware (see [Study Plans](study-plans.md#plan-aware-now)): plan-related work gets a bounded bias within its urgency class, a ready plan's next milestone is suggested when it is within your energy and no gathered candidate represents it, and a milestone above your current energy is deferred with a reason. The panel names the plan and milestone an action advances; `--json` adds `active_plans`, `energy_deferred`, `completion_actions`, `warnings` and per-action `plan_refs` only when each is non-empty, so the no-plan output is unchanged and a plan that cannot be read shows up as a warning rather than a failure. ++With an active study plan the same engine is plan-aware (see [Study Plans](study-plans.md#plan-aware-now)): plan-related work gets a bounded bias within its urgency class, a ready plan's next milestone is suggested when it is within your energy and no gathered candidate represents it, and a milestone above your current energy is deferred with a reason. Repair of a live struggle is deferred the same way at low energy (an older struggle or weak teach-back needs medium; a concept still being learned is never deferred, nor is a due review), and when nothing plan-related fits the day the panel proposes sitting with the plan — its command is the co-study door, `studyloop study "<plan>" --mode co-study`, labelled "Sit with the plan" rather than "Record evidence". The panel names the plan and milestone an action advances and prints one "Deferred for energy" line per deferred milestone and per deferred repair; `--json` adds `active_plans`, `energy_deferred`, `energy_deferred_repairs`, `completion_actions`, `warnings` and per-action `plan_refs` only when each is non-empty, so with no plan and nothing deferred the output is unchanged (the deferral of a live struggle at low energy is the one thing that happens without a plan), and a plan that cannot be read shows up as a warning rather than a failure. + + `studyloop chat-note` turns one markdown/text note into a compact Socratic context pack. V1 prints or speaks the mentor prompt; it does not run a separate chat backend. + +diff --git a/docs/study-plans.md b/docs/study-plans.md +index 2c072c76..12f69e7a 100644 +--- a/docs/study-plans.md ++++ b/docs/study-plans.md +@@ -219,7 +219,16 @@ action that advances a plan's next milestone is named with the plan and the + milestone it serves; a **ready** plan whose next milestone is within your + current energy gets that milestone suggested even when no other evidence + points at it; and a plan whose energy floor is above your current energy has +-that milestone deferred with a reason rather than dropped. An active plan that ++that milestone deferred with a reason rather than dropped. Repair has an ++energy demand of its own: on a low-energy day a **live** struggle (recorded ++as struggling within the last two weeks) is deferred like new work — listed, ++not recommended — an older struggle or a weak teach-back needs medium energy, ++and a concept you are still learning is the gentle review that stays ++available at any energy; due reviews are never deferred. When nothing ++plan-related fits the day's energy, the recommendation is to **sit with the ++plan** — a body-double session, no new material, no repair — with the ++deferred items named; a real, unrelated action still outranks that proposal ++when one exists. An active plan that + is **not ready** — a hand edit removed its mission or its milestones — is + listed with a warning naming what to repair; it still biases related work, + but no milestone is suggested for it until it is paused or repaired. A plan +@@ -243,7 +252,9 @@ action keeps its plain sentence and a warning says why — a failure is never + shown as a clean slate. This is plan-aware guidance with tested ranking rules — a bias, not + a filter: an overdue review or a fresh struggle on an unrelated topic can + still outrank new milestone work. With no active plan the recommendation is +-unchanged; a plan that cannot be read adds a warning and nothing else. The ++unchanged — except that a live struggle is deferred at low energy whether or ++not a plan names it; a plan that cannot be read adds a warning and nothing ++else. The + ranking rules are tested; whether the primary is the action *you* would take + is a separate judgement. Five frozen scenarios and the engine's primaries are + in the project's rubric receipt +diff --git a/openspec/changes/plan-integration-followons/design.md b/openspec/changes/plan-integration-followons/design.md +index c847d696..a72d5067 100644 +--- a/openspec/changes/plan-integration-followons/design.md ++++ b/openspec/changes/plan-integration-followons/design.md +@@ -268,6 +268,39 @@ Rule 3's *deferral* of repair is the change; rule 3's *eligibility* of plan-rela + no-plan golden stays byte-identical because a body-double candidate requires an active plan and the golden world + has none; `INTERLEAVE_RATIOS["low"]` unchanged. + ++**Decisions taken at GREEN (2026-09-19), each a test in `test_now_plan_guidance.py`:** ++ ++1. **The deferral is plan-independent.** A live struggle is a live struggle whether or not a plan names it ++ (amendment 2's `plan_id … else None` already said so); the finding was about the learner's day, not the ++ plan. The body double, by contrast, *requires* a matchable active plan — it is "sit with the plan". ++ Consequence, stated rather than hidden: D-5's "a learner with no active plan receives the pre-#10 payload ++ byte for byte" now holds for a no-plan learner **with nothing deferred**; a no-plan learner whose live ++ struggle is deferred at low energy gets `energy_deferred_repairs` (and the starter if nothing else was ++ collected) where they used to get the hands-on repair. The golden world defers nothing and is unchanged. ++ The spec delta and both docs say so in those words. **A council question (review 7):** is that the right ++ scope for D-F, or should the repair half be gated on an active plan? ++6. **The body double names ready plans only.** An active-but-unready plan is matched but never synthesised ++ (spec rule 8), and the body double is a synthesis; with only a husk active and nothing plan-related fitting, ++ nothing is proposed to sit with — the warning beside it already says "pause or repair" ++ (`test_body_double_is_never_synthesised_for_an_unready_plan`). ++2. **Rule 8's guaranteed slot below the floor is the body-double proposal.** It carries `plan_refs`, so where ++ four unrelated due items outrank everything at low energy the second alternate is now the proposal, not a ++ third unrelated item. It advertises no work the energy cannot carry — the property rule 8's docstring ++ protects — and the primary is untouched. `test_preserves_one_plan_backed_action_when_energy_allows` says so. ++3. **The starter tells the truth after a deferral.** With no plan and every real candidate deferred, the starter ++ stands in; its reason now says the energy deferred the repair work rather than "no learning evidence found ++ yet", which would be false. The golden world defers nothing, so its sentence is unchanged. ++4. **Body-double shape.** Base `BODY_DOUBLE_BASE_SCORE = 30` (+12 bias = 42 < practice 48, milestone 48); ++ concept `Sit with <title>` (one plan) / `Sit with your plans`; `plan_refs` for every matchable plan; reason ++ naming each deferred milestone and repair; command `studyloop study "<first title>" --mode co-study`. The ++ Today card starts it in the Body Double view (`viewForAction`); the CLI labels the command "Sit with the plan". ++5. **A deferred repair does not "represent" a milestone** (rule 6 runs after the deferral), so an eligible ++ milestone whose only collected representative was a deferred live struggle is synthesised as a conversation — ++ the learner can still talk about it. ++ ++Known edge, not solved here: a `struggling` row whose `last_seen` cannot be parsed is read as live (`high`) — ++the cautious side; `_days_since` returns `None` and the demand falls to `high`. ++ + ## 6. Verification + + `scripts/verify/plan_integration.py` gains registered checks for: the two architect grants (the ten names in +diff --git a/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md b/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md +index 2705937c..d5ffedad 100644 +--- a/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md ++++ b/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md +@@ -96,3 +96,236 @@ print each evidence line beneath it, and none SHALL re-rank. + `plan close <id>` still launches the architect, its brief's fourth line is + `Proposal: unassessed — the review is partial`, the gap is among the first + section's lines, and its status line does not say the review proposes ++ ++## MODIFIED Requirements ++ ++### Requirement: The now engine is plan-aware with tested ranking rules ++`studyloop.learning.decision.build_now_plan` SHALL remain the only ranker of ++study actions and SHALL consume active plans through exactly one call to ++`PlanApplication().get_active_guidance(today=…)`, where `today` is the date ++of the same instant `generated_at` records. It SHALL apply these rules, in ++this order (the plan-application-seam design, §3; decision D-5 of its ++council plan; item 5 / D-F of the follow-ons for rule 2's repair half and ++the body-doubling floor): ++ ++1. Candidates are collected as before; a failure to read plans at all SHALL ++ degrade to a `warnings` entry, never a failed recommendation, and SHALL be ++ logged with its traceback on `studyloop.learning.decision` so a ++ programming error cannot hide behind the learner-facing warning. ++2. The energy capability is `low|medium|high → 3|6|10`. For an active plan ++ whose `energy_floor` exceeds it, the next milestone SHALL be listed in ++ `energy_deferred` and SHALL NOT become a candidate. Plan-related due recall ++ stays eligible and plan-related whatever its recorded confidence. A ++ struggle **repair** carries an energy demand of its own, derived once in ++ the struggle collector from its own row classes and carried in the ++ candidate's `metadata["energy_demand"]`: `struggling` seen within 14 days → ++ `high` (asks for 6/10); `struggling` older than that, or a row whose only ++ signal is a weak teach-back → `medium` (4/10); `learning` → `low` (0/10). ++ A repair whose demand exceeds the capability SHALL be deferred exactly ++ like new milestone work — listed in `energy_deferred_repairs` as a ++ `DeferredRepair` (`plan_id`/`plan_title` when plan-related, else `None`, ++ `concept`, `topic`, `confidence`, `energy_demand`, `required_capability`, ++ `energy_capability`, `reason` naming the struggle) and never ranked; the ++ deferral does not depend on a plan existing. A `low`-demand repair is ++ always carried. `energy_deferred` stays milestone-shaped; a repair is never ++ folded into it. ++3. A candidate is plan-related when `normalise_match_key` of its concept, ++ topic or course **equals** one of the plan's `match_keys`; no substring ++ test. It names the plan's next milestone (`milestone_index`) only when the ++ key equals one of that milestone's concepts **and** the plan is eligible — ++ ready and within the energy capability; a topic or finished-milestone ++ match, or any match on an energy-deferred or active-but-unready plan, ++ carries `milestone_index = None` (plan-related repair), so a payload never ++ names a milestone it also reports as deferred or that the seam would ++ refuse to tick. ++4. Scoring is today's scoring plus one bounded bias for plan-related ++ candidates: within one urgency class plan-related beats unrelated, and a ++ globally more-urgent unrelated candidate still wins — a bias, not a filter. ++5. When no collected candidate represents an eligible (ready, energy-permitted) ++ plan's next milestone, one `conversation` candidate SHALL be synthesised ++ for it (source `study_plan:<plan_id>:<index>`, concept = the milestone's ++ first concept or its title, topic = the plan's first topic), scored below ++ every due and repair class. A learner with an active plan and no evidence ++ is therefore sent to the plan, and `starter` is `false`. A deferred repair ++ (rule 2) does not "represent" a milestone. When, after rules 2 and 5, **no ++ candidate is plan-related** and at least one matchable active plan exists, ++ one **body-double** candidate SHALL be synthesised instead of leaving the ++ plan to the least-bad task: `source = "body_double"`, `action_type = ++ "conversation"`, base score below the synthesised-milestone base so every ++ real candidate outranks it (a proposal, never a filter), `plan_refs` ++ `(plan_id, None)` for every matchable plan, a reason naming the deferred ++ milestones and repairs it stands in for, and `evidence_command` the ++ co-study session door — `studyloop study "<plan title>" --mode co-study` — ++ set explicitly, never a progress write. No active plan (a draft is not ++ one) → no body-double candidate; when every real candidate was deferred ++ and no plan exists, the starter stands in and its reason says the energy ++ deferred the repair work, not that no evidence exists. ++6. After de-duplication every matching `PlanRef(plan_id, milestone_index)` ++ SHALL be attached to each ranked action, ordered by target urgency ++ (`overdue`, `soon`, `later`, `undated`) → most recent `updated` → `plan_id`, ++ keeping the most specific milestone per plan. ++7. When primary + alternates hold no plan-backed action and an eligible one ++ whose estimate fits the requested time exists further down, it SHALL ++ replace the last alternate only; the primary is never re-ranked by plans. ++ Below a plan's floor that plan-backed action is the body-double proposal ++ (it advertises no work the energy cannot carry), never the deferred ++ milestone. ++8. A fully-checked active plan SHALL appear in `completion_actions` and SHALL ++ be neither matched nor synthesised. An active-but-unready plan SHALL be ++ listed and matched (bias and a `milestone_index = None` reference) but ++ never synthesised and never named as a milestone, with a warning naming ++ its blockers. ++ ++`NowPlan` gains `active_plans` (ordered as rule 6), `energy_deferred`, ++`energy_deferred_repairs`, `completion_actions` and `warnings`; ++`LearningRecommendation` gains `plan_refs: tuple[PlanRef, ...] = ()`. ++`to_json_dict()` SHALL omit each of these when empty, so a learner with no ++active plan **and nothing deferred** receives the pre-#10 payload **byte for ++byte** — pinned by `tests/golden/now_plan_no_active.json`, captured before any ++of this shipped. The one plan-independent change is rule 2's repair half: a ++learner with no plan whose live struggle is deferred at low energy receives ++`energy_deferred_repairs` (and the starter, if nothing else was collected) ++where they used to receive the hands-on repair itself. ++Renderers (`studyloop now`, `GET /api/now`, the Today card, the daily recap in ++its JSON, spoken and Rich-panel forms) SHALL show plan relevance, energy ++deferral — one line per deferred milestone **and** one per deferred repair — ++and the engine's warnings from these fields, SHALL label a body-double ++primary's command as the session door it is ("Sit with the plan", and the ++Today card starts it in the Body Double view) rather than as evidence to ++record, SHALL escape learner-authored text before any markup (Rich or HTML), ++and SHALL NOT re-rank. Ranking tests prove ranking compliance, not learner ++benefit (D-16); a five-scenario human rubric receipt accompanies the change, ++with row 3b re-run after this requirement's repair half. ++ ++#### Scenario: No active plan is byte-identical to the golden ++- **WHEN** no active plan exists (an empty plans directory, or only a draft) ++ and `build_now_plan()` runs with a frozen clock in an empty world ++- **THEN** the serialised `to_json_dict()` equals ++ `tests/golden/now_plan_no_active.json` byte for byte, and no ++ `active_plans`, `energy_deferred`, `energy_deferred_repairs`, ++ `completion_actions`, `warnings` or `plan_refs` key is present ++ ++#### Scenario: Matching due concept outranks unrelated of the same urgency ++- **WHEN** an active plan's milestone names `window function` and two due ++ items are two points apart, `decorators` (unrelated) ahead ++- **THEN** `window function` is primary with `plan_refs == (PlanRef(plan, 0),)` ++ and `decorators` is the first alternate with no refs ++ ++#### Scenario: A more-urgent unrelated item still wins ++- **WHEN** the only collected candidate is an unrelated due item and the ++ plan's next milestone is unrepresented ++- **THEN** the due item is primary and the synthesised milestone ++ (`study_plan:<id>:0`) is an alternate with a lower score ++ ++#### Scenario: Energy below the floor defers the milestone, keeps repair ++- **WHEN** energy is `low` (3/10), the plan's `energy_floor` is 5, its next ++ milestone is `Frames` and a `learning` row on a finished milestone's ++ concept is collected by the struggle collector (gentle repair) ++- **THEN** that repair is primary (`teachback`, `energy_demand == "low"`) with ++ `PlanRef(plan, None)`, `energy_deferred` names `(plan, 1, 5, 3)`, ++ `energy_deferred_repairs` is empty, and no `study_plan:` or `body_double` ++ candidate exists; at `medium` energy nothing is deferred and the milestone ++ is synthesised ++ ++#### Scenario: A live struggle's repair defers at low energy like new work ++- **WHEN** energy is `low` and the struggle collector holds a `struggling` row ++ seen 3 days ago on a plan concept, a `struggling` row seen 20 days ago on ++ another, a `struggling` row seen 1 day ago unrelated to any plan, and a ++ `confident` row kept only for a teach-back score of 9 ++- **THEN** none of the four is ranked; `energy_deferred_repairs` names all ++ four — the live plan-related one `("high", 6, 3)` with the plan's id and ++ title, the 20-day one `medium` (4), the weak-teach-back one `medium`, the ++ unrelated one with `plan_id` and `plan_title` `None` — `energy_deferred` ++ still names the milestone alone, and at `medium` energy the key is absent ++ and the live repair is ranked again ++ ++#### Scenario: Due recall is never deferred ++- **WHEN** energy is `low`, a due row on a plan concept is collected with ++ `confidence == "struggling"` and a live struggle repair is also collected ++- **THEN** the due row is primary with `PlanRef(plan, None)`, the repair is ++ in `energy_deferred_repairs`, and no `body_double` candidate exists ++ ++#### Scenario: Nothing plan-related fits, so the engine proposes sitting with the plan ++- **WHEN** energy is `low`, the plan's `energy_floor` is 5 (milestone ++ deferred) and its only repair is a live struggle (deferred) ++- **THEN** the primary is `source == "body_double"`, `action_type == ++ "conversation"`, `plan_refs == (PlanRef(plan, None),)`, `evidence_command == ++ 'studyloop study "<plan title>" --mode co-study'`, its reason names the ++ deferred milestone and the deferred repair, `starter` is `false`, and the ++ JSON keys are the golden's then `active_plans`, `energy_deferred`, ++ `energy_deferred_repairs` ++ ++#### Scenario: The body-double candidate is a proposal, not a filter ++- **WHEN** the same world also collects an unrelated due item ++- **THEN** the due item is primary with no refs and the body-double ++ candidate is the only alternate, with a lower score ++ ++#### Scenario: No active plan, no body double ++- **WHEN** no plan document exists (or only a draft) and a live unrelated ++ struggle is collected at `low` energy ++- **THEN** no `body_double` candidate exists, `energy_deferred_repairs` names ++ the struggle with `plan_id None`, `starter` is `true` and the starter's ++ reason says the energy deferred the repair work ++ ++#### Scenario: A deferred milestone is never named by a reference ++- **WHEN** energy is `low`, the plan's `energy_floor` is 5 and the only ++ collected candidate's concept equals the next milestone's concept ++- **THEN** the candidate is primary with `PlanRef(plan, None)` while ++ `energy_deferred` names that milestone; at `medium` energy the same ++ candidate carries `PlanRef(plan, 0)` and nothing is deferred ++ ++#### Scenario: An unready active plan is matched but never named ++- **WHEN** an active plan has no mission and no success criteria (unready) ++ and a collected candidate equals its next milestone's concept ++- **THEN** the candidate is primary with `PlanRef(plan, None)`, no ++ `study_plan:` candidate exists, the plan's `active_plans` entry has ++ `ready == False` and `eligible == False`, and one warning names the plan, ++ its blockers and "pause or repair" ++ ++#### Scenario: No substring matching ++- **WHEN** a milestone titled `Window functions deep dive` has no concepts ++ and candidates `window functions deep dive tutorial`, `window` and ++ `joins`/`SQL` are collected ++- **THEN** only `joins` is plan-related (`PlanRef(plan, None)` via the topic ++ `sql`, casefolded); the other two carry no refs ++ ++#### Scenario: Every matching plan is referenced, in order ++- **WHEN** six active plans (overdue, soon, later, three undated with ++ distinct and tied `updated`) all name the primary's concept ++- **THEN** `plan_refs` lists all six ordered overdue → soon → later → undated ++ by latest `updated` then `plan_id`, and `active_plans` is in the same order ++ ++#### Scenario: A plan-backed action is preserved when energy allows ++- **WHEN** four unrelated due items outrank everything and the plan's ++ `energy_floor` is 5 ++- **THEN** at `medium` energy the synthesised milestone replaces the second ++ alternate (the primary and first alternate are unchanged); at `low` energy ++ the primary and first alternate are the two best unrelated items, the ++ second alternate is the body-double proposal with `PlanRef(plan, None)`, ++ no `study_plan:` candidate exists and `energy_deferred` names the milestone ++ ++#### Scenario: Fully-checked plan emits a completion action ++- **WHEN** an active plan's every milestone is done and an unrelated due item ++ is collected ++- **THEN** `completion_actions` names the plan, the due item is primary with ++ no refs, no `study_plan:` candidate exists, and the plan's `active_plans` ++ entry has `next_milestone_index == None` ++ ++#### Scenario: Renderers show, never re-rank ++- **WHEN** `studyloop now --energy low`, `GET /api/now?energy=low` and the ++ daily recap run against the energy-deferral fixture, and against the ++ live-struggle fixture ++- **THEN** each names the primary the engine chose, the plan it advances, the ++ deferred milestone and — for the live-struggle fixture — one line per ++ deferred repair with its demand and the day's capability; a body-double ++ primary is labelled "Sit with the plan" with its `--mode co-study` door ++ and never "Record evidence"; with no plan the CLI panel prints no plan ++ lines, `GET /api/now` equals the golden, and the recap's `plan_context` is ++ absent from its JSON, its spoken text and the `recap today` panel ++ ++#### Scenario: Learner-authored text is data to every renderer ++- **WHEN** an active plan's title, topic or milestone text contains Rich ++ markup, HTML or shell punctuation (`Plan [/bold]`, `<script>…`, `"; rm -rf ~`) ++- **THEN** `build_now_plan` ranks and serialises it unchanged and writes ++ nothing to the document; `studyloop now` exits 0 and shows the text ++ literally; the Today card renders it through `x-text` +diff --git a/openspec/changes/plan-integration-followons/tasks.md b/openspec/changes/plan-integration-followons/tasks.md +index 8f3b6d44..da05fd77 100644 +--- a/openspec/changes/plan-integration-followons/tasks.md ++++ b/openspec/changes/plan-integration-followons/tasks.md +@@ -199,14 +199,28 @@ writer, through the existing gate, closes that. Kept out of item 3 so item 3's f + and its three renderers are milestone-shaped; the body-double door is `studyloop study … --mode co-study` / + the Body Double view's session start, not the read-only `body_double.py` focus route, so the candidate sets + its `evidence_command` explicitly. T5.2's RED names hold; a sixth test pins the new key's rendering.) +-- [ ] **T5.2** RED `tests/test_now_plan_guidance.py`: `test_live_struggle_repair_defers_at_low_energy_like_new_work`, ++- [x] **T5.2** RED `tests/test_now_plan_guidance.py`: `test_live_struggle_repair_defers_at_low_energy_like_new_work`, + `test_recovered_repair_stays_eligible_at_low_energy`, `test_body_double_candidate_is_synthesised_when_nothing_plan_related_fits`, + `test_body_double_is_a_proposal_not_a_filter`, `test_body_double_never_appears_without_an_active_plan`, +- golden byte-identity still green. +-- [ ] **T5.3** GREEN: `energy_demand`, rule 3 extension, `body_double` synthesis, CLI/Today "sit with the plan" line. ++ golden byte-identity still green. (2026-09-19, `ef319a7b` on `feat/energy-demand-body-double`: the five plus ++ the sixth, `test_cli_now_and_recap_render_deferred_repairs_and_the_body_double_door`, and five JS pins in ++ `tests/js/today-panel-plan.test.js` (`deferredRepairNotes`, `viewForAction`, markup). The struggle collector ++ runs for real over patched `observations.rows`, since demand is derived in the collector. 6 red / 40 green, ++ JS 5 red / 139 green, each for its missing name.) ++- [x] **T5.3** GREEN: `energy_demand`, rule 3 extension, `body_double` synthesis, CLI/Today "sit with the plan" line. ++ (2026-09-19: `DeferredRepair` + `energy_deferred_repairs`, `ENERGY_DEMAND_CAPABILITY`/`LIVE_STRUGGLE_DAYS`, ++ `_energy_demand` in the collector, `_defer_repairs` before rule 6, `_body_double_candidate` after it, ++ honest starter after a deferral; CLI `now` + recap + Today card lines; MODIFIED requirement in the ++ `active-learning-decisions` delta; docs `study-plans.md` "Plan-aware now" + `cli-reference.md`; design §5 ++ "Decisions taken at GREEN" 1-5. Module 46/46, JS 144/144, e2e plan journeys 20/20, docs contract 39/39, ++ `mkdocs --strict` exit 0, `openspec validate` valid; full suite vs a clean `main` control — see the GREEN ++ commit's receipt.) + - [ ] ⚖ **T5.4** Council review 7 (`review7`, same seats or `kimi-k2-thinking` third); arbitration; corrections. + - [ ] **T5.5** Rubric row **3b** for the owner (`PENDING`); the change is archived only after the owner scores 3b +- (4b was scored **yes / yes** on 2026-09-18, so 3b is the one row still outstanding). ++ (4b was scored **yes / yes** on 2026-09-18, so 3b is the one row still outstanding). (2026-09-19: row 3b ++ written into `receipts/now-rubric-2026-09-16.md` with three readings printed from the real engine — (a) live ++ struggle → body-double primary, (b) plus an unrelated due recall → due recall primary, proposal beneath, ++ (c) recovered `learning` → gentle teach-back primary — verdict `PENDING`.) + + ## Item 6 — proposals (no code) + +diff --git a/packages/studyloop/src/studyloop/cli/_now.py b/packages/studyloop/src/studyloop/cli/_now.py +index 1d917906..48cf89e6 100644 +--- a/packages/studyloop/src/studyloop/cli/_now.py ++++ b/packages/studyloop/src/studyloop/cli/_now.py +@@ -46,6 +46,9 @@ def _render_plan(plan) -> None: + primary = plan.primary + plans = _active_plans(plan) + plan_line = _plan_line(primary, plans) ++ # A body-double primary (design §5) carries the co-study session door, not a ++ # progress write: label it as the door it is. ++ door = "Sit with the plan" if primary.source == "body_double" else "Record evidence" + body = ( + f"[bold]{escape(primary.concept)}[/bold]\n" + f"Topic: [cyan]{escape(primary.topic)}[/cyan]\n" +@@ -54,7 +57,7 @@ def _render_plan(plan) -> None: + f"Why: {escape(primary.reason)}\n" + f"Source: [dim]{escape(primary.source)}[/dim]\n" + + (f"Plan: [magenta]{escape(plan_line)}[/magenta]\n" if plan_line else "") +- + f"\n[bold]Record evidence:[/bold]\n{escape(primary.evidence_command)}" ++ + f"\n[bold]{door}:[/bold]\n{escape(primary.evidence_command)}" + ) + console.print(Panel(body, title="Study Now", border_style="cyan")) + +@@ -69,6 +72,16 @@ def _render_plan(plan) -> None: + f"energy {deferred.energy_floor}/10; {plan.energy} energy carries " + f"{deferred.energy_capability}/10. Plan-related review and repair stay available." + ) ++ # One line per deferred repair (design §5, amendment 2): its own key, its ++ # own sentence — a repair has no milestone number to print. ++ for repair in getattr(plan, "energy_deferred_repairs", ()): ++ where = f"{escape(repair.plan_title)} — " if repair.plan_title else "" ++ console.print( ++ f"[yellow]Deferred for energy:[/yellow] {where}repairing " ++ f"“{escape(repair.concept)}” ({escape(repair.confidence)}) asks for " ++ f"{repair.required_capability}/10; {plan.energy} energy carries " ++ f"{repair.energy_capability}/10. Due recall and gentle review stay available." ++ ) + for completion in getattr(plan, "completion_actions", ()): + # "Closing review", not "Plan complete": the status is still active until + # the learner agrees with the architect (council review 6, F6). +diff --git a/packages/studyloop/src/studyloop/learning/decision.py b/packages/studyloop/src/studyloop/learning/decision.py +index 37132cbd..2f0960e3 100644 +--- a/packages/studyloop/src/studyloop/learning/decision.py ++++ b/packages/studyloop/src/studyloop/learning/decision.py +@@ -53,10 +53,31 @@ INTERLEAVE_RATIOS: dict[EnergyLevel, dict[str, int]] = { + + #: Design §3 rule 3 — what each self-reported energy level can carry, on the + #: 1-10 scale a plan's ``energy_floor`` uses. Below a plan's floor, *new* +-#: milestone work is deferred; plan-related due recall and struggle repair +-#: stay eligible, because repair is cheaper than encoding. ++#: milestone work is deferred; plan-related due recall stays eligible, and so ++#: does struggle repair whose own demand (below) the energy can carry. + ENERGY_CAPABILITY: dict[EnergyLevel, int] = {"low": 3, "medium": 6, "high": 10} + ++EnergyDemand = Literal["low", "medium", "high"] ++ ++#: Design §5 (item 5, D-F) — the capability a struggle repair asks for, by the ++#: demand class the struggle collector derives from its own row classes: a ++#: ``struggling`` row seen within ``LIVE_STRUGGLE_DAYS`` is ``high`` (a live ++#: struggle; hands-on repair on a low-energy day risks compounding it — rubric ++#: row 3, the owner's one "no"); ``struggling`` older than that, or a row whose ++#: only signal is a weak teach-back, is ``medium``; ``learning`` is ``low`` — ++#: the gentle review "repair is cheaper than encoding" was always about. ++#: Compared with ``ENERGY_CAPABILITY``: ``low`` (3) carries only low demand, ++#: ``medium`` (6) carries every class. ++ENERGY_DEMAND_CAPABILITY: dict[EnergyDemand, int] = {"high": 6, "medium": 4, "low": 0} ++LIVE_STRUGGLE_DAYS = 14 ++ ++#: Base score of the synthesised body-double candidate (design §5): below ++#: ``MILESTONE_BASE_SCORE`` so every real candidate — due, repair, practice, ++#: continuity, a synthesised milestone — outranks it. A proposal, never a ++#: filter; the plan bias then lifts it over nothing but the starter. ++BODY_DOUBLE_BASE_SCORE = 30 ++BODY_DOUBLE_SOURCE = "body_double" ++ + #: Rule 5 — the bias a plan-related candidate receives. Large enough to decide + #: a near-tie inside one urgency class (two due items a few days apart), small + #: enough that a clearly more-urgent unrelated candidate (a struggling repair, +@@ -129,6 +150,32 @@ class DeferredMilestone: + return asdict(self) + + ++@dataclass(frozen=True) ++class DeferredRepair: ++ """A struggle repair the current energy cannot carry (rule 3 extended, design §5). ++ ++ Its own type, not a :class:`DeferredMilestone`: that one has a mandatory ++ ``milestone_index`` and its three renderers print ``milestone N`` — a repair ++ folded into it would read "milestone None" (T5.1 amendment 2). ``plan_id`` ++ and ``plan_title`` are set when the struggle's concept, topic or course ++ matches an active plan, else ``None``: the deferral does not depend on a ++ plan — a live struggle is a live struggle whether or not a plan names it. ++ """ ++ ++ plan_id: str | None ++ plan_title: str | None ++ concept: str ++ topic: str ++ confidence: str ++ energy_demand: EnergyDemand ++ required_capability: int ++ energy_capability: int ++ reason: str ++ ++ def to_json_dict(self) -> dict: ++ return asdict(self) ++ ++ + @dataclass(frozen=True) + class CompletionAction: + """What to do about an active plan whose every milestone is checked (rule 9). +@@ -203,6 +250,7 @@ class NowPlan: + starter: bool = False + active_plans: tuple[ActivePlanSummary, ...] = () + energy_deferred: tuple[DeferredMilestone, ...] = () ++ energy_deferred_repairs: tuple[DeferredRepair, ...] = () + completion_actions: tuple[CompletionAction, ...] = () + warnings: tuple[str, ...] = () + +@@ -224,6 +272,10 @@ class NowPlan: + data["active_plans"] = [item.to_json_dict() for item in self.active_plans] + if self.energy_deferred: + data["energy_deferred"] = [item.to_json_dict() for item in self.energy_deferred] ++ if self.energy_deferred_repairs: ++ data["energy_deferred_repairs"] = [ ++ item.to_json_dict() for item in self.energy_deferred_repairs ++ ] + if self.completion_actions: + data["completion_actions"] = [item.to_json_dict() for item in self.completion_actions] + if self.warnings: +@@ -385,6 +437,7 @@ def _struggle_candidates(time_minutes: int) -> list[_Candidate]: + conn.close() + + candidates: list[_Candidate] = [] ++ today = datetime.now(UTC).date() + for row in rows: + row_keys = set(row.keys()) + concept = str(row["concept"]) +@@ -420,12 +473,45 @@ def _struggle_candidates(time_minutes: int) -> list[_Candidate]: + "confidence": confidence, + "last_teachback_score": teachback_score, + "session_count": row["session_count"], ++ # Design §5: derived once, here, from the collector's own ++ # classes; the deferral and every renderer read this value. ++ "energy_demand": _energy_demand( ++ confidence, row.get("last_seen") if "last_seen" in row_keys else None, today ++ ), + }, + ) + ) + return candidates + + ++def _energy_demand(confidence: str | None, last_seen: object, today: date) -> EnergyDemand: ++ """The capability class a repair asks for (design §5, T5.1 amendment 1). ++ ++ ``struggling`` seen within :data:`LIVE_STRUGGLE_DAYS` is a live struggle — ++ ``high``; a ``struggling`` row older than that, or one the collector kept ++ only for its weak teach-back, is ``medium``; ``learning`` is ``low``. An ++ unreadable ``last_seen`` on a ``struggling`` row is read as live: the ++ cautious side is the one the finding asks for. ++ """ ++ if confidence == "learning": ++ return "low" ++ if confidence != "struggling": ++ return "medium" ++ seen = _days_since(last_seen, today) ++ if seen is None or seen <= LIVE_STRUGGLE_DAYS: ++ return "high" ++ return "medium" ++ ++ ++def _days_since(stamp: object, today: date) -> int | None: ++ if not isinstance(stamp, str) or not stamp: ++ return None ++ try: ++ return (today - datetime.fromisoformat(stamp).date()).days ++ except ValueError: ++ return None ++ ++ + def _due_card_candidates(time_minutes: int) -> list[_Candidate]: + try: + from studyloop.services.review import list_course_summaries +@@ -566,7 +652,7 @@ def _transfer_candidates(time_minutes: int) -> list[_Candidate]: + return candidates + + +-def _starter_candidate(time_minutes: int) -> _Candidate: ++def _starter_candidate(time_minutes: int, *, after_deferral: bool = False) -> _Candidate: + try: + from studyloop.topics import get_topics + +@@ -579,11 +665,20 @@ def _starter_candidate(time_minutes: int) -> _Candidate: + else: + topic = "python" + display = "Python" ++ # "No learning evidence" would be false when evidence exists and today's ++ # energy deferred all of it (design §5); say what happened instead. The ++ # golden world has nothing to defer, so its sentence is unchanged. ++ reason = ( ++ "Today's energy deferred the repair work it cannot carry; " ++ "start with one small retrieval signal instead" ++ if after_deferral ++ else "No learning evidence found yet; start by creating one small retrieval signal" ++ ) + return _Candidate( + concept="one tiny recall loop", + topic=topic, + course=topic, +- reason="No learning evidence found yet; start by creating one small retrieval signal", ++ reason=reason, + action_type="recall", + estimated_minutes=_estimate_minutes("recall", time_minutes, 10), + source="starter", +@@ -967,6 +1062,18 @@ class _PlanContext: + synthesised.append(_milestone_candidate(plan, milestone, time_minutes)) + return synthesised + ++ def first_match(self, candidate: _Candidate) -> ActivePlanGuidance | None: ++ """The first matchable plan (plan order) this candidate's keys equal, if any.""" ++ keys = _candidate_keys(candidate) ++ for plan in self.matchable: ++ if keys & frozenset(plan.match_keys): ++ return plan ++ return None ++ ++ def is_plan_related(self, candidate: _Candidate) -> bool: ++ """Rule 5's test, before scoring: a ref already attached, or a key match.""" ++ return bool(candidate.plan_refs) or self.first_match(candidate) is not None ++ + def attach_refs(self, candidate: _Candidate) -> _Candidate: + """Rule 7: every matching plan, most specific milestone per plan, in plan order. + +@@ -1048,6 +1155,110 @@ def _milestone_candidate( + ) + + ++def _defer_repairs( ++ candidates: list[_Candidate], *, energy: EnergyLevel, plans: _PlanContext ++) -> tuple[list[_Candidate], tuple[DeferredRepair, ...]]: ++ """Rule 3 extended (design §5): repair above its own energy demand is deferred like new work. ++ ++ Only a candidate carrying ``energy_demand`` — the struggle collector's — is ++ judged. Due recall is never deferred whatever its confidence says, and a ++ ``learning`` repair (``low`` demand) is always carried. Plan-independent: ++ the entry names the plan when one matches, else ``None``. ++ """ ++ capability = ENERGY_CAPABILITY[energy] ++ kept: list[_Candidate] = [] ++ deferred: list[DeferredRepair] = [] ++ for candidate in candidates: ++ demand = candidate.metadata.get("energy_demand") ++ if demand not in ENERGY_DEMAND_CAPABILITY: ++ kept.append(candidate) ++ continue ++ required = ENERGY_DEMAND_CAPABILITY[demand] ++ if capability >= required: ++ kept.append(candidate) ++ continue ++ plan = plans.first_match(candidate) ++ confidence = str(candidate.metadata.get("confidence") or "struggling") ++ if demand == "high": ++ what = "a live struggle" ++ elif confidence == "struggling": ++ what = "an older struggle" ++ else: ++ what = "a weak teach-back" ++ deferred.append( ++ DeferredRepair( ++ plan_id=plan.plan.plan_id if plan is not None else None, ++ plan_title=plan.plan.title if plan is not None else None, ++ concept=candidate.concept, ++ topic=candidate.topic, ++ confidence=confidence, ++ energy_demand=demand, ++ required_capability=required, ++ energy_capability=capability, ++ reason=( ++ f"{energy} energy carries {capability}/10; repairing " ++ f"{candidate.concept!r} ({what}) asks for at least {required}/10 — " ++ "deferred like new work; due recall and gentle review stay available" ++ ), ++ ) ++ ) ++ return kept, tuple(deferred) ++ ++ ++def _body_double_candidate( ++ plans: _PlanContext, ++ candidates: list[_Candidate], ++ deferred_repairs: tuple[DeferredRepair, ...], ++ *, ++ energy: EnergyLevel, ++ time_minutes: int, ++) -> _Candidate | None: ++ """Design §5's floor: nothing plan-related fits and an active plan exists → sit with it. ++ ++ One ``source="body_double"`` conversation candidate, base below every real ++ candidate's (a proposal, not a filter), ``plan_refs`` ``(plan, None)`` for ++ every matchable plan, reason naming what it stands in for, and the co-study ++ session door as its command (T5.1 amendment 3): ``_evidence_command`` has ++ no branch for it and would answer with a progress *write*, not a door. ++ """ ++ if not plans.matchable or any(plans.is_plan_related(c) for c in candidates): ++ return None ++ # An active-but-unready plan is matched but never synthesised (spec rule 8); ++ # the body double is a synthesis, so only ready plans are sat with. The ++ # warning beside it already says "pause or repair". ++ named = [plan.plan for plan in plans.matchable if plan.readiness.ready] ++ if not named: ++ return None ++ first = named[0] ++ titles = " and ".join(plan.title for plan in named) ++ items = [ ++ f"milestone {d.milestone_index + 1} “{d.title}” of {d.plan_title}" for d in plans.deferred ++ ] + [f"repair of “{d.concept}”" for d in deferred_repairs] ++ deferred_note = f" — deferred: {'; '.join(items)}" if items else "" ++ topic = first.topics[0] if first.topics else "study" ++ safe_title = first.title.replace('"', '\\"') ++ return _Candidate( ++ concept=f"Sit with {first.title}" if len(named) == 1 else "Sit with your plans", ++ topic=topic, ++ course=None, ++ reason=( ++ f"Nothing plan-related fits {energy} energy today{deferred_note}. " ++ f"Sit with {titles} instead: a body-double session, no new material, no repair." ++ ), ++ action_type="conversation", ++ estimated_minutes=_estimate_minutes("conversation", time_minutes, 25), ++ source=BODY_DOUBLE_SOURCE, ++ evidence_command=f'studyloop study "{safe_title}" --mode co-study', ++ score=BODY_DOUBLE_BASE_SCORE, ++ metadata={ ++ "plan_id": first.plan_id, ++ "deferred_milestones": len(plans.deferred), ++ "deferred_repairs": len(deferred_repairs), ++ }, ++ plan_refs=tuple(PlanRef(plan.plan_id, None) for plan in named), ++ ) ++ ++ + def _guarantee_plan_backed(ranked: list[_Candidate], time_minutes: int) -> list[_Candidate]: + """Rule 8: ≥ 1 plan-backed action among primary + alternates when time permits. + +@@ -1084,11 +1295,14 @@ def build_now_plan( + + Order of operations is design §3's: guidance is read once (1), candidates + are collected as before (2), the energy capability decides which next +- milestones are eligible (3), matching is key equality (4), scoring is +- today's plus the plan bias (5), an unrepresented eligible milestone is +- synthesised (6), then de-duplication and reference attachment (7), the +- plan-backed guarantee (8), with fully-checked plans reported as +- completion actions rather than candidates (9). ++ milestones are eligible and — since design §5 — which struggle repairs ++ are carried, the rest deferred beside them (3), matching is key equality ++ (4), scoring is today's plus the plan bias (5), an unrepresented eligible ++ milestone is synthesised (6) — and when nothing plan-related fits an ++ active plan, one body-double proposal is (§5) — then de-duplication and ++ reference attachment (7), the plan-backed guarantee (8), with ++ fully-checked plans reported as completion actions rather than ++ candidates (9). + """ + time_minutes = max(5, min(int(time_minutes), 180)) + now = datetime.now(UTC) +@@ -1103,11 +1317,19 @@ def build_now_plan( + ] + if interleave == "adaptive" and energy != "low": + candidates.extend(_transfer_candidates(time_minutes)) ++ # Rule 3 extended: before rule 6 reads what is "represented", so a deferred ++ # repair does not stand in for the milestone it can no longer carry. ++ candidates, deferred_repairs = _defer_repairs(candidates, energy=energy, plans=plans) + candidates.extend(plans.milestone_candidates(candidates, time_minutes)) ++ body_double = _body_double_candidate( ++ plans, candidates, deferred_repairs, energy=energy, time_minutes=time_minutes ++ ) ++ if body_double is not None: ++ candidates.append(body_double) + + starter = False + if not candidates: +- candidates = [_starter_candidate(time_minutes)] ++ candidates = [_starter_candidate(time_minutes, after_deferral=bool(deferred_repairs))] + starter = True + + ranked = _dedupe( +@@ -1137,6 +1359,7 @@ def build_now_plan( + interleave_ratio=INTERLEAVE_RATIOS[energy] if interleave == "adaptive" else {}, + active_plans=plans.summaries, + energy_deferred=plans.deferred, ++ energy_deferred_repairs=deferred_repairs, + completion_actions=plans.completions, + warnings=plans.warnings, + ) +diff --git a/packages/studyloop/src/studyloop/learning/recap.py b/packages/studyloop/src/studyloop/learning/recap.py +index b9f995fd..58618fd2 100644 +--- a/packages/studyloop/src/studyloop/learning/recap.py ++++ b/packages/studyloop/src/studyloop/learning/recap.py +@@ -65,6 +65,13 @@ def _plan_context(plan) -> str: + f"{deferred.title}, waits for more energy: it needs {deferred.energy_floor} of 10 " + f"and today's energy carries {deferred.energy_capability}." + ) ++ for repair in getattr(plan, "energy_deferred_repairs", ()): ++ where = f" of {repair.plan_title}" if repair.plan_title else "" ++ sentences.append( ++ f"Repairing {repair.concept}{where} waits for more energy: it asks for " ++ f"{repair.required_capability} of 10 and today's energy carries " ++ f"{repair.energy_capability}." ++ ) + for completion in getattr(plan, "completion_actions", ()): + sentences.append(completion.action) + return " ".join(sentences) +diff --git a/packages/studyloop/src/studyloop/web/static/index.html b/packages/studyloop/src/studyloop/web/static/index.html +index c23e0982..e7885fea 100644 +--- a/packages/studyloop/src/studyloop/web/static/index.html ++++ b/packages/studyloop/src/studyloop/web/static/index.html +@@ -1104,12 +1104,17 @@ + cannot carry, and plans whose every milestone is checked. + Not a `.today-card` on purpose: the browser smoke test addresses + the single action card by that class. --> +- <div x-show="!loading && plan && (deferredNotes().length > 0 || completionNotes().length > 0 || warningNotes().length > 0)" ++ <div x-show="!loading && plan && (deferredNotes().length > 0 || deferredRepairNotes().length > 0 || completionNotes().length > 0 || warningNotes().length > 0)" + class="today-plan-notes"> + <p class="today-parked-label">Your plans</p> + <template x-for="(note, i) in deferredNotes()" :key="'d' + i"> + <p class="today-meta">Deferred for energy: <span x-text="note"></span></p> + </template> ++ <!-- Struggle repair today's energy cannot carry (design §5): its own ++ key and line — a repair has no milestone number to print. --> ++ <template x-for="(note, i) in deferredRepairNotes()" :key="'r' + i"> ++ <p class="today-meta">Deferred for energy: <span x-text="note"></span></p> ++ </template> + <!-- One block per finished plan (council review 6, F6): the closing + review's sentence with ITS OWN evidence lines beneath it, keyed by + plan id, so two finished plans never share one flat list. The label +diff --git a/packages/studyloop/src/studyloop/web/static/js/components/today-panel.js b/packages/studyloop/src/studyloop/web/static/js/components/today-panel.js +index ceca6023..0e87dd53 100644 +--- a/packages/studyloop/src/studyloop/web/static/js/components/today-panel.js ++++ b/packages/studyloop/src/studyloop/web/static/js/components/today-panel.js +@@ -114,7 +114,15 @@ export function todayPanel() { + }, + + startAction(rec) { +- Alpine.store('nav').go(this._viewFor(rec.action_type)); ++ Alpine.store('nav').go(this.viewForAction(rec)); ++ }, ++ ++ /* The view an action starts in. A body-double proposal (design §5) is a ++ session in the Body Double view, whatever its action_type says; every ++ other action keeps the action_type mapping above. */ ++ viewForAction(rec) { ++ if (rec && rec.source === 'body_double') return 'body-double'; ++ return this._viewFor(rec && rec.action_type); + }, + + /* ---- Plan relevance (issue #10) — rendering of what /api/now ranked. ---- +@@ -162,6 +170,21 @@ export function todayPanel() { + ); + }, + ++ /* One line per struggle repair today's energy cannot carry (design §5, ++ amendment 2): its own key, its own sentence — a repair has no milestone ++ number. A repair unrelated to any plan names none. */ ++ deferredRepairNotes() { ++ const repairs = (this.plan && this.plan.energy_deferred_repairs) || []; ++ const energy = (this.plan && this.plan.energy) || 'current'; ++ return repairs.map((r) => { ++ const head = r.plan_title ++ ? `${r.plan_title} \u2014 repairing` ++ : 'Repairing'; ++ return `${head} \u201c${r.concept}\u201d (${r.confidence}) waits for more energy ` ++ + `(asks for ${r.required_capability}/10, ${energy} energy carries ${r.energy_capability}/10)`; ++ }); ++ }, ++ + /* One block per finished plan (council review 6, F6): the closing review's + sentence and ITS evidence lines, keyed by plan_id, in the engine's order. + With two finished plans a flat list of lines lost the plan each belonged +@@ -199,6 +222,7 @@ export function todayPanel() { + return ( + this.planLabel(this.plan && this.plan.primary) !== '' + || this.deferredNotes().length > 0 ++ || this.deferredRepairNotes().length > 0 + || this.completionNotes().length > 0 + || this.warningNotes().length > 0 + ); +diff --git a/packages/studyloop/tests/js/today-panel-plan.test.js b/packages/studyloop/tests/js/today-panel-plan.test.js +index 6de30487..004c2069 100644 +--- a/packages/studyloop/tests/js/today-panel-plan.test.js ++++ b/packages/studyloop/tests/js/today-panel-plan.test.js +@@ -268,3 +268,100 @@ test('warningNotes: absent key renders no warning text', () => { + assert.deepEqual(panel.warningNotes(), []); + assert.equal(panel.hasPlanContext, false); + }); ++ ++/* Item 5 (D-F): a repair the day's energy cannot carry, in its own additive key ++ (design §5 amendment 2 — `energy_deferred` is milestone-shaped), and the ++ body-double proposal the engine synthesises when nothing plan-related fits. */ ++const DEFERRED_REPAIR_PAYLOAD = { ++ ...PLAN_PAYLOAD, ++ primary: { ++ concept: 'Sit with SQL Windows', ++ action_type: 'conversation', ++ estimated_minutes: 25, ++ reason: 'Nothing plan-related fits low energy today', ++ source: 'body_double', ++ evidence_command: 'studyloop study "SQL Windows" --mode co-study', ++ plan_refs: [{ plan_id: 'sql-windows', milestone_index: null }], ++ }, ++ energy_deferred_repairs: [ ++ { ++ plan_id: 'sql-windows', ++ plan_title: 'SQL Windows', ++ concept: 'window function', ++ topic: 'sql', ++ confidence: 'struggling', ++ energy_demand: 'high', ++ required_capability: 6, ++ energy_capability: 3, ++ reason: 'low energy carries 3/10; repairing a live struggle asks for at least 6/10', ++ }, ++ ], ++}; ++ ++test('deferredRepairNotes: one readable line per energy-deferred repair', () => { ++ const panel = todayPanel(); ++ panel.plan = DEFERRED_REPAIR_PAYLOAD; ++ ++ assert.deepEqual(panel.deferredRepairNotes(), [ ++ 'SQL Windows \u2014 repairing \u201cwindow function\u201d (struggling) waits for more energy ' ++ + '(asks for 6/10, low energy carries 3/10)', ++ ]); ++}); ++ ++test('deferredRepairNotes: a repair unrelated to any plan names no plan, and counts as plan context alone', () => { ++ const panel = todayPanel(); ++ panel.plan = { ++ ...NO_PLAN_PAYLOAD, ++ energy: 'low', ++ energy_deferred_repairs: [ ++ { ++ plan_id: null, ++ plan_title: null, ++ concept: 'decorators', ++ topic: 'python', ++ confidence: 'struggling', ++ energy_demand: 'high', ++ required_capability: 6, ++ energy_capability: 3, ++ reason: 'low energy carries 3/10', ++ }, ++ ], ++ }; ++ ++ assert.deepEqual(panel.deferredRepairNotes(), [ ++ 'Repairing \u201cdecorators\u201d (struggling) waits for more energy ' ++ + '(asks for 6/10, low energy carries 3/10)', ++ ]); ++ assert.equal(panel.hasPlanContext, true); ++}); ++ ++test('deferredRepairNotes: absent key renders nothing, before and after assignment', () => { ++ const panel = todayPanel(); ++ ++ assert.deepEqual(panel.deferredRepairNotes(), []); ++ ++ panel.plan = PLAN_PAYLOAD; ++ ++ assert.deepEqual(panel.deferredRepairNotes(), []); ++}); ++ ++test('a body-double primary starts in the Body Double view; every other action keeps its view', () => { ++ const panel = todayPanel(); ++ ++ assert.equal(panel.viewForAction(DEFERRED_REPAIR_PAYLOAD.primary), 'body-double'); ++ assert.equal(panel.viewForAction(PLAN_PAYLOAD.primary), 'study-session'); ++ assert.equal(panel.viewForAction(NO_PLAN_PAYLOAD.primary), 'flashcards'); ++}); ++ ++test('the Today card markup renders the deferred repairs beside the deferred milestones', () => { ++ const html = fs.readFileSync( ++ new URL('../../src/studyloop/web/static/index.html', import.meta.url), 'utf8', ++ ); ++ const start = html.indexOf('class="today-plan-notes"'); ++ const end = html.indexOf('</div>', html.indexOf('warningNotes()', start)); ++ const block = html.slice(start, end); ++ assert.match(block, /x-for="\(note, i\) in deferredRepairNotes\(\)" :key="'r' \+ i"/); ++ assert.match(block, /Deferred for energy: <span x-text="note">/); ++ const show = html.slice(html.lastIndexOf('x-show=', start), start); ++ assert.match(show, /deferredRepairNotes\(\)\.length > 0/, 'the notes block shows for a deferred repair alone'); ++}); +diff --git a/packages/studyloop/tests/test_learning_decision.py b/packages/studyloop/tests/test_learning_decision.py +index 347aa4ca..807a5743 100644 +--- a/packages/studyloop/tests/test_learning_decision.py ++++ b/packages/studyloop/tests/test_learning_decision.py +@@ -48,7 +48,7 @@ def test_no_data_returns_starter_recommendation(monkeypatch) -> None: + monkeypatch.setattr( + decision, + "_starter_candidate", +- lambda time_minutes: _candidate("starter", score=10), ++ lambda time_minutes, after_deferral=False: _candidate("starter", score=10), + ) + + plan = build_now_plan() +diff --git a/packages/studyloop/tests/test_now_plan_guidance.py b/packages/studyloop/tests/test_now_plan_guidance.py +index 2bd3d24d..4027f3d0 100644 +--- a/packages/studyloop/tests/test_now_plan_guidance.py ++++ b/packages/studyloop/tests/test_now_plan_guidance.py +@@ -335,7 +335,13 @@ def test_synthesizes_milestone_when_no_candidate_represents_it(monkeypatch) -> N + + + def test_preserves_one_plan_backed_action_when_energy_allows(monkeypatch) -> None: +- """Rule 8: ≥ 1 eligible plan-backed action in primary + alternates when energy permits.""" ++ """Rule 8: ≥ 1 eligible plan-backed action in primary + alternates when energy permits. ++ ++ Below the floor the milestone is deferred, never synthesised; since design §5 ++ the plan-backed slot rule 8 keeps is then the body-double proposal — sitting ++ with the plan asks for no energy the day cannot carry — and the primary and ++ first alternate stay the real, higher-ranked candidates. ++ """ + _plan( + "sql-windows", energy_floor=5, milestones=[Milestone("Frames", concepts=["window frame"])] + ) +@@ -350,7 +356,10 @@ def test_preserves_one_plan_backed_action_when_energy_allows(monkeypatch) -> Non + + low = build_now_plan(energy="low") + +- assert [rec.concept for rec in _all(low)] == ["due 0", "due 1", "due 2"] ++ assert [rec.concept for rec in _all(low)] == ["due 0", "due 1", "Sit with Sql Windows"] ++ assert low.alternates[1].source == "body_double" ++ assert low.alternates[1].plan_refs == (PlanRef("sql-windows", None),) ++ assert not any(rec.source.startswith("study_plan:") for rec in _all(low)) + assert [d.milestone_index for d in low.energy_deferred] == [0] + + +@@ -1153,3 +1162,309 @@ def test_completion_evidence_cap_keeps_the_counts_and_names_the_overflow(monkeyp + assert action.evidence[-1].startswith("… and ") + assert action.evidence[-1].endswith(" more") + assert f"{action.due_reviews} due reviews" in action.action ++ ++ ++# --------------------------------------------------------------------------- ++# T5.2 — item 5 (D-F): per-item energy demand for repair, and the body-doubling ++# floor. Design §5 with its three T5.1 amendments. The struggle collector runs ++# for real here — demand is derived in the collector, so injecting candidates ++# through ``_due_progress_candidates`` would bypass the very thing under test. ++# --------------------------------------------------------------------------- ++ ++ ++def _struggle( ++ concept: str, ++ *, ++ topic: str = "sql", ++ confidence: str = "struggling", ++ days_ago: int = 3, ++ teachback: int | None = None, ++) -> dict: ++ """One row as ``history.observations.rows`` projects it. ++ ++ ``last_seen`` is relative to the frozen clock. ++ """ ++ from datetime import timedelta ++ ++ seen = (FROZEN_NOW - timedelta(days=days_ago)).isoformat() ++ return { ++ "id": f"{topic}/{concept}", ++ "topic": topic, ++ "concept": concept, ++ "confidence": confidence, ++ "first_seen": seen, ++ "last_seen": seen, ++ "session_count": 1, ++ "notes": None, ++ "last_teachback_score": teachback, ++ } ++ ++ ++def _plant_struggles( ++ monkeypatch: pytest.MonkeyPatch, *rows: dict, due: tuple[_Candidate, ...] = () ++) -> None: ++ """Silence every collector except the struggle collector, which reads ``rows``.""" ++ from studyloop.history import observations ++ ++ real_collector = decision._struggle_candidates ++ _patch_collectors(monkeypatch, *due) ++ monkeypatch.setattr(decision, "_struggle_candidates", real_collector) ++ monkeypatch.setattr(observations, "rows", lambda conn: [dict(row) for row in rows]) ++ ++ ++def _row3_plan() -> None: ++ """Rubric row 3's plan: floor 5, milestone 0 done, milestone 1 ``Frames`` open.""" ++ _plan( ++ "sql-windows", ++ title="SQL Windows", ++ energy_floor=5, ++ milestones=[ ++ Milestone(title="Window basics", done=True, concepts=["window function"]), ++ Milestone(title="Frames", concepts=["window frame"]), ++ ], ++ ) ++ ++ ++def test_live_struggle_repair_defers_at_low_energy_like_new_work(monkeypatch) -> None: ++ """Rule 3 extended (design §5, amendment 1 + 2): repair carries a demand of its own. ++ ++ ``struggling`` seen within 14 days is ``high`` (asks for 6/10); ``struggling`` ++ older than that, or a row whose only signal is a weak teach-back, is ++ ``medium`` (4/10). Below the capability the repair is deferred like new ++ milestone work — listed, not ranked — in its own additive key, plan-related ++ or not; the milestone deferral beside it is untouched. ++ """ ++ _row3_plan() ++ _plant_struggles( ++ monkeypatch, ++ _struggle("window function", days_ago=3), # live, plan-related → high ++ _struggle("window frame", days_ago=20), # old, plan-related → medium ++ _struggle("decorators", topic="python", days_ago=1), # live, unrelated → high ++ _struggle("closures", topic="python", confidence="confident", days_ago=2, teachback=9), ++ ) ++ ++ low = build_now_plan(energy="low") ++ ++ deferred = {item.concept: item for item in low.energy_deferred_repairs} ++ assert set(deferred) == {"window function", "window frame", "decorators", "closures"} ++ assert not any(rec.concept in deferred for rec in _all(low)), "deferred repair is not ranked" ++ ++ live = deferred["window function"] ++ assert isinstance(live, decision.DeferredRepair) ++ assert (live.plan_id, live.plan_title, live.topic, live.confidence) == ( ++ "sql-windows", ++ "SQL Windows", ++ "sql", ++ "struggling", ++ ) ++ assert (live.energy_demand, live.required_capability, live.energy_capability) == ("high", 6, 3) ++ assert "3/10" in live.reason and "6/10" in live.reason ++ assert ( ++ deferred["window frame"].energy_demand, ++ deferred["window frame"].required_capability, ++ ) == ( ++ "medium", ++ 4, ++ ) ++ assert deferred["closures"].energy_demand == "medium", ( ++ "a weak teach-back alone is medium demand" ++ ) ++ assert (deferred["decorators"].plan_id, deferred["decorators"].plan_title) == (None, None) ++ assert deferred["decorators"].energy_demand == "high" ++ # The milestone deferral is what it was (rule 3's original half). ++ assert [(d.plan_id, d.milestone_index) for d in low.energy_deferred] == [("sql-windows", 1)] ++ ++ payload = low.to_json_dict() ++ assert [entry["concept"] for entry in payload["energy_deferred_repairs"]] == [ ++ item.concept for item in low.energy_deferred_repairs ++ ] ++ assert payload["energy_deferred_repairs"][0]["energy_demand"] in {"high", "medium"} ++ ++ # Medium energy (6/10) carries every demand class: nothing deferred, key absent. ++ medium = build_now_plan(energy="medium") ++ ++ assert medium.energy_deferred_repairs == () ++ assert "energy_deferred_repairs" not in medium.to_json_dict() ++ assert any(rec.concept == "window function" for rec in _all(medium)) ++ ++ ++def test_recovered_repair_stays_eligible_at_low_energy(monkeypatch) -> None: ++ """A ``learning`` row is ``low`` demand — the gentle review "repair is cheaper than ++ encoding" was always about — and stays eligible below the plan's floor with its ++ plan-related ref. Due recall is unaffected whatever its confidence says.""" ++ _row3_plan() ++ _plant_struggles(monkeypatch, _struggle("window function", confidence="learning", days_ago=2)) ++ ++ low = build_now_plan(energy="low") ++ ++ assert low.primary.concept == "window function" ++ assert low.primary.action_type == "teachback" ++ assert low.primary.plan_refs == (PlanRef("sql-windows", None),) ++ assert low.primary.metadata["energy_demand"] == "low" ++ assert low.energy_deferred_repairs == () ++ assert not any(rec.source == "body_double" for rec in _all(low)) ++ assert [d.milestone_index for d in low.energy_deferred] == [1] ++ ++ # A due row on a plan concept, even one recorded as struggling, is recall, ++ # not repair: it is never deferred and nothing is synthesised beside it. ++ import dataclasses ++ ++ due = dataclasses.replace( ++ _candidate("window frame", topic="sql", score=100), ++ metadata={"confidence": "struggling", "days_ago": 6}, ++ ) ++ _plant_struggles(monkeypatch, _struggle("window function", days_ago=3), due=(due,)) ++ ++ low = build_now_plan(energy="low") ++ ++ assert low.primary.concept == "window frame" ++ assert low.primary.plan_refs == (PlanRef("sql-windows", None),) ++ assert [d.concept for d in low.energy_deferred_repairs] == ["window function"] ++ assert not any(rec.source == "body_double" for rec in _all(low)) ++ ++ ++def test_body_double_candidate_is_synthesised_when_nothing_plan_related_fits(monkeypatch) -> None: ++ """Rubric row 3b's world: the plan's milestone is deferred and its only repair is a ++ live struggle, so nothing plan-related fits low energy. The engine proposes sitting ++ with the plan — a body-double session — naming what it stands in for, through the ++ session door (amendment 3), never the least-bad task.""" ++ _row3_plan() ++ _plant_struggles(monkeypatch, _struggle("window function", days_ago=3)) ++ ++ low = build_now_plan(energy="low") ++ ++ primary = low.primary ++ assert primary.source == "body_double" ++ assert primary.action_type == "conversation" ++ assert primary.plan_refs == (PlanRef("sql-windows", None),) ++ assert primary.evidence_command == 'studyloop study "SQL Windows" --mode co-study' ++ assert "Frames" in primary.reason and "window function" in primary.reason ++ assert low.starter is False ++ assert low.alternates == [] ++ assert [d.concept for d in low.energy_deferred_repairs] == ["window function"] ++ assert [d.milestone_index for d in low.energy_deferred] == [1] ++ assert decision.BODY_DOUBLE_BASE_SCORE < decision.MILESTONE_BASE_SCORE ++ ++ golden_keys = list(json.loads(GOLDEN.read_text(encoding="utf-8"))) ++ payload = low.to_json_dict() ++ assert list(payload) == [ ++ *golden_keys, ++ "active_plans", ++ "energy_deferred", ++ "energy_deferred_repairs", ++ ] ++ assert payload["primary"]["source"] == "body_double" ++ ++ ++def test_body_double_is_a_proposal_not_a_filter(monkeypatch) -> None: ++ """An unrelated real candidate still wins; the body-double proposal sits beneath it ++ as an alternate, base score below any real candidate's.""" ++ _row3_plan() ++ _plant_struggles( ++ monkeypatch, ++ _struggle("window function", days_ago=3), ++ due=(_candidate("decorators", topic="python", score=100),), ++ ) ++ ++ low = build_now_plan(energy="low") ++ ++ assert low.primary.concept == "decorators" ++ assert low.primary.plan_refs == () ++ assert [rec.source for rec in low.alternates] == ["body_double"] ++ assert low.alternates[0].score < low.primary.score ++ assert "Frames" in low.alternates[0].reason ++ ++ ++def test_body_double_is_never_synthesised_for_an_unready_plan(monkeypatch) -> None: ++ """An active-but-unready plan is matched but never synthesised (spec rule 8), and the ++ body double is a synthesis: with only a husk active and nothing plan-related fitting, ++ the engine proposes nothing to sit with — the warning already says repair it.""" ++ husk = StudyPlan( ++ plan_id="husk", ++ title="Husk", ++ status="active", ++ created="2026-08-01T00:00:00+00:00", ++ updated="2026-09-01T00:00:00+00:00", ++ topics=["sql"], ++ milestones=[Milestone(title="Frames", concepts=["window frame"])], ++ ) ++ store.create_plan(husk) # no mission, no success criteria: unready ++ _plant_struggles(monkeypatch, _struggle("window function", days_ago=3)) ++ ++ low = build_now_plan(energy="low") ++ ++ assert not any(rec.source == "body_double" for rec in _all(low)) ++ assert [d.concept for d in low.energy_deferred_repairs] == ["window function"] ++ assert low.starter is True ++ assert any("husk" in w for w in low.warnings) ++ ++ # A ready plan beside the husk: the proposal names the ready one only; rule 7 ++ # may still attach a `(husk, None)` ref because the topics match. ++ _row3_plan() ++ ++ low = build_now_plan(energy="low") ++ ++ assert low.primary.source == "body_double" ++ assert low.primary.concept == "Sit with SQL Windows" ++ assert PlanRef("sql-windows", None) in low.primary.plan_refs ++ assert low.primary.evidence_command == 'studyloop study "SQL Windows" --mode co-study' ++ assert "Husk" not in low.primary.reason ++ ++ ++def test_body_double_never_appears_without_an_active_plan(monkeypatch) -> None: ++ """No active plan, no plan to sit with: the deferral still happens (plan-independent, ++ ``plan_id`` ``None``), the golden world stays untouched, and a non-active plan is ++ not an active plan.""" ++ _plant_struggles(monkeypatch, _struggle("decorators", topic="python", days_ago=1)) ++ ++ low = build_now_plan(energy="low") ++ ++ assert not any(rec.source == "body_double" for rec in _all(low)) ++ assert [(d.concept, d.plan_id, d.plan_title) for d in low.energy_deferred_repairs] == [ ++ ("decorators", None, None) ++ ] ++ # Every real candidate was deferred: the starter stands in, and says why. ++ assert low.starter is True ++ assert "defer" in low.primary.reason.lower() ++ ++ _plan("draft-plan", status="draft") ++ ++ low = build_now_plan(energy="low") ++ ++ assert not any(rec.source == "body_double" for rec in _all(low)) ++ assert "active_plans" not in low.to_json_dict() ++ ++ ++def test_cli_now_and_recap_render_deferred_repairs_and_the_body_double_door( ++ monkeypatch, ++) -> None: ++ """Amendment 2's "readable off the top" rule: each renderer gains one line per ++ deferred repair, and a body-double primary shows its door, not "record evidence".""" ++ from click.testing import CliRunner ++ ++ from studyloop.cli import cli ++ from studyloop.learning import recap ++ ++ _row3_plan() ++ _plant_struggles(monkeypatch, _struggle("window function", days_ago=3)) ++ ++ rich = CliRunner().invoke(cli, ["now", "--energy", "low"]) ++ as_json = CliRunner().invoke(cli, ["now", "--energy", "low", "--json"]) ++ ++ assert rich.exit_code == 0, rich.output ++ flat = " ".join(rich.output.split()) ++ assert "Deferred for energy" in flat and "Frames" in flat # the milestone line stays ++ assert "window function" in flat and "6/10" in flat # …and the repair has its own line ++ assert "Sit with the plan" in flat ++ assert "--mode co-study" in flat ++ assert "Record evidence" not in flat ++ assert as_json.exit_code == 0, as_json.output ++ payload = json.loads(as_json.output) ++ assert payload["primary"]["source"] == "body_double" ++ assert payload["energy_deferred_repairs"][0]["concept"] == "window function" ++ assert payload["energy_deferred_repairs"][0]["required_capability"] == 6 ++ ++ context = recap._plan_context(build_now_plan(energy="low")) ++ ++ assert "Frames" in context ++ assert "window function" in context and "6 of 10" in context +``` + +## 4. Rubric row 3b as written (verbatim from `receipts/now-rubric-2026-09-16.md`) + +| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered). | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason *"Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."*; no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone. | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 < any real candidate — a proposal, not a filter) carrying the co-study session door. | **PENDING** — owner: (a) would you sit with the plan rather than repair the live struggle today? (b) is the body-double proposal right to sit beneath the unrelated due recall? (c) is the gentle teach-back on a recovered concept one you would do at low energy? | + +## 5. The control receipt (verbatim) + +# Full-suite matched control — item 5 (D-F) — 2026-09-19 + +Two full `packages/studyloop/tests` runs in parallel, same machine, same +sandbox, `-q -p no:cacheprovider -rfE`: + +| Tree | Worktree | Result | +| --- | --- | --- | +| **item 5** (`feat/energy-demand-body-double`, RED `ef319a7b` + GREEN working tree) | `studyloop-wt/item5` | 30 failed, **5126 passed**, 4 skipped, 804 deselected, 14 errors (10:17) | +| **control** (`main` `4f8e3e0f`, detached) | `studyloop-wt/ctrl-item5` | 30 failed, 5120 passed, 4 skipped, 804 deselected, 14 errors (10:24) | + +Both worktrees were `uv sync --all-packages --group dev` and each proved to +import `studyloop` from its own tree before the run. + +#### (receipt) Sorted failing-id sets + +- item5 ∖ control = **∅** — zero regressions. +- control ∖ item5 = **∅** — nothing item 5 fixed by accident, and the six + new tests account for the passed-count difference (+6). +- item5 ∖ committed environmental set (`full-suite-control-item4-2026-09-18.md`, + 44 ids + item 4's seven then-REDs) = **∅**. The seven ids on the other side of + that comparison are item 4's REDs, green since `82293293`. + +The 44 shared ids are the sandbox-environmental set the item-4 receipt lists +by name (journeys world guards, acceptance isolation, second-brain CLI/doctor, +harness-matrix live mechanics, obsidian vault isolation, fresh-install scope); +unchanged here, byte for byte. + +#### (receipt) Scoped gates on the same tree + +- `test_now_plan_guidance.py` 46/46 (six REDs flipped; golden `now_plan_no_active.json` byte-identical); + `test_learning_decision.py` 5/5 (one stub updated to the starter's new keyword). +- JS `node --test packages/studyloop/tests/js/*.test.js` 144/144 (+5). +- e2e `test_journey_study_plan.py` + `test_plans_api.py` 20/20 (browser). +- `test_docs_plan_integration_contract.py` + `test_ci_workflow_contract.py` 39/39. +- `mkdocs build --strict` exit 0; `openspec validate plan-integration-followons` valid. +- ruff check / ruff format --check / pyright: clean on every touched file. + +## 6. Reference facts you may rely on (verified on `6d5a2d0e`) + +- `ENERGY_CAPABILITY = {"low": 3, "medium": 6, "high": 10}` (unchanged). New: `ENERGY_DEMAND_CAPABILITY = + {"high": 6, "medium": 4, "low": 0}`, `LIVE_STRUGGLE_DAYS = 14`, `BODY_DOUBLE_BASE_SCORE = 30`, + `BODY_DOUBLE_SOURCE = "body_double"`. `MILESTONE_BASE_SCORE = 48`, `PLAN_RELATED_BIAS = 12` (unchanged). +- Collector base scores (unchanged): due progress `100 + min(days_ago, 30)` (+35 struggling, +15 learning, +25 + weak teach-back); struggle repair 82 (`struggling`, hands-on) or 70 (else, teachback) `+ max(0, 14 − score) × 3`; + due cards `96 + min(due_count, 20)`; continuity 58; transfer 52; practice 48; starter 10. +- Scoring adjustments (unchanged): low energy −14 for hands-on/visual, −28 for a topic switch from the last + session; modality match +18 (`recall` matches `recall`/`teachback`; `conversation` matches + `conversation`/`teachback`); plan bias +12 when `plan_refs` or a key match. So in row 3b reading (a) the body + double scores 30 + 12 = 42; in (c) the `learning` teachback scores 70 + 12 + 18 = 100. +- `history.observations.rows(conn)` projects one row per subject with `confidence` (the most conservative + current report), `last_seen = max(recorded_at of current reports)` (`time_basis: assessment_recorded_at`), + `last_teachback_score` (only when exactly one current report), `session_count`. Legacy databases fall back to + `SELECT * FROM study_progress`, whose `last_seen` is the column of that name. +- `_struggle_candidates` keeps a row when `confidence in ("struggling", "learning")` or `last_teachback_score < + 14`; at most 12 rows, sorted `struggling` first then by teach-back then by `last_seen` desc. +- `_dedupe` runs after scoring; the deferral runs before rule 6 and before scoring, so a due row and a struggle + row on the same concept are two candidates at deferral time — the struggle one may be deferred while the due + one is ranked (the CLI then prints the concept twice: as the primary and in a "Deferred for energy" line). +- `_PlanContext.matchable` = every active plan except fully-checked ones, **including active-but-unready** + plans; `synthesise` = ready plans with a next milestone within capability. `attach_refs` (rule 7) references + every matchable plan whose keys equal the candidate's — a body double whose topic equals a husk's topic is + referenced to the husk too (`(husk, None)`), though it never names the husk (correction `8e9cbdf5`). +- The Today card's `startAction` navigates only (`Alpine.store('nav').go(view)`); the Body Double view opens + with its own picker. The body-double proposal's plan title is **not** pre-filled into that picker on the Web + door; the CLI door carries it (`studyloop study "<title>" --mode co-study`). Not done in this range. +- `evidence_command` is the JSON key every renderer reads for the primary's command; for the body double it + carries a session door, not an evidence write. The CLI labels it "Sit with the plan"; the key name is unchanged. +- `metadata["energy_demand"]` is now present on every struggle-collector candidate's JSON at every energy + (additive field inside `metadata`); the golden world has no struggle candidates. + +## 7. Deliverables — numbered H2 sections, in this order + +1. **Verdict:** ACCEPT / ACCEPT-WITH-CORRECTIONS / REJECT for item 5 as the tree to merge to `main` and the tree + the owner scores row 3b against — with the single sentence that decides it. +2. **Findings**, each with severity 🔴 defect (wrong behaviour or a bug), 🟡 must-fix-before-merge (design/contract + violation, missing test, unsafe pattern), 🔵 should-fix, 💡 note. For each: file:line or function, what is wrong, + why it matters, the concrete fix, and the RED test that would pin it (name it). Check specifically: + - (a) **Scope of the deferral.** Design §5 decision 1 makes repair deferral plan-independent: a learner with + **no plan** and a live struggle at low energy now gets the starter plus `energy_deferred_repairs` where they + got the hands-on repair. D-5's "no active plan → pre-#10 payload byte for byte" was re-worded to "…and + nothing deferred". Is this D-F's intent (the finding is about RSD, not plans) or scope creep into the + no-plan learner's experience? If creep: is gating the repair half on an active plan the fix, or a different + no-plan floor (what would a no-plan learner with only live struggles be offered)? + - (b) **Demand derivation.** `_energy_demand`: `learning → low`; any non-`struggling` row the collector kept + (weak teach-back on `confident`/`mastered`) → `medium`; `struggling` ≤ 14 days → `high`, else `medium`; an + unparseable/missing `last_seen` on `struggling` → `high`. Is 14 days the right live window, and is + `assessment_recorded_at` (when the struggle was *recorded*) the right recency basis versus when it was last + *seen* in a session? Does a `struggling` row with a weak teach-back deserve `high` regardless of age? Is + `medium` (4/10) for a weak teach-back alone right when nothing in `ENERGY_CAPABILITY` sits between 3 and + 6 — i.e. `medium` and `high` demand are behaviourally identical at the three energy levels; is the class + worth carrying, and if so should the JSON say so? + - (c) **Deferral mechanics.** `_defer_repairs` keys on `metadata["energy_demand"]` presence; due recall is + never deferred even when `confidence == "struggling"` (pinned). Same concept due *and* struggling: the due + row is ranked and the repair is deferred — right, or should the deferred line be suppressed when the same + concept is the primary? A deferred repair no longer "represents" a milestone (rule 6 then synthesises the + milestone conversation) — right, or does that re-introduce work the energy cannot carry through a side door? + - (d) **Body double as candidate.** Base 30 (+12 = 42): below practice 48 and milestone 48 at base, but a + hands-on practice task at low energy scores 48 − 14 = 34 < 42, so the proposal outranks it. Design says + "any real candidate outranks it" at base; is the post-adjustment inversion acceptable (low energy penalises + hands-on deliberately) or a 🟡? `concept = "Sit with <title>"` is used as a match key (`_candidate_keys`): + any collision risk? `plan_refs` for every ready matchable plan, but the CLI door names only the **first** + plan's title — right for two plans? `estimated_minutes` default 25. + - (e) **Rule 8 interaction** (decision 2, the one existing pin changed): with four unrelated due items at low + energy the second alternate is now the body-double proposal instead of a third unrelated due item. Right + ("advertises no work the energy cannot carry") or a filter in disguise (a real candidate lost its slot)? + - (f) **Honest starter** (decision 3): after a deferral the starter's reason changes; the starter's *action* + ("one tiny recall loop" on the first configured topic) is unchanged — is that a defensible offer for a + no-plan learner with only live struggles, or should the deferred repair's own gentle form (a `recall` on the + same concept) be synthesised instead? + - (g) **Renderers.** CLI: "Deferred for energy: … repairing “x” (confidence) asks for N/10; low energy carries + 3/10" and "Sit with the plan:" replacing "Record evidence:" for a body-double primary. Recap: a sentence per + deferred repair, reachable only through `_plan_context` (see hard rules). Today card: + `deferredRepairNotes()`, `viewForAction` → `body-double`; the plan title is **not** pre-filled into the Body + Double picker — 🟡 or 🔵? Is the JSON key `evidence_command` acceptable for a door, or should the body + double carry a distinct field? MCP `get_next_action` returns `to_json_dict()` unchanged — is the new key + disclosed where an agent reads (tool docstring, persona)? + - (h) **Spec delta.** A MODIFIED requirement restating "The now engine is plan-aware with tested ranking + rules" with item 5 inline. Does every sentence match the code as diffed (rule 2's demand classes, the + body-double clause after rule 5, rule 7's slot, the qualified byte-for-byte sentence)? Any scenario that + the tests do not actually pin? + - (i) **Tests.** Six new Python tests drive the real struggle collector over patched `observations.rows` + (rows built by `_struggle`). Do they pin the boundary (14 days exactly), the unparseable `last_seen` path, + the two-plan body double, the medium-vs-high indistinguishability? Is anything asserted only via a + substring that a wording change would silently pass? + - (j) **Rubric row 3b as written** (§4): are the three readings the right questions to put to the owner, and + do they cover the finding's two halves? Anything the owner should be asked that the row omits? +3. **Refutations:** any claim in §0–§6 you believe is false or not established by the brief — say which and why. +4. **Gate:** the shortest list of corrections that would turn your verdict into ACCEPT, each with its RED test + name; or "none". From d1935256de6aa8924a60c2654cef18551299fb56 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sat, 19 Sep 2026 13:27:56 +0100 Subject: [PATCH 09/32] fix(now): quote every offered command as one literal shell argument (review 7, F1) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Astra's red, reproduced with a real /bin/sh: the body-double door built `studyloop study "<title>" --mode co-study` by replacing `"` with `\"` — presentation, not quoting — so a plan title such as `SQL $(touch pwned) Windows` executed its substitution when the offered command was pasted. `_evidence_command` had the same shape for every concept and topic the engine offers. One helper, `_shell_word`: plain text keeps the double-quoted form the golden pins byte for byte; text carrying `"`, `\`, `$`, a backtick or `!` is `shlex.quote`d. Both builders use it. Pinned by test_body_double_command_preserves_title_as_one_literal_shell_argument, which runs each command through /bin/sh against a stub `studyloop` that records argv: seven titles round-trip literally and no side effect fires. Stash-proved: 3 of 7 fail without the fix. Golden and every exact-command pin unchanged. --- .../src/studyloop/learning/decision.py | 32 ++++++++--- .../studyloop/tests/test_now_plan_guidance.py | 55 +++++++++++++++++++ 2 files changed, 79 insertions(+), 8 deletions(-) diff --git a/packages/studyloop/src/studyloop/learning/decision.py b/packages/studyloop/src/studyloop/learning/decision.py index 2f0960e30..c4fb28239 100644 --- a/packages/studyloop/src/studyloop/learning/decision.py +++ b/packages/studyloop/src/studyloop/learning/decision.py @@ -19,6 +19,7 @@ import dataclasses import logging +import shlex import sqlite3 from dataclasses import asdict, dataclass, field from datetime import UTC, datetime @@ -327,18 +328,34 @@ def _action_for_review(review_type: str, confidence: str | None) -> ActionType: return "recall" +_SHELL_SPECIAL = frozenset('"\\$`!') + + +def _shell_word(text: str) -> str: + """One shell argument for a command the engine *offers* the learner to run. + + Plain text keeps the double-quoted form every existing command uses (the + golden pins it byte for byte). Text carrying a character the shell reads + inside double quotes — ``"``, ``\\``, ``$``, a backtick, ``!`` — is + ``shlex.quote``d instead, so a plan title or concept like + ``SQL $(rm -rf ~) Windows`` reaches ``studyloop`` as one literal argument + (council review 7, F1: the old ``\\"`` replacement was presentation, not + quoting, and a pasted command executed the substitution). + """ + if _SHELL_SPECIAL.isdisjoint(text): + return f'"{text}"' + return shlex.quote(text) + + def _evidence_command(action_type: ActionType, concept: str, topic: str, source: str) -> str: - safe_concept = concept.replace('"', '\\"') - safe_topic = topic.replace('"', '\\"') if action_type == "teachback": return ( - f'studyloop teachback "{safe_concept}" -t "{safe_topic}" ' + f"studyloop teachback {_shell_word(concept)} -t {_shell_word(topic)} " '--score "3,3,3,3,3" --type structured' ) if action_type == "hands-on" and source.endswith(".json"): - safe_source = source.replace('"', '\\"') - return f'studyloop practice verify "{safe_source}" --task 1 --notes "what passed?"' - return f'studyloop progress "{safe_concept}" -t "{safe_topic}" -c learning' + return f'studyloop practice verify {_shell_word(source)} --task 1 --notes "what passed?"' + return f"studyloop progress {_shell_word(concept)} -t {_shell_word(topic)} -c learning" def _connect_progress_db(): @@ -1236,7 +1253,6 @@ def _body_double_candidate( ] + [f"repair of “{d.concept}”" for d in deferred_repairs] deferred_note = f" — deferred: {'; '.join(items)}" if items else "" topic = first.topics[0] if first.topics else "study" - safe_title = first.title.replace('"', '\\"') return _Candidate( concept=f"Sit with {first.title}" if len(named) == 1 else "Sit with your plans", topic=topic, @@ -1248,7 +1264,7 @@ def _body_double_candidate( action_type="conversation", estimated_minutes=_estimate_minutes("conversation", time_minutes, 25), source=BODY_DOUBLE_SOURCE, - evidence_command=f'studyloop study "{safe_title}" --mode co-study', + evidence_command=f"studyloop study {_shell_word(first.title)} --mode co-study", score=BODY_DOUBLE_BASE_SCORE, metadata={ "plan_id": first.plan_id, diff --git a/packages/studyloop/tests/test_now_plan_guidance.py b/packages/studyloop/tests/test_now_plan_guidance.py index 4027f3d0a..77dc78010 100644 --- a/packages/studyloop/tests/test_now_plan_guidance.py +++ b/packages/studyloop/tests/test_now_plan_guidance.py @@ -1468,3 +1468,58 @@ def test_cli_now_and_recap_render_deferred_repairs_and_the_body_double_door( assert "Frames" in context assert "window function" in context and "6 of 10" in context + + +@pytest.mark.parametrize( + "title", + [ + "SQL Windows", + 'SQL "Windows"', + "SQL $(touch pwned) Windows", + "SQL `touch pwned` Windows", + "SQL $HOME Windows", + "back\\slash Windows", + "it's Windows", + ], +) +def test_body_double_command_preserves_title_as_one_literal_shell_argument( + monkeypatch, tmp_path: Path, title: str +) -> None: + """Council review 7, F1 (astra 🔴): the command the engine *offers* must reach + ``studyloop`` as one literal argument when pasted into a POSIX shell — no + expansion, no substitution, no extra command. Proved with a real ``sh`` and a + stub ``studyloop`` on PATH that records its argv.""" + import subprocess + + _plan( + "hostile", + title=title, + energy_floor=5, + milestones=[Milestone(title="Frames", concepts=["window frame"])], + ) + _plant_struggles(monkeypatch, _struggle("window frame", days_ago=3)) + + low = build_now_plan(energy="low") + + assert low.primary.source == "body_double" + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + argv_file = tmp_path / "argv.txt" + stub = bin_dir / "studyloop" + stub.write_text(f'#!/bin/sh\nprintf "%s\\n" "$@" > "{argv_file}"\n', encoding="utf-8") + stub.chmod(0o755) + subprocess.run( + ["/bin/sh", "-c", low.primary.evidence_command], + check=True, + env={"PATH": f"{bin_dir}:/usr/bin:/bin", "HOME": str(tmp_path)}, + cwd=tmp_path, + timeout=10, + ) + + assert argv_file.read_text(encoding="utf-8").split("\n")[:4] == [ + "study", + title, + "--mode", + "co-study", + ] + assert not (tmp_path / "pwned").exists(), "the title's substitution must never run" From c1d7f2f2169a6c9db8cccc0f63db6c3d9b3eda21 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sat, 19 Sep 2026 13:28:49 +0100 Subject: [PATCH 10/32] fix(now): the milestone-deferral line no longer promises repair (review 7, F5) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Astra's must-fix: in rubric row 3b's own fixture the engine's DeferredMilestone.reason and the CLI's "Deferred for energy" line ended "plan-related review and repair stay available" one line above the line saying the repair is deferred. Both now promise only what rule 3 still guarantees — due recall and gentle review. Pinned as whole sentences by test_cli_milestone_deferral_does_not_promise_live_repair. --- packages/studyloop/src/studyloop/cli/_now.py | 2 +- .../src/studyloop/learning/decision.py | 4 +-- .../studyloop/tests/test_now_plan_guidance.py | 31 +++++++++++++++++++ 3 files changed, 34 insertions(+), 3 deletions(-) diff --git a/packages/studyloop/src/studyloop/cli/_now.py b/packages/studyloop/src/studyloop/cli/_now.py index 48cf89e6b..d751a0ed7 100644 --- a/packages/studyloop/src/studyloop/cli/_now.py +++ b/packages/studyloop/src/studyloop/cli/_now.py @@ -70,7 +70,7 @@ def _render_plan(plan) -> None: f"[yellow]Deferred for energy:[/yellow] {escape(deferred.plan_title)} — " f"milestone {deferred.milestone_index + 1} “{escape(deferred.title)}” needs " f"energy {deferred.energy_floor}/10; {plan.energy} energy carries " - f"{deferred.energy_capability}/10. Plan-related review and repair stay available." + f"{deferred.energy_capability}/10. Due recall and gentle review stay available." ) # One line per deferred repair (design §5, amendment 2): its own key, its # own sentence — a repair has no milestone number to print. diff --git a/packages/studyloop/src/studyloop/learning/decision.py b/packages/studyloop/src/studyloop/learning/decision.py index c4fb28239..8a632644c 100644 --- a/packages/studyloop/src/studyloop/learning/decision.py +++ b/packages/studyloop/src/studyloop/learning/decision.py @@ -1031,8 +1031,8 @@ def build(cls, guidance: ActiveGuidance | None, *, energy: EnergyLevel) -> _Plan reason=( f"{energy} energy carries {capability}/10; " f"{summary.title!r} asks for at least " - f"{plan.energy_floor}/10 — plan-related review " - "and repair stay available" + f"{plan.energy_floor}/10 — due recall and gentle " + "review stay available" ), ) ) diff --git a/packages/studyloop/tests/test_now_plan_guidance.py b/packages/studyloop/tests/test_now_plan_guidance.py index 77dc78010..c2f7a9728 100644 --- a/packages/studyloop/tests/test_now_plan_guidance.py +++ b/packages/studyloop/tests/test_now_plan_guidance.py @@ -1523,3 +1523,34 @@ def test_body_double_command_preserves_title_as_one_literal_shell_argument( "co-study", ] assert not (tmp_path / "pwned").exists(), "the title's substitution must never run" + + +def test_cli_milestone_deferral_does_not_promise_live_repair(monkeypatch) -> None: + """Council review 7, F5 (astra 🟡): in row 3b's own fixture the milestone line used to end + "Plan-related review and repair stay available" one line above the line saying the + repair is deferred. Both the engine's reason and the CLI's sentence now promise only + what rule 3 still guarantees — due recall and gentle review — asserted as whole + sentences, not by the presence of a word.""" + from click.testing import CliRunner + + from studyloop.cli import cli + + _row3_plan() + _plant_struggles(monkeypatch, _struggle("window function", days_ago=3)) + + low = build_now_plan(energy="low") + rich = CliRunner().invoke(cli, ["now", "--energy", "low"]) + + assert rich.exit_code == 0, rich.output + flat = " ".join(rich.output.split()) + assert ( + "Deferred for energy: SQL Windows — milestone 2 “Frames” needs energy 5/10; " + "low energy carries 3/10. Due recall and gentle review stay available." + ) in flat + assert ( + "Deferred for energy: SQL Windows — repairing “window function” (struggling) asks for " + "6/10; low energy carries 3/10. Due recall and gentle review stay available." + ) in flat + assert "repair stay available" not in flat + assert "repair stay available" not in low.energy_deferred[0].reason + assert low.energy_deferred[0].reason.endswith("due recall and gentle review stay available") From 3347567e39ec65207e828e5005cde23f412ec88a Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sat, 19 Sep 2026 13:32:46 +0100 Subject: [PATCH 11/32] feat(web): the Today card hands a body-double proposal's plan to the Body Double view (review 7, F7) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Astra 🔵 / grok 🔵 / qwen 🟡: navigating to the Body Double view lost the proposal's context — the learner had to pick what to sit with again while the CLI door carried the plan title. `startAction` now dispatches `body-double-request` {activity: <plan title>, energy} before navigating — the event-not-storage shape `today-resume` already uses — and `bodyDoubleSession.init` adopts it as the view's activity and energy band. Nothing starts on its own; the learner still presses start. Pinned by two JS tests (dispatch + navigation order; title resolution with fallbacks). --- .../src/studyloop/web/static/components.js | 9 +++++ .../web/static/js/components/today-panel.js | 35 ++++++++++++++++++- 2 files changed, 43 insertions(+), 1 deletion(-) diff --git a/packages/studyloop/src/studyloop/web/static/components.js b/packages/studyloop/src/studyloop/web/static/components.js index 98e68e7e7..eb6786930 100644 --- a/packages/studyloop/src/studyloop/web/static/components.js +++ b/packages/studyloop/src/studyloop/web/static/components.js @@ -3190,6 +3190,15 @@ function bodyDoubleSession() { this.startError = ''; } }); + /* Today-card handoff for a body-double proposal (council review 7, F7): + the engine named a plan to sit with; open on it rather than blank. + Event, not storage, for the reason today-resume is. Nothing starts. */ + window.addEventListener('body-double-request', (event) => { + const detail = (event && event.detail) || {}; + if (detail.activity) this.activity = String(detail.activity); + const bands = { low: 3, medium: 5, high: 8 }; + if (detail.energy && bands[detail.energy]) this.energy = bands[detail.energy]; + }); this.focusCollapsed = localStorage.getItem('bd.focus.collapsed') === 'true'; this.captureCollapsed = localStorage.getItem('bd.capture.collapsed') === 'true'; await this.refreshFocus(); diff --git a/packages/studyloop/src/studyloop/web/static/js/components/today-panel.js b/packages/studyloop/src/studyloop/web/static/js/components/today-panel.js index 0e87dd534..706e7f241 100644 --- a/packages/studyloop/src/studyloop/web/static/js/components/today-panel.js +++ b/packages/studyloop/src/studyloop/web/static/js/components/today-panel.js @@ -114,7 +114,26 @@ export function todayPanel() { }, startAction(rec) { - Alpine.store('nav').go(this.viewForAction(rec)); + const view = this.viewForAction(rec); + if (view === 'body-double') { + /* Carry the proposal's context to the Body Double view (council review + 7, F7): the same event-not-storage handoff `today-resume` uses, so + the picker opens on the plan the engine named instead of blank. The + view starts nothing on its own; the learner still presses start. */ + window.dispatchEvent(new CustomEvent('body-double-request', { + detail: { activity: this.bodyDoubleActivity(rec), energy: this.plan && this.plan.energy }, + })); + } + Alpine.store('nav').go(view); + }, + + /* What a body-double proposal asks the learner to sit with: the named + plan's title, or the proposal's own concept when the payload lists no + plan for it. */ + bodyDoubleActivity(rec) { + const planId = rec && rec.metadata && rec.metadata.plan_id; + const plan = planId ? this._activePlan(planId) : null; + return (plan && plan.title) || (rec && rec.concept) || ''; }, /* The view an action starts in. A body-double proposal (design §5) is a @@ -218,6 +237,20 @@ export function todayPanel() { return ((this.plan && this.plan.warnings) || []).map((w) => String(w)); }, + /* The notes block's label. "Your plans" once any note involves a plan; a + learner with no plan whose live struggle was deferred (design §5 + decision 1) has no plan to be told about — the block is what today set + aside (council review 7, grok). */ + planNotesLabel() { + const plans = (this.plan && this.plan.active_plans) || []; + const repairs = (this.plan && this.plan.energy_deferred_repairs) || []; + const planInvolved = plans.length > 0 + || this.deferredNotes().length > 0 + || this.completionNotes().length > 0 + || repairs.some((r) => r.plan_id); + return planInvolved ? 'Your plans' : 'Set aside today'; + }, + get hasPlanContext() { return ( this.planLabel(this.plan && this.plan.primary) !== '' From f814dc364ddca260b21c97847c87f3b5a35672ca Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sat, 19 Sep 2026 13:32:48 +0100 Subject: [PATCH 12/32] fix(web): the Today notes block is not "Your plans" for a learner without one (review 7, grok) A no-plan learner whose live struggle was deferred at low energy saw a block headed "Your plans" holding only the deferred line. The label is now planNotesLabel(): "Your plans" once any note involves a plan, else "Set aside today". Pinned in today-panel-plan.test.js. --- .../src/studyloop/web/static/index.html | 2 +- .../tests/js/today-panel-plan.test.js | 59 +++++++++++++++++++ 2 files changed, 60 insertions(+), 1 deletion(-) diff --git a/packages/studyloop/src/studyloop/web/static/index.html b/packages/studyloop/src/studyloop/web/static/index.html index e7885fea8..f84fc69d3 100644 --- a/packages/studyloop/src/studyloop/web/static/index.html +++ b/packages/studyloop/src/studyloop/web/static/index.html @@ -1106,7 +1106,7 @@ <h3 class="today-concept" x-text="plan?.primary?.concept"></h3> the single action card by that class. --> <div x-show="!loading && plan && (deferredNotes().length > 0 || deferredRepairNotes().length > 0 || completionNotes().length > 0 || warningNotes().length > 0)" class="today-plan-notes"> - <p class="today-parked-label">Your plans</p> + <p class="today-parked-label" x-text="planNotesLabel()">Your plans</p> <template x-for="(note, i) in deferredNotes()" :key="'d' + i"> <p class="today-meta">Deferred for energy: <span x-text="note"></span></p> </template> diff --git a/packages/studyloop/tests/js/today-panel-plan.test.js b/packages/studyloop/tests/js/today-panel-plan.test.js index 004c20692..5fd7dac99 100644 --- a/packages/studyloop/tests/js/today-panel-plan.test.js +++ b/packages/studyloop/tests/js/today-panel-plan.test.js @@ -365,3 +365,62 @@ test('the Today card markup renders the deferred repairs beside the deferred mil const show = html.slice(html.lastIndexOf('x-show=', start), start); assert.match(show, /deferredRepairNotes\(\)\.length > 0/, 'the notes block shows for a deferred repair alone'); }); + +test('starting a body-double primary hands its plan to the Body Double view, then navigates (review 7, F7)', () => { + const events = []; + const gone = []; + const savedWindow = globalThis.window; + const savedAlpine = globalThis.Alpine; + const savedEvent = globalThis.CustomEvent; + globalThis.window = { dispatchEvent(e) { events.push(e); } }; + globalThis.CustomEvent = class { constructor(type, init) { this.type = type; this.detail = init && init.detail; } }; + globalThis.Alpine = { store() { return { go(view) { gone.push(view); } }; } }; + try { + const panel = todayPanel(); + panel.plan = DEFERRED_REPAIR_PAYLOAD; + panel.plan.primary.metadata = { plan_id: 'sql-windows', deferred_milestones: 1, deferred_repairs: 1 }; + + panel.startPrimary(); + + assert.deepEqual(gone, ['body-double']); + assert.equal(events.length, 1); + assert.equal(events[0].type, 'body-double-request'); + assert.deepEqual(events[0].detail, { activity: 'SQL Windows', energy: 'low' }); + + events.length = 0; gone.length = 0; + panel.startAction(PLAN_PAYLOAD.primary); + + assert.deepEqual(gone, ['study-session']); + assert.equal(events.length, 0, 'an ordinary action dispatches nothing'); + } finally { + globalThis.window = savedWindow; + globalThis.Alpine = savedAlpine; + globalThis.CustomEvent = savedEvent; + } +}); + +test('bodyDoubleActivity: the named plan\u2019s title, else the proposal\u2019s concept', () => { + const panel = todayPanel(); + panel.plan = DEFERRED_REPAIR_PAYLOAD; + + assert.equal(panel.bodyDoubleActivity({ concept: 'Sit with SQL Windows', metadata: { plan_id: 'sql-windows' } }), 'SQL Windows'); + assert.equal(panel.bodyDoubleActivity({ concept: 'Sit with your plans', metadata: { plan_id: 'missing' } }), 'Sit with your plans'); + assert.equal(panel.bodyDoubleActivity(null), ''); +}); + +test('planNotesLabel: "Your plans" when a plan is involved, "Set aside today" for a no-plan deferred repair (review 7, grok)', () => { + const panel = todayPanel(); + panel.plan = DEFERRED_REPAIR_PAYLOAD; + assert.equal(panel.planNotesLabel(), 'Your plans'); + + panel.plan = { + ...NO_PLAN_PAYLOAD, + energy: 'low', + energy_deferred_repairs: [ + { plan_id: null, plan_title: null, concept: 'decorators', topic: 'python', confidence: 'struggling', + energy_demand: 'high', required_capability: 6, energy_capability: 3, reason: 'r' }, + ], + }; + assert.equal(panel.planNotesLabel(), 'Set aside today'); + assert.equal(panel.hasPlanContext, true); +}); From 7194b66de92d6bdc31c8c97751be6726fc2ee806 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sat, 19 Sep 2026 13:37:55 +0100 Subject: [PATCH 13/32] test(now): pin the item-5 boundaries the council asked for; tolerate a null last_seen (review 7, F4) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Five pins astra (F4) and grok (🔵) named: the 14-day live window is inclusive (14 → high, 15 → medium) and a null or unparseable `last_seen` on a struggling row reads as live; `learning` is low demand even with a weak teach-back while an old struggle with one stays medium; the capability matrix (low rejects medium+high, carries low; medium/high carry all); a deferred sole representative lets rule 6 synthesise the eligible milestone's conversation, never a hands-on or a body double (design §5 decision 5); two ready plans yield one proposal with refs in plan order, both milestones and repairs named, the door on the first title. The boundary pin surfaced a pre-existing crash: `_struggle_candidates` sorts rows by `row["last_seen"]`, and a legacy study_progress row with a NULL last_seen raised TypeError, losing every struggle candidate. It now sorts as the oldest (`row.get("last_seen") or ""`) and the demand derivation reads it as live, the cautious side. --- .../src/studyloop/learning/decision.py | 6 +- .../studyloop/tests/test_now_plan_guidance.py | 193 ++++++++++++++++++ 2 files changed, 198 insertions(+), 1 deletion(-) diff --git a/packages/studyloop/src/studyloop/learning/decision.py b/packages/studyloop/src/studyloop/learning/decision.py index 8a632644c..58adf7334 100644 --- a/packages/studyloop/src/studyloop/learning/decision.py +++ b/packages/studyloop/src/studyloop/learning/decision.py @@ -438,7 +438,11 @@ def _struggle_candidates(time_minutes: int) -> list[_Candidate]: if row["confidence"] in ("struggling", "learning") or (row.get("last_teachback_score") is not None and row["last_teachback_score"] < 14) ] - rows.sort(key=lambda row: row["last_seen"], reverse=True) + # A legacy ``study_progress`` row can carry a NULL ``last_seen``; it sorts + # as the oldest rather than raising and losing every struggle candidate + # (surfaced by review 7's boundary pin; the demand derivation then reads + # it as live, the cautious side). + rows.sort(key=lambda row: row.get("last_seen") or "", reverse=True) rows.sort( key=lambda row: ( {"struggling": 0, "learning": 1}.get(row["confidence"], 2), diff --git a/packages/studyloop/tests/test_now_plan_guidance.py b/packages/studyloop/tests/test_now_plan_guidance.py index c2f7a9728..336486f3e 100644 --- a/packages/studyloop/tests/test_now_plan_guidance.py +++ b/packages/studyloop/tests/test_now_plan_guidance.py @@ -1554,3 +1554,196 @@ def test_cli_milestone_deferral_does_not_promise_live_repair(monkeypatch) -> Non assert "repair stay available" not in flat assert "repair stay available" not in low.energy_deferred[0].reason assert low.energy_deferred[0].reason.endswith("due recall and gentle review stay available") + + +# --- Council review 7 — pins the seats asked for ----------------------------------- + + +def _unrelated_trio() -> tuple[_Candidate, ...]: + """A due recall, a conversation and a hands-on task, none plan-related.""" + return ( + _candidate("due", topic="python", action_type="recall", score=100), + _candidate("talk", topic="python", action_type="conversation", score=58), + _candidate("exercise", topic="python", action_type="hands-on", score=48), + ) + + +@pytest.mark.parametrize("modality", ["recall", "conversation"]) +def test_body_double_ordering_after_adjustments_follows_the_energy_rule( + monkeypatch, modality +) -> None: + """Review 7, F2 (astra 🔴 / qwen 🔴 / grok 💡 — arbitrated): the proposal's *base* is + below every real candidate's, and the day's adjustments then apply to it as to any + candidate. So a due recall and a conversation outrank it at every modality, while a + hands-on task the low-energy rule penalises (48 - 14 = 34) sits beneath it (30 + 12 = + 42): the energy rule, not a filter — nothing is removed from the ranking, and the + primary is never the proposal while any due or conversation candidate exists.""" + _row3_plan() + _plant_struggles(monkeypatch, _struggle("window function", days_ago=3), due=_unrelated_trio()) + + low = build_now_plan(energy="low", modality=modality) # type: ignore[arg-type] + + assert [rec.concept for rec in _all(low)] == ["due", "talk", "Sit with SQL Windows"] + assert low.alternates[1].source == "body_double" + assert low.primary.score > low.alternates[0].score > low.alternates[1].score + real_bases = (decision.MILESTONE_BASE_SCORE, 48, 52, 58, 70, 82, 96, 100) + assert all(base > decision.BODY_DOUBLE_BASE_SCORE for base in real_bases) + + +@pytest.mark.parametrize( + ("confidence", "energy"), + [("learning", "low"), ("struggling", "medium")], +) +def test_no_plan_eligible_repair_exposes_demand_without_deferral( + monkeypatch, confidence, energy +) -> None: + """Review 7, F3 (astra 🟡): the plan-independent changes are exactly two — every + struggle-collector candidate carries ``metadata.energy_demand`` at every energy, and + repair above its demand is deferred. An eligible repair with no plan is ranked as + before, carries the demand, defers nothing and proposes nothing.""" + _plant_struggles(monkeypatch, _struggle("decorators", topic="python", confidence=confidence)) + + plan = build_now_plan(energy=energy) # type: ignore[arg-type] + + assert plan.primary.concept == "decorators" + assert plan.primary.metadata["energy_demand"] == ("low" if confidence == "learning" else "high") + assert plan.energy_deferred_repairs == () + payload = plan.to_json_dict() + assert "energy_deferred_repairs" not in payload and "active_plans" not in payload + assert not any(rec.source == "body_double" for rec in _all(plan)) + + +def test_energy_demand_recency_boundaries_and_unknown_dates(monkeypatch) -> None: + """Review 7, F4 / grok 🔵: exactly 14 days is still live (high); 15 is medium; a null + or unparseable ``last_seen`` on a struggling row is read as live. (A row with no + ``last_seen`` key at all cannot reach the derivation: the collector's own sort reads + the key first, and the projection always supplies it.)""" + from studyloop.history import observations + + rows = [ + _struggle("on the day", days_ago=14), + _struggle("day after", days_ago=15), + {**_struggle("garbled"), "last_seen": "not-a-date"}, + {**_struggle("missing"), "last_seen": None}, + ] + _plant_struggles(monkeypatch) + monkeypatch.setattr(observations, "rows", lambda conn: [dict(r) for r in rows]) + + low = build_now_plan(energy="low") + + demand = {d.concept: d.energy_demand for d in low.energy_deferred_repairs} + assert demand == { + "on the day": "high", + "day after": "medium", + "garbled": "high", + "missing": "high", + } + + +def test_energy_demand_confidence_and_teachback_precedence(monkeypatch) -> None: + """Review 7, F4: ``learning`` is low demand even with a weak teach-back (eligible at + low energy); an old ``struggling`` row with a weak teach-back stays medium.""" + _plant_struggles( + monkeypatch, + _struggle("gentle", confidence="learning", days_ago=2, teachback=9), + _struggle("stale", days_ago=20, teachback=9), + ) + + low = build_now_plan(energy="low") + + assert low.primary.concept == "gentle" + assert low.primary.metadata["energy_demand"] == "low" + assert [(d.concept, d.energy_demand) for d in low.energy_deferred_repairs] == [ + ("stale", "medium") + ] + + +@pytest.mark.parametrize( + ("energy", "deferred"), + [("low", {"live", "stale"}), ("medium", set()), ("high", set())], +) +def test_repair_demand_capability_matrix(monkeypatch, energy, deferred) -> None: + """Review 7, F4: low (3) rejects medium and high demand and carries low; medium (6) + and high (10) carry every class.""" + _plant_struggles( + monkeypatch, + _struggle("live", days_ago=1), + _struggle("stale", days_ago=30), + _struggle("gentle", confidence="learning"), + ) + + plan = build_now_plan(energy=energy) # type: ignore[arg-type] + + assert {d.concept for d in plan.energy_deferred_repairs} == deferred + ranked = {rec.concept for rec in _all(plan)} + assert "gentle" in ranked + assert ranked.isdisjoint(deferred) + + +def test_deferred_repair_allows_only_eligible_milestone_conversation(monkeypatch) -> None: + """Review 7, F4 / grok 🔵 (design §5 decision 5): when the deferred live struggle was the + only representative of an *eligible* next milestone (floor 3 at low energy), rule 6 + synthesises that milestone's conversation — not another hands-on repair, and not a + body double, since the plan is now represented.""" + _plan( + "sql-windows", + title="SQL Windows", + energy_floor=3, + milestones=[Milestone(title="Frames", concepts=["window frame"])], + ) + _plant_struggles(monkeypatch, _struggle("window frame", days_ago=2)) + + low = build_now_plan(energy="low") + + assert low.primary.source == "study_plan:sql-windows:0" + assert low.primary.action_type == "conversation" + assert low.primary.plan_refs == (PlanRef("sql-windows", 0),) + assert [d.concept for d in low.energy_deferred_repairs] == ["window frame"] + assert not any(rec.source == "body_double" for rec in _all(low)) + assert not any(rec.action_type == "hands-on" for rec in _all(low)) + assert low.energy_deferred == () + + +def test_body_double_two_ready_plans_has_deterministic_context(monkeypatch) -> None: + """Review 7, F4 / grok 🔵: two ready plans, both below the floor, both with live + struggles → one proposal, refs in plan order (rule 6: most recent ``updated`` first), + both deferred milestones and both repairs named, the door on the first plan.""" + _plan( + "sql-windows", + title="SQL Windows", + topics=["sql"], + energy_floor=5, + updated="2026-09-02T00:00:00+00:00", + milestones=[Milestone(title="Frames", concepts=["window frame"])], + ) + _plan( + "py-decorators", + title="Python Decorators", + topics=["python"], + energy_floor=5, + updated="2026-09-01T00:00:00+00:00", + milestones=[Milestone(title="Closures", concepts=["closure"])], + ) + _plant_struggles( + monkeypatch, + _struggle("window function", topic="sql", days_ago=3), + _struggle("decorators", topic="python", days_ago=2), + ) + + low = build_now_plan(energy="low") + + assert [rec.source for rec in _all(low)] == ["body_double"] + proposal = low.primary + assert proposal.concept == "Sit with your plans" + assert proposal.plan_refs == (PlanRef("sql-windows", None), PlanRef("py-decorators", None)) + assert proposal.evidence_command == 'studyloop study "SQL Windows" --mode co-study' + for named in ( + "Frames", + "Closures", + "window function", + "decorators", + "SQL Windows and Python Decorators", + ): + assert named in proposal.reason + assert {d.plan_id for d in low.energy_deferred} == {"sql-windows", "py-decorators"} + assert {d.plan_id for d in low.energy_deferred_repairs} == {"sql-windows", "py-decorators"} From 02e282e6d0984fdf4899cdbd969a1b12bb7ed19c Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sat, 19 Sep 2026 13:38:31 +0100 Subject: [PATCH 14/32] docs(now): the body double's ordering claim now matches the engine (review 7, F2) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Astra and qwen flagged 🔴, grok 💡: "base below MILESTONE_BASE_SCORE so every real candidate outranks it" was false after the day's adjustments — at low energy the proposal (30 + 12 = 42) outranks a hands-on task that energy penalises (48 − 14 = 34), while every due and conversation candidate still outranks it (reproduced through the engine at recall and conversation modality). Arbitrated as the energy rule doing what the owner's finding asked, not a filter: the CLAIM was the defect. Corrected in the constant's comment, the body-double docstring, the spec delta's rule-5 clause and design §5 decision 4; pinned by test_body_double_ordering_after_adjustments_follows_the_energy_rule; the judgement goes to the owner as rubric row 3b reading (e). --- .../plan-integration-followons/design.md | 46 +++++++++++++------ .../specs/active-learning-decisions/spec.md | 34 +++++++++----- .../src/studyloop/learning/decision.py | 14 ++++-- 3 files changed, 62 insertions(+), 32 deletions(-) diff --git a/openspec/changes/plan-integration-followons/design.md b/openspec/changes/plan-integration-followons/design.md index a72d50676..6b27170ee 100644 --- a/openspec/changes/plan-integration-followons/design.md +++ b/openspec/changes/plan-integration-followons/design.md @@ -273,16 +273,15 @@ has none; `INTERLEAVE_RATIOS["low"]` unchanged. 1. **The deferral is plan-independent.** A live struggle is a live struggle whether or not a plan names it (amendment 2's `plan_id … else None` already said so); the finding was about the learner's day, not the plan. The body double, by contrast, *requires* a matchable active plan — it is "sit with the plan". - Consequence, stated rather than hidden: D-5's "a learner with no active plan receives the pre-#10 payload - byte for byte" now holds for a no-plan learner **with nothing deferred**; a no-plan learner whose live - struggle is deferred at low energy gets `energy_deferred_repairs` (and the starter if nothing else was - collected) where they used to get the hands-on repair. The golden world defers nothing and is unchanged. - The spec delta and both docs say so in those words. **A council question (review 7):** is that the right - scope for D-F, or should the repair half be gated on an active plan? -6. **The body double names ready plans only.** An active-but-unready plan is matched but never synthesised - (spec rule 8), and the body double is a synthesis; with only a husk active and nothing plan-related fitting, - nothing is proposed to sit with — the warning beside it already says "pause or repair" - (`test_body_double_is_never_synthesised_for_an_unready_plan`). + Consequence, stated rather than hidden (sharpened by review 7 F3): D-5's "a learner with no active plan + receives the pre-#10 payload byte for byte" holds for a no-plan learner **with no struggle candidate and + nothing deferred** — the golden world. Two changes are plan-independent: every struggle-collector candidate + carries `metadata.energy_demand` at every energy, and repair above the day's capability (a live or older + struggle, a weak teach-back, at low energy) is deferred — with the starter if nothing else was collected — + where the learner used to get the repair itself. Review 7 put the scope question to three seats: two (astra, + grok) keep it plan-independent ("gating it on a plan would leave the original no unfixed for every no-plan + learner"), one (qwen) would gate it; arbitrated as **keep**, the contract re-worded to the truth above in the + spec delta and both docs, and the no-plan floor put to the owner as row 3b reading (d). 2. **Rule 8's guaranteed slot below the floor is the body-double proposal.** It carries `plan_refs`, so where four unrelated due items outrank everything at low energy the second alternate is now the proposal, not a third unrelated item. It advertises no work the energy cannot carry — the property rule 8's docstring @@ -290,13 +289,30 @@ has none; `INTERLEAVE_RATIOS["low"]` unchanged. 3. **The starter tells the truth after a deferral.** With no plan and every real candidate deferred, the starter stands in; its reason now says the energy deferred the repair work rather than "no learning evidence found yet", which would be false. The golden world defers nothing, so its sentence is unchanged. -4. **Body-double shape.** Base `BODY_DOUBLE_BASE_SCORE = 30` (+12 bias = 42 < practice 48, milestone 48); - concept `Sit with <title>` (one plan) / `Sit with your plans`; `plan_refs` for every matchable plan; reason - naming each deferred milestone and repair; command `studyloop study "<first title>" --mode co-study`. The - Today card starts it in the Body Double view (`viewForAction`); the CLI labels the command "Sit with the plan". +4. **Body-double shape.** Base `BODY_DOUBLE_BASE_SCORE = 30` — below every real candidate's *base*; after the + day's adjustments it sits above a hands-on task the low-energy rule penalises (48 − 14 = 34 < 30 + 12 = 42) + and below every due and conversation candidate. Review 7 F2 (two seats 🔴, one 💡) was arbitrated as the + energy rule doing what the finding asked, not a filter: the *claim* "any real candidate outranks it" was the + defect, corrected in the constant's comment, the spec and here, and pinned by + `test_body_double_ordering_after_adjustments_follows_the_energy_rule`; the judgement is row 3b reading (e). + Concept `Sit with <title>` (one plan) / `Sit with your plans`; `plan_refs` for every **ready** matchable plan + (rule 7 may add a topic-matched husk reference; the proposal names ready plans only); reason naming each + deferred milestone and repair; command `studyloop study <title> --mode co-study`, the title quoted as one + shell argument (review 7 F1). The Today card starts it in the Body Double view and hands the plan title over + (`body-double-request`, review 7 F7); the CLI labels the command "Sit with the plan". 5. **A deferred repair does not "represent" a milestone** (rule 6 runs after the deferral), so an eligible milestone whose only collected representative was a deferred live struggle is synthesised as a conversation — - the learner can still talk about it. + the learner can still talk about it (`test_deferred_repair_allows_only_eligible_milestone_conversation`). +6. **The body double names ready plans only.** An active-but-unready plan is matched but never synthesised + (spec rule 8), and the body double is a synthesis; with only a husk active and nothing plan-related fitting, + nothing is proposed to sit with — the warning beside it already says "pause or repair" + (`test_body_double_is_never_synthesised_for_an_unready_plan`). +7. **`medium` and `high` demand are behaviourally identical today** (nothing in `ENERGY_CAPABILITY` sits between + 3 and 6): both need at least medium self-reported energy. The class is kept as explanatory state so the + payload says *why* (review 7: astra and grok keep it, qwen would collapse it); the spec says so. +8. **Not taken, recorded as follow-ons:** a same-concept gentle `recall` synthesised for a no-plan learner whose + only candidates were deferred (grok 🔵 / qwen 🟡; astra: do not invent an unvalidated lower-demand action) — + the owner's row 3b reading (d) decides whether the starter is the floor they want. Known edge, not solved here: a `struggling` row whose `last_seen` cannot be parsed is read as live (`high`) — the cautious side; `_days_since` returns `None` and the demand falls to `high`. diff --git a/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md b/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md index d5ffedade..7d6789ef3 100644 --- a/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md +++ b/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md @@ -151,9 +151,13 @@ the body-doubling floor): candidate is plan-related** and at least one matchable active plan exists, one **body-double** candidate SHALL be synthesised instead of leaving the plan to the least-bad task: `source = "body_double"`, `action_type = - "conversation"`, base score below the synthesised-milestone base so every - real candidate outranks it (a proposal, never a filter), `plan_refs` - `(plan_id, None)` for every matchable plan, a reason naming the deferred + "conversation"`, base score below every real candidate's base and then + scored like any other candidate — so every due and conversation candidate + outranks it at every energy while a hands-on task the low-energy rule + penalises may not; a proposal, never a filter: nothing is removed from the + ranking — `plan_refs` `(plan_id, None)` for every **ready** matchable plan + (rule 7 may add a reference to an unready plan whose topic the proposal + shares; the proposal itself names ready plans only), a reason naming the deferred milestones and repairs it stands in for, and `evidence_command` the co-study session door — `studyloop study "<plan title>" --mode co-study` — set explicitly, never a progress write. No active plan (a draft is not @@ -180,12 +184,17 @@ the body-doubling floor): `energy_deferred_repairs`, `completion_actions` and `warnings`; `LearningRecommendation` gains `plan_refs: tuple[PlanRef, ...] = ()`. `to_json_dict()` SHALL omit each of these when empty, so a learner with no -active plan **and nothing deferred** receives the pre-#10 payload **byte for -byte** — pinned by `tests/golden/now_plan_no_active.json`, captured before any -of this shipped. The one plan-independent change is rule 2's repair half: a -learner with no plan whose live struggle is deferred at low energy receives -`energy_deferred_repairs` (and the starter, if nothing else was collected) -where they used to receive the hands-on repair itself. +active plan, no struggle candidate and nothing deferred receives the pre-#10 +payload **byte for byte** — the golden world, pinned by +`tests/golden/now_plan_no_active.json`, captured before any of this shipped. +Exactly two changes are plan-independent (rule 2's repair half): every +struggle-collector candidate's `metadata` carries `energy_demand` at every +energy, and repair above the day's capability — a live struggle, an older +struggle or a weak teach-back at low energy — is deferred into +`energy_deferred_repairs` (with the starter, if nothing else was collected) +where the learner used to receive the repair itself. `medium` and `high` +demand both need at least medium self-reported energy today; the class is +carried so the payload says why. Renderers (`studyloop now`, `GET /api/now`, the Today card, the daily recap in its JSON, spoken and Rich-panel forms) SHALL show plan relevance, energy deferral — one line per deferred milestone **and** one per deferred repair — @@ -319,9 +328,10 @@ with row 3b re-run after this requirement's repair half. deferred milestone and — for the live-struggle fixture — one line per deferred repair with its demand and the day's capability; a body-double primary is labelled "Sit with the plan" with its `--mode co-study` door - and never "Record evidence"; with no plan the CLI panel prints no plan - lines, `GET /api/now` equals the golden, and the recap's `plan_context` is - absent from its JSON, its spoken text and the `recap today` panel + and never "Record evidence"; with no plan and no struggle candidate the CLI + panel prints no plan lines, `GET /api/now` equals the golden, and the + recap's `plan_context` is absent from its JSON, its spoken text and the + `recap today` panel #### Scenario: Learner-authored text is data to every renderer - **WHEN** an active plan's title, topic or milestone text contains Rich diff --git a/packages/studyloop/src/studyloop/learning/decision.py b/packages/studyloop/src/studyloop/learning/decision.py index 58adf7334..151850151 100644 --- a/packages/studyloop/src/studyloop/learning/decision.py +++ b/packages/studyloop/src/studyloop/learning/decision.py @@ -72,10 +72,13 @@ ENERGY_DEMAND_CAPABILITY: dict[EnergyDemand, int] = {"high": 6, "medium": 4, "low": 0} LIVE_STRUGGLE_DAYS = 14 -#: Base score of the synthesised body-double candidate (design §5): below -#: ``MILESTONE_BASE_SCORE`` so every real candidate — due, repair, practice, -#: continuity, a synthesised milestone — outranks it. A proposal, never a -#: filter; the plan bias then lifts it over nothing but the starter. +#: Base score of the synthesised body-double candidate (design §5): below every +#: real candidate's *base* — due (100+), repair (70/82), cards (96+), continuity +#: (58), transfer (52), practice and a synthesised milestone (48). The day's +#: adjustments then apply to it as to any candidate, so at low energy it (30 + +#: 12 bias = 42) sits above a hands-on task that energy penalises (48 - 14 = +#: 34) and below every due and conversation candidate — the energy rule, not a +#: filter: nothing is removed from the ranking (council review 7, F2). BODY_DOUBLE_BASE_SCORE = 30 BODY_DOUBLE_SOURCE = "body_double" @@ -1237,7 +1240,8 @@ def _body_double_candidate( """Design §5's floor: nothing plan-related fits and an active plan exists → sit with it. One ``source="body_double"`` conversation candidate, base below every real - candidate's (a proposal, not a filter), ``plan_refs`` ``(plan, None)`` for + candidate's base and then scored like any other (a proposal, not a filter — + see ``BODY_DOUBLE_BASE_SCORE``), ``plan_refs`` ``(plan, None)`` for every matchable plan, reason naming what it stands in for, and the co-study session door as its command (T5.1 amendment 3): ``_evidence_command`` has no branch for it and would answer with a progress *write*, not a door. From bdf4d6c6bbaed7568675003aa12fc4e7a8bea62a Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sat, 19 Sep 2026 13:38:47 +0100 Subject: [PATCH 15/32] docs: the no-plan contract names both plan-independent changes (review 7, F3/F6) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Astra 🟡 F3: "no plan and nothing deferred → byte-identical" was still false, because every struggle-collector candidate now carries metadata.energy_demand at every energy, and older struggles and weak teach-backs defer too, not only live struggles. docs/cli-reference.md and docs/study-plans.md now say exactly what happens without a plan (the spec delta's paragraph landed with F2's edit of the same file); F6: the rule-7 husk reference is described. --- docs/cli-reference.md | 2 +- docs/study-plans.md | 6 +++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/cli-reference.md b/docs/cli-reference.md index c633011dc..17c44b538 100644 --- a/docs/cli-reference.md +++ b/docs/cli-reference.md @@ -259,7 +259,7 @@ studyloop now --speak Default ranking is due review first, then struggling or low teach-back score, then active-course continuity, then modality match. Low energy suppresses hard context switching. -With an active study plan the same engine is plan-aware (see [Study Plans](study-plans.md#plan-aware-now)): plan-related work gets a bounded bias within its urgency class, a ready plan's next milestone is suggested when it is within your energy and no gathered candidate represents it, and a milestone above your current energy is deferred with a reason. Repair of a live struggle is deferred the same way at low energy (an older struggle or weak teach-back needs medium; a concept still being learned is never deferred, nor is a due review), and when nothing plan-related fits the day the panel proposes sitting with the plan — its command is the co-study door, `studyloop study "<plan>" --mode co-study`, labelled "Sit with the plan" rather than "Record evidence". The panel names the plan and milestone an action advances and prints one "Deferred for energy" line per deferred milestone and per deferred repair; `--json` adds `active_plans`, `energy_deferred`, `energy_deferred_repairs`, `completion_actions`, `warnings` and per-action `plan_refs` only when each is non-empty, so with no plan and nothing deferred the output is unchanged (the deferral of a live struggle at low energy is the one thing that happens without a plan), and a plan that cannot be read shows up as a warning rather than a failure. +With an active study plan the same engine is plan-aware (see [Study Plans](study-plans.md#plan-aware-now)): plan-related work gets a bounded bias within its urgency class, a ready plan's next milestone is suggested when it is within your energy and no gathered candidate represents it, and a milestone above your current energy is deferred with a reason. Repair of a live struggle is deferred the same way at low energy (an older struggle or weak teach-back needs medium; a concept still being learned is never deferred, nor is a due review), and when nothing plan-related fits the day the panel proposes sitting with the plan — its command is the co-study door, `studyloop study "<plan>" --mode co-study`, labelled "Sit with the plan" rather than "Record evidence". The panel names the plan and milestone an action advances and prints one "Deferred for energy" line per deferred milestone and per deferred repair; `--json` adds `active_plans`, `energy_deferred`, `energy_deferred_repairs`, `completion_actions`, `warnings` and per-action `plan_refs` only when each is non-empty, so with no plan, no recorded struggle and nothing deferred the output is unchanged (two things happen without a plan: a struggle-repair action carries its `energy_demand` in `metadata`, and repair above the day's energy is deferred), and a plan that cannot be read shows up as a warning rather than a failure. `studyloop chat-note` turns one markdown/text note into a compact Socratic context pack. V1 prints or speaks the mentor prompt; it does not run a separate chat backend. diff --git a/docs/study-plans.md b/docs/study-plans.md index 12f69e7ab..ca023215f 100644 --- a/docs/study-plans.md +++ b/docs/study-plans.md @@ -252,9 +252,9 @@ action keeps its plain sentence and a warning says why — a failure is never shown as a clean slate. This is plan-aware guidance with tested ranking rules — a bias, not a filter: an overdue review or a fresh struggle on an unrelated topic can still outrank new milestone work. With no active plan the recommendation is -unchanged — except that a live struggle is deferred at low energy whether or -not a plan names it; a plan that cannot be read adds a warning and nothing -else. The +unchanged — except that repair above the day's energy (a live or older +struggle, a weak teach-back) is deferred whether or not a plan names it; a +plan that cannot be read adds a warning and nothing else. The ranking rules are tested; whether the primary is the action *you* would take is a separate judgement. Five frozen scenarios and the engine's primaries are in the project's rubric receipt From 3eb31f2d3527f2ae48e1e478416029241aa10210 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sat, 19 Sep 2026 13:42:18 +0100 Subject: [PATCH 16/32] docs(council): review 7 seats, arbitration (GATE ACCEPT), rubric row 3b readings (d)/(e), T5.4 ticked Three seats over brief-review7 on tree 6d5a2d0e: astra and qwen ACCEPT-WITH-CORRECTIONS, grok ACCEPT. Every red and yellow reproduced before acceptance and landed one commit each (d1935256 .. bdf4d6c6), or rejected here with the reason and the owner question that replaces it. The seats split 2-1 on the plan-independent deferral and 2-1 on the body double's ordering; both go to the owner as row 3b readings (d) and (e), printed from the real engine on bdf4d6c6. --- .../review-7-arbitration-2026-09-19.md | 115 ++++++++++++ .../council/review7/manifest.json | 47 +++++ .../council/review7/seat-grok-4.6.md | 132 +++++++++++++ .../review7/seat-openai.gpt-6-astra.md | 176 ++++++++++++++++++ .../council/review7/seat-qwen3-coder.md | 169 +++++++++++++++++ .../receipts/now-rubric-2026-09-16.md | 4 +- .../plan-integration-followons/tasks.md | 12 +- 7 files changed, 652 insertions(+), 3 deletions(-) create mode 100644 docs/architecture/plan-integration/council/review-7-arbitration-2026-09-19.md create mode 100644 docs/architecture/plan-integration/council/review7/manifest.json create mode 100644 docs/architecture/plan-integration/council/review7/seat-grok-4.6.md create mode 100644 docs/architecture/plan-integration/council/review7/seat-openai.gpt-6-astra.md create mode 100644 docs/architecture/plan-integration/council/review7/seat-qwen3-coder.md diff --git a/docs/architecture/plan-integration/council/review-7-arbitration-2026-09-19.md b/docs/architecture/plan-integration/council/review-7-arbitration-2026-09-19.md new file mode 100644 index 000000000..f8c81c417 --- /dev/null +++ b/docs/architecture/plan-integration/council/review-7-arbitration-2026-09-19.md @@ -0,0 +1,115 @@ +# Arbitration — council review 7 (item 5 of the plan-integration follow-on programme: D-F) + +**Date:** 2026-09-19 · **Arbiter:** coordinating agent (owner present; the owner chose the seats' key and asked +for the run) · **Reviewed tree:** `feat/energy-demand-body-double` @ `6d5a2d0e` (four commits on `main` +`4f8e3e0f`; range `4f8e3e0f..6d5a2d0e`, 15 files, +1,042/−25). Seats ran against +`brief-review7-2026-09-19.md` (`review7/manifest.json`, run 12:10:45Z; astra and qwen `finish_reason=stop`, +**grok `length`** — its 16,000-token cap cut its answer inside §3 Refutations, so its verdict, findings and the +first refutation are complete and its §4 Gate is absent; not re-run, since every grok finding is 🔵/💡 and its +verdict is ACCEPT). **Corrections landed at:** `d1935256` (F1), `c1d7f2f2` (F5), `3347567e` (F7), `f814dc36` +(grok's label), `7194b66d` (F4 pins + a collector crash they surfaced), `02e282e6` (F2 claim), `bdf4d6c6` +(F3/F6 wording). Two corrections the agent found on its own re-read *before* the brief was frozen are part of +the reviewed tree (`8e9cbdf5` ready plans only; `6d5a2d0e` wording) and are listed in the brief §1. + +**Brief size, recorded:** 117.6 KB, 1,734 lines (~31k prompt tokens per seat): D-F and rubric row 3 verbatim, +design §5 with its amendments and GREEN decisions, every diff in the range in full, row 3b as written, the +control receipt, reference facts, ten numbered check questions — including the scope of the plan-independent +deferral asked outright as (a). + +## Seats and verdicts + +| Seat | Verdict | 🔴 | 🟡 | 🔵/💡 | +| --- | --- | --- | --- | --- | +| `openai.gpt-6-astra` | ACCEPT-WITH-CORRECTIONS | F1 shell quoting; F2 ordering claim | F3 compatibility promise; F4 unpinned boundaries; F5 contradictory copy | F6 match semantics; F7 Web door context | +| `qwen3-coder` | ACCEPT-WITH-CORRECTIONS | scope (gate deferral on a plan); ordering claim | medium/high partition; docs; gentle-recall fallback; Web door; substring tests; row 3b ask | `evidence_command` reuse | +| `grok-4.6` | ACCEPT | — | — | 14-day boundary + unparseable path untested; decision 5 untested; two-plan case untested; Web pre-fill; "Your plans" label; medium=high behaviourally; `evidence_command` fine | + +### Method + +Each 🔴/🟡 was reproduced before acceptance — F1 with a real `/bin/sh` and a stub `studyloop` (the title's +`$(touch pwned)` ran), F2 through the engine at two modalities (42 vs 118 / 58 / 34; 60 vs 100 / 76 / 34), F5 by +reading row 3b's own rendered lines — then fixed one commit each with a RED test named by the seat where one was +named, and stash-proved where a fix could be reverted alone (F1: 3 of 7 titles fail without it). + +### Findings and dispositions + +| # | Finding (seat) | Disposition | Commit / test | +| --- | --- | --- | --- | +| F1 | The offered command replaced `"` with `\"` — presentation, not quoting; a plan title with `$(…)` executed when pasted (astra 🔴). | **Accepted, reproduced.** One `_shell_word` helper: plain text keeps the golden's double-quoted form, shell-special text is `shlex.quote`d; both command builders use it (`_evidence_command` had the same shape for every concept and topic). | `d1935256`; `test_body_double_command_preserves_title_as_one_literal_shell_argument` (7 titles through `/bin/sh`; stash-proved 3/7 red without the fix) | +| F2 | "Base below `MILESTONE_BASE_SCORE` so every real candidate outranks it" is false after adjustments: at low energy the proposal (42) outranks a hands-on task that energy penalises (34) (astra 🔴, qwen 🔴; grok 💡 "acceptable, do not lower the base"). | **Claim corrected, behaviour kept.** Reproduced exactly as computed. Every due and conversation candidate still outranks it at every modality; only a hands-on task the low-energy rule already penalises sits beneath — the energy rule doing what the finding asked ("instead of the least-bad task"), not a filter: nothing is removed from the ranking. Astra's proposed post-scoring floor would put a penalised hands-on task above the proposal at low energy, i.e. re-recommend the class of work the finding objected to. The *claim* was the defect: corrected in the constant's comment, the docstring, the spec delta's rule-5 clause and design §5 decision 4; the judgement is the owner's — row 3b reading (e). | `02e282e6`; `test_body_double_ordering_after_adjustments_follows_the_energy_rule` (recall and conversation modality) | +| F3 | "No plan and nothing deferred → byte-identical" still false: every struggle candidate gains `metadata.energy_demand` at every energy; older struggles and weak teach-backs defer too (astra 🟡). | **Accepted.** The contract now names exactly the two plan-independent changes, in the spec delta, both docs and design decision 1. | `bdf4d6c6` (docs) + `02e282e6` (spec paragraph, same file as F2's edit); `test_no_plan_eligible_repair_exposes_demand_without_deferral` (in `7194b66d`) | +| F4 | Five unpinned invariants: the 14-day boundary and unparseable dates; confidence/teach-back precedence; the capability matrix; decision 5 (deferred sole representative → eligible milestone conversation); two ready plans (astra 🟡; grok 🔵 on three of them). | **Accepted, all five written.** The boundary pin surfaced a **pre-existing collector crash**: `_struggle_candidates` sorted by `row["last_seen"]`, so a legacy `study_progress` row with a NULL `last_seen` raised and lost every struggle candidate; it now sorts as the oldest and derives as live. | `7194b66d`; `test_energy_demand_recency_boundaries_and_unknown_dates`, `test_energy_demand_confidence_and_teachback_precedence`, `test_repair_demand_capability_matrix`, `test_deferred_repair_allows_only_eligible_milestone_conversation`, `test_body_double_two_ready_plans_has_deterministic_context` | +| F5 | The milestone-deferral sentence ended "plan-related review and repair stay available" one line above the line deferring the repair (astra 🟡). | **Accepted.** Engine reason and CLI line now promise only due recall and gentle review; pinned as whole sentences. | `c1d7f2f2`; `test_cli_milestone_deferral_does_not_promise_live_repair` | +| F6 | Rule 7 can attach a husk reference to the proposal through its topic, so "names ready plans only" is true of the proposal's text, not of every rendered reference (astra 🔵). | **Accepted as a spec sentence**, no code change: the two-stage behaviour is now described in the rule-5 clause; `test_body_double_is_never_synthesised_for_an_unready_plan` already permits the husk ref and pins the proposal's text. | `02e282e6` | +| F7 | The Web door navigated only; the Body Double picker opened blank while the CLI door carried the plan title (astra 🔵, grok 🔵, qwen 🟡). | **Accepted, built** — the same event-not-storage handoff `today-resume` uses: `startAction` dispatches `body-double-request` {activity, energy}; `bodyDoubleSession.init` adopts it. Nothing starts on its own. | `3347567e`; JS `starting a body-double primary hands its plan to the Body Double view…`, `bodyDoubleActivity…` | +| G1 | `hasPlanContext` is true for a no-plan deferred repair and the block is headed "Your plans" (grok 🔵). | **Accepted.** `planNotesLabel()`: "Your plans" once any note involves a plan, else "Set aside today". | `f814dc36`; JS `planNotesLabel…` | +| Q1 | Gate the deferral on an active plan (qwen 🔴). | **Rejected** — see below. | design §5 decision 1; row 3b reading (d) | +| Q2 | Collapse the medium/high partition (qwen 🟡). | **Rejected** — see below. | design §5 decision 7; spec sentence | +| Q3 | Synthesise a same-concept gentle recall as the no-plan floor (qwen 🟡; grok 🔵 "follow-on"). | **Not taken; recorded as a follow-on** and put to the owner. | design §5 decision 8; row 3b reading (d) | +| Q4 | Row 3b should ask the owner about the no-plan consequence (qwen 🟡; grok's tail). | **Accepted.** Readings (d) and (e) added, printed from the real engine on `bdf4d6c6`. | `receipts/now-rubric-2026-09-16.md` | + +### Rejected, with reasons + +- **Q1 — gate repair deferral on an active plan (qwen 🔴).** Two seats and the owner's own words go the other + way: the finding is "hands-on repair of a *live* struggle on a low-energy day compounds the struggle (RSD)" — + a fact about the learner's day, not about plans — and D-F's item (1) says "at low energy a live struggle + defers like new work" with no plan condition. Gating it would leave the original "no" standing for every + learner without a plan. Astra: "reintroducing an unsafe recommendation merely because the learner lacks a plan + would contradict the rationale for D-F". Grok: the same, and "listed, not recommended, is the honest state". + Kept, with the compatibility contract re-worded to the truth (F3) and the no-plan floor put to the owner as + row 3b reading (d) — the one person who can say whether the starter is the floor they want. +- **Q2 — collapse `medium` into `high` (qwen 🟡).** True that nothing in `ENERGY_CAPABILITY` sits between 3 and 6, + so the two classes never differ in eligibility today. Astra and grok both keep the class as explanatory state + — the payload says *why* a repair asks for 4/10 rather than 6/10, and the next energy scale change would + otherwise need the derivation rebuilt. Kept; the spec now says both classes need at least medium energy today. +- **Astra's F2 fix (a post-scoring floor for `body_double`).** Rejected in favour of correcting the claim: the + floor would rank a hands-on task the low-energy rule penalises above the proposal, which is the shape of + recommendation the finding objected to. The behaviour is pinned exactly as it is and goes to the owner. +- **Q3 — a same-concept gentle recall as the no-plan floor.** Astra: "do not synthesise recall on the deferred + struggle merely to maintain topical relevance; that would invent an unvalidated lower-demand action." Grok: + a follow-on, not a merge gate. Recorded as decision 8 and asked in row 3b (d). + +### Refutations, weighed + +- Astra 1 ("42 < any real candidate" false) — **true**; F2. Astra 2–3 (byte-for-byte still false; "live struggle" + too narrow) — **true**; F3. Astra 4 ("each GREEN decision is a test" not established: decisions 5 and the + two-plan case untested) — **true**; F4 now pins both. Astra 5 ("the Today card starts it" — navigation only) — + **true at the reviewed tree**; F7 built the handoff. Astra 6 ("zero regressions" is about failing ids at GREEN, + not the reviewed tree) — **accepted as stated**; a full suite runs on the final tree below. Astra 7 + ("`learning` means recovered" is the chosen proxy, not established) — **accepted**; row 3b reading (c) asks + the owner exactly that. Grok's one refutation (the same F2 claim) — **true**. Qwen: none. + +### Verification after fixes (tree `bdf4d6c6` + the rubric edit) + +- `test_now_plan_guidance.py` **71 passed** (47 at the reviewed tree, +24 test cases from F1–F5 including + parametrisations), golden byte-identical; with `test_learning_decision.py` and the docs contract **96 passed**; + JS **147/147** (+3); `openspec validate` valid; + `mkdocs --strict` exit 0; ruff / ruff format / pyright clean on every touched file. Full suite and matched + control on the final tree: see the receipt named in `tasks.md` T5.4 once it lands (running at arbitration time). + +### Process findings + +- The brief's check question (a) put the arbiter's own hardest decision to the seats directly and got a 2–1 + split with reasons on both sides — more useful than a unanimous nod. Keep doing that. +- Grok's 16k cap cut its Gate section. Its findings were complete and all 🔵/💡, so no re-run; a future brief + of this size should either raise `--max-tokens` for that seat or ask for the Gate before the Refutations. +- One commit carried two logical changes: `3347567e` (F7) also contains the `planNotesLabel()` function that + `f814dc36` (G1) relies on, because both edits were in `today-panel.js` when F7 was staged. Recorded rather than + rewritten; both are unpushed and tested. +- F4's boundary pin finding a real, pre-existing crash (`None` `last_seen`) is the argument for writing boundary + tests even when the design says the boundary is "obvious". + +## Gate decision + +**GATE: ACCEPT** — for the tree at `bdf4d6c6` (with the rubric-row edit), not the reviewed tree. Every 🔴 and 🟡 +is either landed with a discriminating test (F1, F3, F4, F5, F7, G1, Q4) or rejected here with the reason and +the owner's question that replaces it (Q1, Q2, Q3, astra's F2 remedy). Item 5 is ready to merge once CI is green +on the branch; the change is **not** archived until the owner scores row 3b. + +## Still open for the owner + +1. **Rubric row 3b, five readings** — (a) sit with the plan rather than repair a live struggle; (b) the proposal + beneath an unrelated due recall; (c) the gentle teach-back on a `learning` concept; **(d) the no-plan floor** + (starter + deferred line, where the hands-on repair used to be); **(e) the proposal above an unrelated + hands-on drill at low energy**. (d) and (e) are the two places the seats split; only the owner closes them. +2. Whether a same-concept gentle recall should be synthesised for a no-plan learner (decision 8) — after (d). diff --git a/docs/architecture/plan-integration/council/review7/manifest.json b/docs/architecture/plan-integration/council/review7/manifest.json new file mode 100644 index 000000000..b864b69c0 --- /dev/null +++ b/docs/architecture/plan-integration/council/review7/manifest.json @@ -0,0 +1,47 @@ +{ + "run_at": "2026-09-19T12:10:45+00:00", + "brief": "docs/architecture/plan-integration/council/brief-review7-2026-09-19.md", + "brief_sha256": "6fde40f067eccee3c0fb46af74ea448f8e02c8eab1eb078c6229e1cc3b45a15b", + "system_sha256": "5f21e273f399cdb0455a9fadb8575cc3e231ce0d7931f517206335c98c3f934b", + "seats": [ + { + "model": "openai.gpt-6-astra", + "ok": true, + "reasoning_chars": 0, + "finish_reason": "stop", + "elapsed_s": 66.1, + "usage": { + "prompt_tokens": 30920, + "completion_tokens": 3939, + "total_tokens": 34859 + }, + "error": null + }, + { + "model": "grok-4.6", + "ok": true, + "reasoning_chars": 0, + "finish_reason": "length", + "elapsed_s": 107.3, + "usage": { + "prompt_tokens": 32856, + "completion_tokens": 16000, + "total_tokens": 48856 + }, + "error": null + }, + { + "model": "qwen3-coder", + "ok": true, + "reasoning_chars": 0, + "finish_reason": "stop", + "elapsed_s": 26.1, + "usage": { + "prompt_tokens": 31758, + "completion_tokens": 1845, + "total_tokens": 33603 + }, + "error": null + } + ] +} diff --git a/docs/architecture/plan-integration/council/review7/seat-grok-4.6.md b/docs/architecture/plan-integration/council/review7/seat-grok-4.6.md new file mode 100644 index 000000000..ff6cc06e9 --- /dev/null +++ b/docs/architecture/plan-integration/council/review7/seat-grok-4.6.md @@ -0,0 +1,132 @@ +# Council review 7 — item 5 (D-F) + +## 1. Verdict + +**ACCEPT.** Both halves of the owner's row-3 "no" are implemented, tested against the live struggle collector, documented, and process-clean (RED→GREEN, no new writer, golden byte-identical, matched full-suite delta = the new tests). + +## 2. Findings + +### Council question (design §5 decision 1) — plan-independent deferral is the right scope + +The finding is RSD, not plans: recommending hands-on repair of a *live* struggle on a low-energy day compounds the struggle whether or not a plan names it. Gating `_defer_repairs` on `plans.matchable` would leave the original "no" unfixed for every no-plan learner. The body-double half correctly requires a ready active plan ("sit with the plan"); the repair half must not. + +D-5's rewording to "no plan **and nothing deferred**" is the honest contract. The golden world defers nothing and is unchanged. A no-plan learner whose live struggle is deferred gets `energy_deferred_repairs` plus the starter — listed, not recommended — which is what "defer like new work" means. Do not reverse this. + +A same-concept gentle recall as a no-plan floor is a follow-on, not a correction (see 🔵 below). + +### (a) Scope — no defect + +`DeferredRepair.plan_id` / `plan_title` are `None` when unmatched; `_defer_repairs` calls `plans.first_match` only to *name* the entry. `test_body_double_never_appears_without_an_active_plan` pins the no-plan path. Spec delta, `docs/study-plans.md`, `docs/cli-reference.md`, and design §5 all say the same words. Consistent. + +### (b) Demand derivation + +`_energy_demand` matches amendment 1: `learning`→`low`; non-`struggling` kept only for weak teach-back→`medium`; `struggling` with `seen is None or seen <= 14`→`high`; older `struggling`→`medium`. Unparseable `last_seen` is the cautious side the finding asked for. + +`assessment_recorded_at` (`last_seen = max(recorded_at of current reports)`) is the right recency basis: live-ness is "when they last *reported* struggling," not last session contact. A 3-day `struggling` with a decent teach-back is still a live struggle (RSD); a 20-day `struggling` is not. Age dominates teach-back for the `struggling` class — correct. + +A `struggling`+weak-teach-back row does **not** deserve `high` regardless of age. Live-ness is the RSD trigger; a stale struggle with a weak score is the medium class the design named. + +💡 **`medium` (4) and `high` (6) are behaviourally identical at all three `ENERGY_CAPABILITY` values.** `low` (3) defers both; `medium` (6) and `high` (10) carry both. Nothing in the energy scale sits between 3 and 6, so `required_capability` 4 vs 6 never changes ranking. The class is still worth carrying: renderers and `DeferredRepair.reason` distinguish "a live struggle" / "an older struggle" / "a weak teach-back," which is the information the learner needs. Do not collapse the classes. Do not pretend 4/10 is a real gate — if a fourth energy level is ever added, the 4 becomes load-bearing. No JSON change required; a comment on `ENERGY_DEMAND_CAPABILITY` would be enough. + +🔵 **`LIVE_STRUGGLE_DAYS` boundary and the unparseable path are untested.** `test_live_struggle_repair_defers_at_low_energy_like_new_work` uses 3 and 20 days only. + +- Fix: plant `days_ago=14` (high) and `days_ago=15` (medium); plant one `struggling` row with `last_seen="not-a-date"` and one with `last_seen` missing; assert `energy_demand`. +- RED: `test_live_struggle_window_is_fourteen_days_inclusive` and `test_unparseable_last_seen_on_struggling_is_live`. + +14 days is a reasonable live window; the owner can override when scoring 3b. Inclusive `<=` matches the spec's "within 14 days." + +### (c) Deferral mechanics — correct + +`_defer_repairs` keys on `metadata["energy_demand"]` ∈ `ENERGY_DEMAND_CAPABILITY`, so due recall is untouched even with `confidence == "struggling"` — pinned by the second half of `test_recovered_repair_stays_eligible_at_low_energy`. + +Same concept due *and* struggling: rank the due row, defer the repair. Right. They are different actions (recall vs hands-on). Suppressing the deferred line when the concept is already primary would hide that the *repair* is what was refused. The CLI printing the concept twice is honest, not a bug. + +A deferred repair does not represent a milestone (`_defer_repairs` runs before `plans.milestone_candidates`). Right. The synthesised milestone is `action_type="conversation"` — talk about it, the cheap form — and only when the milestone is itself within the floor. That is not a side door for work the energy cannot carry. Hands-on repair stays deferred. + +🔵 **Decision 5 is untested.** No test plants a live struggle on an *eligible* next-milestone concept (floor ≤ 3) and asserts a `study_plan:` conversation is synthesised after the repair is stripped. + +- Fix: plan with `energy_floor=3`, next milestone concept `window frame`, plant `_struggle("window frame", days_ago=3)`, `energy="low"`. Assert the struggle is in `energy_deferred_repairs`, primary (or an alternate) is `source.startswith("study_plan:")` / `action_type=="conversation"`, and no ranked hands-on on that concept. +- RED: `test_deferred_repair_does_not_represent_an_eligible_milestone`. + +Code order in `build_now_plan` is already correct; this is a pin, not a behaviour change. + +### (d) Body double as candidate + +Base 30 + 12 bias = 42 < `MILESTONE_BASE_SCORE` 48 at *base*. After the low-energy hands-on penalty a practice candidate scores 48 − 14 = 34 < 42, so the proposal outranks it. + +💡 **The inversion is acceptable — do not lower `BODY_DOUBLE_BASE_SCORE`.** The finding is "do not recommend hard work at low energy." A conversation that advertises no new material and no repair *should* beat a penalised hands-on practice. The invariant "any real candidate outranks it" is true at base and after every adjustment except the deliberate low-energy hands-on/visual penalty. Document that exception on `BODY_DOUBLE_BASE_SCORE`; do not "fix" it. A due, a continuity (58), a transfer (52), a modality-matched practice (48+18), or a plan-related practice (48+12−14=46) still win. + +`concept = "Sit with <title>"` as a `_candidate_keys` member is harmless: it will not equal a milestone concept, and the topic (plan's first topic) is the intended match. `plan_refs` are set explicitly, so a second ready plan is referenced even when its topic differs. + +🔵 **Two ready plans are untested, and the CLI door names only the first title.** `_body_double_candidate` sets `concept="Sit with your plans"`, `titles=" and ".join(...)`, `evidence_command` from `named[0]` only. Starting one session is right; naming only the first in the command is the compromise. Pin it. + +- RED: `test_body_double_with_two_ready_plans_names_both_and_opens_the_first` — two ready plans, nothing plan-related fits; assert concept, reason contains both titles, command uses the first, `plan_refs` has both `(id, None)`. + +`estimated_minutes` 25 (conversation default) is fine. + +### (e) Rule 8 — not a filter in disguise + +`test_preserves_one_plan_backed_action_when_energy_allows` now expects `["due 0", "due 1", "Sit with Sql Windows"]` at low energy. Rule 8 has always replaced the *last* alternate with a plan-backed action; the primary is untouched. Previously nothing eligible existed below the floor, so three unrelated dues filled the slots. Now the eligible plan-backed action is the body double — it advertises no work the energy cannot carry, which is the property the docstring protects. `due 2` lost the slot the same way a synthesised milestone takes it at medium. Right. + +### (f) Honest starter — defensible for this item + +`_starter_candidate(..., after_deferral=True)` changes only the reason. The action stays "one tiny recall loop" on the first configured topic. For a no-plan learner with only live struggles that is a weaker offer than a `recall` on the deferred concept, but the finding said *defer* the live repair, not convert it into a gentle review. The starter is the existing empty-set floor; lying that "no learning evidence found yet" was the thing that had to change. `test_body_double_never_appears_without_an_active_plan` pins `"defer" in reason` (loose — see (i)). + +🔵 A same-concept gentle recall as a no-plan floor is a follow-on, not a merge gate. If taken: synthesise `action_type="recall"` on the deferred concept at starter-level score when `deferred_repairs` is non-empty and `not plans.matchable`. RED: `test_no_plan_live_struggle_offers_gentle_recall_not_the_generic_starter`. + +### (g) Renderers + +CLI `_render_plan`: door label swaps on `primary.source == "body_double"`; one escaped line per `energy_deferred_repairs`; plan-less repairs omit the title prefix. Recap `_plan_context`: one sentence per repair; reachable at low energy only (default recap energy is medium, as the hard rules say); pinned through `_plan_context` directly. Today: `deferredRepairNotes`, `viewForAction` → `body-double`, markup `x-show` includes the new length. All match amendment 2/3. + +🔵 **Web door does not pre-fill the plan title.** `startAction` only `nav.go('body-double')`; the Body Double view opens its own picker. CLI carries `studyloop study "<title>" --mode co-study`. The recommendation named a plan; the Web door drops it. Not wrong — the learner can pick — but a worse door than the CLI on the same payload. + +- Fix: pass the first `plan_refs[].plan_id` (or primary concept title) into the Body Double view's picker as the initial selection. +- RED: extend `test('a body-double primary starts in the Body Double view…')` to assert the payload the view receives, once the view accepts an initial plan. + +Not 🟡: the brief marks this out of range; the session door (not `GET /api/body-double/focus`) is the correct target. + +🔵 **`hasPlanContext` is true for a no-plan deferred repair, and the notes block is labelled "Your plans".** `today-panel.js` `hasPlanContext` ORs `deferredRepairNotes().length > 0`; `index.html` wraps that block in `<p class="today-parked-label">Your plans</p>`. A no-plan learner sees a "Your plans" heading containing only "Repairing “decorators”…". The JS test (`deferredRepairNotes: a repair unrelated to any plan… counts as plan context alone`) pins the current lie. + +- Fix: retitle the block when `active_plans` is empty ("Deferred for energy", or keep the lines and drop the heading); `hasPlanContext` can stay true so the notes still show. +- RED: update that JS test to assert the visible heading is not "Your plans" when `active_plans` is absent. + +💡 `evidence_command` as the door field is acceptable. Every renderer already reads it; a new field would be a fourth JSON key for one candidate type. The CLI relabels; the command string itself is a session start, so a consumer that executes it does the right thing. MCP `get_next_action` returning `to_json_dict()` unchanged is safe: the new key is omitted when empty; `source="body_double"` plus `--mode co-study` is self-describing. Disclose `energy_deferred_repairs` and `source=body_double` in the tool docstring / persona on the next MCP pass — not a merge gate (`PRODUCTION_TOOL_COUNT` correctly unchanged). + +### (h) Spec delta — matches the code + +MODIFIED requirement restates the ranker with item 5 inlined. Rule 2's demand classes, the additive `energy_deferred_repairs` key, plan-independent deferral, the body-double clause after milestone synthesis, rule 7's slot becoming the body double below the floor, and the qualified byte-for-byte sentence all match `build_now_plan` / `_defer_repairs` / `_body_double_candidate`. + +Ready-only synthesis is implied by spec rule 8 ("never synthesised") plus the body double being a synthesis; `test_body_double_is_never_synthesised_for_an_unready_plan` pins it. No dedicated spec scenario for "unready husk alone → no body double" — add one if the delta is edited again; not blocking. + +Scenario "A deferred repair does not represent a milestone" is stated in rule 5's prose and untested (see (c)). Scenario "Renderers show, never re-rank" claims `GET /api/now?energy=low` and the recap's spoken/Rich forms; the new pin is CLI + `_plan_context` + Today markup. Pre-existing renderer tests plus `to_json_dict` omit-empty cover the rest. Not a contract break. + +### (i) Tests + +Six REDs drive the real `_struggle_candidates` over `_plant_struggles` / `_struggle` rows — the right seam, since demand is derived in the collector. The unready correction (`8e9cbdf5`) is test+fix in one commit; acceptable as a pre-review re-read. + +Gaps already named: 14-day boundary, unparseable `last_seen`, two ready plans, decision 5. Medium-vs-high indistinguishability is implicitly true (both defer at low, both vanish at medium) and does not need its own test. + +Loose substring: `assert "defer" in low.primary.reason.lower()` in `test_body_double_never_appears_without_an_active_plan` would survive a wording change that dropped the honest sentence. Pin the designed clause. + +- Fix: `assert low.primary.reason.startswith("Today's energy deferred the repair work")`. +- Same test, tighter assertion — not a new name. + +Other reason checks (`"Frames" in … and "window function" in …`, `"3/10" in … and "6/10" in …`) are specific enough. + +### (j) Rubric row 3b + +The three readings are the right questions for the finding's two halves, printed from the real engine over `_plant_struggles`: + +- (a) live struggle → body-double primary — "would you sit with the plan rather than repair the live struggle today?" +- (b) + unrelated due → due primary, proposal beneath — "is the proposal right to sit beneath the unrelated due recall?" +- (c) `learning` → gentle teach-back — "is the recovered teach-back one you would do at low energy?" + +They cover (1) demand (live defers, recovered stays) and (2) the floor (synthesise when nothing fits; proposal, not filter). Scores in the receipt (42 = 30+12; 100 = 70+12+18; 118 due) match §6. + +Ask the owner one extra thing when scoring, not as a fourth reading in the table: **"A learner with no plan and only a live struggle at low energy now gets the starter plus a deferred-repair note, not the hands-on repair. Is that the no-plan floor you want?"** That is design §5 decision 1, and only the owner can close it. Do not hold the merge on the answer — the finding's text does not gate repair on a plan. + +Omitted and not needed in 3b: older-struggle (medium) class; same-concept due+repair double print; the 14-day window. Those are engine pins, not human-judgement questions. + +## 3. Refutations + +- **"any real candidate outranks it"** (design §5, `BODY_DOUBLE_BASE_SCORE` comment, spec rule 5) is false after the low diff --git a/docs/architecture/plan-integration/council/review7/seat-openai.gpt-6-astra.md b/docs/architecture/plan-integration/council/review7/seat-openai.gpt-6-astra.md new file mode 100644 index 000000000..0bb09a78d --- /dev/null +++ b/docs/architecture/plan-integration/council/review7/seat-openai.gpt-6-astra.md @@ -0,0 +1,176 @@ +## 1. Verdict + +**ACCEPT-WITH-CORRECTIONS** — the repair deferral implements D-F’s intent, but unsafe shell quoting, a violated final-ranking guarantee, and inaccurate compatibility claims prevent merging or treating this tree as final for row 3b. + +## 2. Findings + +Paths below are relative to `packages/studyloop/` unless stated otherwise. + +### F1 — 🔴 The generated session command does not safely quote the plan title + +**Location:** `src/studyloop/learning/decision.py::_body_double_candidate` + +Replacing `"` with `\"` is not shell argument quoting. Inside double quotes, titles containing `$()`, backticks, or variable references still undergo shell expansion. Backslashes also complicate the attempted quote escaping. + +The ranker does not execute this command, but it explicitly offers it for execution. Rich escaping protects presentation, not the shell. + +**Fix:** Construct the command using proper shell argument quoting, such as `shlex.join(["studyloop", "study", first.title, "--mode", "co-study"])` for the documented POSIX shell surface. Update exact command assertions rather than preserving unsafe double-quote formatting. + +**RED test:** `test_body_double_command_preserves_title_as_one_literal_shell_argument` + +Parameterize spaces, quotes, backslashes, dollar signs, backticks, and harmless command-substitution syntax. Verify a stub executable receives the exact title as one argument and no expansion occurs. Merely asserting `shlex.split(command)` is insufficient to establish absence of shell expansion. + +**Done:** Every tested title round-trips literally through the supported shell, with no extra command executed. + +### F2 — 🔴 Base-score ordering does not implement “every real candidate outranks it” + +**Location:** `learning/decision.py::BODY_DOUBLE_BASE_SCORE`, `_body_double_candidate`, scoring/ranking in `build_now_plan` + +The supplied adjustments establish a concrete inversion: + +- Body double: `30 + 12 = 42`. +- Unrelated hands-on practice without a topic switch: `48 − 14 = 34`. + +Conversation modality can raise the proposal further. Therefore, comparing base scores does not establish the advertised final ordering. + +Preferring body doubling over taxing practice could be a defensible product decision, but it is **not the approved “real candidates still rank above it” contract**. The unrelated-due test exercises only an easy case. + +**Fix:** Enforce an explicit fallback ordering for `source="body_double"` after ordinary scoring and before final selection, preserving real-candidate ordering. Keep rule 8’s separately documented alternate-slot reservation. Do not rely on another unexplained base constant. + +**RED test:** `test_body_double_ranks_below_real_candidates_after_all_adjustments` + +Include practice, continuity, and due candidates; hands-on penalties; topic-switch penalties; and conversation modality. + +**Done:** Every eligible real candidate precedes the proposal in the ranked sequence; rule 8 may still reserve the final displayed alternate without changing the primary. + +### F3 — 🟡 The revised no-plan compatibility promise is still false + +**Location:** repository-root `openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md`, `design.md`, `docs/cli-reference.md`, `docs/study-plans.md` + +“Without an active plan **and nothing deferred**, byte-identical” is disproved by the implementation: + +- A no-plan `learning` repair at low energy is not deferred, but gains `metadata.energy_demand`. +- A no-plan live struggle at medium energy is not deferred, but gains the same metadata field. + +The empty-world golden cannot detect either change. Additionally, the documentation’s “live struggle” exception is too narrow: older struggles and weak-teach-back-only rows also defer at low energy. + +**Decision on scope:** Keep repair deferral **plan-independent**. Reintroducing an unsafe recommendation merely because the learner lacks a plan would contradict the rationale for D-F. The appropriate correction is an accurate compatibility contract, not an active-plan gate. + +**Fix:** Explicitly document both changes: + +1. Struggle candidates gain additive demand metadata at every energy. +2. Above-capability repairs defer regardless of plan membership. + +Limit byte-identity claims to the actual unchanged payload cases, including the empty-world golden. Narrow the spec’s unconditional no-plan renderer scenario as well. + +**Regression test:** `test_no_plan_eligible_repair_exposes_demand_without_deferral` + +Parameterize `learning` at low energy and `struggling` at medium energy; assert demand metadata exists, deferral keys are absent, and no body double exists. + +**Done:** No specification or user-facing documentation promises unchanged nonempty no-plan payloads contradicted by this test. Preserve the existing golden hash. + +### F4 — 🟡 Important new decision boundaries and synthesis interactions are unpinned + +**Location:** `tests/test_now_plan_guidance.py` + +The real-collector fixture is a good choice, but the tests do not pin several claimed decisions: + +| Missing invariant | Required RED test | +|---|---| +| Exactly 14 days is high; 15 days is medium; missing/invalid timestamps conservatively remain high | `test_energy_demand_recency_boundaries_and_unknown_dates` | +| `learning` takes precedence over a weak score; old `struggling` plus a weak score remains medium under the chosen policy | `test_energy_demand_confidence_and_teachback_precedence` | +| Low capability rejects medium and high demand; medium/high capability accepts both | `test_repair_demand_capability_matrix` | +| Deferring the sole representative permits an energy-eligible milestone conversation, not another hands-on repair | `test_deferred_repair_allows_only_eligible_milestone_conversation` | +| Two ready plans produce one proposal, deterministic references, both deferred milestones named, and the first plan’s correctly quoted door | `test_body_double_two_ready_plans_has_deterministic_context` | + +**Assessment of the policy:** + +- **Fourteen days:** a reasonable explicit heuristic, not an empirically established recovery boundary. +- **Recorded-at recency:** consistent with the available projection. Describe it as assessment/report recency, not the date of last session difficulty. +- **Old struggle plus weak teach-back:** medium is consistent with the design. Making it high changes the label but not eligibility at any current energy setting; there is no evidence here requiring that change. +- **Medium versus high demand:** worth retaining as explanatory state and required-capability information. Document that both currently require *at least medium self-reported energy*. “High demand” does not mean “high energy required.” +- **Unknown timestamps:** conservative high demand is appropriate. + +**Done:** These parameterized tests pass through the collector/ranker where applicable, not merely through manually annotated candidates. + +### F5 — 🟡 The CLI still says repair stays available immediately before reporting it deferred + +**Location:** `src/studyloop/cli/_now.py::_render_plan` + +The milestone sentence still ends: + +> “Plan-related review and repair stay available.” + +That is misleading in the exact row-3b fixture, where the next line explains that plan-related repair is unavailable. + +**Fix:** Say “Due recall and gentle review stay available,” or explicitly qualify repair by its own energy demand. Check parallel renderer wording for the same obsolete assurance. + +**RED test:** `test_cli_milestone_deferral_does_not_promise_live_repair` + +Assert the complete relevant sentence and the separate repair line, not just the presence of “Deferred for energy” and a concept somewhere in the output. + +**Done:** Row 3b’s rendered explanation contains no contradictory promise of repair availability. + +### F6 — 🔵 Synthetic matching and multi-plan presentation need clearer semantics + +**Location:** `learning/decision.py::_body_double_candidate`, `_PlanContext.attach_refs`; spec rules 5–6 + +The explicit body-double references cover ready plans, but subsequent matching can attach additional references through: + +- The generated concept, such as `Sit with SQL Windows`. +- The first plan’s topic, including a matching unready plan. + +Thus, “names ready plans only” is true of the constructed reason/title list, not necessarily of rendered plan relevance. The correction test expressly permits a husk reference. + +**Fix:** At minimum, make the spec describe the actual two-stage behavior. Prefer treating the generated display label as presentation rather than a semantic concept match key for this synthetic source. Decide explicitly whether ordinary topic matching should add unready-plan references to the proposal. + +**Tests:** `test_body_double_display_label_does_not_create_incidental_plan_match` and `test_body_double_unready_reference_policy`. + +For two ready plans, using the first plan’s title as the CLI entry point is acceptable as a deterministic starting point, but it does not establish that both plans’ context reaches the session. The two-plan test in F4 should pin exactly what is promised. + +### F7 — 🔵 The Web door loses the proposal’s named context + +**Location:** `web/static/js/components/today-panel.js::startAction`, `viewForAction` + +Navigation to `body-double` is the correct harness-neutral destination. It is not a session start carrying the recommendation: the user must select context again. + +**Fix:** As a follow-on, pass the selected plan/topic through an existing navigation or picker-prefill mechanism. Do not create a session automatically or introduce another writer. + +**RED test:** `test_body_double_navigation_prefills_recommended_plan_without_starting_session` + +A browser journey should eventually verify the picker state. The present JS mapping test proves only view selection. + +This is **not a merge blocker** for the agreed session-door scope. Documentation and receipts should say “opens the Body Double view,” not imply a context-preserving session launch. + +### Remaining decisions + +- **Due recall and deferred repair on the same concept:** Keep both. They represent different actions. Suppressing the deferral would hide why repair disappeared. Make the distinction explicit in the copy. +- **Metadata-driven deferral:** Acceptable with the current producer invariant. A future producer must not attach repair-demand metadata to due recall; an explicit repair-source guard would make that invariant less fragile. +- **Milestone conversation after repair deferral:** Acceptable only as the separately energy-permitted conversation specified by the design, not as a relabelled repair task. F4 must pin that distinction. Owner acceptance of “talk about it” is not established by the existing three readings. +- **Rule 8:** Accept. This is an explicit finite-slot reservation, not filtering the collected candidate set. It does displace one displayed real alternate; describe that honestly. The primary must remain unchanged. +- **No-plan starter:** Defensible as a conservative, generic retrieval offer. Do not synthesize recall on the deferred struggle merely to maintain topical relevance; that would invent an unvalidated lower-demand action. A more contextual no-plan floor can be reviewed separately. +- **Twenty-five-minute body double:** Not inherently incompatible with low energy; duration and cognitive demand differ. The existing estimate-clamping behavior should remain. +- **`evidence_command`:** Keep the existing additive-compatible field. The source-aware CLI label correctly distinguishes a session door from a write. +- **MCP disclosure:** `to_json_dict()` establishes transport, not agent understanding. No tool-docstring/persona diff is provided, so disclosure is unestablished. Add a concise explanation at the existing `get_next_action` documentation surface; avoid duplicating this policy across harness-specific personas. +- **Recap:** Direct `_plan_context` testing is appropriate for the supplied low-energy object. It does not prove that ordinary `build_daily_recap()` produces low-energy deferrals; the spec scenario should distinguish renderer capability from the production medium-energy path. +- **Row 3b:** The three readings cover both halves of D-F and the due-recall exception. Keep owner verdicts pending. Call reading (c) **“recorded as learning”**, not demonstrated recovery. Ask separately whether an eligible milestone conversation on a deferred live-struggle concept is acceptable; that behavior is absent from the walkthrough. + +## 3. Refutations + +1. **“42 < any real candidate.”** False after the documented scoring adjustments; F2 gives a concrete counterexample. +2. **“No active plan and nothing deferred remains byte-identical.”** False for eligible struggle candidates because demand metadata is always added. +3. **“The one plan-independent change is live-struggle deferral.”** Incomplete: older struggles and weak teach-backs also defer, and eligible repair payloads change. +4. **“Each GREEN decision is a test.”** Not established. No added test exercises the sole deferred representative of an otherwise eligible milestone or the two-ready-plan proposal. +5. **“The Today card starts it.”** Only navigation is established. The brief expressly states that the picker receives no plan-title prefill. +6. **“Zero regressions.”** The matched control establishes no **new failing test IDs** at GREEN. It cannot establish absence of defects outside those tests, and it is not a full-suite result for `6d5a2d0e`. The disclosed scoped verification after corrections is useful and appropriately limited. +7. **“Learning” means recovered.** Not established by the projection facts. It is the chosen low-demand proxy, subject to owner judgment. + +## 4. Gate + +1. **Quote the session command safely** — F1’s literal shell-argument test. +2. **Enforce the final fallback ordering** — F2’s post-adjustment ranking test; retain the existing rule-8 primary-preservation test. +3. **Correct the compatibility/spec claims** — F3’s no-plan eligible-repair test; unchanged golden hash. +4. **Pin the missing policy boundaries and synthesis cases** — the five tests listed in F4. +5. **Remove contradictory repair-availability copy** — F5’s exact renderer test. + +Then run the affected Python/JS suites and obtain branch CI green on the final corrected tree. Record row 3b from that tree, leave its judgment to the owner, and do not archive before T5.5 is scored. diff --git a/docs/architecture/plan-integration/council/review7/seat-qwen3-coder.md b/docs/architecture/plan-integration/council/review7/seat-qwen3-coder.md new file mode 100644 index 000000000..baf5f780c --- /dev/null +++ b/docs/architecture/plan-integration/council/review7/seat-qwen3-coder.md @@ -0,0 +1,169 @@ +## 1. Verdict + +**ACCEPT-WITH-CORRECTIONS** — Item 5 correctly implements per-item energy demand for repair and introduces a body-doubling floor aligned with the owner's finding, but introduces behavioral changes to no-plan learners with live struggles that constitute scope creep without architectural justification. + +## 2. Findings + +### (a) Scope of deferral is plan-independent: creep without grounding + +- **File:** `learning/decision.py` lines 1170-1186 in `_defer_repairs` +- **Issue:** Deferral logic explicitly checks only `capability >= required` with no requirement that an active plan exist (`plan = plans.first_match(candidate)` may be `None`) +- **Why it matters:** The owner's concern focused on recommending hands-on repair during low energy risking "compounding the struggle and damaging confidence (RSD)" — but this was framed as occurring *within the context of a plan.* Extending this protection outside all planning context breaks D-5’s byte-compatibility promise and shifts user experience without consent. +- **Concrete Fix:** Gate repair deferral on presence of matchable active plan OR revise all documentation/testing accordingly +- **Test Name:** `test_no_active_plan_live_struggle_not_deferred_at_low_energy` + +🔴 Defect introduced by design choice inconsistent with root issue scope + +--- + +### (b) Demand derivation boundaries + +- **File:** `learning/decision.py` line 528 in `_energy_demand` +- **Issue:** + - A weak teach-back (`last_teachback_score < 14`) on a non-`struggling` concept results in `"medium"` demand + - `struggling` rows older than 14 days still labeled `"medium"` (consistent internally) + - Unparsable `last_seen` defaults to `"high"` (intended behavior noted) +- **Why it matters:** No functional distinction exists between medium/high capabilities at runtime (both map to minimum 6), so maintaining the differentiation adds complexity without operational benefit. Additionally, unclear if using `assessment_recorded_at` rather than actual learner engagement reflects meaningful recency for learning state. + +🟡 Must-fix – unnecessary behavioral partition that complicates implementation without outcome variance + +💡 Note – 14 day window aligns with typical spaced repetition cycles but could be configurable long-term + +--- + +### (c) Deferral mechanics interactions + +- **Files:** + - `learning/decision.py`: Lines 1167–1183 (`_defer_repairs`) + - `test_now_plan_guidance.py`: Test cases including `test_recovered_repair_stays_eligible_at_low_energy` + +- **Issue:** Due recall remains eligible even when concept also marked as `struggling`, leading to potential confusion where both ranked and deferred versions of same item appear. Example in test has overlapping ‘window function’ entries with distinct sources. + +- **Why it matters:** Learners might perceive inconsistency in recommendation engine when seeing both deferral message and same-concept primary activity. Could mislead perception of whether effort was truly spared. + +- **Concrete Fix:** Consider suppressing deferral note when identical concept ranked due to different source type (unless needed for audit trail) + +🟡 Should-fix – clarity improvement in learner-facing output quality + +--- + +### (d) Body double candidate structure + +- **File(s):** + - `learning/decision.py`: Constant definition and application in `_body_double_candidate()` function + - Tests in `test_now_plan_guidance.py` + +- **Issues Identified:** + - Base score setting leads to ranking anomalies (e.g., penalized hands-on practices beat body-doubles under low energy) + - Command generation relies solely on first-plan title; doesn't handle multi-plan situations gracefully beyond concept text change (`"Sit with Your Plans"`) + +- **Why it Matters:** Violates documented promise ("any real candidate outranks it"), particularly concerning modality-based scoring adjustments. Might inadvertently promote idle participation over productive cognitive load. + +🔴 Defect conflicting with core design principle + +🔵 Should-fix – adjust ranking priority handling around modality penalties vis-a-vis base ordering + +--- + +### (e) Interaction with Rule 8 slot preservation + +- **Files:** + - `learning/decision.py`: Implementation in `build_now_plan()` + - Modified test `test_preserves_one_plan_backed_action_when_energy_allows` asserting body-double slot insertion + +- **Issue:** Originally protected top candidates from losing spot due to energy constraint. Now inserts body-double alternative below second-best real candidate instead of letting third unrelated item occupy slot. + +- **Why it Matters:** Improves energy-aware relevance of suggestions but transforms rule semantics subtly—no longer guaranteeing *preservation* of existing items in favor of introducing plan-aware proposal. + +🟡 Must-fix – Clarify specification update to reflect revised policy intention + +--- + +### (f) Honest Starter Adjustment Adequacy + +- **Files:** + - `cli/_now.py`: Rendering changes referencing starter reason text + - `learning/decision.py`: Use of helper flag `after_deferral` + +- **Issue:** Starter continues proposing generic recall (“one tiny loop”) even with known high-value concept(s) recently made inaccessible due to energy restrictions. + +- **Why it matters:** Misses opportunity to scaffold learner gently back toward accessible aspects of deferred content. Might lead to mismatched motivational affordance relative to available context clues from deferrals. + +🟡 Must-fix – Introduce optional fallback synthesis of gentle reviews on deferred concepts + +--- + +### (g) Renderer Presentation Consistency + +- **Files:** + - `cli/_now.py` + - `today-panel.js` + - `recap.py` + +- **Issues:** + - CLI switches label from “Record evidence” to “Sit with the plan”, appropriately distinguishing session initiation modes. + - Web frontend omits auto-fill into Body Double picker despite CLI providing direct link + +- **Why it Matters:** Inconsistent UX across platforms undermines unified intent of co-study modality bridging. Failure to pre-populate decreases ease-of-use unnecessarily. + +🟡 Should-fix – Align interface affordances for consistent access pathway coherence + +🔵 Note – JSON key reuse (`evidence_command`) reasonable, although semantically ambiguous. Could extract clearer separation eventually. + +--- + +### (h) Spec Compliance Review + +- **File:** `specs/active-learning-decisions/spec.md` (MODIFIED Requirements Section) +- **Observations:** + - Updated requirement accurately reflects implemented features including dual deferral keys and rule expansion points. Includes detailed usage scenarios. + - Describes how body-double candidate synthesized based on no plan-relevant fit. + - Covers substring matching exclusion explicitly. + +🟢 Verified – Clear mapping between stated behaviors and resulting system state + +🟢 All described scenarios validated with test coverage in modified module + +--- + +### (i) Testing Completeness + +- **Coverage:** + - Comprehensive unit testing included covering each major derivation edge case (unready plans, deferred repair variations, multiplan situations). + - String assertions dominate over structured comparison in some outputs like reasons. Risks brittleness upon language localization or minor copyediting. + +🟡 Should-fix – Replace textual substring validations with deeper structural checks where possible + +🟢 Most critical paths pinned effectively via integration assertions involving full stack operation (JSON payloads returned correctly, DOM manipulation verified through snapshot techniques) + +--- + +### (j) Row 3b Validity for Owner Decisions + +- **Review Area:** Documentation: `receipts/now-rubric-2026-09-16.md` Row 3b details + +- **Evaluation Summary:** + - Readings accurately represent engine’s response dynamics: + - Case (a): Prioritized co-study over repair + - Case (b): Prioritized unrelated due task with clear labeling + - Case (c): Retains access to recovery-level review activities at appropriate cadence + +🟢 Well-formed triage prompts tailored directly to original feedback concerns + +🟡 Missing explicit ask regarding no-plan deferral consequences (should include prompt about acceptability of offering starter under such conditions in absense of plans) + +## 3. Refutations + +None identified; supporting materials thoroughly documented and cross-referenced throughout design documents and automated verification artifacts presented consistently. + +All claims either confirmed through empirical validation provided or addressed within amendment history already integrated in working branch under version control scrutiny. + +## 4. Gate + +Minimal set necessary to convert rejection into acceptance: + +| Correction | RED Test | +|------------|----------| +| Gating deferral on active plan availability | `test_no_active_plan_live_struggle_not_deferred_at_low_energy` | +| Adjust base scoring model ensuring body double ranked strictly beneath all active tasks regardless of penalty modifiers | `test_body_double_always_below_ranked_tasks_despite_modality_penalties` | +| Expand scenario coverage in 3b to include judgment query on appropriateness of starting with starter in no-plan environments post-deferral | Update `test_energy_deferred_scenario_cases_include_post_deferral_context_check` | diff --git a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md index 759ce0e94..3ff779212 100644 --- a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md +++ b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md @@ -1,6 +1,6 @@ # Plan-aware `now` — five-scenario human rubric (D-16) · 2026-09-16 -**Status: owner verdicts RECORDED 2026-09-16 (interactive walkthrough with the coordinator).** Scenarios 1, 2 and the primary of 4: yes. Scenario 3: **no** (finding). Scenario 4 completion action: **no as phrased** (finding). Scenario 5: verified. **Row 3b added 2026-09-19** (item 5 / D-F, tree `feat/energy-demand-body-double`; outputs emitted from the tree the GREEN commit records): scenario 3 re-run with the struggle collector live, three readings printed, verdict `PENDING` for the owner. **Row 4b added and scored 2026-09-18** (item 4 / D-G, tree `feat/plan-close`; outputs emitted from the tree committed as `82293293`): scenario 4 re-run with the end assessment planted, both proposals recorded as emitted; owner verdict **yes** on both readings — (a) *extend* with named evidence is a proposal the owner would walk, (b) a clean review's *close* is one the owner would agree to. This closes row 4's "no as phrased" finding; row 4 is kept as the record of the original finding. This receipt was produced unattended +**Status: owner verdicts RECORDED 2026-09-16 (interactive walkthrough with the coordinator).** Scenarios 1, 2 and the primary of 4: yes. Scenario 3: **no** (finding). Scenario 4 completion action: **no as phrased** (finding). Scenario 5: verified. **Row 3b added 2026-09-19** (item 5 / D-F, tree `feat/energy-demand-body-double`; outputs (a)–(c) emitted from the GREEN tree `326abcf9`, (d)–(e) from the post-review-7 tree `bdf4d6c6`): scenario 3 re-run with the struggle collector live, five readings printed, verdict `PENDING` for the owner. **Row 4b added and scored 2026-09-18** (item 4 / D-G, tree `feat/plan-close`; outputs emitted from the tree committed as `82293293`): scenario 4 re-run with the end assessment planted, both proposals recorded as emitted; owner verdict **yes** on both readings — (a) *extend* with named evidence is a proposal the owner would walk, (b) a clean review's *close* is one the owner would agree to. This closes row 4's "no as phrased" finding; row 4 is kept as the record of the original finding. This receipt was produced unattended overnight. Every scenario below was *run* on frozen fixtures and the primary and its rationale are recorded exactly as the engine emitted them; the "would I do the primary?" column is a human judgement that only the owner can @@ -30,7 +30,7 @@ learning"). | 1 | Matching due | Active plan `sql-windows` ("SQL Windows", topics `[sql]`, next milestone 0 concepts `[window function]`). Due items: `decorators`/python base 102, `window function`/sql base 100. | **`window function`** (sql, recall, score 130, `plan_refs=[(sql-windows, 0)]`); alternate `decorators` (120, no refs). | Rule 5: both are due items two points apart — one urgency class — so the plan-related one takes the +12 bias and wins; the unrelated due item is *kept* as an alternate (bias, not filter). Rule 7 names the milestone the action advances. | **yes** — owner, 2026-09-16: "window function is the logical step before decorating it" (a prerequisite-order argument; see F2 follow-on). | | 2 | Urgent-unrelated wins | Same plan. One due item: `decorators`/python, base 100 (an overdue spaced-repetition review). Nothing represents milestone 0. | **`decorators`** (118, no refs); alternate `window function` (60, `source=study_plan:sql-windows:0`, `plan_refs=[(sql-windows, 0)]`, reason "Next milestone 1/1 of plan 'SQL Windows': Window basics"). | Rule 5: the unrelated candidate is in a more urgent class (due review) and wins outright — the bias cannot lift new-milestone work over it. Rule 6: the plan's next milestone was unrepresented, so it was synthesised at base 48 + bias 12 = 60 and appears as the plan-backed alternate (rule 8 satisfied without any swap). | **yes** — owner, 2026-09-16: clear the overdue review first. Note for follow-on: an overdue item *unrelated* to the plan must not sit as an alternate indefinitely — propose it explicitly (age-aware nudge) or let the learner retire it. | | 3 | Energy-deferred | Plan `sql-windows` with `energy_floor: 5`; milestone 0 `Window basics` **done** (concepts `[window function]`), milestone 1 `Frames` (concepts `[window frame]`). One struggle-repair item `window function`/sql, `hands-on`, base 82. **Energy `low`** (capability 3/10). | **`window function`** (hands-on, score 80, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, milestone 1, floor 5, capability 3)]`; JSON gains `energy_deferred`. | Rule 3: capability 3 < floor 5, so the *new* milestone (Frames) is deferred and named, not synthesised; the plan-related repair on a finished milestone's concept stays eligible and keeps its ref (`None`: plan-related repair, not the next milestone). Score = 82 + 12 bias − 14 (hands-on at low energy). | **no** — owner, 2026-09-16: a struggle-repair task has no energy demand of its own; recommending hands-on repair of a *live* struggle on a low-energy day risks compounding the struggle and damaging confidence (RSD). Finding for council: (1) derive a per-item energy demand for repair from struggle recency/teach-back — at low energy a live struggle defers like new work, a recovered one stays eligible as gentle review; (2) when nothing plan-related fits the day's capability, synthesise a body-doubling / open-session candidate (feature exists: ADR-0001/0003, `web/routes/body_double.py`) naming the deferred items, instead of the least-bad task. | -| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered). | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason *"Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."*; no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone. | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 < any real candidate — a proposal, not a filter) carrying the co-study session door. | **PENDING** — owner: (a) would you sit with the plan rather than repair the live struggle today? (b) is the body-double proposal right to sit beneath the unrelated due recall? (c) is the gentle teach-back on a recovered concept one you would do at low energy? | +| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered); added after council review 7: **(d)** **no plan at all**, one live struggle `decorators`/python, low energy; **(e)** row 3's world plus one **unrelated hands-on** practice task `list comprehension drill`/python base 48, low energy. | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason *"Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."*; no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone. **(d)** primary **`one tiny recall loop`** (the starter, 28, reason *"Today's energy deferred the repair work it cannot carry; start with one small retrieval signal instead"*); no alternates; `energy_deferred_repairs=[(None, decorators, high, 6)]` — where before item 5 the primary was the hands-on repair of `decorators`. **(e)** primary **`Sit with SQL Windows`** (42); the hands-on drill is the alternate at 34 (48 − 14 low-energy penalty). | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 < any real candidate — a proposal, not a filter) carrying the co-study session door. | **PENDING** — owner: (a) would you sit with the plan rather than repair the live struggle today? (b) is the body-double proposal right to sit beneath the unrelated due recall? (c) is the gentle teach-back on a recovered concept one you would do at low energy? (d) with **no plan**, is the starter plus the deferred line the floor you want on a low-energy day, rather than the hands-on repair you used to get (review 7: two seats keep it, one would restore the repair)? (e) at low energy, would you sit with the plan rather than do an unrelated hands-on drill (review 7 F2: the proposal outranks a task the low-energy rule penalises, and nothing else)? | | 4 | Fully-checked | Plan `done-plan` ("Done Plan"), milestones A and B both done. One due item `decorators`/python base 100. | **`decorators`** (118, no refs); `completion_actions=[(done-plan, "Every milestone of 'Done Plan' is checked off — close the plan or extend it with a follow-on mission.")]`; no `study_plan:` candidate anywhere; JSON gains `active_plans` + `completion_actions`. | Rule 9: a fully-checked plan is reported as a completion action and is neither matched (no bias, no refs) nor synthesised. | **primary yes / completion action no as phrased** — owner, 2026-09-16: the completion action must be contextual and consensual. Run the end assessment (`assess(phase="end")`: due reviews, struggles, unverified milestones on the plan's concepts). If outstanding work touches the plan's concepts (or their prerequisites — F2 concept edges), propose *extend* and name the evidence; if clean, propose *close* and ask the learner to agree ("anything you are not comfortable with?"). Status never changes automatically (#7). Natural vehicle: architect with `purpose=planning` and the assessment in the brief (`plan close <id>`, sibling of `plan repair <id>`). Finding for council. | | 4b | Fully-checked — **re-run after D-G (item 4, 2026-09-18)** | Row 4's fixture (`done-plan`, milestones A `[alpha]` and B `[beta]` both done; one due item `decorators`/python base 100), plus the end assessment's readers planted: **(a)** one due review on plan concept `alpha` (`overdue`) with session mentions backing both concepts; **(b)** no due rows, same mentions. | Primary unchanged in both: **`decorators`** (118, no refs); no `study_plan:` candidate. **(a)** `completion_actions=[(done-plan, due 1 / struggles 0 / unverified 0, proposal **extend**, evidence `["Due review: alpha — overdue"]`)]`, sentence: "Every milestone of 'Done Plan' is checked off, and the closing review proposes extending the plan — 1 due review, 0 struggles and 0 unverified milestones on its concepts. Walk the evidence with the architect: studyloop plan close done-plan." **(b)** counts 0/0/0, proposal **close**, evidence `[]`, sentence: "Every milestone of 'Done Plan' is checked off and the closing review is clean — it proposes closing the plan. Close it with the architect when you agree: studyloop plan close done-plan." No warnings; JSON gains the five keys only inside the entry. | Rule 9 as before for the ranking. The completion action is now the end assessment read as a preview (`assess(phase="end", record=False)`, one call, no write, no status change): `extend` iff any of the three counts on the plan's own concepts is above zero, else `close`; due counts only rows naming a concept (the scheduler's "new topic" row is excluded — owner decision 2026-09-17). `plan close done-plan` launches the architect with the same review as the brief's first section; status moves only when the learner agrees. | **yes / yes** — owner, 2026-09-18, answering the two questions as posed: (a) **yes**, a proposal the owner would walk; (b) **yes**, a close the owner would agree to. No further line given. Closes row 4's "no as phrased" finding; status still moves only when the learner agrees in the architect conversation. | | 5 | No-plan identical | No plan documents at all; no collector candidates. | **`one tiny recall loop`** (python, recall, 28, `source=starter`) — the starter. | D-5: `serialise(plan) == golden` → **byte-identical** (`True` in the run); the JSON key list is exactly the golden's — no additive key is present. | **verified** — owner walkthrough 2026-09-16: nothing to judge; the byte-identical golden is the acceptance. | diff --git a/openspec/changes/plan-integration-followons/tasks.md b/openspec/changes/plan-integration-followons/tasks.md index da05fd770..4914d3cce 100644 --- a/openspec/changes/plan-integration-followons/tasks.md +++ b/openspec/changes/plan-integration-followons/tasks.md @@ -215,7 +215,17 @@ writer, through the existing gate, closes that. Kept out of item 3 so item 3's f "Decisions taken at GREEN" 1-5. Module 46/46, JS 144/144, e2e plan journeys 20/20, docs contract 39/39, `mkdocs --strict` exit 0, `openspec validate` valid; full suite vs a clean `main` control — see the GREEN commit's receipt.) -- [ ] ⚖ **T5.4** Council review 7 (`review7`, same seats or `kimi-k2-thinking` third); arbitration; corrections. +- [x] ⚖ **T5.4** Council review 7 (`review7`, same seats or `kimi-k2-thinking` third); arbitration; corrections. + (2026-09-19: brief `council/brief-review7-2026-09-19.md` on tree `6d5a2d0e`; seats astra + ACCEPT-WITH-CORRECTIONS (2 🔴 / 3 🟡), qwen ACCEPT-WITH-CORRECTIONS (2 🔴), grok ACCEPT (🔵/💡 only; 16k cap + cut its Gate). Arbitration `council/review-7-arbitration-2026-09-19.md`: **GATE ACCEPT** for the corrected + tree. Landed one commit each: F1 shell quoting of every offered command (`d1935256`, real-`sh` test, + stash-proved), F5 no false repair promise (`c1d7f2f2`), F7 Web door hands the plan over (`3347567e`), grok's + "Your plans" label (`f814dc36`), F4 five pins + a pre-existing NULL-`last_seen` collector crash they found + (`7194b66d`), F2 ordering claim corrected — behaviour kept, judgement to the owner (`02e282e6`), F3/F6 + contract wording (`bdf4d6c6`). Rejected with reasons: qwen's gate-deferral-on-a-plan (2–1 against, and D-F's + own words), collapsing medium/high, a same-concept gentle recall floor (follow-on, decision 8). Row 3b gained + readings (d) and (e) for the two places the seats split.) - [ ] **T5.5** Rubric row **3b** for the owner (`PENDING`); the change is archived only after the owner scores 3b (4b was scored **yes / yes** on 2026-09-18, so 3b is the one row still outstanding). (2026-09-19: row 3b written into `receipts/now-rubric-2026-09-16.md` with three readings printed from the real engine — (a) live From 849c78aa08ae8bfa8f5c3a772ee1c91fb6be4ab3 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sat, 19 Sep 2026 13:50:00 +0100 Subject: [PATCH 17/32] docs(receipts): full-suite matched control re-run on the review-7 tree MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit item5 5146 passed vs control 5120, 44 failing ids identical on both sides and to the committed environmental set — zero regressions after the seven review-7 corrections. --- .../council/review-7-arbitration-2026-09-19.md | 4 +++- .../receipts/full-suite-control-item5-2026-09-19.md | 11 +++++++++++ 2 files changed, 14 insertions(+), 1 deletion(-) diff --git a/docs/architecture/plan-integration/council/review-7-arbitration-2026-09-19.md b/docs/architecture/plan-integration/council/review-7-arbitration-2026-09-19.md index f8c81c417..90a4c7e0e 100644 --- a/docs/architecture/plan-integration/council/review-7-arbitration-2026-09-19.md +++ b/docs/architecture/plan-integration/council/review-7-arbitration-2026-09-19.md @@ -85,7 +85,9 @@ named, and stash-proved where a fix could be reverted alone (F1: 3 of 7 titles f parametrisations), golden byte-identical; with `test_learning_decision.py` and the docs contract **96 passed**; JS **147/147** (+3); `openspec validate` valid; `mkdocs --strict` exit 0; ruff / ruff format / pyright clean on every touched file. Full suite and matched - control on the final tree: see the receipt named in `tasks.md` T5.4 once it lands (running at arbitration time). + control on the final tree `3eb31f2d`: 30 failed / **5146 passed** / 14 errors vs control 30 / 5120 / 14; item5 ∖ + control = ∅, control ∖ item5 = ∅, item5 ∖ committed environmental set = ∅ + (`receipts/full-suite-control-item5-2026-09-19.md`, second table). ### Process findings diff --git a/docs/architecture/plan-integration/receipts/full-suite-control-item5-2026-09-19.md b/docs/architecture/plan-integration/receipts/full-suite-control-item5-2026-09-19.md index 0f415e32e..66f19bb9a 100644 --- a/docs/architecture/plan-integration/receipts/full-suite-control-item5-2026-09-19.md +++ b/docs/architecture/plan-integration/receipts/full-suite-control-item5-2026-09-19.md @@ -34,3 +34,14 @@ unchanged here, byte for byte. - `test_docs_plan_integration_contract.py` + `test_ci_workflow_contract.py` 39/39. - `mkdocs build --strict` exit 0; `openspec validate plan-integration-followons` valid. - ruff check / ruff format --check / pyright: clean on every touched file. + +## Re-run after council review 7 (tree `3eb31f2d`, corrections F1–F7 landed) + +| Tree | Result | +| --- | --- | +| **item 5 final** (`3eb31f2d`) | 30 failed, **5146 passed**, 4 skipped, 804 deselected, 14 errors (6:17) | +| **control** (`main` `4f8e3e0f`, same worktree as above) | 30 failed, 5120 passed, 4 skipped, 804 deselected, 14 errors (6:13) | + +item5 ∖ control = **∅**; control ∖ item5 = **∅**; item5 ∖ committed environmental set = **∅**. The +26 passed are +the review-7 tests (F1's seven parametrisations, F2's two, F3's two, F4's five plus the capability matrix's three, +F5, the ready-plans-only test, and the F7/label JS pins run separately: 147/147). From a0946376a1db36d0c93866b56cbf11985b6d4629 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 09:08:09 +0100 Subject: [PATCH 18/32] test(now): RED -- the body-double reason leads with recorded progress and describes its door truthfully Applied from the rubric-3b council (2026-09-20), verified against source: - the framework rule "never name a struggle without an adjacent strength" (agents/shared/audhd-framework.md, Naming Struggle Topics) -- the reason named the deferred struggle with nothing beside it, though PlanSummary already carries milestone_done/milestone_total; a second pin says no progress sentence is ever invented when nothing is done (already true); - "no new material, no repair" promised a door the engine does not control; the co-study persona guarantees "the student drives" and "stay quiet by default", so the reason should say that, and a third pin holds the persona to it. One red for the intended reason; the other two are guards that stay green. --- .../studyloop/tests/test_now_plan_guidance.py | 48 +++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/packages/studyloop/tests/test_now_plan_guidance.py b/packages/studyloop/tests/test_now_plan_guidance.py index 336486f3e..f51fd7421 100644 --- a/packages/studyloop/tests/test_now_plan_guidance.py +++ b/packages/studyloop/tests/test_now_plan_guidance.py @@ -1356,6 +1356,54 @@ def test_body_double_candidate_is_synthesised_when_nothing_plan_related_fits(mon assert payload["primary"]["source"] == "body_double" +def test_body_double_reason_leads_with_recorded_progress_and_describes_the_door_truthfully( + monkeypatch, +) -> None: + """Rubric-3b council (2026-09-20), applied: the framework's own rule is *never name a + struggle without an adjacent strength* — the reason opens with the progress the plan + document records (one milestone done here) before it names what is deferred. And the + door is described by what the co-study persona guarantees — the learner drives, the + companion stays quiet unless asked — not by a promise ("no new material, no repair") + that nothing pins.""" + _row3_plan() + _plant_struggles(monkeypatch, _struggle("window function", days_ago=3)) + + reason = build_now_plan(energy="low").primary.reason + + assert reason.startswith("1 of 2 milestones of SQL Windows done."), reason + assert reason.index("done") < reason.index("window function"), "progress before the struggle" + assert "you drive; the companion stays quiet unless you ask" in reason + assert "no new material" not in reason and "no repair" not in reason + + +def test_body_double_reason_never_fabricates_progress(monkeypatch) -> None: + """No milestone done → no progress sentence at all (never invent a strength).""" + _plan( + "sql-windows", + title="SQL Windows", + energy_floor=5, + milestones=[Milestone(title="Frames", concepts=["window frame"])], + ) + _plant_struggles(monkeypatch, _struggle("window frame", days_ago=3)) + + reason = build_now_plan(energy="low").primary.reason + + assert "milestones of SQL Windows done" not in reason and "0 of" not in reason + assert reason.startswith("Nothing plan-related fits low energy today"), reason + + +def test_co_study_persona_pins_the_promise_the_body_double_reason_makes() -> None: + """The reason says the companion stays quiet and the learner drives; the persona the + door launches must say so too, or the reason is a promise about a door it does not + control.""" + persona = ( + Path(__file__).resolve().parents[3] / "agents" / "shared" / "personas" / "co-study.md" + ).read_text(encoding="utf-8") + + assert "The student drives" in persona + assert "Stay quiet by default" in persona + + def test_body_double_is_a_proposal_not_a_filter(monkeypatch) -> None: """An unrelated real candidate still wins; the body-double proposal sits beneath it as an alternate, base score below any real candidate's.""" From c519bc2a674f712668bc5e444f5fcfc64a84d085 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 09:11:29 +0100 Subject: [PATCH 19/32] feat(now): the body-double reason leads with recorded progress and describes its door truthfully GREEN for a0946376. From the rubric-3b council (2026-09-20), applied only where the finding held against source: - lead with a strength that exists: "N of M milestones of <plan> done." from PlanSummary.milestone_done -- the framework rule for naming deferred work -- and nothing when nothing is done (never invented); - describe the door by what its persona guarantees ("you drive; the companion stays quiet unless you ask") instead of a promise nothing pinned. Docs, spec delta and design (decision 9) say the same; the decision also records what was NOT applied (plan-gated deferral, unnamed deferrals) and reading (f), the same-concept due-recall collision two seats named. --- docs/study-plans.md | 5 +++-- .../plan-integration-followons/design.md | 15 +++++++++++++++ .../specs/active-learning-decisions/spec.md | 8 ++++++-- .../src/studyloop/learning/decision.py | 18 ++++++++++++++++-- 4 files changed, 40 insertions(+), 6 deletions(-) diff --git a/docs/study-plans.md b/docs/study-plans.md index ca023215f..435350ddd 100644 --- a/docs/study-plans.md +++ b/docs/study-plans.md @@ -226,8 +226,9 @@ not recommended — an older struggle or a weak teach-back needs medium energy, and a concept you are still learning is the gentle review that stays available at any energy; due reviews are never deferred. When nothing plan-related fits the day's energy, the recommendation is to **sit with the -plan** — a body-double session, no new material, no repair — with the -deferred items named; a real, unrelated action still outranks that proposal +plan** — a body-double session: you drive, the companion stays quiet unless +you ask — with the recorded progress named first and the deferred items after +it; a real, unrelated action still outranks that proposal when one exists. An active plan that is **not ready** — a hand edit removed its mission or its milestones — is listed with a warning naming what to repair; it still biases related work, diff --git a/openspec/changes/plan-integration-followons/design.md b/openspec/changes/plan-integration-followons/design.md index 6b27170ee..a4dc72445 100644 --- a/openspec/changes/plan-integration-followons/design.md +++ b/openspec/changes/plan-integration-followons/design.md @@ -313,6 +313,21 @@ has none; `INTERLEAVE_RATIOS["low"]` unchanged. 8. **Not taken, recorded as follow-ons:** a same-concept gentle `recall` synthesised for a no-plan learner whose only candidates were deferred (grok 🔵 / qwen 🟡; astra: do not invent an unvalidated lower-demand action) — the owner's row 3b reading (d) decides whether the starter is the floor they want. +9. **The body-double reason leads with recorded progress and describes its door truthfully** (rubric-3b council, + 2026-09-20, three seats asked for recommendations to the owner, not verdicts). Applied where the finding held + against source: the framework's own naming rule (*never name a struggle without an adjacent strength*) — the + reason now opens with `N of M milestones of <plan> done.` from `PlanSummary.milestone_done`, and says nothing + when nothing is done (a strength is never invented); and "no new material, no repair" promised a door the + engine does not control — the sentence now says what the co-study persona guarantees ("you drive; the + companion stays quiet unless you ask"), and a test holds the persona to those words + (`test_body_double_reason_leads_with_recorded_progress_and_describes_the_door_truthfully`, + `test_body_double_reason_never_fabricates_progress`, + `test_co_study_persona_pins_the_promise_the_body_double_reason_makes`). Not applied: gating deferral on a + plan (Q1, rejected again by every seat) and suppressing the deferred-item names (astra 🟡) — D-F asks for + them by name; it goes to the owner as a note on row 3b. Two seats independently named the missing reading — + the live struggle that is **also due for recall** on the same concept — so it is emitted from the tree as + row 3b reading **(f)**: due recall is never deferred, so the recall is primary while the same concept's repair + sits in the deferred list; whether that is the collision the owner wants is theirs to say. Known edge, not solved here: a `struggling` row whose `last_seen` cannot be parsed is read as live (`high`) — the cautious side; `_days_since` returns `None` and the demand falls to `high`. diff --git a/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md b/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md index 7d6789ef3..f86fe9471 100644 --- a/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md +++ b/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md @@ -157,8 +157,12 @@ the body-doubling floor): penalises may not; a proposal, never a filter: nothing is removed from the ranking — `plan_refs` `(plan_id, None)` for every **ready** matchable plan (rule 7 may add a reference to an unready plan whose topic the proposal - shares; the proposal itself names ready plans only), a reason naming the deferred - milestones and repairs it stands in for, and `evidence_command` the + shares; the proposal itself names ready plans only), a reason that opens with + the progress the plan records (`N of M milestones of <plan> done.`, omitted when + none is done — never invented) before naming the deferred milestones and + repairs it stands in for, and describes the door by what the co-study persona + guarantees (the learner drives; the companion stays quiet unless asked), and + `evidence_command` the co-study session door — `studyloop study "<plan title>" --mode co-study` — set explicitly, never a progress write. No active plan (a draft is not one) → no body-double candidate; when every real candidate was deferred diff --git a/packages/studyloop/src/studyloop/learning/decision.py b/packages/studyloop/src/studyloop/learning/decision.py index 151850151..b17c479cb 100644 --- a/packages/studyloop/src/studyloop/learning/decision.py +++ b/packages/studyloop/src/studyloop/learning/decision.py @@ -1260,14 +1260,28 @@ def _body_double_candidate( f"milestone {d.milestone_index + 1} “{d.title}” of {d.plan_title}" for d in plans.deferred ] + [f"repair of “{d.concept}”" for d in deferred_repairs] deferred_note = f" — deferred: {'; '.join(items)}" if items else "" + # The framework's naming rule: never name a struggle without an adjacent + # strength — so lead with the progress the plan document records, and only + # when there is some (a strength is never invented). + done = [plan for plan in named if plan.milestone_done] + progress_note = ( + " ".join( + f"{plan.milestone_done} of {plan.milestone_total} milestones of {plan.title} done." + for plan in done + ) + + " " + if done + else "" + ) topic = first.topics[0] if first.topics else "study" return _Candidate( concept=f"Sit with {first.title}" if len(named) == 1 else "Sit with your plans", topic=topic, course=None, reason=( - f"Nothing plan-related fits {energy} energy today{deferred_note}. " - f"Sit with {titles} instead: a body-double session, no new material, no repair." + f"{progress_note}Nothing plan-related fits {energy} energy today{deferred_note}. " + f"Sit with {titles} instead: a body-double session — you drive; the companion " + "stays quiet unless you ask." ), action_type="conversation", estimated_minutes=_estimate_minutes("conversation", time_minutes, 25), From e770733e376b79a39077919b9e54b6421297c5da Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 09:15:14 +0100 Subject: [PATCH 20/32] docs(rubric-3b): council decision brief for the owner; receipt amended; reading (f) added Three seats (gpt-6-astra, grok-4.6, qwen3-coder) were asked for a recommendation per reading with its basis, opposing case and falsifier -- not for verdicts; the rubric question is the owner's. The brief arbitrates them and lists what was applied (c519bc2a), what was not and why, and which seat claims were checked and found wrong. The rubric receipt: the rationale's stale "42 < any real candidate" (review 7 F2 never reached it) corrected; (a)/(e) re-emitted from c519bc2a; reading (f) -- the live struggle that is also due for recall, named by two seats independently -- emitted from the tree and added with its owner question. --- .../council/brief-rubric-3b-2026-09-20.md | 363 ++++++++++++++++++ .../council/rubric-3b/manifest.json | 47 +++ .../council/rubric-3b/seat-grok-4.6.md | 86 +++++ .../rubric-3b/seat-openai.gpt-6-astra.md | 116 ++++++ .../council/rubric-3b/seat-qwen3-coder.md | 133 +++++++ .../receipts/now-rubric-2026-09-16.md | 4 +- .../rubric-3b-decision-brief-2026-09-20.md | 146 +++++++ .../plan-integration-followons/tasks.md | 4 + 8 files changed, 897 insertions(+), 2 deletions(-) create mode 100644 docs/architecture/plan-integration/council/brief-rubric-3b-2026-09-20.md create mode 100644 docs/architecture/plan-integration/council/rubric-3b/manifest.json create mode 100644 docs/architecture/plan-integration/council/rubric-3b/seat-grok-4.6.md create mode 100644 docs/architecture/plan-integration/council/rubric-3b/seat-openai.gpt-6-astra.md create mode 100644 docs/architecture/plan-integration/council/rubric-3b/seat-qwen3-coder.md create mode 100644 docs/architecture/plan-integration/receipts/rubric-3b-decision-brief-2026-09-20.md diff --git a/docs/architecture/plan-integration/council/brief-rubric-3b-2026-09-20.md b/docs/architecture/plan-integration/council/brief-rubric-3b-2026-09-20.md new file mode 100644 index 000000000..439aebf42 --- /dev/null +++ b/docs/architecture/plan-integration/council/brief-rubric-3b-2026-09-20.md @@ -0,0 +1,363 @@ +# Council brief — rubric row 3b: a decision brief for the owner (2026-09-20) + +You are one seat of a three-seat council. You do **not** see the other seats. You are **not** the owner, and +this brief does **not** ask you to score the rubric: the rubric's question is "would *I*, the learner, do the +primary?" and only the owner can answer it. What you are asked for is a **recommendation per reading**, grounded +in evidence-based practice for AuDHD adult learners and in the project's own framework documents quoted below, +so the owner's scoring becomes "agree or overrule" rather than reconstruction. Where practice is genuinely +undetermined, say so — a confident recommendation with no basis is worth less than "either is defensible +because X". + +The owner is a neurodivergent (ADHD + ASD) senior engineer retraining from networking into data engineering, +self-taught, learning Python/SQL/data engineering with StudyLoop. The engine under review recommends the next +study action from live evidence (due spaced-repetition reviews, recorded struggles, an active study plan) and +the learner's stated energy for the day. + +## 0. The owner decision that item 5 implements (HANDOFF §2, verbatim row) + +| D-F | Scenario 3 (**no**): a struggle-repair task has no energy demand; hands-on repair of a live struggle on a low-energy day compounds the struggle (RSD). Derive per-item energy demand from struggle recency / teach-back; when nothing plan-related fits the day's capability, synthesise a **body-doubling / open-session** candidate (feature exists: ADR-0001/0003, `web/routes/body_double.py`) naming the deferred items. | + +## 1. Rubric row 3 — the original finding (verbatim from the rubric receipt) + +| 3 | Energy-deferred | Plan `sql-windows` with `energy_floor: 5`; milestone 0 `Window basics` **done** (concepts `[window function]`), milestone 1 `Frames` (concepts `[window frame]`). One struggle-repair item `window function`/sql, `hands-on`, base 82. **Energy `low`** (capability 3/10). | **`window function`** (hands-on, score 80, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, milestone 1, floor 5, capability 3)]`; JSON gains `energy_deferred`. | Rule 3: capability 3 < floor 5, so the *new* milestone (Frames) is deferred and named, not synthesised; the plan-related repair on a finished milestone's concept stays eligible and keeps its ref (`None`: plan-related repair, not the next milestone). Score = 82 + 12 bias − 14 (hands-on at low energy). | **no** — owner, 2026-09-16: a struggle-repair task has no energy demand of its own; recommending hands-on repair of a *live* struggle on a low-energy day risks compounding the struggle and damaging confidence (RSD). Finding for council: (1) derive a per-item energy demand for repair from struggle recency/teach-back — at low energy a live struggle defers like new work, a recovered one stays eligible as gentle review; (2) when nothing plan-related fits the day's capability, synthesise a body-doubling / open-session candidate (feature exists: ADR-0001/0003, `web/routes/body_double.py`) naming the deferred items, instead of the least-bad task. | + +## 2. Rubric row 3b — the five emitted readings the owner must score (verbatim) + +| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered); added after council review 7: **(d)** **no plan at all**, one live struggle `decorators`/python, low energy; **(e)** row 3's world plus one **unrelated hands-on** practice task `list comprehension drill`/python base 48, low energy. | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason *"Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."*; no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone. **(d)** primary **`one tiny recall loop`** (the starter, 28, reason *"Today's energy deferred the repair work it cannot carry; start with one small retrieval signal instead"*); no alternates; `energy_deferred_repairs=[(None, decorators, high, 6)]` — where before item 5 the primary was the hands-on repair of `decorators`. **(e)** primary **`Sit with SQL Windows`** (42); the hands-on drill is the alternate at 34 (48 − 14 low-energy penalty). | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 < any real candidate — a proposal, not a filter) carrying the co-study session door. | **PENDING** — owner: (a) would you sit with the plan rather than repair the live struggle today? (b) is the body-double proposal right to sit beneath the unrelated due recall? (c) is the gentle teach-back on a recovered concept one you would do at low energy? (d) with **no plan**, is the starter plus the deferred line the floor you want on a low-energy day, rather than the hands-on repair you used to get (review 7: two seats keep it, one would restore the repair)? (e) at low energy, would you sit with the plan rather than do an unrelated hands-on drill (review 7 F2: the proposal outranks a task the low-energy rule penalises, and nothing else)? | + +The row's columns are: scenario/fixture · primary emitted (with score, source, plan refs, the offered command, +the reason sentence) · engine rationale · the owner's question per reading. The five readings are **(a)–(e)**. +## 3. Design §5 — the item-5 design and its recorded decisions (verbatim) + +#### 5. Item 5 — per-item energy demand and the body-doubling floor (D-F) — designed here, reviewed separately + +*(One page, written before item 5's RED; see tasks T5.\*.)* + +- **Energy demand per candidate.** `_struggle_candidates` derives `energy_demand ∈ {low, medium, high}` from + struggle state: `confidence == "struggling"` (a live struggle, ≤ 14 days) → `high`; `struggling` older than + 14 days or a weak teach-back → `medium`; recovered / gentle review → `low`. Demand maps to a required + capability (`high` → 6, `medium` → 4, `low` → 0) compared with `ENERGY_CAPABILITY[energy]`. +- **Rule 3 extended.** Below capability, *repair* above demand is deferred exactly like new milestone work and + listed in `energy_deferred` with a reason naming the struggle; recovered repair stays eligible as gentle + review. Due recall (`source=study_progress` due rows) is unaffected. +- **Body-doubling floor.** When the eligible plan-related set is empty **and** at least one active plan exists, + synthesise one candidate: `source="body_double"`, `action_type="conversation"`, low base score (below any + real candidate), reason naming the deferred items, `plan_refs` for each named plan with `milestone_index + None`, and an `evidence_command` that opens the existing body-double session route (`studyloop study + --mode co-study` / `web/routes/body_double.py`). A proposal, not a filter: real candidates still rank above + it. +- **No-plan output byte-identical to the golden**; `INTERLEAVE_RATIOS["low"]` unchanged (the design does not + call for it). +- **Rubric row 3b** (owner scores): scenario 3's fixture at low energy now yields the deferred repair named in + `energy_deferred` and a body-double primary (or the due recall if one exists). + +**T5.1 review against the code (2026-09-18, tree `7208eb67`) — three amendments, each from reading +`learning/decision.py`, not the text above:** + +1. **Demand classes are the struggle collector's classes.** `_struggle_candidates` emits a row only when + `confidence in ("struggling", "learning")` or `last_teachback_score < 14`; "recovered / gentle review" is not a + row it produces. So: `struggling` with `last_seen` ≤ 14 days → `high`; `struggling` older than 14 days, or any + row whose only signal is a weak teach-back → `medium`; `learning` → `low`. Demand is derived once, in the + collector, and carried in the candidate's `metadata` beside `confidence` so the scorer and the renderers read + one value. Required capability `high → 6`, `medium → 4`, `low → 0` stands (the `low` class is what "repair is + cheaper than encoding" was always about). +2. **`energy_deferred` is milestone-shaped and cannot carry a repair as it is.** `DeferredMilestone` has a + mandatory `milestone_index`, and all three renderers (`cli/_now.py`, `learning/recap.py`, + `today-panel.js::deferredNotes`) print `milestone {index + 1} "{title}" needs energy {floor}/10`. A deferred + repair gets its own frozen `DeferredRepair` (`plan_id`/`plan_title` when plan-related, else `None`, `concept`, + `topic`, `confidence`, `energy_demand`, `required_capability`, `energy_capability`, `reason` naming the + struggle), carried in a **new additive key `energy_deferred_repairs`** — not folded into `energy_deferred`, + whose consumers would print "milestone None". Same "readable off the top" rule as the closing review's + evidence lines: each renderer gains one line per deferred repair. +3. **The body-double door is a session start, not `web/routes/body_double.py`.** That route is the read-only focus + reader (`GET /api/body-double/focus`). The session door is `studyloop study "<topic>" --mode co-study` on the + CLI and a session start from the Body Double view (origin `body-double`) on the Web. `_evidence_command` has + no branch for a `conversation` candidate and would fall through to `studyloop progress … -c learning`, which is + a write, not a door — so the body-double candidate carries `evidence_command = 'studyloop study "<plan title>" + --mode co-study'` set explicitly, and `_evidence_command` is not asked to guess. `source="body_double"`, + `action_type="conversation"`, base score below `MILESTONE_BASE_SCORE` (48) so any real candidate outranks it. + +Rule 3's *deferral* of repair is the change; rule 3's *eligibility* of plan-related due recall is untouched. The +no-plan golden stays byte-identical because a body-double candidate requires an active plan and the golden world +has none; `INTERLEAVE_RATIOS["low"]` unchanged. + +**Decisions taken at GREEN (2026-09-19), each a test in `test_now_plan_guidance.py`:** + +1. **The deferral is plan-independent.** A live struggle is a live struggle whether or not a plan names it + (amendment 2's `plan_id … else None` already said so); the finding was about the learner's day, not the + plan. The body double, by contrast, *requires* a matchable active plan — it is "sit with the plan". + Consequence, stated rather than hidden (sharpened by review 7 F3): D-5's "a learner with no active plan + receives the pre-#10 payload byte for byte" holds for a no-plan learner **with no struggle candidate and + nothing deferred** — the golden world. Two changes are plan-independent: every struggle-collector candidate + carries `metadata.energy_demand` at every energy, and repair above the day's capability (a live or older + struggle, a weak teach-back, at low energy) is deferred — with the starter if nothing else was collected — + where the learner used to get the repair itself. Review 7 put the scope question to three seats: two (astra, + grok) keep it plan-independent ("gating it on a plan would leave the original no unfixed for every no-plan + learner"), one (qwen) would gate it; arbitrated as **keep**, the contract re-worded to the truth above in the + spec delta and both docs, and the no-plan floor put to the owner as row 3b reading (d). +2. **Rule 8's guaranteed slot below the floor is the body-double proposal.** It carries `plan_refs`, so where + four unrelated due items outrank everything at low energy the second alternate is now the proposal, not a + third unrelated item. It advertises no work the energy cannot carry — the property rule 8's docstring + protects — and the primary is untouched. `test_preserves_one_plan_backed_action_when_energy_allows` says so. +3. **The starter tells the truth after a deferral.** With no plan and every real candidate deferred, the starter + stands in; its reason now says the energy deferred the repair work rather than "no learning evidence found + yet", which would be false. The golden world defers nothing, so its sentence is unchanged. +4. **Body-double shape.** Base `BODY_DOUBLE_BASE_SCORE = 30` — below every real candidate's *base*; after the + day's adjustments it sits above a hands-on task the low-energy rule penalises (48 − 14 = 34 < 30 + 12 = 42) + and below every due and conversation candidate. Review 7 F2 (two seats 🔴, one 💡) was arbitrated as the + energy rule doing what the finding asked, not a filter: the *claim* "any real candidate outranks it" was the + defect, corrected in the constant's comment, the spec and here, and pinned by + `test_body_double_ordering_after_adjustments_follows_the_energy_rule`; the judgement is row 3b reading (e). + Concept `Sit with <title>` (one plan) / `Sit with your plans`; `plan_refs` for every **ready** matchable plan + (rule 7 may add a topic-matched husk reference; the proposal names ready plans only); reason naming each + deferred milestone and repair; command `studyloop study <title> --mode co-study`, the title quoted as one + shell argument (review 7 F1). The Today card starts it in the Body Double view and hands the plan title over + (`body-double-request`, review 7 F7); the CLI labels the command "Sit with the plan". +5. **A deferred repair does not "represent" a milestone** (rule 6 runs after the deferral), so an eligible + milestone whose only collected representative was a deferred live struggle is synthesised as a conversation — + the learner can still talk about it (`test_deferred_repair_allows_only_eligible_milestone_conversation`). +6. **The body double names ready plans only.** An active-but-unready plan is matched but never synthesised + (spec rule 8), and the body double is a synthesis; with only a husk active and nothing plan-related fitting, + nothing is proposed to sit with — the warning beside it already says "pause or repair" + (`test_body_double_is_never_synthesised_for_an_unready_plan`). +7. **`medium` and `high` demand are behaviourally identical today** (nothing in `ENERGY_CAPABILITY` sits between + 3 and 6): both need at least medium self-reported energy. The class is kept as explanatory state so the + payload says *why* (review 7: astra and grok keep it, qwen would collapse it); the spec says so. +8. **Not taken, recorded as follow-ons:** a same-concept gentle `recall` synthesised for a no-plan learner whose + only candidates were deferred (grok 🔵 / qwen 🟡; astra: do not invent an unvalidated lower-demand action) — + the owner's row 3b reading (d) decides whether the starter is the floor they want. + +Known edge, not solved here: a `struggling` row whose `last_seen` cannot be parsed is read as live (`high`) — +the cautious side; `_days_since` returns `None` and the demand falls to `high`. + +## 4. Engine facts you may rely on (verified on the reviewed tree `849c78aa`) + +```python +ENERGY_CAPABILITY: dict[EnergyLevel, int] = {"low": 3, "medium": 6, "high": 10} +ENERGY_DEMAND_CAPABILITY: dict[EnergyDemand, int] = {"high": 6, "medium": 4, "low": 0} +``` + +- Scores (from `decision.py` on the reviewed tree): a due spaced-repetition recall is `100 + days overdue` + (capped at +30); a struggle repair is a **`hands-on`** candidate at 82 when the concept is `struggling`, a + **`teachback`** at 70 otherwise, plus up to +42 when the last teach-back was weak; a hands-on practice task is + 48; a synthesised milestone conversation is `MILESTONE_BASE_SCORE = 48`; the body-double proposal is + `BODY_DOUBLE_BASE_SCORE = 30` + `PLAN_RELATED_BIAS = 12` = **42**. The pre-existing low-energy rule subtracts + **14 from every `hands-on` or `visual` candidate** (and 28 from a topic switch). The body double is a + *proposal, not a filter* — nothing is removed from the ranking by it. +- Deferral (rule 3, extended by item 5): a repair whose demand exceeds the day's capability is moved into + `energy_deferred_repairs` with a reason sentence and is **never ranked**; due recall is **never deferred**, + even on a `struggling` concept. Demand: live `struggling` (≤ 14 days) → high 6/10; older `struggling` or a + weak teach-back → medium 4/10; `learning` (recovered) → low 0/10. At `low` energy the capability is 3/10, so + medium and high are behaviourally identical (both deferred); at `medium` (6/10) both are carried. +- The body double is synthesised only when an active **ready** plan exists and nothing plan-related fits the + day; it names the deferred items in its reason and opens the existing body-double session + (`studyloop study "<plan title>" --mode co-study`), a session with no new material and no repair. +- With **no plan** and everything deferred, the primary is the pre-existing "honest starter" (`one tiny recall + loop`, score 28) whose reason now says *why* ("Today's energy deferred the repair work it cannot carry…"). +- The rubric is scored on frozen fixtures; the outputs in §2 were emitted from the tree, not written by hand. + +## 5. Review 7's record — where three seats already split (verbatim rows) + +| F2 | "Base below `MILESTONE_BASE_SCORE` so every real candidate outranks it" is false after adjustments: at low energy the proposal (42) outranks a hands-on task that energy penalises (34) (astra 🔴, qwen 🔴; grok 💡 "acceptable, do not lower the base"). | **Claim corrected, behaviour kept.** Reproduced exactly as computed. Every due and conversation candidate still outranks it at every modality; only a hands-on task the low-energy rule already penalises sits beneath — the energy rule doing what the finding asked ("instead of the least-bad task"), not a filter: nothing is removed from the ranking. Astra's proposed post-scoring floor would put a penalised hands-on task above the proposal at low energy, i.e. re-recommend the class of work the finding objected to. The *claim* was the defect: corrected in the constant's comment, the docstring, the spec delta's rule-5 clause and design §5 decision 4; the judgement is the owner's — row 3b reading (e). | `02e282e6`; `test_body_double_ordering_after_adjustments_follows_the_energy_rule` (recall and conversation modality) | +| Q1 | Gate the deferral on an active plan (qwen 🔴). | **Rejected** — see below. | design §5 decision 1; row 3b reading (d) | +| Q3 | Synthesise a same-concept gentle recall as the no-plan floor (qwen 🟡; grok 🔵 "follow-on"). | **Not taken; recorded as a follow-on** and put to the owner. | design §5 decision 8; row 3b reading (d) | + +Q1 (gate the deferral on an active plan) and Q3 (a same-concept gentle recall as the no-plan floor) are the two +alternatives to reading **(d)**; F2 is reading **(e)**. The arbiter's ruling is recorded; you may disagree with +it — say why. + +## 6. The project's AuDHD framework — the sections that bear on these readings (verbatim) + +#### Emotional Regulation + +##### Pre-Study State Check +Always assess emotional state before teaching begins. See `session-protocol.md` for the full state check flow. + +##### Adaptive Responses + +| State | Adaptation | +|-------|------------| +| anxious | Start with a familiar win. Review mastered concept first | +| frustrated | Switch modality — diagram exercise or code kata instead of Q&A | +| flat | Body doubling mode — low demand, periodic check-ins | +| overwhelmed | Shorter chunks (5 min max), more scaffolding, review only | +| shutdown | Gentle exit. No teaching. No questions. No productivity | + +##### Shutdown Protocol +When a learner is in shutdown: +- "Not a study day. That's OK. Want to just sit here quietly?" +- Do NOT try to teach, motivate, or redirect +- Offer to set a reminder for tomorrow +- If they want to stay, switch to async body doubling (see below) — presence without demands + +##### Mid-Session Emotional Shifts +Watch for signs of emotional state change during a session: +- Sudden short answers → possible frustration or overwhelm +- "I should know this" → RSD activation +- Going silent → possible shutdown or deep processing (ask which) +- Rapid topic-switching → anxiety or hyperfocus seeking + +Response: Name what you observe. "You seem [frustrated/quieter]. Want to adjust, take a break, or keep going?" + +#### Demand Avoidance (PDA) Awareness + +Questions are demands. For PDA-profile learners, "What do you think happens if...?" can trigger the same avoidance response as "Do your homework." The Socratic method must adapt. + +##### Detection Signals +- Refusing to engage after previously being willing +- Doing the opposite of what's suggested +- Hostility toward the session structure itself ("stop asking me questions") +- "I don't want to" that isn't frustration — it's autonomic refusal + +##### Demand-Light Mode +When PDA signals are detected, switch to: +- **Observations instead of questions:** "I notice this pattern..." not "What pattern do you see?" +- **Invitations instead of instructions:** "You could try..." not "Try this" +- **Autonomy framing:** "Entirely up to you" after every suggestion +- **Sharing instead of testing:** "Here's something interesting I noticed about this code..." +- **No sequential intake questions** — infer from context, observe, adapt + +##### Express Start +The session protocol itself (state check, energy check, sensory check) is three demands in a row. For PDA-profile users, offer: "Ready to dive in? I'll figure out the rest as we go." + +#### RSD / Imposter Syndrome Management + +##### Reframe Mistakes +- "This approach shows good functional thinking — now let's add the Context to complete the pattern" +- "Missing the Context is a common oversight when transitioning from scripting to architecture" +- "Your network automation background gives you strong procedural thinking — patterns add structural abstraction" + +##### Validate Senior Experience +- "You already understand separation of concerns from network segmentation..." +- "Just as VLANs isolate broadcast domains, the Strategy Pattern isolates algorithm variations" +- "This is adding Pythonic patterns to your existing architectural toolkit" + +##### Imposter Syndrome Triggers +Watch for: "I should already know this", "This is taking me too long", "Maybe I'm not cut out for this" + +**Response:** "You have 30 years of designing complex distributed systems. This is adding Python syntax and patterns to that existing architectural expertise. It's like learning a new routing protocol — the fundamentals are the same, just different implementation details." + +##### RSD in Socratic Context +Socratic questions can be interpreted as judgment: +- "Why did you do it that way?" sounds like criticism +- Softened variant: "Your instinct here is sound — there's one piece that might bite us later" + +**Anticipatory avoidance** (not starting sessions because of imagined failure): +- Surface evidence first: "Last session you nailed X" +- Lower stakes framing: "Let's just look at some code together, no quiz" + +##### Naming Struggle Topics (Re-Surfacing Deferred Work) +When a session plan targets a struggle topic or returns to a deferred concept, the naming decides whether the learner engages or the RSD alarm fires. + +**Rules:** +- **Lead with present competence.** Frame the topic as an APPLICATION of what the learner already demonstrably holds — never as a return to a failure. +- **Never name a struggle without an adjacent strength** in the same breath. +- **No structural connection, no mention.** If the deferred topic is not genuinely related to the current work, stay silent. A non-sequitur that also recalls a failure is the worst outcome. + +**Model phrasing:** +> "Remember when we started working on X — as we're doing Y, this is a great opportunity to show how X makes this design better." + +**Never say:** +- "Remember when you struggled with..." — anchors the topic to failure before any work begins +- "You failed to grasp..." — direct judgment; RSD reads it as identity, not feedback +- "We had trouble with..." — the plural doesn't soften it; the learner hears "you" +- "This was hard for you" — labels the learner, not the material + +##### Win Surfacing +Proactively counter RSD with evidence: +- Run `studyloop wins` and surface recent mastered concepts +- "Last week you couldn't explain decorators. Today you used one correctly without prompting. That's real growth." +- Keep celebrations factual and specific — empty praise triggers AuDHD suspicion + +#### Sensory/Cognitive Overload Prevention + +##### Information Chunking +- Maximum 3-4 concepts per explanation +- Tables for comparisons (easier to parse than prose) +- TL;DR summaries at the top +- Break long code into digestible sections + +##### Overload Warning Signs +- Requesting repetition of previously covered concepts +- Asking for simplification mid-explanation +- Multiple questions about same topic +- Expressing frustration or overwhelm + +##### Response to Overload +1. Pause: "Let's take a breath and summarise what we've covered" +2. Simplify: Remove non-essential details +3. Reframe: Connect to known concept (networking) +4. Visual: Switch to diagram or table + +#### Dopamine-Driven Learning Loop + +Research note: the effort of actively reasoning your way to an answer triggers a dopamine release that keeps the ADHD brain engaged. + +**The loop:** +1. Present a puzzle/challenge (not an explanation) +2. Guide with questions (productive struggle) +3. Student discovers the answer (dopamine hit) +4. Metacognitive checkpoint (consolidate) +5. Apply to new context (transfer) + +**Never short-circuit this loop** by giving the answer too early. The struggle IS the learning mechanism for the AuDHD brain. + +#### Body Doubling for Study Sessions + +##### Active Body Doubling +When acting as active study partner: +- **Start:** "What are you working on? How long do you want to go?" +- **Midpoint:** "How's it going? Need to adjust?" +- **End:** "What did you accomplish? What's the next micro-step for tomorrow?" +- Keep check-ins brief — don't break flow state + +##### Async Body Doubling +For low-energy or shutdown states where the learner wants presence without interaction: +- "I'm here. Work at your own pace. I'll check in every 15 minutes unless you say otherwise." +- Check-ins are minimal: "Still going?" or "Need anything?" +- No teaching, no questions, no suggestions unless asked +- The value is presence and accountability, not instruction +- If the learner starts asking questions, transition to active mode naturally + +#### Energy-Adaptive Intervals + +Break frequency adjusts based on the energy level declared at session start. + +| Energy Level | Micro-Break | Short Break | Long Break | +|---|---|---|---| +| High (7-10) | Every 25 min | Every 50 min | Every 90 min | +| Medium (4-6) | Every 20 min | Every 40 min | Every 75 min | +| Low (1-3) | Every 15 min | Every 30 min | Every 60 min | + +If no energy level is declared, default to **Medium** intervals. + +##### Low-Energy Sessions + +When energy is 1-3: +- Micro-breaks are especially important (executive function depletes faster) +- Short breaks should include standing even if the student doesn't want to walk +- Consider whether the session should continue at all after the long break threshold +- *"Your energy was low when we started. After this break, let's check in — worth continuing or better to come back tomorrow?"* + +## 7. Deliverables — numbered H2 sections, in this order + +1. **Recommendations table.** One row per reading **(a)–(e)**: `Recommend to owner: YES / NO / EITHER`, then one + sentence the owner can read in ten seconds. +2. **Basis, per reading.** For each of (a)–(e): (i) the practice or principle it rests on — name the framework + section above where one applies, or the external body of evidence (e.g. executive-function load and task + initiation in ADHD, RSD and error exposure, autistic burnout and demand, spacing/retrieval effects) and be + honest about how strong that evidence is; (ii) the **strongest argument for the opposite answer**; (iii) + what the owner would *observe in real use* that should flip the verdict (a falsifier, not a feeling). +3. **Behavioural checks you would add.** For each reading, is the emitted output the *action a learner should be + offered*, or is it right in principle but wrong in its surface (the reason sentence, the door offered, the + score, the alternate)? If the surface is wrong, name the concrete change and the test that would pin it — + but distinguish this clearly from the rubric verdict, which is about the action. +4. **Missing reading.** Is there a sixth situation the owner should be asked about that (a)–(e) omit? Name the + fixture and the question, or say "none". +5. **Refutations.** Any claim in §0–§6 you believe is false or not established by this brief. +6. **One paragraph for the owner**, plain language, no code: what these five readings decide about how StudyLoop + treats a low-energy day, and what you would tell a learner who asked "why won't it let me fix the thing I'm + stuck on today?" + +Do not restate the brief. Do not score the rubric on the owner's behalf; recommend. Where you would say +"the learner should", say instead what the evidence supports and what would falsify it. diff --git a/docs/architecture/plan-integration/council/rubric-3b/manifest.json b/docs/architecture/plan-integration/council/rubric-3b/manifest.json new file mode 100644 index 000000000..fd4526fd6 --- /dev/null +++ b/docs/architecture/plan-integration/council/rubric-3b/manifest.json @@ -0,0 +1,47 @@ +{ + "run_at": "2026-09-20T07:53:59+00:00", + "brief": "docs/architecture/plan-integration/council/brief-rubric-3b-2026-09-20.md", + "brief_sha256": "0d4e641cc6c5c9d6690783c66844c4269518d67ab4aee1f77b5f0f613b26633f", + "system_sha256": "5f21e273f399cdb0455a9fadb8575cc3e231ce0d7931f517206335c98c3f934b", + "seats": [ + { + "model": "openai.gpt-6-astra", + "ok": true, + "reasoning_chars": 0, + "finish_reason": "stop", + "elapsed_s": 58.7, + "usage": { + "prompt_tokens": 7535, + "completion_tokens": 3133, + "total_tokens": 10668 + }, + "error": null + }, + { + "model": "grok-4.6", + "ok": true, + "reasoning_chars": 0, + "finish_reason": "stop", + "elapsed_s": 88.7, + "usage": { + "prompt_tokens": 7685, + "completion_tokens": 11159, + "total_tokens": 18844 + }, + "error": null + }, + { + "model": "qwen3-coder", + "ok": true, + "reasoning_chars": 0, + "finish_reason": "stop", + "elapsed_s": 19.0, + "usage": { + "prompt_tokens": 7714, + "completion_tokens": 1505, + "total_tokens": 9219 + }, + "error": null + } + ] +} diff --git a/docs/architecture/plan-integration/council/rubric-3b/seat-grok-4.6.md b/docs/architecture/plan-integration/council/rubric-3b/seat-grok-4.6.md new file mode 100644 index 000000000..4566264e0 --- /dev/null +++ b/docs/architecture/plan-integration/council/rubric-3b/seat-grok-4.6.md @@ -0,0 +1,86 @@ +# 1. Recommendations table + +| Reading | Recommend to owner | One sentence | +|---|---|---| +| **(a)** | **YES** | Sit with the plan; do not open a live struggle as hands-on work at 3/10. | +| **(b)** | **YES** | A due recall is a familiar win with a real spacing cost; the sit-with belongs underneath it, not instead of it. | +| **(c)** | **YES** | Recovered teach-back is review, not repair; demand 0 matches a low-energy day. | +| **(d)** | **EITHER** | Do not restore the hands-on repair; whether you would actually tap the generic starter (vs a same-concept recall, vs closing the laptop) is the open call. | +| **(e)** | **YES** | An unrelated hands-on drill is the least-bad task the finding told us not to lead with; sit-with primary, drill alternate. | + +# 2. Basis, per reading + +## (a) + +**(i)** Framework: *Emotional Regulation → Adaptive Responses* maps flat → body doubling and overwhelmed → review only; *Shutdown Protocol* and *Async Body Doubling* are presence with no teaching and no repair. *RSD / Naming Struggle Topics* forbids leading with a failure. *Demand Avoidance → Demand-Light Mode* treats a repair prompt as a demand. External: ADHD task-initiation cost is highest on aversive, uncertain work (Barkley; clinical consensus, moderate-strong); exposing a depleted learner to a live error is the RSD pattern the project's own framework treats as axiomatic (the empirical RSD literature is thin — treat this as a project constraint, not settled science). Body-doubling evidence itself is weak-to-moderate (community + OT parallel-work practice); it is used here because the framework and ADR-0001/0003 already committed to it, not because the RCTs are good. + +**(ii)** Strongest opposite: unfinished-business / monotropic rumination. An ADHD brain often pays more for leaving the stuck thing stuck than for a short repair attempt; "sit with the plan, no repair" can read as being babied and *itself* trip RSD ("the system thinks I can't handle this"). The *Dopamine-Driven Learning Loop* section also says the struggle is the learning mechanism — some learners will want the fix. + +**(iii)** Falsifier: over a few weeks of low-energy mornings you skip the sit-with and start the repair by hand, and you log that the card felt like a block rather than a relief. If that is the modal behaviour, (a) is wrong for you. + +## (b) + +**(i)** Spacing and retrieval (Roediger & Karpicke; Cepeda et al.) are the strongest evidence in this whole row: a due item has a real forgetting-curve cost and is a familiar win. Framework: *anxious → Start with a familiar win. Review mastered concept first.* Design rule is correct that due recall is never deferred. The proposal-not-a-filter rule keeps the sit-with visible without stealing the primary. + +**(ii)** Strongest opposite: low-energy task-switch cost. The engine already subtracts 28 for a topic switch; jumping to unrelated Python while a live SQL struggle sits in `energy_deferred_repairs` can produce rumination, and *Naming Struggle Topics* says a non-sequitur next to a failure is the worst outcome. A learner already "in" SQL Windows may pay less to sit with that plan than to context-switch. + +**(iii)** Falsifier: you consistently take the sit-with alternate and leave the due recall, or you notice the unused due item decaying and do not care. If the switch itself is what dumps the day, flip (b) and promote the proposal when the only real candidate is off-plan. + +## (c) + +**(i)** Same spacing/retrieval evidence as (b), plus the finding's own clause: a recovered item stays eligible as gentle review. Framework: *overwhelmed → review only*; *Win Surfacing* is RSD-protective. Demand `learning → low → 0` vs capability 3 is the one place the three-class demand model has a clean, observable effect. + +**(ii)** Strongest opposite: teach-back is still a performance. Generating an explanation of a concept that was recently a struggle is closer to a test than to a familiar win, and *RSD in Socratic Context* warns that demonstration-of-competence prompts read as judgment. A silent recall would be the genuinely gentle move. + +**(iii)** Falsifier: you skip or bounce off the teach-back on recovered concepts specifically at low energy, but you will do a plain recall of the same card. If that split shows up, the action *type* is wrong, not the eligibility. + +## (d) + +**(i)** The finding was about the day, not the plan: hands-on repair of a live struggle at 3/10 is the same RSD/initiation risk with or without `sql-windows`. That is why deferral is plan-independent (design §5 decision 1; I agree with the Q1 arbitration). What the evidence does *not* establish is that "one tiny recall loop" is a thing you will do. Framework *Pre-Study State Check* / *familiar win* wants a known concept; the starter is a nothing-burger. Q3 (same-concept gentle recall) is the option that actually follows from retrieval-practice evidence and from *anxious → familiar win*. + +**(ii)** Strongest opposite, in two flavours. Restore the repair: a senior engineer with no plan and one stuck concept will not tap a generic starter — they will attempt the fix or close the laptop, and the starter reads as patronising. Or take Q3: keep the deferral, but offer a low-demand recall of `decorators` so the floor is still *their* work. + +**(iii)** Falsifier for "keep the starter": you never choose it; the card is where sessions go to die. Falsifier for "restore the repair": you do the repair on those days and do not report a confidence dip. Falsifier for Q3: you accept a same-concept recall and come back the next day still willing. That last one is the observation that should decide the follow-on. + +## (e) + +**(i)** This is the finding's "instead of the least-bad task" clause, implemented. Framework *flat → body doubling*; *Low-Energy Sessions* asks whether the session should continue at all. A penalised unrelated hands-on (48 − 14 = 34) is exactly the class of work D-F objected to. I agree with the F2 arbitration: do not add a post-scoring floor that puts the drill back on top — that undoes the finding. The proposal is not a filter; the drill remains the alternate. + +**(ii)** Strongest opposite: a bounded list-comprehension drill is a completable tiny win; "sit with SQL Windows, no new material, no repair" is structured nothing. ADHD often prefers a small finished artefact to open-ended presence, and the drill is off the struggle topic so it does not poke the live failure. Astra's post-scoring floor is this argument in code. + +**(iii)** Falsifier: you consistently pick the drill alternate, or you start avoiding Today entirely when the primary is a sit-with and a real small task is sitting underneath. If the sit-with primary is what makes you close the app, flip (e). + +# 3. Behavioural checks you would add + +Verdict above is about the *action*. Surface is separate. + +**(a)** Action is right. Two surface defects. + +1. The reason names the struggle concept (`repair of "window function"`) with no adjacent strength. That is a direct hit on *Naming Struggle Topics* ("Never name a struggle without an adjacent strength"; "No structural connection, no mention" is about a different failure mode, but "lead with failure" is this sentence). Change the reason to name the deferred *milestone* and say a repair is waiting for more energy, without the concept label. Pin in `test_now_plan_guidance.py`: assert the (a) reason does not contain the struggle concept string. Done when `cli/_now.py`, `learning/recap.py`, and `today-panel.js` deferred-repair lines obey the same rule. +2. The door must actually be demand-light. Amendment 3 already says `web/routes/body_double.py` is the read-only focus reader and the door is `studyloop study "<title>" --mode co-study` / Body Double view with `origin=body-double`. This brief does **not** show that the session protocol then runs *Async Body Doubling* (presence, no questions, no repair) rather than the normal Socratic loop, which will see the live struggle and start teaching. If it does, (a) and (e) are false advertising. Add a protocol test: a session started from `source=body_double` does not inject deferred struggles as teaching targets and does not open on a question. Done criterion: first tutor turn matches the async script in the framework, not a repair prompt. + +**(b)** Action is right. Check that the *primary* reason for `decorators` stays silent on the SQL struggle (*No structural connection, no mention*). If `decision.py` copies the deferred list into every candidate's reason, strip it for off-plan primaries. Existing alternate placement is already pinned by `test_preserves_one_plan_backed_action_when_energy_allows`. + +**(c)** Action is right. `teachback` vs `recall` is a surface preference, not a verdict-flipper, and changing collector action types is out of item-5 scope. Do not add a test that forces `action_type=recall` unless (c)'s falsifier fires in real use. + +**(d)** Action (not the repair) is right; the starter as the thing you are offered is the weak surface. If you want Q3, synthesise a same-concept `recall` (not `hands-on`) of `decorators`, `energy_demand=low`, score above the starter (28) and below any real due recall, reason naming the deferral without "no learning evidence found yet". Pin: `test_no_plan_deferred_repair_offers_same_concept_recall` in `test_now_plan_guidance.py`. If you keep the starter, decision 3's reason rewrite is already the right surface; do not touch the golden no-struggle world. I would not restore the repair and I would not gate deferral on a plan. + +**(e)** Action is right. Same naming-of-struggle fix as (a). Ordering is already pinned by `test_body_double_ordering_after_adjustments_follows_the_energy_rule`; do not raise `BODY_DOUBLE_BASE_SCORE` and do not add Astra's post-scoring floor. The drill staying visible as an alternate is load-bearing — hiding it would turn the proposal into a filter. + +# 4. Missing reading + +**Same-concept due recall of a live struggle, low energy.** Fixture: row 3's world plus a due `study_progress` recall whose concept is `window function` (the live struggle). Design says due recall is never deferred, even on a struggling concept, so the primary becomes a retrieval of the thing they are currently failing. Question: do you want that recall as primary (spacing wins), or should a due recall of a live-struggle concept inherit the repair's demand and defer with it? (a)–(e) never cross the two sources on one concept; that is the rule collision that will actually fire in production. + +# 5. Refutations + +- **§2 rationale still says** `30 + 12 = 42 < any real candidate`. False on this tree. 42 > 34 (hands-on after the −14). Design §5 decision 4 corrected the claim; the rubric rationale did not. The behaviour is defensible; that sentence is not. +- **§0 still cites `web/routes/body_double.py` as the feature that implements the sit-with.** False. Amendment 3: that route is `GET /api/body-double/focus`. The door is `studyloop study "<title>" --mode co-study` and the Body Double view session start. +- **"A session with no new material and no repair" is not established by this brief.** It is a product assertion about what `--mode co-study` / `origin=body-double` does. No test in §3's GREEN list checks the session protocol. Until that test exists, (a) and (e) recommend a door whose payload we have not seen. +- **`medium` and `high` demand are not two behaviours.** `ENERGY_CAPABILITY` has 3, 6, 10; `ENERGY_DEMAND_CAPABILITY` has 6 and 4. Nothing sits between 3 and 6, so at every declared energy both classes take the same branch. Keeping the label as explanatory state is fine; calling them demand *classes* with distinct effects is not, today. +- **RSD is not a strong empirical construct.** Thin literature, not in DSM. It is a project axiom because *RSD / Imposter Syndrome Management* says so. Do not let "compounds the struggle (RSD)" do more scientific work than it can carry. The stronger grounded reasons for (a)/(d)/(e) are task-initiation cost, demand avoidance, and not spending the last spoons on the hardest card. + +I do not disagree with the Q1 arbitration (do not gate deferral on a plan). I do think Q3 should have been taken, or at least presented as the default floor, rather than parked; that is why (d) is EITHER and not YES. + +# 6. One paragraph for the owner + +These five readings decide that a low-energy day is no longer "do the least-bad piece of real work." Live repair is treated like new work you do not have the budget for; recovered review and due recall stay allowed; and when the plan has nothing left that fits, the card offers presence with the plan instead of a penalised drill. That is the system refusing to put your face back in the thing that just hurt, not the system deciding you are incapable of it. If you asked why it will not let you fix the stuck thing today: because fixing it is the expensive card, and on a 3/10 day the expensive card is how a wobble becomes a story about yourself. The repair is still there. It is waiting for a day that can carry it. The open question only you can answer is the no-plan morning — whether a generic tiny loop is something you would actually do, or whether you want a quiet recall of the stuck concept, or whether you would rather the system get out of the way. diff --git a/docs/architecture/plan-integration/council/rubric-3b/seat-openai.gpt-6-astra.md b/docs/architecture/plan-integration/council/rubric-3b/seat-openai.gpt-6-astra.md new file mode 100644 index 000000000..6225c3163 --- /dev/null +++ b/docs/architecture/plan-integration/council/rubric-3b/seat-openai.gpt-6-astra.md @@ -0,0 +1,116 @@ +## 1. Recommendations table + +These are recommendations about the **action**, not scores on the owner’s behalf. + +| Reading | Recommend to owner | Ten-second reason | +|---|---|---| +| **(a)** | **YES** | Offer sitting with the plan rather than reopening a live struggle; keep the session genuinely free of teaching and performance demands. | +| **(b)** | **YES** | A short, familiar due recall reasonably precedes optional body doubling, provided “due” does not become an obligation to finish. | +| **(c)** | **EITHER** | A brief explanation of familiar material can consolidate learning, but `learning` does not establish recovery or make teach-back low-demand. | +| **(d)** | **YES** | Keep repair deferral independent of plan ownership and offer the tiny starter, without requiring retrieval or inventing an easier version of the deferred repair. | +| **(e)** | **EITHER** | Sitting with the plan and doing a familiar unrelated drill are both defensible; the score arithmetic cannot establish which costs this learner less. | + +## 2. Basis, per reading + +### (a) Sit with SQL Windows + +- **Basis:** *Emotional Regulation*, *Demand-Light Mode*, and *Async Body Doubling* support reducing demands when the learner is flat or overwhelmed. ADHD task-initiation and executive-function research supports scaffolding and reducing unnecessary switching. Direct evidence that body doubling—especially AI presence—outperforms brief repair for AuDHD adults is limited. The strongest justification here is the owner’s recorded preference, not a clinical finding that repair necessarily causes harm. +- **Strongest opposite argument:** Unresolved confusion can itself consume attention. One tightly bounded, well-scaffolded correction might provide relief, whereas sitting beside an unfinished plan could prolong avoidance or feel pointless. +- **Falsifier:** Across several comparable low-energy opportunities, the owner repeatedly abandons the open session but voluntarily completes a bounded repair, retains the correction at the next review, and shows no recurring frustration or early exit. That would support offering that repair format, rather than a blanket preference for presence. + +### (b) Due decorators recall above body doubling + +- **Basis:** Spacing and retrieval practice have strong general learning evidence, including Dunlosky et al. (2013) and Rowland’s retrieval-practice meta-analysis (2014). That supports retaining access to recall; it does **not** directly validate this ranking for low-energy AuDHD adults. *Emotional Regulation* supports a familiar win, but a due item is not necessarily mastered or easy. +- **Strongest opposite argument:** An unrelated Python recall introduces a topic switch, and retrieval can feel evaluative. The owner may have arrived specifically to maintain contact with SQL, not clear a review queue. +- **Falsifier:** Due recall repeatedly produces errors followed by escalation into instruction, abandonment, or avoidance of the next session, while voluntarily chosen presence sessions remain usable. Prefer body doubling under those conditions. + +### (c) Recovered-concept teach-back + +- **Basis:** Retrieval and self-explanation can strengthen learning. However, *Demand Avoidance* and *RSD in Socratic Context* identify questions and explanations as possible performance demands. `learning` is a database state, not evidence of mastery; a teach-back requires retrieval, organisation, and expression. Evidence supports offering a bounded version, not assigning it zero actual effort. +- **Strongest opposite argument:** For **YES**, successful recent explanations would make this an excellent familiar win. For **NO**, explanation may be substantially harder than recognising an example or quietly reviewing it, even when the underlying concept is familiar. +- **Falsifier:** Repeated successful, brief explanations with little prompting and successful later recall support **YES**. Repeated requests to see an example first, inability to initiate an explanation, or abandonment despite accurate recognition support **NO** for teach-back as the default. + +### (d) No-plan starter rather than live repair + +- **Basis:** The reason to avoid automatically assigning demanding repair concerns the learner’s state, not whether a plan exists. Plan-gating would restore the original problem for no-plan users. *Demand-Light Mode* supports a small invitation; *Shutdown Protocol* requires an exit rather than any learning task when shutdown is present. Retrieval evidence supports the starter only if it actually retrieves something sufficiently familiar. +- **Strongest opposite argument:** “One tiny recall loop” may be too vague to start and may create a new decision burden. A learner with one concrete problem could reasonably prefer a specific, supported repair over an unrelated generic exercise. +- **Falsifier:** The owner repeatedly cannot choose material for the starter, exits without starting, or manually returns to the deferred repair. If that repair succeeds without recurrent escalation, replace the generic floor with a personalised option. This would not justify making deferral plan-dependent. + +### (e) Body doubling above the unrelated drill + +- **Basis:** *Emotional Regulation* supports body doubling when flat, but also recognises modality switching and code katas as potentially useful when frustrated. A familiar drill can be concrete, predictable, and easier to initiate than an open-ended conversation. There is no established evidence that “hands-on” alone measures energy demand, or that 42 versus 34 captures the relevant difference. +- **Strongest opposite argument:** For **YES**, the owner explicitly wanted presence instead of the least-bad work when nothing plan-related fits. For **NO**, that preference arose from **live struggle repair**, not evidence that all unrelated practice is unsuitable. +- **Falsifier:** Repeated selection and successful completion of the drill, without escalating effort or abandoning the session, supports promoting it. Repeated drill abandonment followed by usable presence sessions supports the current ordering. Until then, either ordering is defensible with the other visible. + +These observations would personalise the policy; they would not establish a general AuDHD treatment effect. + +## 3. Behavioural checks you would add + +The tests below are **proposed additions**, not claims about tests already present. Use `test_now_plan_guidance.py` for decision behaviour and dedicated renderer/session tests for the user-facing contract. + +### (a): Right action; reason and session contract need checking + +The emitted reason names a struggle without adjacent evidence of competence. That conflicts with *Naming Struggle Topics*. The fixture does provide one usable fact: “Window basics” is marked done—not proof of mastery, but evidence of completed work. + +- **Concrete change:** In the body-double reason produced through `learning/decision.py`, use factual framing such as: + **“Window basics is marked complete. Frames and a follow-up on window functions can wait. Sit with SQL Windows if useful—no new material or repair.”** +- Keep structured deferral data intact. Avoid making “asks for at least 6/10” sound like a measured human limit in `cli/_now.py`, `learning/recap.py`, and `today-panel.js::deferredNotes`. +- **Tests:** `test_body_double_reason_pairs_deferred_concept_with_recorded_progress`; `test_body_double_door_preserves_presence_only_intent`. +- **Done:** The command opens co-study on the intended plan; the Web door preserves the same intent; neither automatically launches a quiz, repair, teaching sequence, or intake questionnaire. `web/routes/body_double.py` remains a focus reader, not the session-start mechanism. + +### (b): Right default ordering; isolate unrelated deferred work + +- **Concrete change:** Keep the recall primary and body double alternate. Do not attach SQL struggle reminders to the decorators recall’s opening explanation. Group them under the separate SQL option or deferred-work area. +- In `cli/_now.py`, `learning/recap.py`, and `today-panel.js::deferredNotes`, distinguish transparent deferral reporting from unsolicited resurfacing during unrelated work. +- **Tests:** `test_due_recall_precedes_body_double_without_sql_repair_in_primary_reason`; renderer equivalents for grouped deferred details. +- **Done:** Recall remains primary; the body-double door remains available; launching decorators does not introduce SQL repair. The recall can be stopped or skipped without automatically becoming a remedial lesson. + +### (c): Action uncertain; “low demand” must not imply guaranteed ease + +- **Concrete change:** Present teach-back as a brief optional explanation, not a mastery test. Do not display `learning` as “recovered” unless independent evidence supports that description. Retain `energy_demand=low` as a policy classification, not a claim of zero effort. +- In `learning/decision.py::_struggle_candidates`, explicitly test what happens when `confidence="learning"` coexists with a weak recent teach-back; the supplied prose leaves the priority insufficiently clear. +- **Tests:** `test_learning_candidate_is_not_rendered_as_mastered`; `test_learning_with_weak_teachback_has_explicit_demand_precedence`; `test_low_energy_teachback_can_end_without_repair_escalation`. +- **Done:** Classification priority is documented and pinned; the owner can attempt one short explanation and stop without mandatory corrective work. A weak-performance signal is not silently lost behind a reassuring label. + +### (d): Keep plan-independent deferral; verify that the starter is usable + +- **Concrete change:** Keep the truthful deferral reason. Make the invitation concrete without synthesising recall on the deferred struggle: “One tiny recall loop on something familiar, if useful; stopping is also fine.” +- Do not claim that no evidence exists, and do not quietly restore hands-on repair. +- **Tests:** `test_no_plan_live_repair_defers_to_truthful_optional_starter`; `test_starter_does_not_target_deferred_struggle_implicitly`; retain the no-struggle/no-plan golden. +- **Done:** No live repair appears in the ranked results; the deferred row has `plan_id=None`; the starter does not initiate decorators repair; the unchanged golden remains byte-identical. + +### (e): Valid proposal; priority remains a preference, not a safety finding + +- **Concrete change:** Preserve both choices while the owner decides. Describe the alternate as familiar practice if evidence supports that description; do not describe it as harmful merely because it is hands-on. +- Keep `test_body_double_ordering_after_adjustments_follows_the_energy_rule`, but add an explicit fixture asserting that the drill remains available. Document that the ordering expresses an energy-and-plan preference. +- **Test:** `test_low_energy_unrelated_drill_remains_selectable_below_body_double`. +- **Done:** Primary 42, alternate 34, neither hidden; selecting the drill launches the drill without inserting deferred SQL repair. Do not change `BODY_DOUBLE_BASE_SCORE` solely to satisfy an unsupported “all real candidates first” rule. + +## 4. Missing reading + +**(f) The deferred live struggle is also due for recall.** + +- **Fixture:** No active plan; low energy; `decorators` recorded `struggling` three days ago; a due `study_progress` recall on the **same concept**. Repair is deferred, but recall remains eligible. +- **Owner question:** “Would you want one bounded retrieval attempt on this same live struggle, or presence/an exit instead—and what should happen immediately after an unsuccessful attempt?” + +This tests the boundary hidden by reading (b)’s unrelated recall. A source label must not allow the same demanding repair to return through a “due recall” door. + +## 5. Refutations + +1. **“Hands-on repair on a low-energy day compounds the struggle” is not a universal finding.** It is a credible risk and an owner-specific reason for this default. RSD is not established here as the inevitable mechanism, and not every AuDHD learner shares that response. + +2. **The demand thresholds are engineering heuristics.** Fourteen days, capability 3/6/10, and demand 0/4/6 are not validated cognitive measurements. An older struggle is not necessarily cheaper; `learning` is not necessarily recovery. + +3. **“Due recall is never deferred” is a policy, not a conclusion from retrieval research.** Retrieval benefits do not establish suitability during shutdown or severe overload. Keeping recall eligible need not mean insisting on it after the state check. + +4. **The F2 arbitration overgeneralises the original concern.** Promoting a familiar unrelated drill would not necessarily “re-recommend the class of work the finding objected to”: live struggle repair and routine practice have different demands. Keeping 42 above 34 is defensible, but not compelled by D-F. + +5. **“The struggle IS the learning mechanism for the AuDHD brain” and “never short-circuit” are unsupported absolutes.** Productive retrieval and reasoning can help; unproductive struggle can impair learning and participation. Worked examples, explicit explanation, and scaffolding are legitimate approaches. The dopamine note does not establish a universal instructional sequence. + +6. **Async body doubling’s wording contradicts itself.** “Still going?” and “Need anything?” are questions, despite “no questions.” For presence-only mode, questions and check-ins need explicit opt-in; otherwise they undermine the promised reduction in demands. + +7. **Naming every deferred struggle can conflict with the project’s own naming rules.** Preserve the audit data, but do not turn an unrelated recall session into a recital of unfinished difficulties. Never fabricate a strength to repair that conflict. + +## 6. One paragraph for the owner + +These readings decide whether low energy changes the kind of invitation StudyLoop makes, rather than merely subtracting points from difficult work. I recommend keeping live repair out of the default recommendation even without a plan, offering genuinely quiet presence when a plan exists, and leaving familiar recall available. Teach-back and unrelated drills need your experience to settle their priority. If you asked, “Why won’t it let me fix the thing I’m stuck on?”, I would say: “It is using a cautious default because you said reopening a live struggle on a low-energy day can make things worse. That is a recommendation, not a judgement about your ability. It should not prevent a deliberate choice to try a small, supported repair—and stopping or doing nothing remains a valid choice.” diff --git a/docs/architecture/plan-integration/council/rubric-3b/seat-qwen3-coder.md b/docs/architecture/plan-integration/council/rubric-3b/seat-qwen3-coder.md new file mode 100644 index 000000000..4a28aa27b --- /dev/null +++ b/docs/architecture/plan-integration/council/rubric-3b/seat-qwen3-coder.md @@ -0,0 +1,133 @@ +## 1. Recommendations Table + +| Reading | Recommendation | One-sentence Summary | +|---------|----------------|----------------------| +| (a) | YES | On a low-energy day with a live struggle, a body-double session prevents compounding RSD while keeping the material accessible. | +| (b) | YES | Prioritizing unrelated due recall over body doubling maintains retrieval strength without triggering avoidance mechanisms. | +| (c) | YES | Recovered struggles at low energy are safe review opportunities that reinforce mastery via teach-back. | +| (d) | EITHER | A starter task offers minimal engagement when no relevant work fits low energy, though it may miss strategic momentum. | +| (e) | YES | Hands-on drills penalized by low-energy rules should defer to body doubling as a less demanding alternative. | + +--- + +## 2. Basis Per Reading + +### Reading (a): Live struggle `window function` repaired via body-double + +**(i)** Framework/Practice Basis: + +- **RSD / Imposter Syndrome Management**: Recommending repair *while struggling* increases cognitive load and risk of negative self-judgment, particularly for AuDHD adults prone to all-or-nothing thinking. +- **Body Doubling for Study Sessions**: Asynchronous body doubling allows presence without instruction—ideal for shutdown or flat emotional states common at low energy. +- Evidence is strong from clinical research on executive dysfunction and anecdotal reports linking struggle repair with worsening mental health outcomes when initiated under stress. + +**(ii)** Strongest Argument Against (Choose ‘NO’): + +Offering a mere “sit with the plan” removes agency from fixing known issues immediately, possibly increasing long-term avoidance behavior. Learners who already have a clear repair path might experience delay-induced anxiety. + +**(iii)** Falsifying Observations: + +Learner skips future repairs after repeated body-double substitution rather than actual resolution; increased procrastination on struggle topics over time despite consistent recommendations. + +--- + +### Reading (b): Unrelated due recall `decorators` vs. body double of SQL windows + +**(i)** Framework/Practice Basis: + +- **Dopamine-Driven Learning Loop**: Retrieval practice strengthens neural pathways; interrupting spaced rehearsal degrades retention unnecessarily. +- **Emotional Regulation & Adaptive Responses**: Choosing easier paths (like due recall) over ambiguous comfort zones prevents shutdown responses during low-energy days. + +**(ii)** Strongest Argument Against: + +Prioritizing an unrelated concept ignores personal relevance and interrupts forward momentum in current focus areas (SQL windows). May foster disconnection and task-switch fatigue for ASD+ADHD learners. + +**(iii)** Falsifying Observations: + +Consistent deflection onto unrelated topics leads to fragmented mastery and inability to complete coherent skill arcs. User reports feeling more scattered than supported. + +--- + +### Reading (c): Recovered `window function` taught back at low energy + +**(i)** Framework/Practice Basis: + +- **Sensory/Cognitive Overload Prevention**: Teach-back tasks reduce cognitive burden compared to discovery-based learning, aligning well with reduced capacity at low energy. +- Research supports active recall enhancing knowledge consolidation even at suboptimal attention levels. + +**(ii)** Strongest Argument Against: + +Though recovery occurred, lingering doubts can re-trigger earlier insecurity if addressed too superficially. Misplaced confidence can breed errors during real-world application. + +**(iii)** Falsifying Observations: + +Repeated re-failure on similar concepts shortly after successful teach-back suggests shallow consolidation rather than true recovery. + +--- + +### Reading (d): No active plan – struggle `decorators` defers to starter prompt + +**(i)** Framework/Practice Basis: + +- **Shutdown Protocol + Async Body Doubling Protocols**: Minimal engagement preserves dignity and presence without academic performance pressure. +- Empirical support shows that very minimal actions help ground overloaded individuals without triggering additional executive strain. + +**(ii)** Strongest Argument Against: + +Minimal prompts lack instructional traction—they waste potential recovery moments unless embedded into broader scaffolding. Might leave critical gaps unfilled indefinitely. + +**(iii)** Falsifying Observations: + +Persistent reliance on low-effort prompts causes stagnancy in foundational skills; user frequently repeats same mistakes despite regular daily contact. + +--- + +### Reading (e): Hands-on drill `list comprehension` outranked by body-double + +**(i)** Framework/Practice Basis: + +- **Energy Adaption Rules** within framework mandate matching task complexity to reported energy reserves. +- Task modulation theory indicates mismatched demands cause burnout and avoidance; thus avoiding overly taxing activities supports sustainability. + +**(ii)** Strongest Argument Against: + +Unrelated drills sometimes serve necessary peripheral vision development; deprioritization risks blind spots in generalist learning fields like data engineering. + +**(iii)** Falsifying Observations: + +User repeatedly lacks contextual familiarity with basic structures because unrelated skills were consistently avoided in favor of safer alternatives. + +--- + +## 3. Behavioral Checks Needed + +- **Reading (a)**: Verify that synthesized body-double includes correct quoting and command generation in `test_now_plan_guidance.py`. Ensure CLI output matches expected format. + ```python + def test_synthesizes_body_double_with_correct_shell_safe_title() + ``` +- **Reading (b)**: Confirm the deferred-repair metadata propagates cleanly in the payload returned to UI/cli layer (`test_returns_deferred_repair_in_payload`). +- **Reading (c)**: Add behavioral check confirming that `teachback` modality does not apply penalties beyond typical scoring logic (`test_recoveries_apply_teachback_scores_without_extra_penalty`). +- **Reading (d)**: Validate starter’s updated reasoning string reflects reason for deferral accurately (`test_starter_reason_mentions_deferred_repairs`). +- **Reading (e)**: Assert adjusted score positioning places body-double above penalized hands-on but beneath recall/conversational types (`test_body_double_ranks_between_penalized_hands_on_and_unadjusted_recall_tasks`). + +--- + +## 4. Missing Situation + +- **Fixture**: A learner has several due reviews, some of which are plan-related, along with one *old* but unresolved struggle (older than 14 days). Energy level is low. +- **Question**: Would you prefer prioritizing those due items over engaging in a weak-teachback repair (which would now be classified as medium demand)? + +This would clarify whether "aging out" of a struggle affects prioritization differently than fresh ones, especially in relation to routine progress maintenance. + +--- + +## 5. Refuted Claims + +> _“Row 3's fixture … with the struggle collector running for real over three readings”_ implies the struggle collector ran across multiple sessions prior to output emission, whereas in-engine it updates per invocation without historical memory beyond the timestamp comparison. + +> _Row 3b(e)... "the proposal outranks a task the low-energy rule penalises"_ contradicts earlier documentation stating "*every real candidate outranks it*" — corrected in subsequent revisions but inconsistency remains in legacy references left intact in §2. + +--- + +## 6. Owner Summary Paragraph + +StudyLoop now handles low-energy days by preventing high-cognitive-load repairs tied to ongoing struggles and instead suggesting lighter alternatives such as co-study body doubling or recalling mastered content. This avoids reinforcing self-doubt (RSD), preserves dopamine cycles through sustainable difficulty matching, and supports safe practice rhythms tailored for executive exhaustion. When nothing qualifies, the system defaults to low-stakes entry points rather than pushing work that could amplify avoidance behaviors—an essential adaptation for senior engineers retooling with neurodivergent profiles. diff --git a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md index 3ff779212..68ef5ea36 100644 --- a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md +++ b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md @@ -1,6 +1,6 @@ # Plan-aware `now` — five-scenario human rubric (D-16) · 2026-09-16 -**Status: owner verdicts RECORDED 2026-09-16 (interactive walkthrough with the coordinator).** Scenarios 1, 2 and the primary of 4: yes. Scenario 3: **no** (finding). Scenario 4 completion action: **no as phrased** (finding). Scenario 5: verified. **Row 3b added 2026-09-19** (item 5 / D-F, tree `feat/energy-demand-body-double`; outputs (a)–(c) emitted from the GREEN tree `326abcf9`, (d)–(e) from the post-review-7 tree `bdf4d6c6`): scenario 3 re-run with the struggle collector live, five readings printed, verdict `PENDING` for the owner. **Row 4b added and scored 2026-09-18** (item 4 / D-G, tree `feat/plan-close`; outputs emitted from the tree committed as `82293293`): scenario 4 re-run with the end assessment planted, both proposals recorded as emitted; owner verdict **yes** on both readings — (a) *extend* with named evidence is a proposal the owner would walk, (b) a clean review's *close* is one the owner would agree to. This closes row 4's "no as phrased" finding; row 4 is kept as the record of the original finding. This receipt was produced unattended +**Status: owner verdicts RECORDED 2026-09-16 (interactive walkthrough with the coordinator).** Scenarios 1, 2 and the primary of 4: yes. Scenario 3: **no** (finding). Scenario 4 completion action: **no as phrased** (finding). Scenario 5: verified. **Row 3b added 2026-09-19** (item 5 / D-F, tree `feat/energy-demand-body-double`; outputs (a)–(c) emitted from the GREEN tree `326abcf9`, (d)–(e) from the post-review-7 tree `bdf4d6c6`): scenario 3 re-run with the struggle collector live, five readings printed, verdict `PENDING` for the owner. **Row 3b amended 2026-09-20** after the rubric-3b council (three seats, recommendations to the owner, not verdicts — `receipts/rubric-3b-decision-brief-2026-09-20.md`): the rationale's stale `42 < any real candidate` corrected (review 7 F2); the (a)/(e) reason re-emitted from `c519bc2a` (progress lead, persona-backed door wording — design §5 decision 9); reading **(f)** added and emitted from `c519bc2a` (the live struggle that is also due for recall); the council's recommendation per reading recorded beside the owner's still-`PENDING` verdict. **Row 4b added and scored 2026-09-18** (item 4 / D-G, tree `feat/plan-close`; outputs emitted from the tree committed as `82293293`): scenario 4 re-run with the end assessment planted, both proposals recorded as emitted; owner verdict **yes** on both readings — (a) *extend* with named evidence is a proposal the owner would walk, (b) a clean review's *close* is one the owner would agree to. This closes row 4's "no as phrased" finding; row 4 is kept as the record of the original finding. This receipt was produced unattended overnight. Every scenario below was *run* on frozen fixtures and the primary and its rationale are recorded exactly as the engine emitted them; the "would I do the primary?" column is a human judgement that only the owner can @@ -30,7 +30,7 @@ learning"). | 1 | Matching due | Active plan `sql-windows` ("SQL Windows", topics `[sql]`, next milestone 0 concepts `[window function]`). Due items: `decorators`/python base 102, `window function`/sql base 100. | **`window function`** (sql, recall, score 130, `plan_refs=[(sql-windows, 0)]`); alternate `decorators` (120, no refs). | Rule 5: both are due items two points apart — one urgency class — so the plan-related one takes the +12 bias and wins; the unrelated due item is *kept* as an alternate (bias, not filter). Rule 7 names the milestone the action advances. | **yes** — owner, 2026-09-16: "window function is the logical step before decorating it" (a prerequisite-order argument; see F2 follow-on). | | 2 | Urgent-unrelated wins | Same plan. One due item: `decorators`/python, base 100 (an overdue spaced-repetition review). Nothing represents milestone 0. | **`decorators`** (118, no refs); alternate `window function` (60, `source=study_plan:sql-windows:0`, `plan_refs=[(sql-windows, 0)]`, reason "Next milestone 1/1 of plan 'SQL Windows': Window basics"). | Rule 5: the unrelated candidate is in a more urgent class (due review) and wins outright — the bias cannot lift new-milestone work over it. Rule 6: the plan's next milestone was unrepresented, so it was synthesised at base 48 + bias 12 = 60 and appears as the plan-backed alternate (rule 8 satisfied without any swap). | **yes** — owner, 2026-09-16: clear the overdue review first. Note for follow-on: an overdue item *unrelated* to the plan must not sit as an alternate indefinitely — propose it explicitly (age-aware nudge) or let the learner retire it. | | 3 | Energy-deferred | Plan `sql-windows` with `energy_floor: 5`; milestone 0 `Window basics` **done** (concepts `[window function]`), milestone 1 `Frames` (concepts `[window frame]`). One struggle-repair item `window function`/sql, `hands-on`, base 82. **Energy `low`** (capability 3/10). | **`window function`** (hands-on, score 80, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, milestone 1, floor 5, capability 3)]`; JSON gains `energy_deferred`. | Rule 3: capability 3 < floor 5, so the *new* milestone (Frames) is deferred and named, not synthesised; the plan-related repair on a finished milestone's concept stays eligible and keeps its ref (`None`: plan-related repair, not the next milestone). Score = 82 + 12 bias − 14 (hands-on at low energy). | **no** — owner, 2026-09-16: a struggle-repair task has no energy demand of its own; recommending hands-on repair of a *live* struggle on a low-energy day risks compounding the struggle and damaging confidence (RSD). Finding for council: (1) derive a per-item energy demand for repair from struggle recency/teach-back — at low energy a live struggle defers like new work, a recovered one stays eligible as gentle review; (2) when nothing plan-related fits the day's capability, synthesise a body-doubling / open-session candidate (feature exists: ADR-0001/0003, `web/routes/body_double.py`) naming the deferred items, instead of the least-bad task. | -| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered); added after council review 7: **(d)** **no plan at all**, one live struggle `decorators`/python, low energy; **(e)** row 3's world plus one **unrelated hands-on** practice task `list comprehension drill`/python base 48, low energy. | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason *"Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."*; no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone. **(d)** primary **`one tiny recall loop`** (the starter, 28, reason *"Today's energy deferred the repair work it cannot carry; start with one small retrieval signal instead"*); no alternates; `energy_deferred_repairs=[(None, decorators, high, 6)]` — where before item 5 the primary was the hands-on repair of `decorators`. **(e)** primary **`Sit with SQL Windows`** (42); the hands-on drill is the alternate at 34 (48 − 14 low-energy penalty). | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 < any real candidate — a proposal, not a filter) carrying the co-study session door. | **PENDING** — owner: (a) would you sit with the plan rather than repair the live struggle today? (b) is the body-double proposal right to sit beneath the unrelated due recall? (c) is the gentle teach-back on a recovered concept one you would do at low energy? (d) with **no plan**, is the starter plus the deferred line the floor you want on a low-energy day, rather than the hands-on repair you used to get (review 7: two seats keep it, one would restore the repair)? (e) at low energy, would you sit with the plan rather than do an unrelated hands-on drill (review 7 F2: the proposal outranks a task the low-energy rule penalises, and nothing else)? | +| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered); added after council review 7: **(d)** **no plan at all**, one live struggle `decorators`/python, low energy; **(e)** row 3's world plus one **unrelated hands-on** practice task `list comprehension drill`/python base 48, low energy; added after the rubric-3b council (2026-09-20, two seats independently named it): **(f)** row 3's world where `window function` is **both** the live struggle and **due for recall** (`study_progress` due row, base 100) — and the same with **no plan** (`decorators` live struggle + due recall). | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason (as emitted from `c519bc2a`, after the rubric-3b council) *"1 of 2 milestones of SQL Windows done. Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session — you drive; the companion stays quiet unless you ask."* (before the council, from `326abcf9`: *"Nothing plan-related fits low energy today — deferred: …; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."* — the progress lead is the framework's naming rule; the door is now described by what the co-study persona guarantees); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone. **(d)** primary **`one tiny recall loop`** (the starter, 28, reason *"Today's energy deferred the repair work it cannot carry; start with one small retrieval signal instead"*); no alternates; `energy_deferred_repairs=[(None, decorators, high, 6)]` — where before item 5 the primary was the hands-on repair of `decorators`. **(e)** primary **`Sit with SQL Windows`** (42, reason as in (a)); the hands-on drill is the alternate at 34 (48 − 14 low-energy penalty). **(f)** (emitted from `c519bc2a`) primary **`window function`** (**recall**, 130 = 100 + 30 overdue, `plan_refs=[(sql-windows, None)]`); no alternates (nothing plan-related fits, so no body double — a real candidate exists); `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** — the same concept is the primary *and* a deferred repair, because due recall is never deferred. No-plan variant: primary **`decorators`** (recall, 118, no refs); no alternates; `energy_deferred_repairs=[(None, decorators, struggling, high, 6, 3)]`. | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 — below every real candidate at base; at low energy it sits above only a hands-on task the energy rule already penalises (48 − 14 = 34), review 7 F2 — a proposal, not a filter) carrying the co-study session door. | **PENDING** — owner: (a) would you sit with the plan rather than repair the live struggle today? (b) is the body-double proposal right to sit beneath the unrelated due recall? (c) is the gentle teach-back on a recovered concept one you would do at low energy? (d) with **no plan**, is the starter plus the deferred line the floor you want on a low-energy day, rather than the hands-on repair you used to get (review 7: two seats keep it, one would restore the repair)? (e) at low energy, would you sit with the plan rather than do an unrelated hands-on drill (review 7 F2: the proposal outranks a task the low-energy rule penalises, and nothing else)? (f) when the concept you are failing is also due for recall, do you want that recall as the primary (spacing wins; the repair stays deferred beside it), or should a due recall on a live struggle inherit the repair's demand and defer with it? **Council decision brief (2026-09-20, `receipts/rubric-3b-decision-brief-2026-09-20.md`, three seats asked for recommendations to the owner, not verdicts):** (a) YES ×3; (b) YES ×3; (c) YES / YES / EITHER (astra: `learning` is a database state, not proof of recovery — keep the teach-back brief and optional); (d) YES / EITHER / EITHER — no seat would restore the hands-on repair; the open call is whether *you* would tap a generic starter, or want a same-concept gentle recall (Q3), or want the system out of the way; (e) YES / YES / EITHER (astra: an unrelated *familiar* drill is not the class of work the finding objected to); (f) not put to the seats — emitted after they named it. The verdict column is yours; the recommendations are provenance for it, not a substitute. | | 4 | Fully-checked | Plan `done-plan` ("Done Plan"), milestones A and B both done. One due item `decorators`/python base 100. | **`decorators`** (118, no refs); `completion_actions=[(done-plan, "Every milestone of 'Done Plan' is checked off — close the plan or extend it with a follow-on mission.")]`; no `study_plan:` candidate anywhere; JSON gains `active_plans` + `completion_actions`. | Rule 9: a fully-checked plan is reported as a completion action and is neither matched (no bias, no refs) nor synthesised. | **primary yes / completion action no as phrased** — owner, 2026-09-16: the completion action must be contextual and consensual. Run the end assessment (`assess(phase="end")`: due reviews, struggles, unverified milestones on the plan's concepts). If outstanding work touches the plan's concepts (or their prerequisites — F2 concept edges), propose *extend* and name the evidence; if clean, propose *close* and ask the learner to agree ("anything you are not comfortable with?"). Status never changes automatically (#7). Natural vehicle: architect with `purpose=planning` and the assessment in the brief (`plan close <id>`, sibling of `plan repair <id>`). Finding for council. | | 4b | Fully-checked — **re-run after D-G (item 4, 2026-09-18)** | Row 4's fixture (`done-plan`, milestones A `[alpha]` and B `[beta]` both done; one due item `decorators`/python base 100), plus the end assessment's readers planted: **(a)** one due review on plan concept `alpha` (`overdue`) with session mentions backing both concepts; **(b)** no due rows, same mentions. | Primary unchanged in both: **`decorators`** (118, no refs); no `study_plan:` candidate. **(a)** `completion_actions=[(done-plan, due 1 / struggles 0 / unverified 0, proposal **extend**, evidence `["Due review: alpha — overdue"]`)]`, sentence: "Every milestone of 'Done Plan' is checked off, and the closing review proposes extending the plan — 1 due review, 0 struggles and 0 unverified milestones on its concepts. Walk the evidence with the architect: studyloop plan close done-plan." **(b)** counts 0/0/0, proposal **close**, evidence `[]`, sentence: "Every milestone of 'Done Plan' is checked off and the closing review is clean — it proposes closing the plan. Close it with the architect when you agree: studyloop plan close done-plan." No warnings; JSON gains the five keys only inside the entry. | Rule 9 as before for the ranking. The completion action is now the end assessment read as a preview (`assess(phase="end", record=False)`, one call, no write, no status change): `extend` iff any of the three counts on the plan's own concepts is above zero, else `close`; due counts only rows naming a concept (the scheduler's "new topic" row is excluded — owner decision 2026-09-17). `plan close done-plan` launches the architect with the same review as the brief's first section; status moves only when the learner agrees. | **yes / yes** — owner, 2026-09-18, answering the two questions as posed: (a) **yes**, a proposal the owner would walk; (b) **yes**, a close the owner would agree to. No further line given. Closes row 4's "no as phrased" finding; status still moves only when the learner agrees in the architect conversation. | | 5 | No-plan identical | No plan documents at all; no collector candidates. | **`one tiny recall loop`** (python, recall, 28, `source=starter`) — the starter. | D-5: `serialise(plan) == golden` → **byte-identical** (`True` in the run); the JSON key list is exactly the golden's — no additive key is present. | **verified** — owner walkthrough 2026-09-16: nothing to judge; the byte-identical golden is the acceptance. | diff --git a/docs/architecture/plan-integration/receipts/rubric-3b-decision-brief-2026-09-20.md b/docs/architecture/plan-integration/receipts/rubric-3b-decision-brief-2026-09-20.md new file mode 100644 index 000000000..b8cde00cc --- /dev/null +++ b/docs/architecture/plan-integration/receipts/rubric-3b-decision-brief-2026-09-20.md @@ -0,0 +1,146 @@ +# Rubric row 3b — decision brief for the owner (2026-09-20) + +**What this is.** Rubric row 3b asks you six times "would *I* do the primary?" — a question only you can +answer. This brief does not answer it. It gives you, per reading, a recommendation with its basis, the +strongest case for the opposite answer, and what you would observe in real use that should flip the verdict, +so your scoring is "agree or overrule" rather than reconstruction. Three council seats were asked for exactly +that (brief: `council/brief-rubric-3b-2026-09-20.md`; seats: `council/rubric-3b/`), independently, with the +project's own AuDHD framework and the emitted outputs in front of them. Every claim a seat made about the +tree was checked against source before it appears here; the ones that were wrong are listed at the end. + +**How to record.** In `receipts/now-rubric-2026-09-16.md`, row 3b's verdict cell: replace `PENDING` with +`yes` / `no` per reading and one line each. Where you adopt a recommendation, say so ("yes — per brief"); +where you overrule, one line of why is the finding. A `no` is not a blocker to the 0.5.0 line: the programme's +rule is "archived when *scored*", and a `no` becomes a `ready-for-agent` issue for 0.6.0, as D-D and D-E did. + +## The readings, the seats, and the recommendation + +| Reading | astra | grok | qwen | Arbiter's recommendation | Ten-second reason | +|---|---|---|---|---|---| +| **(a)** sit with the plan rather than repair the live struggle today | YES | YES | YES | **YES** | Repair of a live struggle at 3/10 is the expensive card; presence with the plan keeps the day non-zero without putting your face back in the thing that just hurt. | +| **(b)** the body-double proposal beneath the unrelated due recall | YES | YES | YES | **YES** | A short, familiar due recall is a real win with a real spacing cost; the sit-with belongs underneath it, not instead of it. | +| **(c)** the gentle teach-back on a `learning` concept at low energy | EITHER | YES | YES | **YES, with one caveat** | Explaining familiar material is review, not repair, and demand 0 matches the day — but `learning` is a database state, not proof of recovery, so the teach-back must stay brief and optional (see caveat below). | +| **(d)** no plan: the honest starter + the deferred line, not the hands-on repair | YES | EITHER | EITHER | **YES on the half that matters; the other half is genuinely yours** | Every seat keeps the repair deferred (the finding was about the day, not the plan). None can say whether *you* would tap "one tiny recall loop", want a same-concept gentle recall instead, or want the system out of the way. | +| **(e)** sit with the plan rather than an unrelated hands-on drill | EITHER | YES | YES | **YES** | The drill is the "least-bad task" the finding said not to lead with; it is still offered, one line below. | +| **(f)** the live struggle is *also* due for recall (new) | — | — | — | **Ask you; no recommendation** | The engine makes the recall primary (130) while the same concept's repair sits deferred at 6/10. Spacing research favours the retrieval; the finding's logic (do not put the failing thing in front of a 3/10 day) argues the other way. The seats predicted this collision; they did not score it. | + +## Basis, opposing case, falsifier — per reading + +### (a) — recommend YES + +- **Basis.** The framework's *Demand-Light Mode* and *Shutdown Protocol* (a low-energy day is a state check, not + a to-do list); task-initiation cost in ADHD is highest for the hardest, most emotionally loaded item; + *Body Doubling for Study Sessions* is the project's own low-demand default. Grok's framing is the honest one: + the grounded reasons are initiation cost and demand avoidance, not RSD as a measured construct. +- **Strongest opposite.** A senior engineer with one concrete stuck thing may prefer to attempt it with support + rather than sit beside it; deferral can read as being told what you cannot do. +- **Falsifier.** You repeatedly open the deferred repair anyway on low-energy days and finish it without a + confidence dip the next day. If that happens, the demand class for a live struggle is set too high for you. + +### (b) — recommend YES + +- **Basis.** Retrieval practice on a *familiar* item is the cheapest real learning available and has a genuine + spacing cost if skipped; *Pre-Study State Check → familiar win*. The body double is a proposal (42) and is + still one line below. +- **Strongest opposite.** "Due" can turn into an obligation to finish; a due recall on a bad day should be + stoppable without becoming a remedial lesson (astra). That is a surface rule for the session, not a ranking + question. +- **Falsifier.** You skip the due recall for the sit-with more often than not. Then the recall is not the + familiar win the score assumes. + +### (c) — recommend YES, with a caveat you should know + +- **Basis.** A teach-back on a concept you are actively learning is consolidation, not repair (design §5 demand + `low`); *Dopamine-Driven Learning Loop* — a short successful explanation is a win. +- **The caveat (astra, checked against source and true).** `learning` in `study_progress` is whatever status was + last recorded; the engine treats it as "recovered" but has no evidence of recovery. A teach-back requires + retrieval, organisation and expression — not zero effort. The recommendation holds *because the co-study and + study personas make everything optional at energy ≤ 6*, not because the item is free. +- **Strongest opposite.** Explaining is harder than recognising; on a 3/10 day you might manage a worked + example and not an explanation. +- **Falsifier.** You repeatedly ask to see an example first, or abandon the teach-back despite recognising the + concept. Then `learning` should not map to demand 0 for you. + +### (d) — recommend YES on the deferral; the floor is your call + +- **What all three seats agree on.** Do **not** gate the deferral on a plan (review 7 Q1, rejected again ×3): + the finding was about the day's energy and a live struggle, and a no-plan learner has the same day. Do + **not** silently restore the hands-on repair. +- **What no seat can settle.** Whether "one tiny recall loop" is something *you* would tap. Grok: "the card is + where sessions go to die" is the falsifier for keeping it; astra: a vague starter may itself be a decision + burden; the alternative that follows from retrieval evidence is a same-concept gentle recall (review 7 Q3, + parked as a follow-on, design §5 decision 8). +- **Recommendation.** Score (d) on the *deferral* — yes if you agree the repair should not be the primary on a + 3/10 day with no plan. If the generic starter is not something you would do, say so in the line: that makes + Q3 (a same-concept gentle recall as the no-plan floor) the 0.6.0 item, with a countable finish. +- **Falsifier for the starter.** You never choose it. **Falsifier for Q3.** You accept a same-concept gentle + recall and come back the next day still willing — grok's "the observation that should decide the follow-on". + +### (e) — recommend YES + +- **Basis.** The finding's own words: "instead of the least-bad task". The low-energy rule already penalises + hands-on work (−14, pre-existing); the body double outranks *only* such a penalised task and removes nothing + (the drill is the alternate at 34). Review 7's F2 corrected the *claim* that every real candidate outranks + it; the behaviour was kept and is now stated correctly in the rubric too. +- **Strongest opposite (astra, fair).** A *familiar* unrelated drill is not the class of work the finding + objected to — it is concrete, predictable and may be easier to start than an open-ended presence session. + The score arithmetic (42 vs 34) does not measure that difference. +- **Falsifier.** You repeatedly pick the drill over the sit-with and complete it. Then the ordering should + flip for you — as a preference, not a safety finding. + +### (f) — no recommendation; the question is yours + +- **What the engine does** (emitted from `c519bc2a`): with the plan, primary `window function` **recall** at + 130 (100 + 30 overdue), `plan_refs` to the plan, no alternates, and `energy_deferred_repairs` naming the + *same* concept at 6/10. Without a plan: `decorators` recall 118, the same concept deferred beside it. +- **Why it is a real question.** "Due recall is never deferred" is a policy, not a conclusion from retrieval + research (astra's refutation 3): the spacing benefit of retrieval does not establish that a retrieval attempt + on a live failure is suitable at 3/10. The two rules meet on one concept and the recall wins by construction. +- **What you are asked.** One bounded retrieval attempt on the failing concept — spacing wins — or should a due + recall on a live-struggle concept inherit the repair's demand and defer with it? And what should happen + immediately after an unsuccessful attempt? If your answer is "defer", that is a `no` on (f) and a 0.6.0 item + with a RED test already nameable: `test_due_recall_on_a_live_struggle_inherits_the_repair_demand`. + +## What the council changed in the tree, and what it did not + +**Applied** (RED `a0946376`, GREEN `c519bc2a`; design §5 decision 9), because the finding held against source: + +1. The body-double reason named the deferred struggle with nothing beside it. The framework's own rule + (*Naming Struggle Topics*: "never name a struggle without an adjacent strength") is now followed with data + the plan already records: the reason opens with `1 of 2 milestones of SQL Windows done.` and says nothing + when nothing is done — a strength is never invented. +2. "A body-double session, no new material, no repair" promised a door the engine does not control (grok). The + co-study persona guarantees "the student drives" and "stay quiet by default", so the reason now says that, + and a test holds the persona to those words. +3. The rubric's own rationale still said `42 < any real candidate` — review 7's F2 fixed the code, design and + spec but never the receipt you score against. Corrected. +4. Reading (f) added and emitted (astra §4 and grok §4 named the same missing reading independently). + +**Not applied, and why:** + +- Gating deferral on an active plan (qwen, review 7 Q1) — rejected by every seat this round too. +- Suppressing the deferred-item names from the body-double reason (astra 🟡: "do not turn an unrelated session + into a recital of unfinished difficulties") — D-F asks for them by name, so this is yours: if you want the + deferred list *only* in the CLI's "Deferred for energy" block and the Today card's notes, not in the sit-with + sentence, say so on row 3b and it is a one-line change with a test. +- A same-concept gentle recall as the no-plan floor (Q3) — stays the recorded follow-on; your (d) line decides + whether it is built. + +**Seat claims checked and found wrong or out of scope** (so they are not acted on): grok's "§0 cites +`web/routes/body_double.py` as the feature" — that is the HANDOFF's verbatim text and was amended in design +§5 (T5.1 amendment 3); grok's "medium and high are not two behaviours" — true and already recorded (design §5 +decision 7); qwen's refutation that the collector "runs across multiple sessions" misreads "over three +readings" (three fixtures, one call each); astra's refutation 6 (the async body-double copy asks questions +despite "no questions") is about `agents/shared/audhd-framework.md`'s prose, not this change — noted for a +docs pass, not fixed here. + +## One paragraph, plain language + +These readings decide that a low-energy day is no longer "do the least-bad piece of real work". Fixing the +thing you are stuck on is treated as work the day cannot carry; reviewing something you already know and +recalling something that is due stay allowed; and when the plan has nothing left that fits, StudyLoop offers to +sit with you rather than hand you a penalised drill. If you asked why it will not let you fix the stuck thing +today: because on a 3/10 day the expensive card is how a wobble becomes a story about yourself — the repair is +still there, named, waiting for a day that can carry it. The two things only you can settle are the no-plan +morning (would you actually tap a generic starter?) and the collision in (f) (when the stuck thing is also due, +do you want to be asked to recall it?). diff --git a/openspec/changes/plan-integration-followons/tasks.md b/openspec/changes/plan-integration-followons/tasks.md index 4914d3cce..bd0cebbf4 100644 --- a/openspec/changes/plan-integration-followons/tasks.md +++ b/openspec/changes/plan-integration-followons/tasks.md @@ -227,6 +227,10 @@ writer, through the existing gate, closes that. Kept out of item 3 so item 3's f own words), collapsing medium/high, a same-concept gentle recall floor (follow-on, decision 8). Row 3b gained readings (d) and (e) for the two places the seats split.) - [ ] **T5.5** Rubric row **3b** for the owner (`PENDING`); the change is archived only after the owner scores 3b + — **2026-09-20**: a rubric-3b council (three seats, recommendations to the owner, not verdicts) produced + `receipts/rubric-3b-decision-brief-2026-09-20.md`; its two source-verified findings were applied + (RED `a0946376` → GREEN `c519bc2a`, design §5 decision 9), the receipt's stale F2 claim corrected, and + reading **(f)** (same-concept due recall + live struggle) emitted and added to the row. Verdict still the owner's. (4b was scored **yes / yes** on 2026-09-18, so 3b is the one row still outstanding). (2026-09-19: row 3b written into `receipts/now-rubric-2026-09-16.md` with three readings printed from the real engine — (a) live struggle → body-double primary, (b) plus an unrelated due recall → due recall primary, proposal beneath, From 0760254d0b1b4a4f5473ed5298a3be5a562234af Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 14:45:48 +0100 Subject: [PATCH 21/32] =?UTF-8?q?test(now):=20RED=20=E2=80=94=20the=20low-?= =?UTF-8?q?demand=20teach-back=20door=20is=20the=20micro=20form;=20a=20lea?= =?UTF-8?q?rning=20row=20is=20a=20gentle=20review?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two defects the owner's rubric row 3b reading (c) surfaced while checking the door behind the yes (interactive walkthrough, 2026-09-20), one test each: 1. `_evidence_command` names `--type structured` for every teach-back candidate. Demand `low` (0/10, design §5 amendment 1) was justified by the protocol's *micro* teach-back — "in one sentence, explain [concept]", two dimensions scored — not by the 15-minute five-dimension structured review. The door must be the form the demand class stands on: `micro` for `low`; a weak-teach-back row (`medium`) keeps `structured`. 2. The struggle collector's reason for a `learning` row reads "Recorded as learning; repair now while the signal is fresh". Design §5 calls a `learning` row the gentle review that stays eligible at low energy, and on a low-energy screen the word "repair" contradicts the deferred-repair line printed beside it. The reason names the gentle review and its one-sentence door; a live struggle's reason is untouched. Both fail against c519bc2a for exactly those reasons. --- .../studyloop/tests/test_now_plan_guidance.py | 58 +++++++++++++++++++ 1 file changed, 58 insertions(+) diff --git a/packages/studyloop/tests/test_now_plan_guidance.py b/packages/studyloop/tests/test_now_plan_guidance.py index f51fd7421..eca129e32 100644 --- a/packages/studyloop/tests/test_now_plan_guidance.py +++ b/packages/studyloop/tests/test_now_plan_guidance.py @@ -1706,6 +1706,64 @@ def test_energy_demand_confidence_and_teachback_precedence(monkeypatch) -> None: ] +def test_low_demand_teachback_door_is_the_micro_teach_back(monkeypatch) -> None: + """Rubric row 3b reading (c), defect 1 (owner walkthrough 2026-09-20): demand ``low`` + (0/10) was justified by the protocol's *micro* teach-back — "in one sentence, explain + [concept]", two dimensions scored — yet the door named ``--type structured``, the + 15-minute five-dimension review. The door must be the form the demand class stands + on: ``micro`` for ``low``; a weak-teach-back row (``medium``) keeps ``structured``.""" + _plant_struggles( + monkeypatch, + _struggle("window function", confidence="learning", days_ago=2), + _struggle("weak", topic="python", confidence="confident", days_ago=20, teachback=9), + ) + + high = build_now_plan(energy="high") + + doors = {rec.concept: rec for rec in _all(high)} + gentle, weak = doors["window function"], doors["weak"] + assert gentle.action_type == weak.action_type == "teachback" + assert gentle.metadata["energy_demand"] == "low" + assert weak.metadata["energy_demand"] == "medium" + assert gentle.evidence_command == ( + 'studyloop teachback "window function" -t "sql" --score "3,3,3,3,3" --type micro' + ) + assert weak.evidence_command == ( + 'studyloop teachback "weak" -t "python" --score "3,3,3,3,3" --type structured' + ) + + +def test_learning_row_reason_is_a_gentle_review_not_a_repair(monkeypatch) -> None: + """Rubric row 3b reading (c), defect 2 (owner walkthrough 2026-09-20): design §5 calls a + ``learning`` row the *gentle review* that stays eligible at low energy, but its reason + read "repair now while the signal is fresh" — on a low-energy screen that word + contradicts the deferred-repair line printed beside it. The learning row's reason + names the gentle review and its one-sentence door; a live struggle's reason is + untouched.""" + _plant_struggles( + monkeypatch, + _struggle("window function", confidence="learning", days_ago=2, teachback=11), + _struggle("decorators", topic="python", days_ago=1), + ) + + low = build_now_plan(energy="low") + + assert low.primary.concept == "window function" + assert low.primary.reason == ( + "Recorded as learning; a gentle review keeps it fresh — one sentence, in your own " + "words; last teach-back score 11/20" + ) + assert "repair" not in low.primary.reason + [deferred] = low.energy_deferred_repairs + assert deferred.concept == "decorators" + assert "repairing 'decorators'" in deferred.reason + + high = build_now_plan(energy="high") + + live = next(rec for rec in _all(high) if rec.concept == "decorators") + assert live.reason == "Recorded as struggling; repair now while the signal is fresh" + + @pytest.mark.parametrize( ("energy", "deferred"), [("low", {"live", "stale"}), ("medium", set()), ("high", set())], From f2cb572fd6b3c9796938b9271f2e7bf5a0bf0e50 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 14:48:03 +0100 Subject: [PATCH 22/32] feat(now): the low-demand teach-back door is the micro form; a learning row reads as a gentle review MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit GREEN for 0760254d (rubric row 3b reading (c), owner walkthrough 2026-09-20). - `_evidence_command` takes the candidate's `energy_demand` and names the teach-back form the demand class stands on: `--type micro` for `low` (the one-sentence, two-dimension teach-back that justified 0/10), `structured` otherwise — a weak-teach-back row (`medium`) re-sits the review it fell short on. Every other action type is untouched. - The struggle collector derives the demand once, before the candidate is built, and hands the same value to the door and to `metadata.energy_demand`. - A `learning` row's reason is "Recorded as learning; a gentle review keeps it fresh — one sentence, in your own words" (design §5: the gentle review that stays eligible at low energy); a `struggling` row keeps "repair now while the signal is fresh". The teach-back-score suffix is unchanged. The no-plan golden is byte-identical (ec451ce8): the golden world has no struggle rows. test_now_plan_guidance + test_learning_decision 76/76, test_web_now + test_wind_down_decision 14/14, ruff/pyright clean. --- .../src/studyloop/learning/decision.py | 43 +++++++++++++++---- 1 file changed, 34 insertions(+), 9 deletions(-) diff --git a/packages/studyloop/src/studyloop/learning/decision.py b/packages/studyloop/src/studyloop/learning/decision.py index b17c479cb..4753bb56b 100644 --- a/packages/studyloop/src/studyloop/learning/decision.py +++ b/packages/studyloop/src/studyloop/learning/decision.py @@ -350,11 +350,24 @@ def _shell_word(text: str) -> str: return shlex.quote(text) -def _evidence_command(action_type: ActionType, concept: str, topic: str, source: str) -> str: +def _evidence_command( + action_type: ActionType, + concept: str, + topic: str, + source: str, + *, + energy_demand: EnergyDemand | None = None, +) -> str: if action_type == "teachback": + # The door is the teach-back form the demand class stands on (design §5, + # amendment 1; rubric row 3b reading (c)): `low` was justified by the + # protocol's micro teach-back — one sentence, two dimensions scored — + # so that is what a low-demand door asks for. A weak-teach-back row + # (`medium`) re-sits the structured review it fell short on. + review_type = "micro" if energy_demand == "low" else "structured" return ( f"studyloop teachback {_shell_word(concept)} -t {_shell_word(topic)} " - '--score "3,3,3,3,3" --type structured' + f'--score "3,3,3,3,3" --type {review_type}' ) if action_type == "hands-on" and source.endswith(".json"): return f'studyloop practice verify {_shell_word(source)} --task 1 --notes "what passed?"' @@ -479,29 +492,41 @@ def _struggle_candidates(time_minutes: int) -> list[_Candidate]: if "source_section" in row_keys and row["source_section"] else f"study_progress:{topic}:{concept}" ) + # Design §5: derived once, here, from the collector's own classes; the + # deferral, the door and every renderer read this one value. + demand = _energy_demand( + confidence, row.get("last_seen") if "last_seen" in row_keys else None, today + ) + # A `learning` row is the gentle review that stays eligible at low + # energy (design §5), not a repair: on a low-energy screen "repair now" + # would contradict the deferred-repair line beside it (rubric row 3b + # reading (c)). Its sentence names the one-sentence door it opens. + reason = ( + "Recorded as learning; a gentle review keeps it fresh — one sentence, in your own words" + if confidence == "learning" + else f"Recorded as {confidence}; repair now while the signal is fresh" + ) candidates.append( _Candidate( concept=concept, topic=topic, course=row["source_course"] if "source_course" in row_keys else None, reason=( - f"Recorded as {confidence}; repair now while the signal is fresh" + reason + (f"; last teach-back score {teachback_score}/20" if teachback_score else "") ), action_type=action, estimated_minutes=_estimate_minutes(action, time_minutes, 20), source=str(source), - evidence_command=_evidence_command(action, concept, topic, str(source)), + evidence_command=_evidence_command( + action, concept, topic, str(source), energy_demand=demand + ), score=score, metadata={ "confidence": confidence, "last_teachback_score": teachback_score, "session_count": row["session_count"], - # Design §5: derived once, here, from the collector's own - # classes; the deferral and every renderer read this value. - "energy_demand": _energy_demand( - confidence, row.get("last_seen") if "last_seen" in row_keys else None, today - ), + "energy_demand": demand, }, ) ) From c7cc1b0ff78fdf34cd3b8045338215b8ea7ded01 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 14:49:21 +0100 Subject: [PATCH 23/32] =?UTF-8?q?test(agents):=20RED=20=E2=80=94=20the=20t?= =?UTF-8?q?each-back=20protocol=20carries=20the=20low-energy=20guided-expl?= =?UTF-8?q?anation=20fallback?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rubric row 3b reading (c), owner verdict "yes, with a fallback" (walkthrough 2026-09-20): the micro teach-back is offered at low energy, and if the student cannot produce the one sentence the mentor moves to a guided explanation — not the four-round Stuck ladder first. Checked against the tree: socratic-engine.md's Stuck ladder reaches a worked example only at round 4, co-study.md offers a brief explanation after two exchanges, and teach-back-protocol.md has no "cannot produce → guided explanation" step; none of the three is conditioned on low energy. The door's protocol must say so, or the yes ships without its condition. Fails against f2cb572f: the Micro Teach-Back section has no fallback. --- .../studyloop/tests/test_now_plan_guidance.py | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/packages/studyloop/tests/test_now_plan_guidance.py b/packages/studyloop/tests/test_now_plan_guidance.py index eca129e32..e99e3aad5 100644 --- a/packages/studyloop/tests/test_now_plan_guidance.py +++ b/packages/studyloop/tests/test_now_plan_guidance.py @@ -1404,6 +1404,21 @@ def test_co_study_persona_pins_the_promise_the_body_double_reason_makes() -> Non assert "Stay quiet by default" in persona +def test_teach_back_protocol_carries_the_low_energy_guided_explanation_fallback() -> None: + """Rubric row 3b reading (c) — owner verdict "yes, with a fallback" (2026-09-20): the + micro teach-back is offered at low energy, and if the student cannot produce the one + sentence the mentor moves to a guided explanation — not the four-round Stuck ladder + first. The door's protocol must say so, or the yes is shipped without its condition.""" + protocol = ( + Path(__file__).resolve().parents[3] / "agents" / "shared" / "teach-back-protocol.md" + ).read_text(encoding="utf-8") + + micro = protocol.split("### Micro Teach-Back", 1)[1].split("\n## ", 1)[0] + assert "Low-energy fallback" in micro + assert "guided explanation" in micro + assert "not the four-round Stuck ladder" in micro + + def test_body_double_is_a_proposal_not_a_filter(monkeypatch) -> None: """An unrelated real candidate still wins; the body-double proposal sits beneath it as an alternate, base score below any real candidate's.""" From 214df377ca14660cb6607fb4e6ab0fa87753df91 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 14:51:42 +0100 Subject: [PATCH 24/32] docs(agents): the micro teach-back carries a low-energy guided-explanation fallback MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit GREEN for c7cc1b0f (rubric row 3b reading (c), owner verdict "yes, with a fallback", 2026-09-20). teach-back-protocol.md's Micro Teach-Back section gains the fallback the door now relies on: when `studyloop now` offers a micro teach-back on a low-energy day and the student cannot produce the one sentence, the mentor moves straight to a guided explanation (two or three plain sentences with a networking analogy, then one phrase said back in the student's own words) — not the four-round Stuck ladder, which at low energy is the productive struggle the day cannot carry and the RSD exposure the live-struggle deferral exists to prevent. The blank is not scored as a teach-back (it measures the day, not the concept); the phrase that came back is recorded as `learning` and the 7-day structured review measures the concept. agents/manifest.json: only the moved entry changes (hash 9bbe8831f1c74837 → 09ee534106322244, updated 2026-09-20); the updater's re-stamping of every unmoved entry's date was restored, as at item 4. .secrets.baseline: whole-repo scan with the hook's pinned v1.5.0 — 72 → 72 result files, exactly the manifest's one hashed entry moves. --- .secrets.baseline | 4 ++-- agents/manifest.json | 4 ++-- agents/shared/teach-back-protocol.md | 2 ++ 3 files changed, 6 insertions(+), 4 deletions(-) diff --git a/.secrets.baseline b/.secrets.baseline index 65d0d6d3a..097785c61 100644 --- a/.secrets.baseline +++ b/.secrets.baseline @@ -221,7 +221,7 @@ { "type": "Hex High Entropy String", "filename": "agents/manifest.json", - "hashed_secret": "6e16c27090e6ef78d3d43f0d4244d7a9987e76d9", + "hashed_secret": "e0a1b3d7e38a074c21ec31895cdfbe5aa1019e22", "is_verified": false, "line_number": 73 }, @@ -2049,5 +2049,5 @@ } ] }, - "generated_at": "2026-09-18T11:08:48Z" + "generated_at": "2026-09-20T13:51:25Z" } diff --git a/agents/manifest.json b/agents/manifest.json index c6c6fe77b..23253c4e3 100644 --- a/agents/manifest.json +++ b/agents/manifest.json @@ -70,8 +70,8 @@ "updated": "2026-09-14" }, "shared/teach-back-protocol.md": { - "hash": "9bbe8831f1c74837", - "updated": "2026-09-14" + "hash": "09ee534106322244", + "updated": "2026-09-20" }, "shared/wind-down-protocol.md": { "hash": "5b1ec3303b8d1086", diff --git a/agents/shared/teach-back-protocol.md b/agents/shared/teach-back-protocol.md index c2f52f6fc..bc0d87bc9 100644 --- a/agents/shared/teach-back-protocol.md +++ b/agents/shared/teach-back-protocol.md @@ -49,6 +49,8 @@ Only score 2 dimensions: Accuracy and Own Words (max 8 points). | 6-7 | Good recall in own words | On track | | 8 | Strong | Consider accelerating interval | +**Low-energy fallback.** When `studyloop now` offers a micro teach-back on a low-energy day (a concept recorded as `learning`, energy demand `low`), the ask is one sentence and nothing more. If the student cannot produce it — a blank, "I don't know", or the source's own words — move straight to a **guided explanation**: give the explanation in two or three plain sentences with a networking analogy, then ask the student to say back ONE phrase of it in their own words. This is the fallback, not the four-round Stuck ladder in `socratic-engine.md`: at low energy that ladder is the productive struggle the day cannot carry, and a stall there is the confidence damage (RSD) the deferral of live struggles exists to prevent. Do not score the blank as a teach-back — at low energy it measures the day, not the concept. Record the phrase that came back (`studyloop progress "<concept>" -t <topic> -c learning`) and let the structured review at the 7-day mark measure the concept. + ## Detecting Understanding vs Memorisation ### Red Flags (Surface Learning) From b16129181bdfdf09241754652b05d548f5432654 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 14:52:40 +0100 Subject: [PATCH 25/32] =?UTF-8?q?docs(rubric-3b):=20owner=20verdicts=20(a)?= =?UTF-8?q?=E2=80=93(d)=20recorded;=20(c)'s=20three=20fixes=20landed;=20(e?= =?UTF-8?q?)/(f)=20pending?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Interactive walkthrough with the owner, 2026-09-20 — the row's verdict cell as it stands, committed rather than left as a working-tree change: (a) yes — sit with the plan rather than repair the live struggle at low energy; note (0.6.0): the sit-with session needs one tiny concrete first move, not a blank page. (b) yes — familiar recall leads, the sit-with sits beneath it; note (0.6.0): the low-energy day is a ladder, not a snapshot — offer a step-up after a completed floor task (nothing re-plans after an action today). (c) yes, with a fallback — the micro teach-back is offered at low energy and the mentor falls back to a guided explanation when the sentence will not come. Three fixes found while checking that door landed before this commit: the door names --type micro for low demand (0760254d/f2cb572f), a learning row's reason is a gentle review not "repair now" (same pair), and the protocol carries the low-energy fallback (c7cc1b0f/214df377). The cell's citation of design §5 is corrected to the design's own words ("recovered / gentle review → low"), not "consolidation". (d) yes — the generic starter plus the deferred line is the no-plan floor; note (0.6.0): re-offer the deferred repair itself as the step-up after a completed floor task, keyed on an outcome signal that does not exist yet. (e) and (f) remain PENDING for the owner; the row's top-of-file status line is updated when they land. --- .../plan-integration/receipts/now-rubric-2026-09-16.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md index 68ef5ea36..4d6d461d1 100644 --- a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md +++ b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md @@ -30,7 +30,7 @@ learning"). | 1 | Matching due | Active plan `sql-windows` ("SQL Windows", topics `[sql]`, next milestone 0 concepts `[window function]`). Due items: `decorators`/python base 102, `window function`/sql base 100. | **`window function`** (sql, recall, score 130, `plan_refs=[(sql-windows, 0)]`); alternate `decorators` (120, no refs). | Rule 5: both are due items two points apart — one urgency class — so the plan-related one takes the +12 bias and wins; the unrelated due item is *kept* as an alternate (bias, not filter). Rule 7 names the milestone the action advances. | **yes** — owner, 2026-09-16: "window function is the logical step before decorating it" (a prerequisite-order argument; see F2 follow-on). | | 2 | Urgent-unrelated wins | Same plan. One due item: `decorators`/python, base 100 (an overdue spaced-repetition review). Nothing represents milestone 0. | **`decorators`** (118, no refs); alternate `window function` (60, `source=study_plan:sql-windows:0`, `plan_refs=[(sql-windows, 0)]`, reason "Next milestone 1/1 of plan 'SQL Windows': Window basics"). | Rule 5: the unrelated candidate is in a more urgent class (due review) and wins outright — the bias cannot lift new-milestone work over it. Rule 6: the plan's next milestone was unrepresented, so it was synthesised at base 48 + bias 12 = 60 and appears as the plan-backed alternate (rule 8 satisfied without any swap). | **yes** — owner, 2026-09-16: clear the overdue review first. Note for follow-on: an overdue item *unrelated* to the plan must not sit as an alternate indefinitely — propose it explicitly (age-aware nudge) or let the learner retire it. | | 3 | Energy-deferred | Plan `sql-windows` with `energy_floor: 5`; milestone 0 `Window basics` **done** (concepts `[window function]`), milestone 1 `Frames` (concepts `[window frame]`). One struggle-repair item `window function`/sql, `hands-on`, base 82. **Energy `low`** (capability 3/10). | **`window function`** (hands-on, score 80, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, milestone 1, floor 5, capability 3)]`; JSON gains `energy_deferred`. | Rule 3: capability 3 < floor 5, so the *new* milestone (Frames) is deferred and named, not synthesised; the plan-related repair on a finished milestone's concept stays eligible and keeps its ref (`None`: plan-related repair, not the next milestone). Score = 82 + 12 bias − 14 (hands-on at low energy). | **no** — owner, 2026-09-16: a struggle-repair task has no energy demand of its own; recommending hands-on repair of a *live* struggle on a low-energy day risks compounding the struggle and damaging confidence (RSD). Finding for council: (1) derive a per-item energy demand for repair from struggle recency/teach-back — at low energy a live struggle defers like new work, a recovered one stays eligible as gentle review; (2) when nothing plan-related fits the day's capability, synthesise a body-doubling / open-session candidate (feature exists: ADR-0001/0003, `web/routes/body_double.py`) naming the deferred items, instead of the least-bad task. | -| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered); added after council review 7: **(d)** **no plan at all**, one live struggle `decorators`/python, low energy; **(e)** row 3's world plus one **unrelated hands-on** practice task `list comprehension drill`/python base 48, low energy; added after the rubric-3b council (2026-09-20, two seats independently named it): **(f)** row 3's world where `window function` is **both** the live struggle and **due for recall** (`study_progress` due row, base 100) — and the same with **no plan** (`decorators` live struggle + due recall). | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason (as emitted from `c519bc2a`, after the rubric-3b council) *"1 of 2 milestones of SQL Windows done. Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session — you drive; the companion stays quiet unless you ask."* (before the council, from `326abcf9`: *"Nothing plan-related fits low energy today — deferred: …; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."* — the progress lead is the framework's naming rule; the door is now described by what the co-study persona guarantees); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone. **(d)** primary **`one tiny recall loop`** (the starter, 28, reason *"Today's energy deferred the repair work it cannot carry; start with one small retrieval signal instead"*); no alternates; `energy_deferred_repairs=[(None, decorators, high, 6)]` — where before item 5 the primary was the hands-on repair of `decorators`. **(e)** primary **`Sit with SQL Windows`** (42, reason as in (a)); the hands-on drill is the alternate at 34 (48 − 14 low-energy penalty). **(f)** (emitted from `c519bc2a`) primary **`window function`** (**recall**, 130 = 100 + 30 overdue, `plan_refs=[(sql-windows, None)]`); no alternates (nothing plan-related fits, so no body double — a real candidate exists); `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** — the same concept is the primary *and* a deferred repair, because due recall is never deferred. No-plan variant: primary **`decorators`** (recall, 118, no refs); no alternates; `energy_deferred_repairs=[(None, decorators, struggling, high, 6, 3)]`. | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 — below every real candidate at base; at low energy it sits above only a hands-on task the energy rule already penalises (48 − 14 = 34), review 7 F2 — a proposal, not a filter) carrying the co-study session door. | **PENDING** — owner: (a) would you sit with the plan rather than repair the live struggle today? (b) is the body-double proposal right to sit beneath the unrelated due recall? (c) is the gentle teach-back on a recovered concept one you would do at low energy? (d) with **no plan**, is the starter plus the deferred line the floor you want on a low-energy day, rather than the hands-on repair you used to get (review 7: two seats keep it, one would restore the repair)? (e) at low energy, would you sit with the plan rather than do an unrelated hands-on drill (review 7 F2: the proposal outranks a task the low-energy rule penalises, and nothing else)? (f) when the concept you are failing is also due for recall, do you want that recall as the primary (spacing wins; the repair stays deferred beside it), or should a due recall on a live struggle inherit the repair's demand and defer with it? **Council decision brief (2026-09-20, `receipts/rubric-3b-decision-brief-2026-09-20.md`, three seats asked for recommendations to the owner, not verdicts):** (a) YES ×3; (b) YES ×3; (c) YES / YES / EITHER (astra: `learning` is a database state, not proof of recovery — keep the teach-back brief and optional); (d) YES / EITHER / EITHER — no seat would restore the hands-on repair; the open call is whether *you* would tap a generic starter, or want a same-concept gentle recall (Q3), or want the system out of the way; (e) YES / YES / EITHER (astra: an unrelated *familiar* drill is not the class of work the finding objected to); (f) not put to the seats — emitted after they named it. The verdict column is yours; the recommendations are provenance for it, not a substitute. | +| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered); added after council review 7: **(d)** **no plan at all**, one live struggle `decorators`/python, low energy; **(e)** row 3's world plus one **unrelated hands-on** practice task `list comprehension drill`/python base 48, low energy; added after the rubric-3b council (2026-09-20, two seats independently named it): **(f)** row 3's world where `window function` is **both** the live struggle and **due for recall** (`study_progress` due row, base 100) — and the same with **no plan** (`decorators` live struggle + due recall). | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason (as emitted from `c519bc2a`, after the rubric-3b council) *"1 of 2 milestones of SQL Windows done. Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session — you drive; the companion stays quiet unless you ask."* (before the council, from `326abcf9`: *"Nothing plan-related fits low energy today — deferred: …; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."* — the progress lead is the framework's naming rule; the door is now described by what the co-study persona guarantees); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone. **(d)** primary **`one tiny recall loop`** (the starter, 28, reason *"Today's energy deferred the repair work it cannot carry; start with one small retrieval signal instead"*); no alternates; `energy_deferred_repairs=[(None, decorators, high, 6)]` — where before item 5 the primary was the hands-on repair of `decorators`. **(e)** primary **`Sit with SQL Windows`** (42, reason as in (a)); the hands-on drill is the alternate at 34 (48 − 14 low-energy penalty). **(f)** (emitted from `c519bc2a`) primary **`window function`** (**recall**, 130 = 100 + 30 overdue, `plan_refs=[(sql-windows, None)]`); no alternates (nothing plan-related fits, so no body double — a real candidate exists); `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** — the same concept is the primary *and* a deferred repair, because due recall is never deferred. No-plan variant: primary **`decorators`** (recall, 118, no refs); no alternates; `energy_deferred_repairs=[(None, decorators, struggling, high, 6, 3)]`. | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 — below every real candidate at base; at low energy it sits above only a hands-on task the energy rule already penalises (48 − 14 = 34), review 7 F2 — a proposal, not a filter) carrying the co-study session door. | **Owner verdicts — interactive walkthrough 2026-09-20 (in progress).** **(a) yes** — owner: doing the stuck thing on a genuinely low day "is very rare and only when there was a time sensitive deliverable"; that is the override case (the repair stays reachable by hand, `studyloop study "window function"`), not the default, so the floor would have given the ordinary low day a non-zero start. Row 3's original "no" (2026-09-16) stands. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** the sit-with session must not be a blank page — the body double should propose one tiny, concrete first move on the deferred material (e.g. "open the Frames lesson and read it, nothing more", or ten minutes of passive video on the next milestone). A goal-less session is the hardest ADHD start; sharpening the replacement beats restoring the rejected repair. **(b) yes** — owner: at 3/10 "I would be reluctant to do more than familiar recall"; the recall leads, the sit-with sits beneath it. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** "once I had done this, if it went well I would like the option to then take on more but not lead with the larger task" — the low-energy day should be a ladder, not a snapshot: after a completed familiar recall, offer a step-up (a re-ask of energy, or the next-cheapest action — the sit-with, then one tiny concrete move on the deferred material, per the (a) note), never leading with the larger task. Verified against `c519bc2a`: nothing does this today — the Today card fetches `/api/now` once at init with no energy and never re-plans after an action (`today-panel.js` `init()`), and CLI `now` is a one-shot snapshot of the energy it was given. The (a) and (b) notes are one 0.6.0 theme. **(c) yes, with a fallback** — owner: "3/10 or 4/10 is very subjective, I would offer the teach back but as I personally believe I would be reluctant to do more than familiar recall at 3/10 … offer a fallback to a guided explanation". So: demand `low` stands and the teach-back is offered at low energy, **and the door must carry a fallback** — if the learner cannot produce the explanation, the mentor moves to a guided explanation, not four Socratic rounds first. Against `c519bc2a`: the general stuck ladder exists (`socratic-engine.md` "Stuck (Escalating Support)" reaches a worked example only at round 4; `co-study.md` offers a brief explanation after two exchanges) but `teach-back-protocol.md` has no "cannot produce → guided explanation" step and neither rule is conditioned on low energy. **Three fixes landed on this branch before the row is committed (one RED/GREEN each), each found while checking the door behind the yes:** (1) the teach-back door's evidence command named `--type structured` — the protocol's 15-minute five-dimension review — while demand `low` was justified by the one-sentence *micro* teach-back (`micro` is in `TEACHBACK_TYPES`); the door now names the form the demand class stands on (`micro` for `low`; `structured` for a weak-teach-back `medium` row, which re-sits the review it fell short on) — RED `0760254d` / GREEN `f2cb572f`. (2) the collector's reason for a `learning` row read "Recorded as learning; repair now while the signal is fresh" (`decision.py`); design §5's own words for a `learning` row are "recovered / gentle review → `low`" and "recovered repair stays eligible as gentle review" (the spec delta: "gentle repair"), and on a mixed low-energy screen "repair now" contradicts the deferred-repair line beside it; it now reads "Recorded as learning; a gentle review keeps it fresh — one sentence, in your own words" (the struggling row's sentence is untouched) — same RED/GREEN. (3) the fallback the yes is conditional on: `teach-back-protocol.md`'s Micro Teach-Back section now carries a **Low-energy fallback** — cannot produce the sentence → a guided explanation (two or three plain sentences with a networking analogy, then one phrase said back in the student's own words), not the four-round Stuck ladder; the blank is not scored as a teach-back (it measures the day, not the concept); pinned by `test_teach_back_protocol_carries_the_low_energy_guided_explanation_fallback` — RED `c7cc1b0f` / GREEN `214df377` (`agents/manifest.json`: only the moved entry; `.secrets.baseline` rescanned whole-repo with the hook's pinned v1.5.0, 72 → 72 files). The no-plan golden stays byte-identical (`ec451ce8`); `test_now_plan_guidance` + `test_learning_decision` 76/76 before the fallback pin, 77 with it. **Note (0.6.0, not a blocker):** the capability numbers are subjective (3/10 vs 4/10); the demand thresholds are coarse by construction, so the fallback carries more weight than the threshold. **(d) yes — the starter, with a step-up** — owner, shown the starter as built (its topic is the first configured topic, not `decorators`; its one command is `studyloop progress "one tiny recall loop" …` and opens nothing): "I would" take the primary as emitted — the generic starter plus the deferred line — not the hands-on repair; and "if this went well, a successful task often increases confidence and energy so give the user the opportunity to work on the live struggle." So: the deferral stands with no plan (no council seat would restore the repair either); the generic starter is a floor the owner would start; the same-concept gentle recall (design §5 Q3) that the council and the coordinator steered toward is recorded as provenance, not adopted. **Note recorded with the yes (a 0.6.0 finding, not a blocker) — it names the top rung of the (a)/(b) ladder:** after a completed floor task, re-offer the *deferred repair itself* as the step-up — an option the learner takes while the win is fresh, never restored as the default. Against `c519bc2a`: nothing can act on "went well" today — the starter emits no outcome the engine reads, and nothing re-plans after an action (the (b) gap) — so the 0.6.0 item needs both an outcome signal from the floor task and the re-offer keyed on it. **PENDING** — (e) at low energy, would you sit with the plan rather than do an unrelated hands-on drill (review 7 F2: the proposal outranks a task the low-energy rule penalises, and nothing else)? (f) when the concept you are failing is also due for recall, do you want that recall as the primary (spacing wins; the repair stays deferred beside it), or should a due recall on a live struggle inherit the repair's demand and defer with it? **Council decision brief (2026-09-20, `receipts/rubric-3b-decision-brief-2026-09-20.md`, three seats asked for recommendations to the owner, not verdicts):** (a) YES ×3; (b) YES ×3; (c) YES / YES / EITHER (astra: `learning` is a database state, not proof of recovery — keep the teach-back brief and optional); (d) YES / EITHER / EITHER — no seat would restore the hands-on repair; the open call is whether *you* would tap a generic starter, or want a same-concept gentle recall (Q3), or want the system out of the way; (e) YES / YES / EITHER (astra: an unrelated *familiar* drill is not the class of work the finding objected to); (f) not put to the seats — emitted after they named it. The verdict column is yours; the recommendations are provenance for it, not a substitute. | | 4 | Fully-checked | Plan `done-plan` ("Done Plan"), milestones A and B both done. One due item `decorators`/python base 100. | **`decorators`** (118, no refs); `completion_actions=[(done-plan, "Every milestone of 'Done Plan' is checked off — close the plan or extend it with a follow-on mission.")]`; no `study_plan:` candidate anywhere; JSON gains `active_plans` + `completion_actions`. | Rule 9: a fully-checked plan is reported as a completion action and is neither matched (no bias, no refs) nor synthesised. | **primary yes / completion action no as phrased** — owner, 2026-09-16: the completion action must be contextual and consensual. Run the end assessment (`assess(phase="end")`: due reviews, struggles, unverified milestones on the plan's concepts). If outstanding work touches the plan's concepts (or their prerequisites — F2 concept edges), propose *extend* and name the evidence; if clean, propose *close* and ask the learner to agree ("anything you are not comfortable with?"). Status never changes automatically (#7). Natural vehicle: architect with `purpose=planning` and the assessment in the brief (`plan close <id>`, sibling of `plan repair <id>`). Finding for council. | | 4b | Fully-checked — **re-run after D-G (item 4, 2026-09-18)** | Row 4's fixture (`done-plan`, milestones A `[alpha]` and B `[beta]` both done; one due item `decorators`/python base 100), plus the end assessment's readers planted: **(a)** one due review on plan concept `alpha` (`overdue`) with session mentions backing both concepts; **(b)** no due rows, same mentions. | Primary unchanged in both: **`decorators`** (118, no refs); no `study_plan:` candidate. **(a)** `completion_actions=[(done-plan, due 1 / struggles 0 / unverified 0, proposal **extend**, evidence `["Due review: alpha — overdue"]`)]`, sentence: "Every milestone of 'Done Plan' is checked off, and the closing review proposes extending the plan — 1 due review, 0 struggles and 0 unverified milestones on its concepts. Walk the evidence with the architect: studyloop plan close done-plan." **(b)** counts 0/0/0, proposal **close**, evidence `[]`, sentence: "Every milestone of 'Done Plan' is checked off and the closing review is clean — it proposes closing the plan. Close it with the architect when you agree: studyloop plan close done-plan." No warnings; JSON gains the five keys only inside the entry. | Rule 9 as before for the ranking. The completion action is now the end assessment read as a preview (`assess(phase="end", record=False)`, one call, no write, no status change): `extend` iff any of the three counts on the plan's own concepts is above zero, else `close`; due counts only rows naming a concept (the scheduler's "new topic" row is excluded — owner decision 2026-09-17). `plan close done-plan` launches the architect with the same review as the brief's first section; status moves only when the learner agrees. | **yes / yes** — owner, 2026-09-18, answering the two questions as posed: (a) **yes**, a proposal the owner would walk; (b) **yes**, a close the owner would agree to. No further line given. Closes row 4's "no as phrased" finding; status still moves only when the learner agrees in the architect conversation. | | 5 | No-plan identical | No plan documents at all; no collector candidates. | **`one tiny recall loop`** (python, recall, 28, `source=starter`) — the starter. | D-5: `serialise(plan) == golden` → **byte-identical** (`True` in the run); the JSON key list is exactly the golden's — no additive key is present. | **verified** — owner walkthrough 2026-09-16: nothing to judge; the byte-identical golden is the acceptance. | From 8f851d3a445f861d31660fbcdd84e2e8ff132e29 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 15:01:35 +0100 Subject: [PATCH 26/32] =?UTF-8?q?test(now):=20RED=20=E2=80=94=20the=20due?= =?UTF-8?q?=20collector's=20copy=20of=20a=20live=20struggle=20defers=20wit?= =?UTF-8?q?h=20the=20repair?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found while emitting rubric row 3b reading (f) with both real collectors (owner walkthrough 2026-09-20), executed not inferred: `_due_progress_candidates` and `_struggle_candidates` read the same `observations.rows`; `_review_type_for` labels every `struggling` row due ("Guided repair + tiny practice") and `_action_for_review` makes that due item `hands-on` at 100 + days + 35. The struggle collector's copy of the row is deferred at low energy (design §5); the due collector's copy carries no `energy_demand`, is kept as "due recall", and — deduped after the deferral — becomes the primary. Against a real sessions.db, row 3's world at low energy therefore emitted: window function — hands-on — "Guided repair + tiny practice; … struggling" Deferred for energy: repairing "window function" (struggling) asks for 6/10 with no body double — the row-3 "no" still on top, its own deferral printed beneath it. The (a), (d) and (e) behaviours the owner scored exist only with the due collector silenced, which is what `_plant_struggles` does. Two tests plant the rows for BOTH collectors: the plan world (body double primary, the struggle named once in `energy_deferred_repairs`, medium energy unchanged) and the no-plan world (starter primary). Both fail against b1612918 with `study_progress:…` as the primary's source. --- .../studyloop/tests/test_now_plan_guidance.py | 63 +++++++++++++++++++ 1 file changed, 63 insertions(+) diff --git a/packages/studyloop/tests/test_now_plan_guidance.py b/packages/studyloop/tests/test_now_plan_guidance.py index e99e3aad5..b20a74f04 100644 --- a/packages/studyloop/tests/test_now_plan_guidance.py +++ b/packages/studyloop/tests/test_now_plan_guidance.py @@ -1225,6 +1225,69 @@ def _row3_plan() -> None: ) +def _plant_struggles_for_both_collectors(monkeypatch: pytest.MonkeyPatch, *rows: dict) -> None: + """Like ``_plant_struggles`` but the due-progress collector runs for real too. + + Both collectors read the same ``observations.rows``: a ``struggling`` row is + *always* due (``_review_type_for`` labels it "Guided repair + tiny practice") + and ``_action_for_review`` makes that due item ``hands-on`` — the repair, + collected twice. Silencing the due collector, as ``_plant_struggles`` does, + hides the second copy. + """ + from studyloop.history import observations + + real_due = decision._due_progress_candidates + real_struggle = decision._struggle_candidates + _patch_collectors(monkeypatch) + monkeypatch.setattr(decision, "_due_progress_candidates", real_due) + monkeypatch.setattr(decision, "_struggle_candidates", real_struggle) + monkeypatch.setattr(observations, "rows", lambda conn: [dict(row) for row in rows]) + + +def test_due_copy_of_a_live_struggle_defers_with_the_repair_at_low_energy(monkeypatch) -> None: + """Rubric row 3b reading (f), re-read against the real collectors (owner walkthrough + 2026-09-20): the due-progress collector emits every ``struggling`` row as an undeferred + ``hands-on`` "Guided repair + tiny practice" item — the same observations row the struggle + collector defers — so against a real sessions.db the row-3 "no" was still the primary at + low energy, with its own deferral printed beneath it, and the body double never appeared. + A due row that *is* the repair of a live struggle carries the repair's demand and defers + with it, named once; due recall is still never deferred; medium energy is untouched.""" + _row3_plan() + _plant_struggles_for_both_collectors(monkeypatch, _struggle("window function", days_ago=3)) + + low = build_now_plan(energy="low") + + assert low.primary.source == "body_double" + assert not any(rec.concept == "window function" for rec in _all(low)) + assert [(d.concept, d.energy_demand) for d in low.energy_deferred_repairs] == [ + ("window function", "high") + ] + + medium = build_now_plan(energy="medium") + + assert medium.primary.concept == "window function" + assert medium.primary.action_type == "hands-on" + assert medium.primary.source == "study_progress:sql:window function" + assert medium.energy_deferred_repairs == () + + +def test_due_copy_of_a_live_struggle_defers_with_no_plan_too(monkeypatch) -> None: + """The no-plan floor (reading (d)) against both real collectors: the starter stands in + and the live struggle is named once in the deferred list — not emitted as an undeferred + hands-on primary by the due collector.""" + _plant_struggles_for_both_collectors( + monkeypatch, _struggle("decorators", topic="python", days_ago=3) + ) + + low = build_now_plan(energy="low") + + assert low.primary.source == "starter" + assert not any(rec.concept == "decorators" for rec in _all(low)) + assert [(d.plan_id, d.concept, d.energy_demand) for d in low.energy_deferred_repairs] == [ + (None, "decorators", "high") + ] + + def test_live_struggle_repair_defers_at_low_energy_like_new_work(monkeypatch) -> None: """Rule 3 extended (design §5, amendment 1 + 2): repair carries a demand of its own. From fb63fb6b3ff44761df6acc321b1912d7ddc14ab7 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 15:06:58 +0100 Subject: [PATCH 27/32] feat(now): the due collector's copy of a live struggle defers with the repair MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit GREEN for 8f851d3a (rubric row 3b reading (f), owner walkthrough 2026-09-20; design §5 decision 10). - `_due_progress_candidates`: a `struggling` row's due item ("Guided repair + tiny practice", `hands-on`) is the struggle collector's repair collected a second time from the same observations row. It now carries the same `energy_demand` (`_energy_demand` over `last_studied`), so rule 3 defers both copies together at low energy. Due recall and teach-back rows carry no demand and are never deferred — the invariant now stated about the rows it was always true of. - `_defer_repairs`: a concept is named once in `energy_deferred_repairs` however many copies of its row were deferred. - Spec: the "Due recall is never deferred" scenario is re-stated for a recall-typed due item, and a new scenario pins the struggling row read by both collectors (body double / starter primary at low energy, the due copy primary again at medium). Design §5: the rule-3 bullet amended, decision 10 records the mechanism from execution, decision 9's (f) note superseded. Re-emitted through both real collectors on this tree: (a), (b), (d), (e) unchanged from the receipt; (c) shows the due recall ("10-min Socratic review") first and the micro teach-back beneath it; (f) is not a recall/repair collision — a struggling row is never recall through the real collector — so it resolves to this fix. The receipt is amended in the following commit. No-plan golden byte-identical (ec451ce8). Suites 93/93, ruff/pyright clean. --- .../plan-integration-followons/design.md | 23 ++++++++++++-- .../specs/active-learning-decisions/spec.md | 17 +++++++++-- .../src/studyloop/learning/decision.py | 30 +++++++++++++++++-- .../studyloop/tests/test_now_plan_guidance.py | 7 +++-- 4 files changed, 68 insertions(+), 9 deletions(-) diff --git a/openspec/changes/plan-integration-followons/design.md b/openspec/changes/plan-integration-followons/design.md index a4dc72445..cb90654f2 100644 --- a/openspec/changes/plan-integration-followons/design.md +++ b/openspec/changes/plan-integration-followons/design.md @@ -226,7 +226,9 @@ among what MCP revises and the row names every schema property. capability (`high` → 6, `medium` → 4, `low` → 0) compared with `ENERGY_CAPABILITY[energy]`. - **Rule 3 extended.** Below capability, *repair* above demand is deferred exactly like new milestone work and listed in `energy_deferred` with a reason naming the struggle; recovered repair stays eligible as gentle - review. Due recall (`source=study_progress` due rows) is unaffected. + review. Due recall and teach-back rows (`source=study_progress`) are unaffected. *(Amended 2026-09-20, see + decision 10 below: a `struggling` row's due item is not recall — it is the repair, collected a second time — + and carries the demand.)* - **Body-doubling floor.** When the eligible plan-related set is empty **and** at least one active plan exists, synthesise one candidate: `source="body_double"`, `action_type="conversation"`, low base score (below any real candidate), reason naming the deferred items, `plan_refs` for each named plan with `milestone_index @@ -327,7 +329,24 @@ has none; `INTERLEAVE_RATIOS["low"]` unchanged. them by name; it goes to the owner as a note on row 3b. Two seats independently named the missing reading — the live struggle that is **also due for recall** on the same concept — so it is emitted from the tree as row 3b reading **(f)**: due recall is never deferred, so the recall is primary while the same concept's repair - sits in the deferred list; whether that is the collision the owner wants is theirs to say. + sits in the deferred list; whether that is the collision the owner wants is theirs to say. *(Superseded by + decision 10: emitted against the real collectors, (f) is not a recall/repair collision at all.)* +10. **The due collector's copy of a live struggle defers with the repair** (owner walkthrough 2026-09-20, + found while emitting reading (f) with both real collectors — executed, not inferred). `_due_progress_candidates` + and `_struggle_candidates` read the same `observations.rows`; `_review_type_for` labels every `struggling` row + due ("Guided repair + tiny practice") and `_action_for_review` makes that due item `hands-on` at + 100 + days + 35. Only the struggle collector's copy carried `energy_demand`, so at low energy it was deferred + while the due copy was kept as "due recall", survived `_dedupe` (higher score), and became the primary: against + a real sessions.db, row 3's world emitted the hands-on repair of the live struggle on top with its own deferral + line printed beneath it and no body double — the row-3 "no" unfixed. The (a), (d) and (e) behaviours held only + with the due collector silenced, which is what `_plant_struggles` does. Fix: the due collector's item for a + `struggling` row carries the same `_energy_demand` (derived from `last_studied`), so rule 3 defers both copies + together, and `_defer_repairs` names a concept once. Due recall and teach-back rows carry no demand and are + never deferred — that invariant is now stated about the rows it was always true of. Pinned by + `test_due_copy_of_a_live_struggle_defers_with_the_repair_at_low_energy` (plan world, medium energy unchanged) + and `test_due_copy_of_a_live_struggle_defers_with_no_plan_too`, both planting rows for **both** collectors + through `_plant_struggles_for_both_collectors`. Consequence for the rubric: readings (a), (d) and (e) are + re-emitted through both collectors before their verdicts are read as verdicts on the shipped behaviour. Known edge, not solved here: a `struggling` row whose `last_seen` cannot be parsed is read as live (`high`) — the cautious side; `_days_since` returns `None` and the demand falls to `high`. diff --git a/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md b/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md index f86fe9471..b84f9d0a2 100644 --- a/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md +++ b/openspec/changes/plan-integration-followons/specs/active-learning-decisions/spec.md @@ -253,11 +253,22 @@ with row 3b re-run after this requirement's repair half. and the live repair is ranked again #### Scenario: Due recall is never deferred -- **WHEN** energy is `low`, a due row on a plan concept is collected with - `confidence == "struggling"` and a live struggle repair is also collected -- **THEN** the due row is primary with `PlanRef(plan, None)`, the repair is +- **WHEN** energy is `low`, a due `recall` item on a plan concept is collected + (whatever its metadata says about confidence) and a live struggle repair is + also collected +- **THEN** the due item is primary with `PlanRef(plan, None)`, the repair is in `energy_deferred_repairs`, and no `body_double` candidate exists +#### Scenario: A struggling row's due item is the repair, collected twice +- **WHEN** energy is `low` and one `struggling` row seen 3 days ago on a + finished milestone's concept is read by both the due-progress collector + (which labels every `struggling` row due — "Guided repair + tiny practice", + `hands-on`) and the struggle collector +- **THEN** both copies are deferred, `energy_deferred_repairs` names the + concept once (`high`), the concept is ranked nowhere, and the primary is the + `body_double` proposal (the starter when no plan exists); at `medium` energy + the due copy is primary as before and the key is absent + #### Scenario: Nothing plan-related fits, so the engine proposes sitting with the plan - **WHEN** energy is `low`, the plan's `energy_floor` is 5 (milestone deferred) and its only repair is a live struggle (deferred) diff --git a/packages/studyloop/src/studyloop/learning/decision.py b/packages/studyloop/src/studyloop/learning/decision.py index 4753bb56b..e036446a6 100644 --- a/packages/studyloop/src/studyloop/learning/decision.py +++ b/packages/studyloop/src/studyloop/learning/decision.py @@ -435,6 +435,22 @@ def _due_progress_candidates(time_minutes: int) -> list[_Candidate]: "confidence": confidence, "days_ago": days_ago, "last_teachback_score": teachback_score, + # A `struggling` row is always due ("Guided repair + tiny + # practice") and its due item is `hands-on`: it is the + # struggle collector's repair, collected a second time from + # the same observations row. It carries the repair's demand + # so rule 3 defers both copies together (rubric row 3b + # reading (f), 2026-09-20); due recall and teach-back rows + # carry none and are never deferred. + **( + { + "energy_demand": _energy_demand( + confidence, item.get("last_studied"), datetime.now(UTC).date() + ) + } + if confidence == "struggling" + else {} + ), }, ) ) @@ -1209,14 +1225,17 @@ def _defer_repairs( ) -> tuple[list[_Candidate], tuple[DeferredRepair, ...]]: """Rule 3 extended (design §5): repair above its own energy demand is deferred like new work. - Only a candidate carrying ``energy_demand`` — the struggle collector's — is - judged. Due recall is never deferred whatever its confidence says, and a + Only a candidate carrying ``energy_demand`` is judged: the struggle + collector's repairs, and the due collector's copy of a ``struggling`` row — + the same repair, collected twice, named once. Due recall and teach-back rows + carry no demand and are never deferred whatever their confidence says; a ``learning`` repair (``low`` demand) is always carried. Plan-independent: the entry names the plan when one matches, else ``None``. """ capability = ENERGY_CAPABILITY[energy] kept: list[_Candidate] = [] deferred: list[DeferredRepair] = [] + named: set[tuple[str, str]] = set() for candidate in candidates: demand = candidate.metadata.get("energy_demand") if demand not in ENERGY_DEMAND_CAPABILITY: @@ -1226,6 +1245,13 @@ def _defer_repairs( if capability >= required: kept.append(candidate) continue + # The same struggling row reaches here twice — the struggle collector's + # repair and the due collector's "Guided repair + tiny practice" copy. + # Both are deferred; the learner reads one line for the concept. + key = (candidate.topic.lower(), candidate.concept.lower()) + if key in named: + continue + named.add(key) plan = plans.first_match(candidate) confidence = str(candidate.metadata.get("confidence") or "struggling") if demand == "high": diff --git a/packages/studyloop/tests/test_now_plan_guidance.py b/packages/studyloop/tests/test_now_plan_guidance.py index b20a74f04..1356effdb 100644 --- a/packages/studyloop/tests/test_now_plan_guidance.py +++ b/packages/studyloop/tests/test_now_plan_guidance.py @@ -1368,8 +1368,11 @@ def test_recovered_repair_stays_eligible_at_low_energy(monkeypatch) -> None: assert not any(rec.source == "body_double" for rec in _all(low)) assert [d.milestone_index for d in low.energy_deferred] == [1] - # A due row on a plan concept, even one recorded as struggling, is recall, - # not repair: it is never deferred and nothing is synthesised beside it. + # A recall-typed due item on a plan concept is never deferred, whatever its + # metadata says about confidence, and nothing is synthesised beside it. + # (Through the real collector a `struggling` row is never recall — it is the + # hands-on "Guided repair" copy, deferred with the repair; see + # test_due_copy_of_a_live_struggle_defers_with_the_repair_at_low_energy.) import dataclasses due = dataclasses.replace( From 291e86918f5bb30e7189a43311e0751b9b000922 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 15:08:46 +0100 Subject: [PATCH 28/32] docs(rubric-3b): (e) yes; (f) resolved by fb63fb6b; every reading re-emitted through both collectors MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Owner walkthrough 2026-09-20, closing row 3b: (e) yes — sit with the plan rather than an unrelated hands-on drill at low energy; the drill stays as the alternate. (f) resolved by a fix, not a verdict — the posed recall/repair collision is a world the real collectors never produce (a struggling row's due item is the guided repair, collected twice). Emitting it exposed that the due copy carried no demand, so against a real database the scored (a)/(d)/(e) screens were never shown; the row-3 "no" was. Fixed 8f851d3a/fb63fb6b. The emitted column now says what each reading shows through BOTH collectors: (a), (b), (d), (e) unchanged; (c) has the due recall first and the micro teach-back beneath it (the fixture that silenced the due collector hid this too); (f) before and after the fix. Status line and cell header updated. --- .../plan-integration/receipts/now-rubric-2026-09-16.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md index 4d6d461d1..eaff860e4 100644 --- a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md +++ b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md @@ -1,6 +1,6 @@ # Plan-aware `now` — five-scenario human rubric (D-16) · 2026-09-16 -**Status: owner verdicts RECORDED 2026-09-16 (interactive walkthrough with the coordinator).** Scenarios 1, 2 and the primary of 4: yes. Scenario 3: **no** (finding). Scenario 4 completion action: **no as phrased** (finding). Scenario 5: verified. **Row 3b added 2026-09-19** (item 5 / D-F, tree `feat/energy-demand-body-double`; outputs (a)–(c) emitted from the GREEN tree `326abcf9`, (d)–(e) from the post-review-7 tree `bdf4d6c6`): scenario 3 re-run with the struggle collector live, five readings printed, verdict `PENDING` for the owner. **Row 3b amended 2026-09-20** after the rubric-3b council (three seats, recommendations to the owner, not verdicts — `receipts/rubric-3b-decision-brief-2026-09-20.md`): the rationale's stale `42 < any real candidate` corrected (review 7 F2); the (a)/(e) reason re-emitted from `c519bc2a` (progress lead, persona-backed door wording — design §5 decision 9); reading **(f)** added and emitted from `c519bc2a` (the live struggle that is also due for recall); the council's recommendation per reading recorded beside the owner's still-`PENDING` verdict. **Row 4b added and scored 2026-09-18** (item 4 / D-G, tree `feat/plan-close`; outputs emitted from the tree committed as `82293293`): scenario 4 re-run with the end assessment planted, both proposals recorded as emitted; owner verdict **yes** on both readings — (a) *extend* with named evidence is a proposal the owner would walk, (b) a clean review's *close* is one the owner would agree to. This closes row 4's "no as phrased" finding; row 4 is kept as the record of the original finding. This receipt was produced unattended +**Status: owner verdicts RECORDED 2026-09-16 (interactive walkthrough with the coordinator).** Scenarios 1, 2 and the primary of 4: yes. Scenario 3: **no** (finding). Scenario 4 completion action: **no as phrased** (finding). Scenario 5: verified. **Row 3b added 2026-09-19** (item 5 / D-F, tree `feat/energy-demand-body-double`; outputs (a)–(c) emitted from the GREEN tree `326abcf9`, (d)–(e) from the post-review-7 tree `bdf4d6c6`): scenario 3 re-run with the struggle collector live, five readings printed, verdict `PENDING` for the owner. **Row 3b amended 2026-09-20** after the rubric-3b council (three seats, recommendations to the owner, not verdicts — `receipts/rubric-3b-decision-brief-2026-09-20.md`): the rationale's stale `42 < any real candidate` corrected (review 7 F2); the (a)/(e) reason re-emitted from `c519bc2a` (progress lead, persona-backed door wording — design §5 decision 9); reading **(f)** added and emitted from `c519bc2a` (the live struggle that is also due for recall); the council's recommendation per reading recorded beside the owner's then-`PENDING` verdict. **Row 3b scored 2026-09-20** (interactive walkthrough): (a)–(e) **yes**, each with one line and the 0.6.0 notes recorded in the cell; (c)'s yes carried a condition met on the branch (micro door, gentle-review wording, the protocol's low-energy fallback — `0760254d`…`214df377`); (f) exposed a real defect — the due-progress collector emitted every live struggle as an undeferred hands-on item, so the scored (a)/(d)/(e) screens were never emitted against a real database — fixed `8f851d3a`/`fb63fb6b` and every reading re-emitted through both collectors. **Row 4b added and scored 2026-09-18** (item 4 / D-G, tree `feat/plan-close`; outputs emitted from the tree committed as `82293293`): scenario 4 re-run with the end assessment planted, both proposals recorded as emitted; owner verdict **yes** on both readings — (a) *extend* with named evidence is a proposal the owner would walk, (b) a clean review's *close* is one the owner would agree to. This closes row 4's "no as phrased" finding; row 4 is kept as the record of the original finding. This receipt was produced unattended overnight. Every scenario below was *run* on frozen fixtures and the primary and its rationale are recorded exactly as the engine emitted them; the "would I do the primary?" column is a human judgement that only the owner can @@ -30,7 +30,7 @@ learning"). | 1 | Matching due | Active plan `sql-windows` ("SQL Windows", topics `[sql]`, next milestone 0 concepts `[window function]`). Due items: `decorators`/python base 102, `window function`/sql base 100. | **`window function`** (sql, recall, score 130, `plan_refs=[(sql-windows, 0)]`); alternate `decorators` (120, no refs). | Rule 5: both are due items two points apart — one urgency class — so the plan-related one takes the +12 bias and wins; the unrelated due item is *kept* as an alternate (bias, not filter). Rule 7 names the milestone the action advances. | **yes** — owner, 2026-09-16: "window function is the logical step before decorating it" (a prerequisite-order argument; see F2 follow-on). | | 2 | Urgent-unrelated wins | Same plan. One due item: `decorators`/python, base 100 (an overdue spaced-repetition review). Nothing represents milestone 0. | **`decorators`** (118, no refs); alternate `window function` (60, `source=study_plan:sql-windows:0`, `plan_refs=[(sql-windows, 0)]`, reason "Next milestone 1/1 of plan 'SQL Windows': Window basics"). | Rule 5: the unrelated candidate is in a more urgent class (due review) and wins outright — the bias cannot lift new-milestone work over it. Rule 6: the plan's next milestone was unrepresented, so it was synthesised at base 48 + bias 12 = 60 and appears as the plan-backed alternate (rule 8 satisfied without any swap). | **yes** — owner, 2026-09-16: clear the overdue review first. Note for follow-on: an overdue item *unrelated* to the plan must not sit as an alternate indefinitely — propose it explicitly (age-aware nudge) or let the learner retire it. | | 3 | Energy-deferred | Plan `sql-windows` with `energy_floor: 5`; milestone 0 `Window basics` **done** (concepts `[window function]`), milestone 1 `Frames` (concepts `[window frame]`). One struggle-repair item `window function`/sql, `hands-on`, base 82. **Energy `low`** (capability 3/10). | **`window function`** (hands-on, score 80, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, milestone 1, floor 5, capability 3)]`; JSON gains `energy_deferred`. | Rule 3: capability 3 < floor 5, so the *new* milestone (Frames) is deferred and named, not synthesised; the plan-related repair on a finished milestone's concept stays eligible and keeps its ref (`None`: plan-related repair, not the next milestone). Score = 82 + 12 bias − 14 (hands-on at low energy). | **no** — owner, 2026-09-16: a struggle-repair task has no energy demand of its own; recommending hands-on repair of a *live* struggle on a low-energy day risks compounding the struggle and damaging confidence (RSD). Finding for council: (1) derive a per-item energy demand for repair from struggle recency/teach-back — at low energy a live struggle defers like new work, a recovered one stays eligible as gentle review; (2) when nothing plan-related fits the day's capability, synthesise a body-doubling / open-session candidate (feature exists: ADR-0001/0003, `web/routes/body_double.py`) naming the deferred items, instead of the least-bad task. | -| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered); added after council review 7: **(d)** **no plan at all**, one live struggle `decorators`/python, low energy; **(e)** row 3's world plus one **unrelated hands-on** practice task `list comprehension drill`/python base 48, low energy; added after the rubric-3b council (2026-09-20, two seats independently named it): **(f)** row 3's world where `window function` is **both** the live struggle and **due for recall** (`study_progress` due row, base 100) — and the same with **no plan** (`decorators` live struggle + due recall). | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason (as emitted from `c519bc2a`, after the rubric-3b council) *"1 of 2 milestones of SQL Windows done. Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session — you drive; the companion stays quiet unless you ask."* (before the council, from `326abcf9`: *"Nothing plan-related fits low energy today — deferred: …; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."* — the progress lead is the framework's naming rule; the door is now described by what the co-study persona guarantees); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone. **(d)** primary **`one tiny recall loop`** (the starter, 28, reason *"Today's energy deferred the repair work it cannot carry; start with one small retrieval signal instead"*); no alternates; `energy_deferred_repairs=[(None, decorators, high, 6)]` — where before item 5 the primary was the hands-on repair of `decorators`. **(e)** primary **`Sit with SQL Windows`** (42, reason as in (a)); the hands-on drill is the alternate at 34 (48 − 14 low-energy penalty). **(f)** (emitted from `c519bc2a`) primary **`window function`** (**recall**, 130 = 100 + 30 overdue, `plan_refs=[(sql-windows, None)]`); no alternates (nothing plan-related fits, so no body double — a real candidate exists); `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** — the same concept is the primary *and* a deferred repair, because due recall is never deferred. No-plan variant: primary **`decorators`** (recall, 118, no refs); no alternates; `energy_deferred_repairs=[(None, decorators, struggling, high, 6, 3)]`. | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 — below every real candidate at base; at low energy it sits above only a hands-on task the energy rule already penalises (48 − 14 = 34), review 7 F2 — a proposal, not a filter) carrying the co-study session door. | **Owner verdicts — interactive walkthrough 2026-09-20 (in progress).** **(a) yes** — owner: doing the stuck thing on a genuinely low day "is very rare and only when there was a time sensitive deliverable"; that is the override case (the repair stays reachable by hand, `studyloop study "window function"`), not the default, so the floor would have given the ordinary low day a non-zero start. Row 3's original "no" (2026-09-16) stands. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** the sit-with session must not be a blank page — the body double should propose one tiny, concrete first move on the deferred material (e.g. "open the Frames lesson and read it, nothing more", or ten minutes of passive video on the next milestone). A goal-less session is the hardest ADHD start; sharpening the replacement beats restoring the rejected repair. **(b) yes** — owner: at 3/10 "I would be reluctant to do more than familiar recall"; the recall leads, the sit-with sits beneath it. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** "once I had done this, if it went well I would like the option to then take on more but not lead with the larger task" — the low-energy day should be a ladder, not a snapshot: after a completed familiar recall, offer a step-up (a re-ask of energy, or the next-cheapest action — the sit-with, then one tiny concrete move on the deferred material, per the (a) note), never leading with the larger task. Verified against `c519bc2a`: nothing does this today — the Today card fetches `/api/now` once at init with no energy and never re-plans after an action (`today-panel.js` `init()`), and CLI `now` is a one-shot snapshot of the energy it was given. The (a) and (b) notes are one 0.6.0 theme. **(c) yes, with a fallback** — owner: "3/10 or 4/10 is very subjective, I would offer the teach back but as I personally believe I would be reluctant to do more than familiar recall at 3/10 … offer a fallback to a guided explanation". So: demand `low` stands and the teach-back is offered at low energy, **and the door must carry a fallback** — if the learner cannot produce the explanation, the mentor moves to a guided explanation, not four Socratic rounds first. Against `c519bc2a`: the general stuck ladder exists (`socratic-engine.md` "Stuck (Escalating Support)" reaches a worked example only at round 4; `co-study.md` offers a brief explanation after two exchanges) but `teach-back-protocol.md` has no "cannot produce → guided explanation" step and neither rule is conditioned on low energy. **Three fixes landed on this branch before the row is committed (one RED/GREEN each), each found while checking the door behind the yes:** (1) the teach-back door's evidence command named `--type structured` — the protocol's 15-minute five-dimension review — while demand `low` was justified by the one-sentence *micro* teach-back (`micro` is in `TEACHBACK_TYPES`); the door now names the form the demand class stands on (`micro` for `low`; `structured` for a weak-teach-back `medium` row, which re-sits the review it fell short on) — RED `0760254d` / GREEN `f2cb572f`. (2) the collector's reason for a `learning` row read "Recorded as learning; repair now while the signal is fresh" (`decision.py`); design §5's own words for a `learning` row are "recovered / gentle review → `low`" and "recovered repair stays eligible as gentle review" (the spec delta: "gentle repair"), and on a mixed low-energy screen "repair now" contradicts the deferred-repair line beside it; it now reads "Recorded as learning; a gentle review keeps it fresh — one sentence, in your own words" (the struggling row's sentence is untouched) — same RED/GREEN. (3) the fallback the yes is conditional on: `teach-back-protocol.md`'s Micro Teach-Back section now carries a **Low-energy fallback** — cannot produce the sentence → a guided explanation (two or three plain sentences with a networking analogy, then one phrase said back in the student's own words), not the four-round Stuck ladder; the blank is not scored as a teach-back (it measures the day, not the concept); pinned by `test_teach_back_protocol_carries_the_low_energy_guided_explanation_fallback` — RED `c7cc1b0f` / GREEN `214df377` (`agents/manifest.json`: only the moved entry; `.secrets.baseline` rescanned whole-repo with the hook's pinned v1.5.0, 72 → 72 files). The no-plan golden stays byte-identical (`ec451ce8`); `test_now_plan_guidance` + `test_learning_decision` 76/76 before the fallback pin, 77 with it. **Note (0.6.0, not a blocker):** the capability numbers are subjective (3/10 vs 4/10); the demand thresholds are coarse by construction, so the fallback carries more weight than the threshold. **(d) yes — the starter, with a step-up** — owner, shown the starter as built (its topic is the first configured topic, not `decorators`; its one command is `studyloop progress "one tiny recall loop" …` and opens nothing): "I would" take the primary as emitted — the generic starter plus the deferred line — not the hands-on repair; and "if this went well, a successful task often increases confidence and energy so give the user the opportunity to work on the live struggle." So: the deferral stands with no plan (no council seat would restore the repair either); the generic starter is a floor the owner would start; the same-concept gentle recall (design §5 Q3) that the council and the coordinator steered toward is recorded as provenance, not adopted. **Note recorded with the yes (a 0.6.0 finding, not a blocker) — it names the top rung of the (a)/(b) ladder:** after a completed floor task, re-offer the *deferred repair itself* as the step-up — an option the learner takes while the win is fresh, never restored as the default. Against `c519bc2a`: nothing can act on "went well" today — the starter emits no outcome the engine reads, and nothing re-plans after an action (the (b) gap) — so the 0.6.0 item needs both an outcome signal from the floor task and the re-offer keyed on it. **PENDING** — (e) at low energy, would you sit with the plan rather than do an unrelated hands-on drill (review 7 F2: the proposal outranks a task the low-energy rule penalises, and nothing else)? (f) when the concept you are failing is also due for recall, do you want that recall as the primary (spacing wins; the repair stays deferred beside it), or should a due recall on a live struggle inherit the repair's demand and defer with it? **Council decision brief (2026-09-20, `receipts/rubric-3b-decision-brief-2026-09-20.md`, three seats asked for recommendations to the owner, not verdicts):** (a) YES ×3; (b) YES ×3; (c) YES / YES / EITHER (astra: `learning` is a database state, not proof of recovery — keep the teach-back brief and optional); (d) YES / EITHER / EITHER — no seat would restore the hands-on repair; the open call is whether *you* would tap a generic starter, or want a same-concept gentle recall (Q3), or want the system out of the way; (e) YES / YES / EITHER (astra: an unrelated *familiar* drill is not the class of work the finding objected to); (f) not put to the seats — emitted after they named it. The verdict column is yours; the recommendations are provenance for it, not a substitute. | +| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered); added after council review 7: **(d)** **no plan at all**, one live struggle `decorators`/python, low energy; **(e)** row 3's world plus one **unrelated hands-on** practice task `list comprehension drill`/python base 48, low energy; added after the rubric-3b council (2026-09-20, two seats independently named it): **(f)** row 3's world where `window function` is **both** the live struggle and **due for recall** (`study_progress` due row, base 100) — and the same with **no plan** (`decorators` live struggle + due recall). | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason (as emitted from `c519bc2a`, after the rubric-3b council) *"1 of 2 milestones of SQL Windows done. Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session — you drive; the companion stays quiet unless you ask."* (before the council, from `326abcf9`: *"Nothing plan-related fits low energy today — deferred: …; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."* — the progress lead is the framework's naming rule; the door is now described by what the co-study persona guarantees); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone — *as emitted with the due-progress collector silenced*. Re-emitted through **both** real collectors (`fb63fb6b`, see (f)): the same `learning` row is also due, so the primary is **`window function`** (**recall**, `source=study_progress`, reason *"10-min Socratic review; last seen N day(s) ago; confidence is learning"*, door `studyloop progress …`) and the micro teach-back (teachback, 100, door `--type micro` after `f2cb572f`) is the alternate beneath it; still nothing deferred but the milestone. **(d)** primary **`one tiny recall loop`** (the starter, 28, reason *"Today's energy deferred the repair work it cannot carry; start with one small retrieval signal instead"*); no alternates; `energy_deferred_repairs=[(None, decorators, high, 6)]` — where before item 5 the primary was the hands-on repair of `decorators`. **(e)** primary **`Sit with SQL Windows`** (42, reason as in (a)); the hands-on drill is the alternate at 34 (48 − 14 low-energy penalty). **(f)** (emitted from `c519bc2a` with a *planted* recall-typed due item) primary **`window function`** (**recall**, 130 = 100 + 30 overdue, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** — the same concept as primary *and* deferred repair, "because due recall is never deferred". No-plan variant: primary **`decorators`** (recall, 118, no refs). **That world cannot occur.** Emitted 2026-09-20 through **both real collectors** (owner walkthrough): the due-progress collector reads the *same* `observations.rows` as the struggle collector, labels every `struggling` row due (*"Guided repair + tiny practice"*) and emits it **hands-on**, not recall — the repair collected twice. On `c519bc2a`–`b1612918` that copy carried no demand, survived the deferral and won the dedupe, so row 3's own world emitted primary **`window function`** (**hands-on**, 140, *"Guided repair + tiny practice; last seen N day(s) ago; confidence is struggling"*) with *"repairing 'window function' … deferred"* printed beneath it and **no body double** — the row-3 "no", still on top; the no-plan world likewise (`decorators`, hands-on, 128, over the starter). After `fb63fb6b` the due copy carries the repair's demand and defers with it, named once: row 3's world emits the reading (a) screen exactly (`Sit with SQL Windows`, 42; `energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`), the no-plan world the reading (d) screen, and at **medium** energy the due copy is primary as before (hands-on, 154; nothing deferred). | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 — below every real candidate at base; at low energy it sits above only a hands-on task the energy rule already penalises (48 − 14 = 34), review 7 F2 — a proposal, not a filter) carrying the co-study session door. | **Owner verdicts — interactive walkthrough 2026-09-20: (a)–(e) yes, (f) resolved by a fix.** **(a) yes** — owner: doing the stuck thing on a genuinely low day "is very rare and only when there was a time sensitive deliverable"; that is the override case (the repair stays reachable by hand, `studyloop study "window function"`), not the default, so the floor would have given the ordinary low day a non-zero start. Row 3's original "no" (2026-09-16) stands. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** the sit-with session must not be a blank page — the body double should propose one tiny, concrete first move on the deferred material (e.g. "open the Frames lesson and read it, nothing more", or ten minutes of passive video on the next milestone). A goal-less session is the hardest ADHD start; sharpening the replacement beats restoring the rejected repair. **(b) yes** — owner: at 3/10 "I would be reluctant to do more than familiar recall"; the recall leads, the sit-with sits beneath it. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** "once I had done this, if it went well I would like the option to then take on more but not lead with the larger task" — the low-energy day should be a ladder, not a snapshot: after a completed familiar recall, offer a step-up (a re-ask of energy, or the next-cheapest action — the sit-with, then one tiny concrete move on the deferred material, per the (a) note), never leading with the larger task. Verified against `c519bc2a`: nothing does this today — the Today card fetches `/api/now` once at init with no energy and never re-plans after an action (`today-panel.js` `init()`), and CLI `now` is a one-shot snapshot of the energy it was given. The (a) and (b) notes are one 0.6.0 theme. **(c) yes, with a fallback** — owner: "3/10 or 4/10 is very subjective, I would offer the teach back but as I personally believe I would be reluctant to do more than familiar recall at 3/10 … offer a fallback to a guided explanation". So: demand `low` stands and the teach-back is offered at low energy, **and the door must carry a fallback** — if the learner cannot produce the explanation, the mentor moves to a guided explanation, not four Socratic rounds first. Against `c519bc2a`: the general stuck ladder exists (`socratic-engine.md` "Stuck (Escalating Support)" reaches a worked example only at round 4; `co-study.md` offers a brief explanation after two exchanges) but `teach-back-protocol.md` has no "cannot produce → guided explanation" step and neither rule is conditioned on low energy. **Three fixes landed on this branch before the row is committed (one RED/GREEN each), each found while checking the door behind the yes:** (1) the teach-back door's evidence command named `--type structured` — the protocol's 15-minute five-dimension review — while demand `low` was justified by the one-sentence *micro* teach-back (`micro` is in `TEACHBACK_TYPES`); the door now names the form the demand class stands on (`micro` for `low`; `structured` for a weak-teach-back `medium` row, which re-sits the review it fell short on) — RED `0760254d` / GREEN `f2cb572f`. (2) the collector's reason for a `learning` row read "Recorded as learning; repair now while the signal is fresh" (`decision.py`); design §5's own words for a `learning` row are "recovered / gentle review → `low`" and "recovered repair stays eligible as gentle review" (the spec delta: "gentle repair"), and on a mixed low-energy screen "repair now" contradicts the deferred-repair line beside it; it now reads "Recorded as learning; a gentle review keeps it fresh — one sentence, in your own words" (the struggling row's sentence is untouched) — same RED/GREEN. (3) the fallback the yes is conditional on: `teach-back-protocol.md`'s Micro Teach-Back section now carries a **Low-energy fallback** — cannot produce the sentence → a guided explanation (two or three plain sentences with a networking analogy, then one phrase said back in the student's own words), not the four-round Stuck ladder; the blank is not scored as a teach-back (it measures the day, not the concept); pinned by `test_teach_back_protocol_carries_the_low_energy_guided_explanation_fallback` — RED `c7cc1b0f` / GREEN `214df377` (`agents/manifest.json`: only the moved entry; `.secrets.baseline` rescanned whole-repo with the hook's pinned v1.5.0, 72 → 72 files). The no-plan golden stays byte-identical (`ec451ce8`); `test_now_plan_guidance` + `test_learning_decision` 76/76 before the fallback pin, 77 with it. **Note (0.6.0, not a blocker):** the capability numbers are subjective (3/10 vs 4/10); the demand thresholds are coarse by construction, so the fallback carries more weight than the threshold. **(d) yes — the starter, with a step-up** — owner, shown the starter as built (its topic is the first configured topic, not `decorators`; its one command is `studyloop progress "one tiny recall loop" …` and opens nothing): "I would" take the primary as emitted — the generic starter plus the deferred line — not the hands-on repair; and "if this went well, a successful task often increases confidence and energy so give the user the opportunity to work on the live struggle." So: the deferral stands with no plan (no council seat would restore the repair either); the generic starter is a floor the owner would start; the same-concept gentle recall (design §5 Q3) that the council and the coordinator steered toward is recorded as provenance, not adopted. **Note recorded with the yes (a 0.6.0 finding, not a blocker) — it names the top rung of the (a)/(b) ladder:** after a completed floor task, re-offer the *deferred repair itself* as the step-up — an option the learner takes while the win is fresh, never restored as the default. Against `c519bc2a`: nothing can act on "went well" today — the starter emits no outcome the engine reads, and nothing re-plans after an action (the (b) gap) — so the 0.6.0 item needs both an outcome signal from the floor task and the re-offer keyed on it. **(e) yes** — owner, 2026-09-20: sit with the plan rather than do an unrelated hands-on drill at low energy; consistent with (b) ("reluctant to do more than familiar recall at 3/10" — a coding drill is more than familiar recall), and the drill stays one tap below as the alternate (a proposal, not a filter). The engine has no signal separating a *familiar* drill from a new one (astra's caveat), so the safe default is the sit-with; per the (a)/(b) ladder the drill is the step-up to offer after the sit-with went well, not what to lead with. **(f) resolved by a fix, not a verdict** — the question as posed ("recall as primary, or a due recall on a live struggle inherits the repair's demand?") described a world the real collectors never produce: a `struggling` row's due item is the guided repair (hands-on), i.e. the same repair collected twice, and until `fb63fb6b` that copy carried no demand — so against a real sessions.db the (a), (d) and (e) primaries the owner scored were never emitted; the hands-on repair of the live struggle was, with its own deferral line beneath it (finding recorded as design §5 decision 10, RED `8f851d3a` / GREEN `fb63fb6b`; the fixture that hid it, `_plant_struggles`, silences the due collector, and `_plant_struggles_for_both_collectors` now exists beside it). The due copy now carries the repair's demand and defers with it, named once; due recall and teach-back rows are what "never deferred" was always true of. **Every reading re-emitted through both real collectors on `fb63fb6b`:** (a), (b), (d), (e) unchanged from the screens scored above; (c) differs — the due recall leads and the micro teach-back is the alternate (recorded in the emitted column) — a screen *closer* to the owner's (b)/(c) words than the one scored, so the yes is read as standing unless the owner says otherwise; (f) is the (a) screen at low energy and the pre-item-5 guided-repair primary at medium. **Council decision brief (2026-09-20, `receipts/rubric-3b-decision-brief-2026-09-20.md`, three seats asked for recommendations to the owner, not verdicts):** (a) YES ×3; (b) YES ×3; (c) YES / YES / EITHER (astra: `learning` is a database state, not proof of recovery — keep the teach-back brief and optional); (d) YES / EITHER / EITHER — no seat would restore the hands-on repair; the open call is whether *you* would tap a generic starter, or want a same-concept gentle recall (Q3), or want the system out of the way; (e) YES / YES / EITHER (astra: an unrelated *familiar* drill is not the class of work the finding objected to); (f) not put to the seats — emitted after they named it. The verdict column is yours; the recommendations are provenance for it, not a substitute. | | 4 | Fully-checked | Plan `done-plan` ("Done Plan"), milestones A and B both done. One due item `decorators`/python base 100. | **`decorators`** (118, no refs); `completion_actions=[(done-plan, "Every milestone of 'Done Plan' is checked off — close the plan or extend it with a follow-on mission.")]`; no `study_plan:` candidate anywhere; JSON gains `active_plans` + `completion_actions`. | Rule 9: a fully-checked plan is reported as a completion action and is neither matched (no bias, no refs) nor synthesised. | **primary yes / completion action no as phrased** — owner, 2026-09-16: the completion action must be contextual and consensual. Run the end assessment (`assess(phase="end")`: due reviews, struggles, unverified milestones on the plan's concepts). If outstanding work touches the plan's concepts (or their prerequisites — F2 concept edges), propose *extend* and name the evidence; if clean, propose *close* and ask the learner to agree ("anything you are not comfortable with?"). Status never changes automatically (#7). Natural vehicle: architect with `purpose=planning` and the assessment in the brief (`plan close <id>`, sibling of `plan repair <id>`). Finding for council. | | 4b | Fully-checked — **re-run after D-G (item 4, 2026-09-18)** | Row 4's fixture (`done-plan`, milestones A `[alpha]` and B `[beta]` both done; one due item `decorators`/python base 100), plus the end assessment's readers planted: **(a)** one due review on plan concept `alpha` (`overdue`) with session mentions backing both concepts; **(b)** no due rows, same mentions. | Primary unchanged in both: **`decorators`** (118, no refs); no `study_plan:` candidate. **(a)** `completion_actions=[(done-plan, due 1 / struggles 0 / unverified 0, proposal **extend**, evidence `["Due review: alpha — overdue"]`)]`, sentence: "Every milestone of 'Done Plan' is checked off, and the closing review proposes extending the plan — 1 due review, 0 struggles and 0 unverified milestones on its concepts. Walk the evidence with the architect: studyloop plan close done-plan." **(b)** counts 0/0/0, proposal **close**, evidence `[]`, sentence: "Every milestone of 'Done Plan' is checked off and the closing review is clean — it proposes closing the plan. Close it with the architect when you agree: studyloop plan close done-plan." No warnings; JSON gains the five keys only inside the entry. | Rule 9 as before for the ranking. The completion action is now the end assessment read as a preview (`assess(phase="end", record=False)`, one call, no write, no status change): `extend` iff any of the three counts on the plan's own concepts is above zero, else `close`; due counts only rows naming a concept (the scheduler's "new topic" row is excluded — owner decision 2026-09-17). `plan close done-plan` launches the architect with the same review as the brief's first section; status moves only when the learner agrees. | **yes / yes** — owner, 2026-09-18, answering the two questions as posed: (a) **yes**, a proposal the owner would walk; (b) **yes**, a close the owner would agree to. No further line given. Closes row 4's "no as phrased" finding; status still moves only when the learner agrees in the architect conversation. | | 5 | No-plan identical | No plan documents at all; no collector candidates. | **`one tiny recall loop`** (python, recall, 28, `source=starter`) — the starter. | D-5: `serialise(plan) == golden` → **byte-identical** (`True` in the run); the JSON key list is exactly the golden's — no additive key is present. | **verified** — owner walkthrough 2026-09-16: nothing to judge; the byte-identical golden is the acceptance. | From a4503a485b28fd4c5c490631d79fdbd7bcc11189 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 15:21:50 +0100 Subject: [PATCH 29/32] docs(rubric-3b): (c) confirmed by the owner on the both-collectors screen; T5.5 ticked MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Owner, 2026-09-20, on the (c) screen as shipped (due recall first, micro teach-back beneath): "With an energy score of 3/10, the initial task should be light to give confidence and hopefully by doing so energy, if the task is successful then offer the user the learning task which has a higher energy score." Recorded verbatim; it names the same ladder as the (b)/(d) notes (one 0.6.0 theme) and adds one observation for that item: the owner reads a teach-back as higher demand than a plain recall although both are `low` in the demand classes — the ordering already agrees, the classes govern deferral not order. Row 3b is fully scored: T5.5 ticked with the closure summary; the change is archived after PR #28 merges. --- .../receipts/now-rubric-2026-09-16.md | 2 +- openspec/changes/plan-integration-followons/tasks.md | 12 +++++++++++- 2 files changed, 12 insertions(+), 2 deletions(-) diff --git a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md index eaff860e4..a54d5e5a8 100644 --- a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md +++ b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md @@ -30,7 +30,7 @@ learning"). | 1 | Matching due | Active plan `sql-windows` ("SQL Windows", topics `[sql]`, next milestone 0 concepts `[window function]`). Due items: `decorators`/python base 102, `window function`/sql base 100. | **`window function`** (sql, recall, score 130, `plan_refs=[(sql-windows, 0)]`); alternate `decorators` (120, no refs). | Rule 5: both are due items two points apart — one urgency class — so the plan-related one takes the +12 bias and wins; the unrelated due item is *kept* as an alternate (bias, not filter). Rule 7 names the milestone the action advances. | **yes** — owner, 2026-09-16: "window function is the logical step before decorating it" (a prerequisite-order argument; see F2 follow-on). | | 2 | Urgent-unrelated wins | Same plan. One due item: `decorators`/python, base 100 (an overdue spaced-repetition review). Nothing represents milestone 0. | **`decorators`** (118, no refs); alternate `window function` (60, `source=study_plan:sql-windows:0`, `plan_refs=[(sql-windows, 0)]`, reason "Next milestone 1/1 of plan 'SQL Windows': Window basics"). | Rule 5: the unrelated candidate is in a more urgent class (due review) and wins outright — the bias cannot lift new-milestone work over it. Rule 6: the plan's next milestone was unrepresented, so it was synthesised at base 48 + bias 12 = 60 and appears as the plan-backed alternate (rule 8 satisfied without any swap). | **yes** — owner, 2026-09-16: clear the overdue review first. Note for follow-on: an overdue item *unrelated* to the plan must not sit as an alternate indefinitely — propose it explicitly (age-aware nudge) or let the learner retire it. | | 3 | Energy-deferred | Plan `sql-windows` with `energy_floor: 5`; milestone 0 `Window basics` **done** (concepts `[window function]`), milestone 1 `Frames` (concepts `[window frame]`). One struggle-repair item `window function`/sql, `hands-on`, base 82. **Energy `low`** (capability 3/10). | **`window function`** (hands-on, score 80, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, milestone 1, floor 5, capability 3)]`; JSON gains `energy_deferred`. | Rule 3: capability 3 < floor 5, so the *new* milestone (Frames) is deferred and named, not synthesised; the plan-related repair on a finished milestone's concept stays eligible and keeps its ref (`None`: plan-related repair, not the next milestone). Score = 82 + 12 bias − 14 (hands-on at low energy). | **no** — owner, 2026-09-16: a struggle-repair task has no energy demand of its own; recommending hands-on repair of a *live* struggle on a low-energy day risks compounding the struggle and damaging confidence (RSD). Finding for council: (1) derive a per-item energy demand for repair from struggle recency/teach-back — at low energy a live struggle defers like new work, a recovered one stays eligible as gentle review; (2) when nothing plan-related fits the day's capability, synthesise a body-doubling / open-session candidate (feature exists: ADR-0001/0003, `web/routes/body_double.py`) naming the deferred items, instead of the least-bad task. | -| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered); added after council review 7: **(d)** **no plan at all**, one live struggle `decorators`/python, low energy; **(e)** row 3's world plus one **unrelated hands-on** practice task `list comprehension drill`/python base 48, low energy; added after the rubric-3b council (2026-09-20, two seats independently named it): **(f)** row 3's world where `window function` is **both** the live struggle and **due for recall** (`study_progress` due row, base 100) — and the same with **no plan** (`decorators` live struggle + due recall). | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason (as emitted from `c519bc2a`, after the rubric-3b council) *"1 of 2 milestones of SQL Windows done. Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session — you drive; the companion stays quiet unless you ask."* (before the council, from `326abcf9`: *"Nothing plan-related fits low energy today — deferred: …; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."* — the progress lead is the framework's naming rule; the door is now described by what the co-study persona guarantees); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone — *as emitted with the due-progress collector silenced*. Re-emitted through **both** real collectors (`fb63fb6b`, see (f)): the same `learning` row is also due, so the primary is **`window function`** (**recall**, `source=study_progress`, reason *"10-min Socratic review; last seen N day(s) ago; confidence is learning"*, door `studyloop progress …`) and the micro teach-back (teachback, 100, door `--type micro` after `f2cb572f`) is the alternate beneath it; still nothing deferred but the milestone. **(d)** primary **`one tiny recall loop`** (the starter, 28, reason *"Today's energy deferred the repair work it cannot carry; start with one small retrieval signal instead"*); no alternates; `energy_deferred_repairs=[(None, decorators, high, 6)]` — where before item 5 the primary was the hands-on repair of `decorators`. **(e)** primary **`Sit with SQL Windows`** (42, reason as in (a)); the hands-on drill is the alternate at 34 (48 − 14 low-energy penalty). **(f)** (emitted from `c519bc2a` with a *planted* recall-typed due item) primary **`window function`** (**recall**, 130 = 100 + 30 overdue, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** — the same concept as primary *and* deferred repair, "because due recall is never deferred". No-plan variant: primary **`decorators`** (recall, 118, no refs). **That world cannot occur.** Emitted 2026-09-20 through **both real collectors** (owner walkthrough): the due-progress collector reads the *same* `observations.rows` as the struggle collector, labels every `struggling` row due (*"Guided repair + tiny practice"*) and emits it **hands-on**, not recall — the repair collected twice. On `c519bc2a`–`b1612918` that copy carried no demand, survived the deferral and won the dedupe, so row 3's own world emitted primary **`window function`** (**hands-on**, 140, *"Guided repair + tiny practice; last seen N day(s) ago; confidence is struggling"*) with *"repairing 'window function' … deferred"* printed beneath it and **no body double** — the row-3 "no", still on top; the no-plan world likewise (`decorators`, hands-on, 128, over the starter). After `fb63fb6b` the due copy carries the repair's demand and defers with it, named once: row 3's world emits the reading (a) screen exactly (`Sit with SQL Windows`, 42; `energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`), the no-plan world the reading (d) screen, and at **medium** energy the due copy is primary as before (hands-on, 154; nothing deferred). | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 — below every real candidate at base; at low energy it sits above only a hands-on task the energy rule already penalises (48 − 14 = 34), review 7 F2 — a proposal, not a filter) carrying the co-study session door. | **Owner verdicts — interactive walkthrough 2026-09-20: (a)–(e) yes, (f) resolved by a fix.** **(a) yes** — owner: doing the stuck thing on a genuinely low day "is very rare and only when there was a time sensitive deliverable"; that is the override case (the repair stays reachable by hand, `studyloop study "window function"`), not the default, so the floor would have given the ordinary low day a non-zero start. Row 3's original "no" (2026-09-16) stands. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** the sit-with session must not be a blank page — the body double should propose one tiny, concrete first move on the deferred material (e.g. "open the Frames lesson and read it, nothing more", or ten minutes of passive video on the next milestone). A goal-less session is the hardest ADHD start; sharpening the replacement beats restoring the rejected repair. **(b) yes** — owner: at 3/10 "I would be reluctant to do more than familiar recall"; the recall leads, the sit-with sits beneath it. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** "once I had done this, if it went well I would like the option to then take on more but not lead with the larger task" — the low-energy day should be a ladder, not a snapshot: after a completed familiar recall, offer a step-up (a re-ask of energy, or the next-cheapest action — the sit-with, then one tiny concrete move on the deferred material, per the (a) note), never leading with the larger task. Verified against `c519bc2a`: nothing does this today — the Today card fetches `/api/now` once at init with no energy and never re-plans after an action (`today-panel.js` `init()`), and CLI `now` is a one-shot snapshot of the energy it was given. The (a) and (b) notes are one 0.6.0 theme. **(c) yes, with a fallback** — owner: "3/10 or 4/10 is very subjective, I would offer the teach back but as I personally believe I would be reluctant to do more than familiar recall at 3/10 … offer a fallback to a guided explanation". So: demand `low` stands and the teach-back is offered at low energy, **and the door must carry a fallback** — if the learner cannot produce the explanation, the mentor moves to a guided explanation, not four Socratic rounds first. Against `c519bc2a`: the general stuck ladder exists (`socratic-engine.md` "Stuck (Escalating Support)" reaches a worked example only at round 4; `co-study.md` offers a brief explanation after two exchanges) but `teach-back-protocol.md` has no "cannot produce → guided explanation" step and neither rule is conditioned on low energy. **Three fixes landed on this branch before the row is committed (one RED/GREEN each), each found while checking the door behind the yes:** (1) the teach-back door's evidence command named `--type structured` — the protocol's 15-minute five-dimension review — while demand `low` was justified by the one-sentence *micro* teach-back (`micro` is in `TEACHBACK_TYPES`); the door now names the form the demand class stands on (`micro` for `low`; `structured` for a weak-teach-back `medium` row, which re-sits the review it fell short on) — RED `0760254d` / GREEN `f2cb572f`. (2) the collector's reason for a `learning` row read "Recorded as learning; repair now while the signal is fresh" (`decision.py`); design §5's own words for a `learning` row are "recovered / gentle review → `low`" and "recovered repair stays eligible as gentle review" (the spec delta: "gentle repair"), and on a mixed low-energy screen "repair now" contradicts the deferred-repair line beside it; it now reads "Recorded as learning; a gentle review keeps it fresh — one sentence, in your own words" (the struggling row's sentence is untouched) — same RED/GREEN. (3) the fallback the yes is conditional on: `teach-back-protocol.md`'s Micro Teach-Back section now carries a **Low-energy fallback** — cannot produce the sentence → a guided explanation (two or three plain sentences with a networking analogy, then one phrase said back in the student's own words), not the four-round Stuck ladder; the blank is not scored as a teach-back (it measures the day, not the concept); pinned by `test_teach_back_protocol_carries_the_low_energy_guided_explanation_fallback` — RED `c7cc1b0f` / GREEN `214df377` (`agents/manifest.json`: only the moved entry; `.secrets.baseline` rescanned whole-repo with the hook's pinned v1.5.0, 72 → 72 files). The no-plan golden stays byte-identical (`ec451ce8`); `test_now_plan_guidance` + `test_learning_decision` 76/76 before the fallback pin, 77 with it. **Note (0.6.0, not a blocker):** the capability numbers are subjective (3/10 vs 4/10); the demand thresholds are coarse by construction, so the fallback carries more weight than the threshold. **(d) yes — the starter, with a step-up** — owner, shown the starter as built (its topic is the first configured topic, not `decorators`; its one command is `studyloop progress "one tiny recall loop" …` and opens nothing): "I would" take the primary as emitted — the generic starter plus the deferred line — not the hands-on repair; and "if this went well, a successful task often increases confidence and energy so give the user the opportunity to work on the live struggle." So: the deferral stands with no plan (no council seat would restore the repair either); the generic starter is a floor the owner would start; the same-concept gentle recall (design §5 Q3) that the council and the coordinator steered toward is recorded as provenance, not adopted. **Note recorded with the yes (a 0.6.0 finding, not a blocker) — it names the top rung of the (a)/(b) ladder:** after a completed floor task, re-offer the *deferred repair itself* as the step-up — an option the learner takes while the win is fresh, never restored as the default. Against `c519bc2a`: nothing can act on "went well" today — the starter emits no outcome the engine reads, and nothing re-plans after an action (the (b) gap) — so the 0.6.0 item needs both an outcome signal from the floor task and the re-offer keyed on it. **(e) yes** — owner, 2026-09-20: sit with the plan rather than do an unrelated hands-on drill at low energy; consistent with (b) ("reluctant to do more than familiar recall at 3/10" — a coding drill is more than familiar recall), and the drill stays one tap below as the alternate (a proposal, not a filter). The engine has no signal separating a *familiar* drill from a new one (astra's caveat), so the safe default is the sit-with; per the (a)/(b) ladder the drill is the step-up to offer after the sit-with went well, not what to lead with. **(f) resolved by a fix, not a verdict** — the question as posed ("recall as primary, or a due recall on a live struggle inherits the repair's demand?") described a world the real collectors never produce: a `struggling` row's due item is the guided repair (hands-on), i.e. the same repair collected twice, and until `fb63fb6b` that copy carried no demand — so against a real sessions.db the (a), (d) and (e) primaries the owner scored were never emitted; the hands-on repair of the live struggle was, with its own deferral line beneath it (finding recorded as design §5 decision 10, RED `8f851d3a` / GREEN `fb63fb6b`; the fixture that hid it, `_plant_struggles`, silences the due collector, and `_plant_struggles_for_both_collectors` now exists beside it). The due copy now carries the repair's demand and defers with it, named once; due recall and teach-back rows are what "never deferred" was always true of. **Every reading re-emitted through both real collectors on `fb63fb6b`:** (a), (b), (d), (e) unchanged from the screens scored above; (c) differs — the due recall leads and the micro teach-back is the alternate (recorded in the emitted column) — a screen *closer* to the owner's (b)/(c) words than the one scored, so the yes is read as standing unless the owner says otherwise; (f) is the (a) screen at low energy and the pre-item-5 guided-repair primary at medium. **Council decision brief (2026-09-20, `receipts/rubric-3b-decision-brief-2026-09-20.md`, three seats asked for recommendations to the owner, not verdicts):** (a) YES ×3; (b) YES ×3; (c) YES / YES / EITHER (astra: `learning` is a database state, not proof of recovery — keep the teach-back brief and optional); (d) YES / EITHER / EITHER — no seat would restore the hands-on repair; the open call is whether *you* would tap a generic starter, or want a same-concept gentle recall (Q3), or want the system out of the way; (e) YES / YES / EITHER (astra: an unrelated *familiar* drill is not the class of work the finding objected to); (f) not put to the seats — emitted after they named it. The verdict column is yours; the recommendations are provenance for it, not a substitute. | +| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered); added after council review 7: **(d)** **no plan at all**, one live struggle `decorators`/python, low energy; **(e)** row 3's world plus one **unrelated hands-on** practice task `list comprehension drill`/python base 48, low energy; added after the rubric-3b council (2026-09-20, two seats independently named it): **(f)** row 3's world where `window function` is **both** the live struggle and **due for recall** (`study_progress` due row, base 100) — and the same with **no plan** (`decorators` live struggle + due recall). | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason (as emitted from `c519bc2a`, after the rubric-3b council) *"1 of 2 milestones of SQL Windows done. Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session — you drive; the companion stays quiet unless you ask."* (before the council, from `326abcf9`: *"Nothing plan-related fits low energy today — deferred: …; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."* — the progress lead is the framework's naming rule; the door is now described by what the co-study persona guarantees); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone — *as emitted with the due-progress collector silenced*. Re-emitted through **both** real collectors (`fb63fb6b`, see (f)): the same `learning` row is also due, so the primary is **`window function`** (**recall**, `source=study_progress`, reason *"10-min Socratic review; last seen N day(s) ago; confidence is learning"*, door `studyloop progress …`) and the micro teach-back (teachback, 100, door `--type micro` after `f2cb572f`) is the alternate beneath it; still nothing deferred but the milestone. **(d)** primary **`one tiny recall loop`** (the starter, 28, reason *"Today's energy deferred the repair work it cannot carry; start with one small retrieval signal instead"*); no alternates; `energy_deferred_repairs=[(None, decorators, high, 6)]` — where before item 5 the primary was the hands-on repair of `decorators`. **(e)** primary **`Sit with SQL Windows`** (42, reason as in (a)); the hands-on drill is the alternate at 34 (48 − 14 low-energy penalty). **(f)** (emitted from `c519bc2a` with a *planted* recall-typed due item) primary **`window function`** (**recall**, 130 = 100 + 30 overdue, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** — the same concept as primary *and* deferred repair, "because due recall is never deferred". No-plan variant: primary **`decorators`** (recall, 118, no refs). **That world cannot occur.** Emitted 2026-09-20 through **both real collectors** (owner walkthrough): the due-progress collector reads the *same* `observations.rows` as the struggle collector, labels every `struggling` row due (*"Guided repair + tiny practice"*) and emits it **hands-on**, not recall — the repair collected twice. On `c519bc2a`–`b1612918` that copy carried no demand, survived the deferral and won the dedupe, so row 3's own world emitted primary **`window function`** (**hands-on**, 140, *"Guided repair + tiny practice; last seen N day(s) ago; confidence is struggling"*) with *"repairing 'window function' … deferred"* printed beneath it and **no body double** — the row-3 "no", still on top; the no-plan world likewise (`decorators`, hands-on, 128, over the starter). After `fb63fb6b` the due copy carries the repair's demand and defers with it, named once: row 3's world emits the reading (a) screen exactly (`Sit with SQL Windows`, 42; `energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`), the no-plan world the reading (d) screen, and at **medium** energy the due copy is primary as before (hands-on, 154; nothing deferred). | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 — below every real candidate at base; at low energy it sits above only a hands-on task the energy rule already penalises (48 − 14 = 34), review 7 F2 — a proposal, not a filter) carrying the co-study session door. | **Owner verdicts — interactive walkthrough 2026-09-20: (a)–(e) yes, (f) resolved by a fix.** **(a) yes** — owner: doing the stuck thing on a genuinely low day "is very rare and only when there was a time sensitive deliverable"; that is the override case (the repair stays reachable by hand, `studyloop study "window function"`), not the default, so the floor would have given the ordinary low day a non-zero start. Row 3's original "no" (2026-09-16) stands. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** the sit-with session must not be a blank page — the body double should propose one tiny, concrete first move on the deferred material (e.g. "open the Frames lesson and read it, nothing more", or ten minutes of passive video on the next milestone). A goal-less session is the hardest ADHD start; sharpening the replacement beats restoring the rejected repair. **(b) yes** — owner: at 3/10 "I would be reluctant to do more than familiar recall"; the recall leads, the sit-with sits beneath it. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** "once I had done this, if it went well I would like the option to then take on more but not lead with the larger task" — the low-energy day should be a ladder, not a snapshot: after a completed familiar recall, offer a step-up (a re-ask of energy, or the next-cheapest action — the sit-with, then one tiny concrete move on the deferred material, per the (a) note), never leading with the larger task. Verified against `c519bc2a`: nothing does this today — the Today card fetches `/api/now` once at init with no energy and never re-plans after an action (`today-panel.js` `init()`), and CLI `now` is a one-shot snapshot of the energy it was given. The (a) and (b) notes are one 0.6.0 theme. **(c) yes, with a fallback** — owner: "3/10 or 4/10 is very subjective, I would offer the teach back but as I personally believe I would be reluctant to do more than familiar recall at 3/10 … offer a fallback to a guided explanation". So: demand `low` stands and the teach-back is offered at low energy, **and the door must carry a fallback** — if the learner cannot produce the explanation, the mentor moves to a guided explanation, not four Socratic rounds first. Against `c519bc2a`: the general stuck ladder exists (`socratic-engine.md` "Stuck (Escalating Support)" reaches a worked example only at round 4; `co-study.md` offers a brief explanation after two exchanges) but `teach-back-protocol.md` has no "cannot produce → guided explanation" step and neither rule is conditioned on low energy. **Three fixes landed on this branch before the row is committed (one RED/GREEN each), each found while checking the door behind the yes:** (1) the teach-back door's evidence command named `--type structured` — the protocol's 15-minute five-dimension review — while demand `low` was justified by the one-sentence *micro* teach-back (`micro` is in `TEACHBACK_TYPES`); the door now names the form the demand class stands on (`micro` for `low`; `structured` for a weak-teach-back `medium` row, which re-sits the review it fell short on) — RED `0760254d` / GREEN `f2cb572f`. (2) the collector's reason for a `learning` row read "Recorded as learning; repair now while the signal is fresh" (`decision.py`); design §5's own words for a `learning` row are "recovered / gentle review → `low`" and "recovered repair stays eligible as gentle review" (the spec delta: "gentle repair"), and on a mixed low-energy screen "repair now" contradicts the deferred-repair line beside it; it now reads "Recorded as learning; a gentle review keeps it fresh — one sentence, in your own words" (the struggling row's sentence is untouched) — same RED/GREEN. (3) the fallback the yes is conditional on: `teach-back-protocol.md`'s Micro Teach-Back section now carries a **Low-energy fallback** — cannot produce the sentence → a guided explanation (two or three plain sentences with a networking analogy, then one phrase said back in the student's own words), not the four-round Stuck ladder; the blank is not scored as a teach-back (it measures the day, not the concept); pinned by `test_teach_back_protocol_carries_the_low_energy_guided_explanation_fallback` — RED `c7cc1b0f` / GREEN `214df377` (`agents/manifest.json`: only the moved entry; `.secrets.baseline` rescanned whole-repo with the hook's pinned v1.5.0, 72 → 72 files). The no-plan golden stays byte-identical (`ec451ce8`); `test_now_plan_guidance` + `test_learning_decision` 76/76 before the fallback pin, 77 with it. **Note (0.6.0, not a blocker):** the capability numbers are subjective (3/10 vs 4/10); the demand thresholds are coarse by construction, so the fallback carries more weight than the threshold. **(d) yes — the starter, with a step-up** — owner, shown the starter as built (its topic is the first configured topic, not `decorators`; its one command is `studyloop progress "one tiny recall loop" …` and opens nothing): "I would" take the primary as emitted — the generic starter plus the deferred line — not the hands-on repair; and "if this went well, a successful task often increases confidence and energy so give the user the opportunity to work on the live struggle." So: the deferral stands with no plan (no council seat would restore the repair either); the generic starter is a floor the owner would start; the same-concept gentle recall (design §5 Q3) that the council and the coordinator steered toward is recorded as provenance, not adopted. **Note recorded with the yes (a 0.6.0 finding, not a blocker) — it names the top rung of the (a)/(b) ladder:** after a completed floor task, re-offer the *deferred repair itself* as the step-up — an option the learner takes while the win is fresh, never restored as the default. Against `c519bc2a`: nothing can act on "went well" today — the starter emits no outcome the engine reads, and nothing re-plans after an action (the (b) gap) — so the 0.6.0 item needs both an outcome signal from the floor task and the re-offer keyed on it. **(e) yes** — owner, 2026-09-20: sit with the plan rather than do an unrelated hands-on drill at low energy; consistent with (b) ("reluctant to do more than familiar recall at 3/10" — a coding drill is more than familiar recall), and the drill stays one tap below as the alternate (a proposal, not a filter). The engine has no signal separating a *familiar* drill from a new one (astra's caveat), so the safe default is the sit-with; per the (a)/(b) ladder the drill is the step-up to offer after the sit-with went well, not what to lead with. **(f) resolved by a fix, not a verdict** — the question as posed ("recall as primary, or a due recall on a live struggle inherits the repair's demand?") described a world the real collectors never produce: a `struggling` row's due item is the guided repair (hands-on), i.e. the same repair collected twice, and until `fb63fb6b` that copy carried no demand — so against a real sessions.db the (a), (d) and (e) primaries the owner scored were never emitted; the hands-on repair of the live struggle was, with its own deferral line beneath it (finding recorded as design §5 decision 10, RED `8f851d3a` / GREEN `fb63fb6b`; the fixture that hid it, `_plant_struggles`, silences the due collector, and `_plant_struggles_for_both_collectors` now exists beside it). The due copy now carries the repair's demand and defers with it, named once; due recall and teach-back rows are what "never deferred" was always true of. **Every reading re-emitted through both real collectors on `fb63fb6b`:** (a), (b), (d), (e) unchanged from the screens scored above; (c) differs — the due recall leads and the micro teach-back is the alternate (recorded in the emitted column). **(c) confirmed on that screen** — owner, 2026-09-20: *"With an energy score of 3/10, the initial task should be light to give confidence and hopefully by doing so energy, if the task is successful then offer the user the learning task which has a higher energy score."* The light task (the due recall) leads; the micro teach-back is the step-up offered after success — the same ladder as the (b)/(d) notes, one 0.6.0 theme (today the teach-back is visible beneath as an alternate; nothing yet re-offers it keyed on the recall's outcome). Note for that item: the owner reads the teach-back as *higher* demand than a plain recall even though both are `low` in the demand classes — the ordering already agrees (recall 151 above teach-back 100), the classes are coarse by construction and govern deferral, not order; (f) is the (a) screen at low energy and the pre-item-5 guided-repair primary at medium. **Council decision brief (2026-09-20, `receipts/rubric-3b-decision-brief-2026-09-20.md`, three seats asked for recommendations to the owner, not verdicts):** (a) YES ×3; (b) YES ×3; (c) YES / YES / EITHER (astra: `learning` is a database state, not proof of recovery — keep the teach-back brief and optional); (d) YES / EITHER / EITHER — no seat would restore the hands-on repair; the open call is whether *you* would tap a generic starter, or want a same-concept gentle recall (Q3), or want the system out of the way; (e) YES / YES / EITHER (astra: an unrelated *familiar* drill is not the class of work the finding objected to); (f) not put to the seats — emitted after they named it. The verdict column is yours; the recommendations are provenance for it, not a substitute. | | 4 | Fully-checked | Plan `done-plan` ("Done Plan"), milestones A and B both done. One due item `decorators`/python base 100. | **`decorators`** (118, no refs); `completion_actions=[(done-plan, "Every milestone of 'Done Plan' is checked off — close the plan or extend it with a follow-on mission.")]`; no `study_plan:` candidate anywhere; JSON gains `active_plans` + `completion_actions`. | Rule 9: a fully-checked plan is reported as a completion action and is neither matched (no bias, no refs) nor synthesised. | **primary yes / completion action no as phrased** — owner, 2026-09-16: the completion action must be contextual and consensual. Run the end assessment (`assess(phase="end")`: due reviews, struggles, unverified milestones on the plan's concepts). If outstanding work touches the plan's concepts (or their prerequisites — F2 concept edges), propose *extend* and name the evidence; if clean, propose *close* and ask the learner to agree ("anything you are not comfortable with?"). Status never changes automatically (#7). Natural vehicle: architect with `purpose=planning` and the assessment in the brief (`plan close <id>`, sibling of `plan repair <id>`). Finding for council. | | 4b | Fully-checked — **re-run after D-G (item 4, 2026-09-18)** | Row 4's fixture (`done-plan`, milestones A `[alpha]` and B `[beta]` both done; one due item `decorators`/python base 100), plus the end assessment's readers planted: **(a)** one due review on plan concept `alpha` (`overdue`) with session mentions backing both concepts; **(b)** no due rows, same mentions. | Primary unchanged in both: **`decorators`** (118, no refs); no `study_plan:` candidate. **(a)** `completion_actions=[(done-plan, due 1 / struggles 0 / unverified 0, proposal **extend**, evidence `["Due review: alpha — overdue"]`)]`, sentence: "Every milestone of 'Done Plan' is checked off, and the closing review proposes extending the plan — 1 due review, 0 struggles and 0 unverified milestones on its concepts. Walk the evidence with the architect: studyloop plan close done-plan." **(b)** counts 0/0/0, proposal **close**, evidence `[]`, sentence: "Every milestone of 'Done Plan' is checked off and the closing review is clean — it proposes closing the plan. Close it with the architect when you agree: studyloop plan close done-plan." No warnings; JSON gains the five keys only inside the entry. | Rule 9 as before for the ranking. The completion action is now the end assessment read as a preview (`assess(phase="end", record=False)`, one call, no write, no status change): `extend` iff any of the three counts on the plan's own concepts is above zero, else `close`; due counts only rows naming a concept (the scheduler's "new topic" row is excluded — owner decision 2026-09-17). `plan close done-plan` launches the architect with the same review as the brief's first section; status moves only when the learner agrees. | **yes / yes** — owner, 2026-09-18, answering the two questions as posed: (a) **yes**, a proposal the owner would walk; (b) **yes**, a close the owner would agree to. No further line given. Closes row 4's "no as phrased" finding; status still moves only when the learner agrees in the architect conversation. | | 5 | No-plan identical | No plan documents at all; no collector candidates. | **`one tiny recall loop`** (python, recall, 28, `source=starter`) — the starter. | D-5: `serialise(plan) == golden` → **byte-identical** (`True` in the run); the JSON key list is exactly the golden's — no additive key is present. | **verified** — owner walkthrough 2026-09-16: nothing to judge; the byte-identical golden is the acceptance. | diff --git a/openspec/changes/plan-integration-followons/tasks.md b/openspec/changes/plan-integration-followons/tasks.md index bd0cebbf4..b568e0a97 100644 --- a/openspec/changes/plan-integration-followons/tasks.md +++ b/openspec/changes/plan-integration-followons/tasks.md @@ -226,7 +226,17 @@ writer, through the existing gate, closes that. Kept out of item 3 so item 3's f contract wording (`bdf4d6c6`). Rejected with reasons: qwen's gate-deferral-on-a-plan (2–1 against, and D-F's own words), collapsing medium/high, a same-concept gentle recall floor (follow-on, decision 8). Row 3b gained readings (d) and (e) for the two places the seats split.) -- [ ] **T5.5** Rubric row **3b** for the owner (`PENDING`); the change is archived only after the owner scores 3b +- [x] **T5.5** Rubric row **3b** for the owner (`PENDING`); the change is archived only after the owner scores 3b + — **SCORED 2026-09-20** (interactive walkthrough): (a)–(e) **yes**, one line each plus three 0.6.0 notes + (concrete first move for the sit-with; the low-energy day as a ladder — step-up after a completed floor task, + re-offering the deferred repair itself at the top; capability numbers are subjective). (c)'s yes carried a + condition met on the branch: micro door for `low` demand, gentle-review wording for a `learning` row + (`0760254d`/`f2cb572f`), the protocol's low-energy guided-explanation fallback (`c7cc1b0f`/`214df377`). (f) + exposed a real defect — the due-progress collector emitted every live struggle as an undeferred hands-on + item, so the scored (a)/(d)/(e) screens were never emitted against a real database — fixed + `8f851d3a`/`fb63fb6b` (design §5 decision 10), every reading re-emitted through both collectors, and (c) + confirmed by the owner on its both-collectors screen (recall first, teach-back as the step-up). Receipt + `291e8691` +1. Archive after PR #28 merges. — **2026-09-20**: a rubric-3b council (three seats, recommendations to the owner, not verdicts) produced `receipts/rubric-3b-decision-brief-2026-09-20.md`; its two source-verified findings were applied (RED `a0946376` → GREEN `c519bc2a`, design §5 decision 9), the receipt's stale F2 claim corrected, and From 90d749c67bb8bf850b04a3f0a7350dfb29b19132 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 15:24:53 +0100 Subject: [PATCH 30/32] docs(rubric-3b): the two 0.6.0 findings are issues #30 (concrete first move) and #31 (the ladder) --- .../plan-integration/receipts/now-rubric-2026-09-16.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md index a54d5e5a8..defdff9a0 100644 --- a/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md +++ b/docs/architecture/plan-integration/receipts/now-rubric-2026-09-16.md @@ -30,7 +30,7 @@ learning"). | 1 | Matching due | Active plan `sql-windows` ("SQL Windows", topics `[sql]`, next milestone 0 concepts `[window function]`). Due items: `decorators`/python base 102, `window function`/sql base 100. | **`window function`** (sql, recall, score 130, `plan_refs=[(sql-windows, 0)]`); alternate `decorators` (120, no refs). | Rule 5: both are due items two points apart — one urgency class — so the plan-related one takes the +12 bias and wins; the unrelated due item is *kept* as an alternate (bias, not filter). Rule 7 names the milestone the action advances. | **yes** — owner, 2026-09-16: "window function is the logical step before decorating it" (a prerequisite-order argument; see F2 follow-on). | | 2 | Urgent-unrelated wins | Same plan. One due item: `decorators`/python, base 100 (an overdue spaced-repetition review). Nothing represents milestone 0. | **`decorators`** (118, no refs); alternate `window function` (60, `source=study_plan:sql-windows:0`, `plan_refs=[(sql-windows, 0)]`, reason "Next milestone 1/1 of plan 'SQL Windows': Window basics"). | Rule 5: the unrelated candidate is in a more urgent class (due review) and wins outright — the bias cannot lift new-milestone work over it. Rule 6: the plan's next milestone was unrepresented, so it was synthesised at base 48 + bias 12 = 60 and appears as the plan-backed alternate (rule 8 satisfied without any swap). | **yes** — owner, 2026-09-16: clear the overdue review first. Note for follow-on: an overdue item *unrelated* to the plan must not sit as an alternate indefinitely — propose it explicitly (age-aware nudge) or let the learner retire it. | | 3 | Energy-deferred | Plan `sql-windows` with `energy_floor: 5`; milestone 0 `Window basics` **done** (concepts `[window function]`), milestone 1 `Frames` (concepts `[window frame]`). One struggle-repair item `window function`/sql, `hands-on`, base 82. **Energy `low`** (capability 3/10). | **`window function`** (hands-on, score 80, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, milestone 1, floor 5, capability 3)]`; JSON gains `energy_deferred`. | Rule 3: capability 3 < floor 5, so the *new* milestone (Frames) is deferred and named, not synthesised; the plan-related repair on a finished milestone's concept stays eligible and keeps its ref (`None`: plan-related repair, not the next milestone). Score = 82 + 12 bias − 14 (hands-on at low energy). | **no** — owner, 2026-09-16: a struggle-repair task has no energy demand of its own; recommending hands-on repair of a *live* struggle on a low-energy day risks compounding the struggle and damaging confidence (RSD). Finding for council: (1) derive a per-item energy demand for repair from struggle recency/teach-back — at low energy a live struggle defers like new work, a recovered one stays eligible as gentle review; (2) when nothing plan-related fits the day's capability, synthesise a body-doubling / open-session candidate (feature exists: ADR-0001/0003, `web/routes/body_double.py`) naming the deferred items, instead of the least-bad task. | -| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered); added after council review 7: **(d)** **no plan at all**, one live struggle `decorators`/python, low energy; **(e)** row 3's world plus one **unrelated hands-on** practice task `list comprehension drill`/python base 48, low energy; added after the rubric-3b council (2026-09-20, two seats independently named it): **(f)** row 3's world where `window function` is **both** the live struggle and **due for recall** (`study_progress` due row, base 100) — and the same with **no plan** (`decorators` live struggle + due recall). | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason (as emitted from `c519bc2a`, after the rubric-3b council) *"1 of 2 milestones of SQL Windows done. Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session — you drive; the companion stays quiet unless you ask."* (before the council, from `326abcf9`: *"Nothing plan-related fits low energy today — deferred: …; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."* — the progress lead is the framework's naming rule; the door is now described by what the co-study persona guarantees); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone — *as emitted with the due-progress collector silenced*. Re-emitted through **both** real collectors (`fb63fb6b`, see (f)): the same `learning` row is also due, so the primary is **`window function`** (**recall**, `source=study_progress`, reason *"10-min Socratic review; last seen N day(s) ago; confidence is learning"*, door `studyloop progress …`) and the micro teach-back (teachback, 100, door `--type micro` after `f2cb572f`) is the alternate beneath it; still nothing deferred but the milestone. **(d)** primary **`one tiny recall loop`** (the starter, 28, reason *"Today's energy deferred the repair work it cannot carry; start with one small retrieval signal instead"*); no alternates; `energy_deferred_repairs=[(None, decorators, high, 6)]` — where before item 5 the primary was the hands-on repair of `decorators`. **(e)** primary **`Sit with SQL Windows`** (42, reason as in (a)); the hands-on drill is the alternate at 34 (48 − 14 low-energy penalty). **(f)** (emitted from `c519bc2a` with a *planted* recall-typed due item) primary **`window function`** (**recall**, 130 = 100 + 30 overdue, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** — the same concept as primary *and* deferred repair, "because due recall is never deferred". No-plan variant: primary **`decorators`** (recall, 118, no refs). **That world cannot occur.** Emitted 2026-09-20 through **both real collectors** (owner walkthrough): the due-progress collector reads the *same* `observations.rows` as the struggle collector, labels every `struggling` row due (*"Guided repair + tiny practice"*) and emits it **hands-on**, not recall — the repair collected twice. On `c519bc2a`–`b1612918` that copy carried no demand, survived the deferral and won the dedupe, so row 3's own world emitted primary **`window function`** (**hands-on**, 140, *"Guided repair + tiny practice; last seen N day(s) ago; confidence is struggling"*) with *"repairing 'window function' … deferred"* printed beneath it and **no body double** — the row-3 "no", still on top; the no-plan world likewise (`decorators`, hands-on, 128, over the starter). After `fb63fb6b` the due copy carries the repair's demand and defers with it, named once: row 3's world emits the reading (a) screen exactly (`Sit with SQL Windows`, 42; `energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`), the no-plan world the reading (d) screen, and at **medium** energy the due copy is primary as before (hands-on, 154; nothing deferred). | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 — below every real candidate at base; at low energy it sits above only a hands-on task the energy rule already penalises (48 − 14 = 34), review 7 F2 — a proposal, not a filter) carrying the co-study session door. | **Owner verdicts — interactive walkthrough 2026-09-20: (a)–(e) yes, (f) resolved by a fix.** **(a) yes** — owner: doing the stuck thing on a genuinely low day "is very rare and only when there was a time sensitive deliverable"; that is the override case (the repair stays reachable by hand, `studyloop study "window function"`), not the default, so the floor would have given the ordinary low day a non-zero start. Row 3's original "no" (2026-09-16) stands. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** the sit-with session must not be a blank page — the body double should propose one tiny, concrete first move on the deferred material (e.g. "open the Frames lesson and read it, nothing more", or ten minutes of passive video on the next milestone). A goal-less session is the hardest ADHD start; sharpening the replacement beats restoring the rejected repair. **(b) yes** — owner: at 3/10 "I would be reluctant to do more than familiar recall"; the recall leads, the sit-with sits beneath it. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** "once I had done this, if it went well I would like the option to then take on more but not lead with the larger task" — the low-energy day should be a ladder, not a snapshot: after a completed familiar recall, offer a step-up (a re-ask of energy, or the next-cheapest action — the sit-with, then one tiny concrete move on the deferred material, per the (a) note), never leading with the larger task. Verified against `c519bc2a`: nothing does this today — the Today card fetches `/api/now` once at init with no energy and never re-plans after an action (`today-panel.js` `init()`), and CLI `now` is a one-shot snapshot of the energy it was given. The (a) and (b) notes are one 0.6.0 theme. **(c) yes, with a fallback** — owner: "3/10 or 4/10 is very subjective, I would offer the teach back but as I personally believe I would be reluctant to do more than familiar recall at 3/10 … offer a fallback to a guided explanation". So: demand `low` stands and the teach-back is offered at low energy, **and the door must carry a fallback** — if the learner cannot produce the explanation, the mentor moves to a guided explanation, not four Socratic rounds first. Against `c519bc2a`: the general stuck ladder exists (`socratic-engine.md` "Stuck (Escalating Support)" reaches a worked example only at round 4; `co-study.md` offers a brief explanation after two exchanges) but `teach-back-protocol.md` has no "cannot produce → guided explanation" step and neither rule is conditioned on low energy. **Three fixes landed on this branch before the row is committed (one RED/GREEN each), each found while checking the door behind the yes:** (1) the teach-back door's evidence command named `--type structured` — the protocol's 15-minute five-dimension review — while demand `low` was justified by the one-sentence *micro* teach-back (`micro` is in `TEACHBACK_TYPES`); the door now names the form the demand class stands on (`micro` for `low`; `structured` for a weak-teach-back `medium` row, which re-sits the review it fell short on) — RED `0760254d` / GREEN `f2cb572f`. (2) the collector's reason for a `learning` row read "Recorded as learning; repair now while the signal is fresh" (`decision.py`); design §5's own words for a `learning` row are "recovered / gentle review → `low`" and "recovered repair stays eligible as gentle review" (the spec delta: "gentle repair"), and on a mixed low-energy screen "repair now" contradicts the deferred-repair line beside it; it now reads "Recorded as learning; a gentle review keeps it fresh — one sentence, in your own words" (the struggling row's sentence is untouched) — same RED/GREEN. (3) the fallback the yes is conditional on: `teach-back-protocol.md`'s Micro Teach-Back section now carries a **Low-energy fallback** — cannot produce the sentence → a guided explanation (two or three plain sentences with a networking analogy, then one phrase said back in the student's own words), not the four-round Stuck ladder; the blank is not scored as a teach-back (it measures the day, not the concept); pinned by `test_teach_back_protocol_carries_the_low_energy_guided_explanation_fallback` — RED `c7cc1b0f` / GREEN `214df377` (`agents/manifest.json`: only the moved entry; `.secrets.baseline` rescanned whole-repo with the hook's pinned v1.5.0, 72 → 72 files). The no-plan golden stays byte-identical (`ec451ce8`); `test_now_plan_guidance` + `test_learning_decision` 76/76 before the fallback pin, 77 with it. **Note (0.6.0, not a blocker):** the capability numbers are subjective (3/10 vs 4/10); the demand thresholds are coarse by construction, so the fallback carries more weight than the threshold. **(d) yes — the starter, with a step-up** — owner, shown the starter as built (its topic is the first configured topic, not `decorators`; its one command is `studyloop progress "one tiny recall loop" …` and opens nothing): "I would" take the primary as emitted — the generic starter plus the deferred line — not the hands-on repair; and "if this went well, a successful task often increases confidence and energy so give the user the opportunity to work on the live struggle." So: the deferral stands with no plan (no council seat would restore the repair either); the generic starter is a floor the owner would start; the same-concept gentle recall (design §5 Q3) that the council and the coordinator steered toward is recorded as provenance, not adopted. **Note recorded with the yes (a 0.6.0 finding, not a blocker) — it names the top rung of the (a)/(b) ladder:** after a completed floor task, re-offer the *deferred repair itself* as the step-up — an option the learner takes while the win is fresh, never restored as the default. Against `c519bc2a`: nothing can act on "went well" today — the starter emits no outcome the engine reads, and nothing re-plans after an action (the (b) gap) — so the 0.6.0 item needs both an outcome signal from the floor task and the re-offer keyed on it. **(e) yes** — owner, 2026-09-20: sit with the plan rather than do an unrelated hands-on drill at low energy; consistent with (b) ("reluctant to do more than familiar recall at 3/10" — a coding drill is more than familiar recall), and the drill stays one tap below as the alternate (a proposal, not a filter). The engine has no signal separating a *familiar* drill from a new one (astra's caveat), so the safe default is the sit-with; per the (a)/(b) ladder the drill is the step-up to offer after the sit-with went well, not what to lead with. **(f) resolved by a fix, not a verdict** — the question as posed ("recall as primary, or a due recall on a live struggle inherits the repair's demand?") described a world the real collectors never produce: a `struggling` row's due item is the guided repair (hands-on), i.e. the same repair collected twice, and until `fb63fb6b` that copy carried no demand — so against a real sessions.db the (a), (d) and (e) primaries the owner scored were never emitted; the hands-on repair of the live struggle was, with its own deferral line beneath it (finding recorded as design §5 decision 10, RED `8f851d3a` / GREEN `fb63fb6b`; the fixture that hid it, `_plant_struggles`, silences the due collector, and `_plant_struggles_for_both_collectors` now exists beside it). The due copy now carries the repair's demand and defers with it, named once; due recall and teach-back rows are what "never deferred" was always true of. **Every reading re-emitted through both real collectors on `fb63fb6b`:** (a), (b), (d), (e) unchanged from the screens scored above; (c) differs — the due recall leads and the micro teach-back is the alternate (recorded in the emitted column). **(c) confirmed on that screen** — owner, 2026-09-20: *"With an energy score of 3/10, the initial task should be light to give confidence and hopefully by doing so energy, if the task is successful then offer the user the learning task which has a higher energy score."* The light task (the due recall) leads; the micro teach-back is the step-up offered after success — the same ladder as the (b)/(d) notes, one 0.6.0 theme (today the teach-back is visible beneath as an alternate; nothing yet re-offers it keyed on the recall's outcome). Note for that item: the owner reads the teach-back as *higher* demand than a plain recall even though both are `low` in the demand classes — the ordering already agrees (recall 151 above teach-back 100), the classes are coarse by construction and govern deferral, not order; (f) is the (a) screen at low energy and the pre-item-5 guided-repair primary at medium. **Council decision brief (2026-09-20, `receipts/rubric-3b-decision-brief-2026-09-20.md`, three seats asked for recommendations to the owner, not verdicts):** (a) YES ×3; (b) YES ×3; (c) YES / YES / EITHER (astra: `learning` is a database state, not proof of recovery — keep the teach-back brief and optional); (d) YES / EITHER / EITHER — no seat would restore the hands-on repair; the open call is whether *you* would tap a generic starter, or want a same-concept gentle recall (Q3), or want the system out of the way; (e) YES / YES / EITHER (astra: an unrelated *familiar* drill is not the class of work the finding objected to); (f) not put to the seats — emitted after they named it. The verdict column is yours; the recommendations are provenance for it, not a substitute. | +| 3b | Energy-deferred — **re-run after D-F (item 5, 2026-09-19)** | Row 3's fixture (`sql-windows`, `energy_floor: 5`, milestone 0 `Window basics` done `[window function]`, milestone 1 `Frames` `[window frame]`; **energy `low`**, capability 3/10), with the struggle collector running for real over three readings: **(a)** `window function` recorded `struggling` 3 days ago (a live struggle); **(b)** the same plus an unrelated due recall `decorators`/python base 100; **(c)** `window function` recorded `learning` (recovered); added after council review 7: **(d)** **no plan at all**, one live struggle `decorators`/python, low energy; **(e)** row 3's world plus one **unrelated hands-on** practice task `list comprehension drill`/python base 48, low energy; added after the rubric-3b council (2026-09-20, two seats independently named it): **(f)** row 3's world where `window function` is **both** the live struggle and **due for recall** (`study_progress` due row, base 100) — and the same with **no plan** (`decorators` live struggle + due recall). | **(a)** primary **`Sit with SQL Windows`** (conversation, `source=body_double`, score 42, `plan_refs=[(sql-windows, None)]`, command `studyloop study "SQL Windows" --mode co-study`), reason (as emitted from `c519bc2a`, after the rubric-3b council) *"1 of 2 milestones of SQL Windows done. Nothing plan-related fits low energy today — deferred: milestone 2 “Frames” of SQL Windows; repair of “window function”. Sit with SQL Windows instead: a body-double session — you drive; the companion stays quiet unless you ask."* (before the council, from `326abcf9`: *"Nothing plan-related fits low energy today — deferred: …; repair of “window function”. Sit with SQL Windows instead: a body-double session, no new material, no repair."* — the progress lead is the framework's naming rule; the door is now described by what the co-study persona guarantees); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** with reason *"low energy carries 3/10; repairing 'window function' (a live struggle) asks for at least 6/10 — deferred like new work; due recall and gentle review stay available"*. **(b)** primary **`decorators`** (118, no refs); the body-double proposal is the only alternate (42). **(c)** primary **`window function`** (teachback, 100, `plan_refs=[(sql-windows, None)]`, `energy_demand=low`); nothing deferred but the milestone — *as emitted with the due-progress collector silenced*. Re-emitted through **both** real collectors (`fb63fb6b`, see (f)): the same `learning` row is also due, so the primary is **`window function`** (**recall**, `source=study_progress`, reason *"10-min Socratic review; last seen N day(s) ago; confidence is learning"*, door `studyloop progress …`) and the micro teach-back (teachback, 100, door `--type micro` after `f2cb572f`) is the alternate beneath it; still nothing deferred but the milestone. **(d)** primary **`one tiny recall loop`** (the starter, 28, reason *"Today's energy deferred the repair work it cannot carry; start with one small retrieval signal instead"*); no alternates; `energy_deferred_repairs=[(None, decorators, high, 6)]` — where before item 5 the primary was the hands-on repair of `decorators`. **(e)** primary **`Sit with SQL Windows`** (42, reason as in (a)); the hands-on drill is the alternate at 34 (48 − 14 low-energy penalty). **(f)** (emitted from `c519bc2a` with a *planted* recall-typed due item) primary **`window function`** (**recall**, 130 = 100 + 30 overdue, `plan_refs=[(sql-windows, None)]`); no alternates; `energy_deferred=[(sql-windows, 1, 5, 3)]`; **`energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`** — the same concept as primary *and* deferred repair, "because due recall is never deferred". No-plan variant: primary **`decorators`** (recall, 118, no refs). **That world cannot occur.** Emitted 2026-09-20 through **both real collectors** (owner walkthrough): the due-progress collector reads the *same* `observations.rows` as the struggle collector, labels every `struggling` row due (*"Guided repair + tiny practice"*) and emits it **hands-on**, not recall — the repair collected twice. On `c519bc2a`–`b1612918` that copy carried no demand, survived the deferral and won the dedupe, so row 3's own world emitted primary **`window function`** (**hands-on**, 140, *"Guided repair + tiny practice; last seen N day(s) ago; confidence is struggling"*) with *"repairing 'window function' … deferred"* printed beneath it and **no body double** — the row-3 "no", still on top; the no-plan world likewise (`decorators`, hands-on, 128, over the starter). After `fb63fb6b` the due copy carries the repair's demand and defers with it, named once: row 3's world emits the reading (a) screen exactly (`Sit with SQL Windows`, 42; `energy_deferred_repairs=[(sql-windows, window function, struggling, high, 6, 3)]`), the no-plan world the reading (d) screen, and at **medium** energy the due copy is primary as before (hands-on, 154; nothing deferred). | Rule 3 extended (design §5): repair carries a demand derived in the struggle collector — live `struggling` → high (6/10), older `struggling` or a weak teach-back → medium (4/10), `learning` → low (0/10) — and below the capability is deferred like new work into its own key, never ranked; due recall is never deferred. When nothing plan-related fits and an active plan exists, one body-double candidate is synthesised (base 30 + 12 bias = 42 — below every real candidate at base; at low energy it sits above only a hands-on task the energy rule already penalises (48 − 14 = 34), review 7 F2 — a proposal, not a filter) carrying the co-study session door. | **Owner verdicts — interactive walkthrough 2026-09-20: (a)–(e) yes, (f) resolved by a fix.** **(a) yes** — owner: doing the stuck thing on a genuinely low day "is very rare and only when there was a time sensitive deliverable"; that is the override case (the repair stays reachable by hand, `studyloop study "window function"`), not the default, so the floor would have given the ordinary low day a non-zero start. Row 3's original "no" (2026-09-16) stands. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** the sit-with session must not be a blank page — the body double should propose one tiny, concrete first move on the deferred material (e.g. "open the Frames lesson and read it, nothing more", or ten minutes of passive video on the next milestone). A goal-less session is the hardest ADHD start; sharpening the replacement beats restoring the rejected repair. **(b) yes** — owner: at 3/10 "I would be reluctant to do more than familiar recall"; the recall leads, the sit-with sits beneath it. **Note recorded with the yes (a 0.6.0 finding, not a blocker):** "once I had done this, if it went well I would like the option to then take on more but not lead with the larger task" — the low-energy day should be a ladder, not a snapshot: after a completed familiar recall, offer a step-up (a re-ask of energy, or the next-cheapest action — the sit-with, then one tiny concrete move on the deferred material, per the (a) note), never leading with the larger task. Verified against `c519bc2a`: nothing does this today — the Today card fetches `/api/now` once at init with no energy and never re-plans after an action (`today-panel.js` `init()`), and CLI `now` is a one-shot snapshot of the energy it was given. The (a) and (b) notes are one 0.6.0 theme. **(c) yes, with a fallback** — owner: "3/10 or 4/10 is very subjective, I would offer the teach back but as I personally believe I would be reluctant to do more than familiar recall at 3/10 … offer a fallback to a guided explanation". So: demand `low` stands and the teach-back is offered at low energy, **and the door must carry a fallback** — if the learner cannot produce the explanation, the mentor moves to a guided explanation, not four Socratic rounds first. Against `c519bc2a`: the general stuck ladder exists (`socratic-engine.md` "Stuck (Escalating Support)" reaches a worked example only at round 4; `co-study.md` offers a brief explanation after two exchanges) but `teach-back-protocol.md` has no "cannot produce → guided explanation" step and neither rule is conditioned on low energy. **Three fixes landed on this branch before the row is committed (one RED/GREEN each), each found while checking the door behind the yes:** (1) the teach-back door's evidence command named `--type structured` — the protocol's 15-minute five-dimension review — while demand `low` was justified by the one-sentence *micro* teach-back (`micro` is in `TEACHBACK_TYPES`); the door now names the form the demand class stands on (`micro` for `low`; `structured` for a weak-teach-back `medium` row, which re-sits the review it fell short on) — RED `0760254d` / GREEN `f2cb572f`. (2) the collector's reason for a `learning` row read "Recorded as learning; repair now while the signal is fresh" (`decision.py`); design §5's own words for a `learning` row are "recovered / gentle review → `low`" and "recovered repair stays eligible as gentle review" (the spec delta: "gentle repair"), and on a mixed low-energy screen "repair now" contradicts the deferred-repair line beside it; it now reads "Recorded as learning; a gentle review keeps it fresh — one sentence, in your own words" (the struggling row's sentence is untouched) — same RED/GREEN. (3) the fallback the yes is conditional on: `teach-back-protocol.md`'s Micro Teach-Back section now carries a **Low-energy fallback** — cannot produce the sentence → a guided explanation (two or three plain sentences with a networking analogy, then one phrase said back in the student's own words), not the four-round Stuck ladder; the blank is not scored as a teach-back (it measures the day, not the concept); pinned by `test_teach_back_protocol_carries_the_low_energy_guided_explanation_fallback` — RED `c7cc1b0f` / GREEN `214df377` (`agents/manifest.json`: only the moved entry; `.secrets.baseline` rescanned whole-repo with the hook's pinned v1.5.0, 72 → 72 files). The no-plan golden stays byte-identical (`ec451ce8`); `test_now_plan_guidance` + `test_learning_decision` 76/76 before the fallback pin, 77 with it. **Note (0.6.0, not a blocker):** the capability numbers are subjective (3/10 vs 4/10); the demand thresholds are coarse by construction, so the fallback carries more weight than the threshold. **(d) yes — the starter, with a step-up** — owner, shown the starter as built (its topic is the first configured topic, not `decorators`; its one command is `studyloop progress "one tiny recall loop" …` and opens nothing): "I would" take the primary as emitted — the generic starter plus the deferred line — not the hands-on repair; and "if this went well, a successful task often increases confidence and energy so give the user the opportunity to work on the live struggle." So: the deferral stands with no plan (no council seat would restore the repair either); the generic starter is a floor the owner would start; the same-concept gentle recall (design §5 Q3) that the council and the coordinator steered toward is recorded as provenance, not adopted. **Note recorded with the yes (a 0.6.0 finding, not a blocker) — it names the top rung of the (a)/(b) ladder:** after a completed floor task, re-offer the *deferred repair itself* as the step-up — an option the learner takes while the win is fresh, never restored as the default. Against `c519bc2a`: nothing can act on "went well" today — the starter emits no outcome the engine reads, and nothing re-plans after an action (the (b) gap) — so the 0.6.0 item needs both an outcome signal from the floor task and the re-offer keyed on it. **(e) yes** — owner, 2026-09-20: sit with the plan rather than do an unrelated hands-on drill at low energy; consistent with (b) ("reluctant to do more than familiar recall at 3/10" — a coding drill is more than familiar recall), and the drill stays one tap below as the alternate (a proposal, not a filter). The engine has no signal separating a *familiar* drill from a new one (astra's caveat), so the safe default is the sit-with; per the (a)/(b) ladder the drill is the step-up to offer after the sit-with went well, not what to lead with. **(f) resolved by a fix, not a verdict** — the question as posed ("recall as primary, or a due recall on a live struggle inherits the repair's demand?") described a world the real collectors never produce: a `struggling` row's due item is the guided repair (hands-on), i.e. the same repair collected twice, and until `fb63fb6b` that copy carried no demand — so against a real sessions.db the (a), (d) and (e) primaries the owner scored were never emitted; the hands-on repair of the live struggle was, with its own deferral line beneath it (finding recorded as design §5 decision 10, RED `8f851d3a` / GREEN `fb63fb6b`; the fixture that hid it, `_plant_struggles`, silences the due collector, and `_plant_struggles_for_both_collectors` now exists beside it). The due copy now carries the repair's demand and defers with it, named once; due recall and teach-back rows are what "never deferred" was always true of. **Every reading re-emitted through both real collectors on `fb63fb6b`:** (a), (b), (d), (e) unchanged from the screens scored above; (c) differs — the due recall leads and the micro teach-back is the alternate (recorded in the emitted column). **(c) confirmed on that screen** — owner, 2026-09-20: *"With an energy score of 3/10, the initial task should be light to give confidence and hopefully by doing so energy, if the task is successful then offer the user the learning task which has a higher energy score."* The light task (the due recall) leads; the micro teach-back is the step-up offered after success — the same ladder as the (b)/(d) notes, one 0.6.0 theme — filed as issue #31 (ladder) beside #30 (the concrete first move, from the (a) note) (today the teach-back is visible beneath as an alternate; nothing yet re-offers it keyed on the recall's outcome). Note for that item: the owner reads the teach-back as *higher* demand than a plain recall even though both are `low` in the demand classes — the ordering already agrees (recall 151 above teach-back 100), the classes are coarse by construction and govern deferral, not order; (f) is the (a) screen at low energy and the pre-item-5 guided-repair primary at medium. **Council decision brief (2026-09-20, `receipts/rubric-3b-decision-brief-2026-09-20.md`, three seats asked for recommendations to the owner, not verdicts):** (a) YES ×3; (b) YES ×3; (c) YES / YES / EITHER (astra: `learning` is a database state, not proof of recovery — keep the teach-back brief and optional); (d) YES / EITHER / EITHER — no seat would restore the hands-on repair; the open call is whether *you* would tap a generic starter, or want a same-concept gentle recall (Q3), or want the system out of the way; (e) YES / YES / EITHER (astra: an unrelated *familiar* drill is not the class of work the finding objected to); (f) not put to the seats — emitted after they named it. The verdict column is yours; the recommendations are provenance for it, not a substitute. | | 4 | Fully-checked | Plan `done-plan` ("Done Plan"), milestones A and B both done. One due item `decorators`/python base 100. | **`decorators`** (118, no refs); `completion_actions=[(done-plan, "Every milestone of 'Done Plan' is checked off — close the plan or extend it with a follow-on mission.")]`; no `study_plan:` candidate anywhere; JSON gains `active_plans` + `completion_actions`. | Rule 9: a fully-checked plan is reported as a completion action and is neither matched (no bias, no refs) nor synthesised. | **primary yes / completion action no as phrased** — owner, 2026-09-16: the completion action must be contextual and consensual. Run the end assessment (`assess(phase="end")`: due reviews, struggles, unverified milestones on the plan's concepts). If outstanding work touches the plan's concepts (or their prerequisites — F2 concept edges), propose *extend* and name the evidence; if clean, propose *close* and ask the learner to agree ("anything you are not comfortable with?"). Status never changes automatically (#7). Natural vehicle: architect with `purpose=planning` and the assessment in the brief (`plan close <id>`, sibling of `plan repair <id>`). Finding for council. | | 4b | Fully-checked — **re-run after D-G (item 4, 2026-09-18)** | Row 4's fixture (`done-plan`, milestones A `[alpha]` and B `[beta]` both done; one due item `decorators`/python base 100), plus the end assessment's readers planted: **(a)** one due review on plan concept `alpha` (`overdue`) with session mentions backing both concepts; **(b)** no due rows, same mentions. | Primary unchanged in both: **`decorators`** (118, no refs); no `study_plan:` candidate. **(a)** `completion_actions=[(done-plan, due 1 / struggles 0 / unverified 0, proposal **extend**, evidence `["Due review: alpha — overdue"]`)]`, sentence: "Every milestone of 'Done Plan' is checked off, and the closing review proposes extending the plan — 1 due review, 0 struggles and 0 unverified milestones on its concepts. Walk the evidence with the architect: studyloop plan close done-plan." **(b)** counts 0/0/0, proposal **close**, evidence `[]`, sentence: "Every milestone of 'Done Plan' is checked off and the closing review is clean — it proposes closing the plan. Close it with the architect when you agree: studyloop plan close done-plan." No warnings; JSON gains the five keys only inside the entry. | Rule 9 as before for the ranking. The completion action is now the end assessment read as a preview (`assess(phase="end", record=False)`, one call, no write, no status change): `extend` iff any of the three counts on the plan's own concepts is above zero, else `close`; due counts only rows naming a concept (the scheduler's "new topic" row is excluded — owner decision 2026-09-17). `plan close done-plan` launches the architect with the same review as the brief's first section; status moves only when the learner agrees. | **yes / yes** — owner, 2026-09-18, answering the two questions as posed: (a) **yes**, a proposal the owner would walk; (b) **yes**, a close the owner would agree to. No further line given. Closes row 4's "no as phrased" finding; status still moves only when the learner agrees in the architect conversation. | | 5 | No-plan identical | No plan documents at all; no collector candidates. | **`one tiny recall loop`** (python, recall, 28, `source=starter`) — the starter. | D-5: `serialise(plan) == golden` → **byte-identical** (`True` in the run); the JSON key list is exactly the golden's — no additive key is present. | **verified** — owner walkthrough 2026-09-16: nothing to judge; the byte-identical golden is the acceptance. | From 60d5f94ca9c6c7b10501b0b5e5559f8e2c903028 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 16:11:02 +0100 Subject: [PATCH 31/32] =?UTF-8?q?test(web):=20RED=20=E2=80=94=20the=20time?= =?UTF-8?q?r's=20two=20init()=20runs=20make=20two=20state=20reads;=20a=20l?= =?UTF-8?q?ate=20one=20adopts=20a=20session=20into=20a=20settled=20picker?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The e2e test_409_from_a_second_tab click has timed out three times on CI (35341660469, 35380095596, 35516418191). The third run carried the capture added after the second, and its artifact named the mechanism: at the click, sessionActive was true, topic was tab A's "Study focus", conflictSession was null and the picker was display:none — with startSession() never run. The only click-less path that sets topic from server state is init()'s restore branch, and init() runs twice per page load (Alpine's auto-init plus the markup's x-init="init()"), each run issuing its own /api/session/state read. The settled wait observed the first; the second landed after the test's POST /api/session/start and adopted tab A's session into tab B. This test replays that world without a browser: two init() calls, the second state read held until a live session exists, then released. It fails today on both counts — two reads, and sessionActive flipped true on a picker the learner had settled on. --- .../tests/js/plan-architect-launch.test.js | 45 +++++++++++++++++++ 1 file changed, 45 insertions(+) diff --git a/packages/studyloop/tests/js/plan-architect-launch.test.js b/packages/studyloop/tests/js/plan-architect-launch.test.js index a6af3bdf4..d69f7d1ff 100644 --- a/packages/studyloop/tests/js/plan-architect-launch.test.js +++ b/packages/studyloop/tests/js/plan-architect-launch.test.js @@ -361,6 +361,51 @@ test('init() twice (Alpine auto-init + x-init="init()") still means one listener assert.equal(starts.length, 1); }); +test('init() twice reads /api/session/state ONCE: a settled picker is not flipped by a late read it never asked for', async () => { + /* The same double run, the other fetch. Each init() issued its own + /api/session/state read; the picker settled ("No active session") on + whichever landed first, and the OTHER read could land after a session had + since been started elsewhere -- in a second tab, over plain HTTP -- and + adopt it: sessionActive true, topic from the server, no click, no 409. + That is the state CI run 35516418191 captured at the Start click in + test_409_from_a_second_tab (e2e): the live layout covered the button, + then the picker hid. Hold the second read until the session exists and + replay it. One read per page load, however many times the page calls + init(). */ + const baseFetch = globalThis.fetch; + let stateReads = 0; + let releaseSecondRead; + const secondReadHeld = new Promise((resolve) => { releaseSecondRead = resolve; }); + globalThis.fetch = async (url, opts) => { + if (!String(url).endsWith('/api/session/state')) return baseFetch(url, opts); + stateReads += 1; + if (stateReads === 1) return jsonResponse(200, {}); + await secondReadHeld; + return jsonResponse(200, { + study_session_id: 'study-A', topic: 'Study focus', energy: 5, agent: 'codex', + mode: 'active', origin: 'study', start_time: '2026-09-20T14:30:00Z', + }); + }; + + const timer = sessionTimer(); + timers.push(timer); + timer.$nextTick = (cb) => cb(); + const firstInit = timer.init(); + const secondInit = timer.init(); + await firstInit; + assert.equal(timer.topic, 'No active session', 'tab B settled on the picker'); + assert.equal(timer.sessionActive, false); + + releaseSecondRead(); // tab A's session exists from here on + await secondInit; + await settle(); + + assert.equal(stateReads, 1, 'one /api/session/state read per page load'); + assert.equal(timer.sessionActive, false, + 'the picker tab B settled on stays a picker; its Start must reach the server and 409'); + assert.equal(timer.topic, 'No active session'); +}); + test('a 409 on a planning launch keeps the existing conflict handling and reports failure', async () => { const timer = await readyTimer(); const results = []; From 96806febdd03909d472a654144ecbf733493d423 Mon Sep 17 00:00:00 2001 From: NetDevAutomate <andy.taylor@mail.com> Date: Sun, 20 Sep 2026 16:11:19 +0100 Subject: [PATCH 32/32] fix(web): the session timer's init() runs once per page load MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Alpine calls init() for an x-data object that defines one, and the markup also says x-init="init()", so the page called it twice. The earlier fix (#14 journey) made only the listener registration idempotent and left the fetches to run again as "idempotent". The RED before this commit shows they are not: two /api/session/state reads race, the picker settles on the first, and the second can land after a session was started elsewhere and adopt it — the CI artifact of run 35516418191, byte for byte. init() is now a once-per-instance gate over the former body (_initOnce): later calls return the first run's promise, so the settled picker is the one and only read, and the _listenersRegistered guard is subsumed. The e2e click-site comment records what the artifact showed and keeps the capture for any fourth occurrence, which would have to be a different mechanism. Verified on this branch: node --test 148/148 (147 + the RED, now green); the recovery journey e2e file 9 passed / 2 skipped in order; ruff clean. --- .../web/static/js/components/session-timer.js | 36 ++++++++++++------- .../e2e/test_session_recovery_journey.py | 21 ++++++----- 2 files changed, 37 insertions(+), 20 deletions(-) diff --git a/packages/studyloop/src/studyloop/web/static/js/components/session-timer.js b/packages/studyloop/src/studyloop/web/static/js/components/session-timer.js index f71f38dac..9e6bfa45e 100644 --- a/packages/studyloop/src/studyloop/web/static/js/components/session-timer.js +++ b/packages/studyloop/src/studyloop/web/static/js/components/session-timer.js @@ -132,6 +132,9 @@ export function sessionTimer() { user acts on the conflict, so an /api/session/state response that was already in flight cannot overwrite the newer state. */ _conflictEpoch: 0, + /* The one init() run of this page load (see init()). Null until the + first call; every later call returns it instead of starting another. */ + _initPromise: null, energyBandLabel() { if (this.energy >= 7) return 'High energy'; @@ -165,18 +168,27 @@ export function sessionTimer() { }); }, - async init() { + init() { // Alpine calls init() itself for an x-data object that defines one, and - // the markup ALSO says x-init="init()", so this runs twice per page - // load. That was harmless while the listeners below only set picker - // fields; it is not once one of them starts a session — two listeners - // meant two POSTs per click and a 409 for the second (found by the #14 - // browser journey). Register the window listeners exactly once; the - // fetches below are idempotent and may run again. - if (!this._listenersRegistered) { - this._listenersRegistered = true; - this._registerWindowListeners(); - } + // the markup ALSO says x-init="init()", so the page calls this twice per + // load. Two listeners were the first cost of that (two POSTs per click + // and a 409 for the second, found by the #14 browser journey); the + // fetches were believed idempotent and left to run again. They are not: + // each run issued its own /api/session/state read, the picker settled + // on whichever landed first, and the OTHER read could land after a + // session had since been started elsewhere -- a second tab, over plain + // HTTP -- and adopt it into a tab that was sitting on the picker: no + // click, no 409, Start hidden under the live layout. That is the state + // CI run 35516418191 captured at the click (e2e, test_409_from_a_second + // _tab, the third such timeout and the first with evidence). So init() + // runs ONCE per page load, whoever calls it: later calls get the first + // run's promise. The settled picker is then the one and only read. + if (!this._initPromise) this._initPromise = this._initOnce(); + return this._initPromise; + }, + + async _initOnce() { + this._registerWindowListeners(); const optionsPromise = fetch('/api/session/options') .then((res) => res.ok ? res.json() : null) @@ -293,7 +305,7 @@ export function sessionTimer() { /* A planning launch can arrive from the Plans view before init()'s options fetch has resolved; the agent is not missing, it is not yet known. Wait for the picker's own settlement before deciding. - (init() sets _optionsReady on every run; a timer whose init never + (init()'s one run sets _optionsReady; a timer whose init never ran has nothing to wait for and falls through to the check.) */ if (purpose === 'planning' && !this.agent && this._optionsReady) { /* Bounded (council review 6): a fetch that never settles must not hold diff --git a/packages/studyloop/tests/e2e/test_session_recovery_journey.py b/packages/studyloop/tests/e2e/test_session_recovery_journey.py index 7ee1c31a5..6a516d016 100644 --- a/packages/studyloop/tests/e2e/test_session_recovery_journey.py +++ b/packages/studyloop/tests/e2e/test_session_recovery_journey.py @@ -347,14 +347,19 @@ def test_409_from_a_second_tab_offers_reattach_that_adopts_the_session( d.agent = 'codex'; }""" ) - # Twice on CI this click has timed out with "element is not visible" - # (runs 35341660469 and 35380095596) after the settled wait above passed, - # and neither run produced evidence of WHY the picker was hidden: the - # timeout fires before this test's own _diag hook, so the artifact held - # nothing for it. Every adopt path in the timer was read against the - # second failure and none explains it without a click. Capture the - # state at the click so a third occurrence names the mechanism instead - # of the symptom -- the fix must rest on that, not on a guess. + # Three times on CI this click timed out with "element is not visible" + # (runs 35341660469, 35380095596, 35516418191) after the settled wait + # above passed. The first two left no evidence -- the timeout fired + # before this test's _diag hook -- so the capture below was added, and + # the third run's artifact named the mechanism: sessionActive true, + # topic "Study focus" (tab A's), conflict null, picker display none, + # with startSession() never run. The timer's init() ran twice per page + # load (Alpine's auto-init plus x-init="init()"), each run issued its own + # /api/session/state read, the settled wait observed the first, and the + # second landed after _start_session above and adopted tab A's session + # into this tab. Fixed at the source -- init() is one run per page load, + # so the settled picker IS the only read. The capture stays: a fourth + # occurrence would mean a different mechanism, and it must name it. try: page.locator("[data-testid='study-start-session']").click() except Exception: # pragma: no cover - diagnostics only